google scraper
This commit is contained in:
parent
5060643a3f
commit
a195d3b8fe
36 changed files with 988 additions and 238 deletions
2
.env
2
.env
|
|
@ -13,4 +13,4 @@ OPENAI_API_KEY=sk-GtdLX9YBOCsBEgBL4rBONA
|
||||||
OPENAI_BASE_URL=https://hubai.loe.gg/v1
|
OPENAI_BASE_URL=https://hubai.loe.gg/v1
|
||||||
OPENAI_MODEL=gpt-4o-mini
|
OPENAI_MODEL=gpt-4o-mini
|
||||||
CV_AI_MODEL=your_ai_model
|
CV_AI_MODEL=your_ai_model
|
||||||
OPENROUTER_API_KEY=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
|
OPENROUTER_API_KEY=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
|
||||||
18
Makefile
18
Makefile
|
|
@ -5,6 +5,7 @@ APP_NAME=jobs-scraper
|
||||||
API_DIR=./services/api
|
API_DIR=./services/api
|
||||||
SCRAPER_LINKEDIN_DIR=./services/scraper-linkedin
|
SCRAPER_LINKEDIN_DIR=./services/scraper-linkedin
|
||||||
SCRAPER_GLASSDOOR_DIR=./services/scraper-glassdoor
|
SCRAPER_GLASSDOOR_DIR=./services/scraper-glassdoor
|
||||||
|
SCRAPER_PLAYWRIGHT_DIR=./services/scraper-playwright
|
||||||
CRON_ANALYZER_DIR=./apps/cron-analyzer
|
CRON_ANALYZER_DIR=./apps/cron-analyzer
|
||||||
CV_ANALYZER_DIR=./apps/cv-analyzer
|
CV_ANALYZER_DIR=./apps/cv-analyzer
|
||||||
DOCS_DIR=./services/api/pkg/swagger
|
DOCS_DIR=./services/api/pkg/swagger
|
||||||
|
|
@ -53,6 +54,23 @@ build-scraper-glassdoor: ## Build the Glassdoor scraper service
|
||||||
@echo "Building Glassdoor scraper service..."
|
@echo "Building Glassdoor scraper service..."
|
||||||
@cd $(SCRAPER_GLASSDOOR_DIR) && npm run build
|
@cd $(SCRAPER_GLASSDOOR_DIR) && npm run build
|
||||||
|
|
||||||
|
.PHONY: setup-scraper-playwright
|
||||||
|
setup-scraper-playwright: ## Setup the Playwright scraper service
|
||||||
|
@echo "Setting up Playwright scraper service..."
|
||||||
|
@cd $(SCRAPER_PLAYWRIGHT_DIR) && bash setup.sh
|
||||||
|
|
||||||
|
.PHONY: run-scraper-playwright
|
||||||
|
run-scraper-playwright: ## Run the Playwright scraper service
|
||||||
|
@echo "Running Playwright scraper service..."
|
||||||
|
@cd $(SCRAPER_PLAYWRIGHT_DIR) && source venv/bin/activate && python playwright_worker.py
|
||||||
|
|
||||||
|
.PHONY: docker-scraper-playwright
|
||||||
|
docker-scraper-playwright: ## Build and run Playwright scraper in Docker
|
||||||
|
@echo "Building Docker image for Playwright scraper..."
|
||||||
|
@cd $(SCRAPER_PLAYWRIGHT_DIR) && docker build -t playwright-scraper .
|
||||||
|
@echo "Running Playwright scraper in Docker..."
|
||||||
|
@docker run --rm --name playwright-scraper playwright-scraper
|
||||||
|
|
||||||
# Apps
|
# Apps
|
||||||
.PHONY: run-cron-analyzer
|
.PHONY: run-cron-analyzer
|
||||||
run-cron-analyzer: ## Run the job analysis cron service
|
run-cron-analyzer: ## Run the job analysis cron service
|
||||||
|
|
|
||||||
|
|
@ -9,7 +9,7 @@ SERVER_PORT=8080
|
||||||
SERVER_HOST=localhost
|
SERVER_HOST=localhost
|
||||||
|
|
||||||
CV_AI_MODEL=your_ai_model
|
CV_AI_MODEL=your_ai_model
|
||||||
OPENROUTER_API_KEY=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
|
OPENROUTER_API_KEY=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
|
||||||
|
|
||||||
# OpenAI Configuration
|
# OpenAI Configuration
|
||||||
OPENAI_API_KEY=sk-GtdLX9YBOCsBEgBL4rBONA
|
OPENAI_API_KEY=sk-GtdLX9YBOCsBEgBL4rBONA
|
||||||
|
|
|
||||||
|
|
@ -9,7 +9,7 @@ SERVER_PORT=8080
|
||||||
SERVER_HOST=localhost
|
SERVER_HOST=localhost
|
||||||
|
|
||||||
CV_AI_MODEL=your_ai_model
|
CV_AI_MODEL=your_ai_model
|
||||||
OPENROUTER_API_KEY=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
|
OPENROUTER_API_KEY=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
|
||||||
|
|
||||||
# OpenAI Configuration
|
# OpenAI Configuration
|
||||||
OPENAI_API_KEY=sk-GtdLX9YBOCsBEgBL4rBONA
|
OPENAI_API_KEY=sk-GtdLX9YBOCsBEgBL4rBONA
|
||||||
|
|
|
||||||
|
|
@ -22,6 +22,7 @@ services:
|
||||||
environment:
|
environment:
|
||||||
RABBITMQ_DEFAULT_USER: guest
|
RABBITMQ_DEFAULT_USER: guest
|
||||||
RABBITMQ_DEFAULT_PASS: guest
|
RABBITMQ_DEFAULT_PASS: guest
|
||||||
|
RABBITMQ_ERLANG_COOKIE: "SWQOKODSQALRPCLNMEQGABCDEFGHIJKLMNOPQRST"
|
||||||
ports:
|
ports:
|
||||||
- "5672:5672"
|
- "5672:5672"
|
||||||
- "15672:15672"
|
- "15672:15672"
|
||||||
|
|
@ -32,6 +33,34 @@ services:
|
||||||
interval: 10s
|
interval: 10s
|
||||||
timeout: 5s
|
timeout: 5s
|
||||||
retries: 5
|
retries: 5
|
||||||
|
|
||||||
|
redis:
|
||||||
|
image: redis:7-alpine
|
||||||
|
ports:
|
||||||
|
- "6379:6379"
|
||||||
|
volumes:
|
||||||
|
- redis_data:/data
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD", "redis-cli", "ping"]
|
||||||
|
interval: 10s
|
||||||
|
timeout: 5s
|
||||||
|
retries: 5
|
||||||
|
|
||||||
|
scraper-playwright:
|
||||||
|
build:
|
||||||
|
context: ./services/scraper-playwright
|
||||||
|
dockerfile: Dockerfile
|
||||||
|
environment:
|
||||||
|
REDIS_HOST: redis
|
||||||
|
REDIS_PORT: 6379
|
||||||
|
WORKER_ID: playwright_1
|
||||||
|
LOG_LEVEL: INFO
|
||||||
|
depends_on:
|
||||||
|
redis:
|
||||||
|
condition: service_healthy
|
||||||
|
restart: unless-stopped
|
||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
postgres_data:
|
postgres_data:
|
||||||
rabbitmq_data:
|
rabbitmq_data:
|
||||||
|
redis_data:
|
||||||
|
|
|
||||||
1
go.work
1
go.work
|
|
@ -12,5 +12,6 @@ use (
|
||||||
./libs/repo
|
./libs/repo
|
||||||
./libs/server
|
./libs/server
|
||||||
./services/api
|
./services/api
|
||||||
|
./services/scraper-google
|
||||||
./services/scraper-linkedin
|
./services/scraper-linkedin
|
||||||
)
|
)
|
||||||
|
|
|
||||||
9
internal/pkg/domain/google-link.go
Normal file
9
internal/pkg/domain/google-link.go
Normal file
|
|
@ -0,0 +1,9 @@
|
||||||
|
package domain
|
||||||
|
|
||||||
|
type GoogleLink struct {
|
||||||
|
ID int64
|
||||||
|
Title string
|
||||||
|
Description string
|
||||||
|
Link string
|
||||||
|
CompanyName string
|
||||||
|
}
|
||||||
|
|
@ -14,6 +14,7 @@ const (
|
||||||
Bayt
|
Bayt
|
||||||
TokyoDev
|
TokyoDev
|
||||||
JapanDev
|
JapanDev
|
||||||
|
Google
|
||||||
)
|
)
|
||||||
|
|
||||||
const (
|
const (
|
||||||
|
|
|
||||||
|
|
@ -17,6 +17,7 @@ const (
|
||||||
TokyoDevQueue = "scraper.tokyodev"
|
TokyoDevQueue = "scraper.tokyodev"
|
||||||
GlassDoorQueue = "scraper.glassdoor"
|
GlassDoorQueue = "scraper.glassdoor"
|
||||||
JapanDevQueue = "scraper.japandev"
|
JapanDevQueue = "scraper.japandev"
|
||||||
|
GoogleQueue = "scraper.google"
|
||||||
DeadLetterExchange = "scraper_dlx"
|
DeadLetterExchange = "scraper_dlx"
|
||||||
CvAnalyzeExchange = "cv_exchange"
|
CvAnalyzeExchange = "cv_exchange"
|
||||||
CvAnalyzeQueue = "cv.analyze"
|
CvAnalyzeQueue = "cv.analyze"
|
||||||
|
|
@ -112,6 +113,7 @@ func (r *RabbitMQClient) setupInfrastructure() error {
|
||||||
TokyoDevQueue: "scraper.tokyodev",
|
TokyoDevQueue: "scraper.tokyodev",
|
||||||
JapanDevQueue: "scraper.japandev",
|
JapanDevQueue: "scraper.japandev",
|
||||||
GlassDoorQueue: "scraper.glassdoor",
|
GlassDoorQueue: "scraper.glassdoor",
|
||||||
|
GoogleQueue: "scraper.google",
|
||||||
}
|
}
|
||||||
|
|
||||||
// Declare queues with dead letter exchange and TTL
|
// Declare queues with dead letter exchange and TTL
|
||||||
|
|
|
||||||
|
|
@ -256,6 +256,88 @@ func min(a, b int) int {
|
||||||
return b
|
return b
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// CreateJobDescription extracts structured job information from HTML content
|
||||||
|
func (s *OpenAIService) CreateJobDescription(htmlContent string) (*domain.JobDescription, error) {
|
||||||
|
prompt := fmt.Sprintf(`Extract job posting information from the following HTML content.
|
||||||
|
|
||||||
|
CRITICAL RULES:
|
||||||
|
1. Return ONLY valid JSON - no markdown, no code blocks, no backticks, no explanations
|
||||||
|
2. Use SINGLE-LINE JSON (no pretty printing, no newlines inside the JSON)
|
||||||
|
3. All criteria values MUST be strings (never use booleans, numbers, or arrays)
|
||||||
|
4. Properly escape all quotes inside strings using \"
|
||||||
|
5. Use only standard ASCII quotes ("), never smart quotes (" " ' ')
|
||||||
|
|
||||||
|
Required fields:
|
||||||
|
- description: Full job description as a single string
|
||||||
|
- criteria: Object with keys like title, company, location, salary, skills, experience, job_type, remote
|
||||||
|
|
||||||
|
Example (SINGLE LINE):
|
||||||
|
{"description":"Full job description text here...","criteria":{"title":"Software Engineer","location":"City, Country","salary":"$100k-$150k","skills":"Go, Python, Docker","experience":"3-5 years","job_type":"Full-time","remote":"Remote"}}
|
||||||
|
|
||||||
|
HTML Content:
|
||||||
|
%s`, htmlContent)
|
||||||
|
|
||||||
|
req := openai.ChatCompletionRequest{
|
||||||
|
Model: s.model,
|
||||||
|
Messages: []openai.ChatCompletionMessage{
|
||||||
|
{
|
||||||
|
Role: openai.ChatMessageRoleSystem,
|
||||||
|
Content: "You are a JSON extraction bot. You ONLY output valid single-line JSON. No markdown. No code blocks. No explanations. No pretty printing. Just raw minified JSON starting with { and ending with }. Always use standard ASCII double quotes (\"), never smart quotes.",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
Role: openai.ChatMessageRoleUser,
|
||||||
|
Content: prompt,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
Temperature: 0.1, // Low temperature for consistent structured output
|
||||||
|
MaxTokens: 2000,
|
||||||
|
}
|
||||||
|
|
||||||
|
resp, err := s.client.CreateChatCompletion(context.Background(), req)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("failed to execute completion: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(resp.Choices) == 0 {
|
||||||
|
return nil, fmt.Errorf("no response choices received from API")
|
||||||
|
}
|
||||||
|
|
||||||
|
jsonContent := resp.Choices[0].Message.Content
|
||||||
|
|
||||||
|
log.Printf("Raw API response for job description: %s", jsonContent)
|
||||||
|
|
||||||
|
jsonContent = cleanMarkdownCodeBlocks(jsonContent)
|
||||||
|
|
||||||
|
log.Printf("Cleaned JSON (first 500 chars): %s", jsonContent[:min(500, len(jsonContent))])
|
||||||
|
|
||||||
|
var result struct {
|
||||||
|
Description string `json:"description"`
|
||||||
|
Criteria map[string]string `json:"criteria"`
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := json.Unmarshal([]byte(jsonContent), &result); err != nil {
|
||||||
|
if syntaxErr, ok := err.(*json.SyntaxError); ok {
|
||||||
|
start := int(syntaxErr.Offset) - 50
|
||||||
|
if start < 0 {
|
||||||
|
start = 0
|
||||||
|
}
|
||||||
|
end := int(syntaxErr.Offset) + 50
|
||||||
|
if end > len(jsonContent) {
|
||||||
|
end = len(jsonContent)
|
||||||
|
}
|
||||||
|
log.Printf("JSON parse error at position %d, context: ...%s...", syntaxErr.Offset, jsonContent[start:end])
|
||||||
|
}
|
||||||
|
return nil, fmt.Errorf("failed to parse JSON response: %v. Cleaned JSON length: %d", err, len(jsonContent))
|
||||||
|
}
|
||||||
|
|
||||||
|
jobDescription := &domain.JobDescription{
|
||||||
|
Description: result.Description,
|
||||||
|
Criteria: result.Criteria,
|
||||||
|
}
|
||||||
|
|
||||||
|
return jobDescription, nil
|
||||||
|
}
|
||||||
|
|
||||||
// cleanMarkdownCodeBlocks removes markdown code block formatting from JSON response
|
// cleanMarkdownCodeBlocks removes markdown code block formatting from JSON response
|
||||||
func cleanMarkdownCodeBlocks(content string) string {
|
func cleanMarkdownCodeBlocks(content string) string {
|
||||||
// Remove ```json prefix and ``` suffix if present
|
// Remove ```json prefix and ``` suffix if present
|
||||||
|
|
|
||||||
|
|
@ -131,23 +131,97 @@ func min(a, b int) int {
|
||||||
return b
|
return b
|
||||||
}
|
}
|
||||||
|
|
||||||
// cleanMarkdownCodeBlocks removes markdown code block formatting from JSON response
|
// cleanMarkdownCodeBlocks removes only markdown code block formatting
|
||||||
func cleanMarkdownCodeBlocks(content string) string {
|
func cleanMarkdownCodeBlocks(content string) string {
|
||||||
// Remove ```json prefix and ``` suffix if present
|
|
||||||
content = strings.TrimSpace(content)
|
content = strings.TrimSpace(content)
|
||||||
|
|
||||||
// Remove opening code block markers
|
// Remove markdown code blocks
|
||||||
if strings.HasPrefix(content, "```json") {
|
if strings.HasPrefix(content, "```json") {
|
||||||
content = strings.TrimPrefix(content, "```json")
|
content = strings.TrimPrefix(content, "```json")
|
||||||
} else if strings.HasPrefix(content, "```") {
|
} else if strings.HasPrefix(content, "```") {
|
||||||
content = strings.TrimPrefix(content, "```")
|
content = strings.TrimPrefix(content, "```")
|
||||||
}
|
}
|
||||||
|
|
||||||
// Remove closing code block markers
|
|
||||||
content = strings.TrimSuffix(content, "```")
|
content = strings.TrimSuffix(content, "```")
|
||||||
|
|
||||||
return strings.TrimSpace(content)
|
return strings.TrimSpace(content)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (s *OpenRouterService) CreateJobDescription(htmlContent string) (*domain.JobDescription, error) {
|
||||||
|
client, err := openroutergo.
|
||||||
|
NewClient().
|
||||||
|
WithAPIKey(s.apiKey).
|
||||||
|
Create()
|
||||||
|
if err != nil {
|
||||||
|
log.Fatalf("Failed to create client: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
prompt := fmt.Sprintf(`Extract job posting information from the following HTML content.
|
||||||
|
|
||||||
|
CRITICAL RULES:
|
||||||
|
1. Return ONLY valid JSON - no markdown, no code blocks, no backticks, no explanations
|
||||||
|
2. Use SINGLE-LINE JSON (no pretty printing, no newlines inside the JSON)
|
||||||
|
3. All criteria values MUST be strings (never use booleans, numbers, or arrays)
|
||||||
|
4. Properly escape all quotes inside strings using \"
|
||||||
|
5. Use only standard ASCII quotes ("), never smart quotes (" " ' ')
|
||||||
|
|
||||||
|
Required fields:
|
||||||
|
- description: Full job description as a single string
|
||||||
|
- criteria: Object with keys like title, company, location, salary, skills, experience, job_type, remote
|
||||||
|
|
||||||
|
Example (SINGLE LINE):
|
||||||
|
{"description":"Full job description text here...","criteria":{"title":"Software Engineer","location":"City, Country","salary":"$100k-$150k","skills":"Go, Python, Docker","experience":"3-5 years","job_type":"Full-time","remote":"Remote"}}
|
||||||
|
|
||||||
|
HTML Content:
|
||||||
|
%s`, htmlContent)
|
||||||
|
|
||||||
|
_, resp, err := client.
|
||||||
|
NewChatCompletion().
|
||||||
|
WithModel(s.model).
|
||||||
|
WithSystemMessage("You are an expert job description extractor. You convert unstructured job description text into a structured JSON format. CRITICAL: You must respond with ONLY raw JSON - no markdown, no code blocks, no backticks, no additional text. Start directly with { and end with }.").
|
||||||
|
WithUserMessage(prompt).
|
||||||
|
Execute()
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("failed to execute completion: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(resp.Choices) == 0 {
|
||||||
|
return nil, fmt.Errorf("no response choices received from API")
|
||||||
|
}
|
||||||
|
|
||||||
|
jsonContent := resp.Choices[0].Message.Content
|
||||||
|
|
||||||
|
log.Printf("Raw API response for job description: %s", jsonContent)
|
||||||
|
|
||||||
|
jsonContent = cleanMarkdownCodeBlocks(jsonContent)
|
||||||
|
|
||||||
|
log.Printf("Cleaned JSON (first 500 chars): %s", jsonContent[:min(500, len(jsonContent))])
|
||||||
|
|
||||||
|
var result struct {
|
||||||
|
Description string `json:"description"`
|
||||||
|
Criteria map[string]string `json:"criteria"`
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := json.Unmarshal([]byte(jsonContent), &result); err != nil {
|
||||||
|
if syntaxErr, ok := err.(*json.SyntaxError); ok {
|
||||||
|
start := int(syntaxErr.Offset) - 50
|
||||||
|
if start < 0 {
|
||||||
|
start = 0
|
||||||
|
}
|
||||||
|
end := int(syntaxErr.Offset) + 50
|
||||||
|
if end > len(jsonContent) {
|
||||||
|
end = len(jsonContent)
|
||||||
|
}
|
||||||
|
log.Printf("JSON parse error at position %d, context: ...%s...", syntaxErr.Offset, jsonContent[start:end])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
jobDescription := &domain.JobDescription{
|
||||||
|
Description: result.Description,
|
||||||
|
Criteria: result.Criteria,
|
||||||
|
}
|
||||||
|
|
||||||
|
return jobDescription, nil
|
||||||
|
}
|
||||||
|
|
||||||
// func (s *OpenRouterService) CreateCV(cv string, jobDesc domain.JobDescription) (*JobAnalysisResult, error) {
|
// func (s *OpenRouterService) CreateCV(cv string, jobDesc domain.JobDescription) (*JobAnalysisResult, error) {
|
||||||
// }
|
// }
|
||||||
|
|
|
||||||
|
|
@ -36,15 +36,16 @@ func (r *JobRepository) SaveJobs(jobs []domain.Job) error {
|
||||||
}
|
}
|
||||||
|
|
||||||
sqlTemplate := `
|
sqlTemplate := `
|
||||||
INSERT INTO jobs (id, title, company, company_link, location, job_link, job_timestamp)
|
INSERT INTO jobs (id, title, company, company_link, location, job_link, job_timestamp, provider)
|
||||||
VALUES %s
|
VALUES %s
|
||||||
ON CONFLICT (id) DO UPDATE SET
|
ON CONFLICT (job_link) DO UPDATE SET
|
||||||
|
id = EXCLUDED.id,
|
||||||
title = EXCLUDED.title,
|
title = EXCLUDED.title,
|
||||||
company = EXCLUDED.company,
|
company = EXCLUDED.company,
|
||||||
company_link = EXCLUDED.company_link,
|
company_link = EXCLUDED.company_link,
|
||||||
location = EXCLUDED.location,
|
location = EXCLUDED.location,
|
||||||
job_link = EXCLUDED.job_link,
|
job_timestamp = EXCLUDED.job_timestamp,
|
||||||
job_timestamp = EXCLUDED.job_timestamp
|
provider = EXCLUDED.provider
|
||||||
`
|
`
|
||||||
|
|
||||||
sqlStatement, vals := prepareQueryCreateBulk(sqlTemplate, uniqueJobs)
|
sqlStatement, vals := prepareQueryCreateBulk(sqlTemplate, uniqueJobs)
|
||||||
|
|
@ -111,14 +112,14 @@ func (r *JobRepository) GetJobByID(id int) (*domain.Job, error) {
|
||||||
|
|
||||||
func prepareQueryCreateBulk(s string, models []*domain.Job) (string, []interface{}) {
|
func prepareQueryCreateBulk(s string, models []*domain.Job) (string, []interface{}) {
|
||||||
bf := bytes.Buffer{}
|
bf := bytes.Buffer{}
|
||||||
values := make([]interface{}, 0, len(models)*7)
|
values := make([]interface{}, 0, len(models)*8)
|
||||||
|
|
||||||
for i, v := range models {
|
for i, v := range models {
|
||||||
values = append(values, v.ID, v.Title, v.Company,
|
values = append(values, v.ID, v.Title, v.Company,
|
||||||
v.CompanyLink, v.Location, v.JobLink, v.JobPostTime,
|
v.CompanyLink, v.Location, v.JobLink, v.JobPostTime, v.Provider,
|
||||||
)
|
)
|
||||||
|
|
||||||
numFields := 7 // the number of fields you are inserting
|
numFields := 8
|
||||||
n := i * numFields
|
n := i * numFields
|
||||||
|
|
||||||
bf.WriteString("(")
|
bf.WriteString("(")
|
||||||
|
|
|
||||||
|
|
@ -9,4 +9,4 @@ SERVER_PORT=8080
|
||||||
SERVER_HOST=localhost
|
SERVER_HOST=localhost
|
||||||
|
|
||||||
CV_AI_MODEL=your_ai_model
|
CV_AI_MODEL=your_ai_model
|
||||||
OPENROUTER_API_KEY=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
|
OPENROUTER_API_KEY=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
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
DB_HOST=localhost
|
DB_HOST=localhost
|
||||||
DB_PORT=5432
|
DB_PORT=5432
|
||||||
DB_USER=your_db_user
|
DB_USER=postgres
|
||||||
DB_PASSWORD=your_db_password
|
DB_PASSWORD=password
|
||||||
DB_NAME=linkedin_jobs
|
DB_NAME=linkedin_jobs
|
||||||
DB_SSLMODE=disable
|
DB_SSLMODE=disable
|
||||||
|
|
||||||
|
|
@ -9,4 +9,4 @@ SERVER_PORT=8080
|
||||||
SERVER_HOST=localhost
|
SERVER_HOST=localhost
|
||||||
|
|
||||||
CV_AI_MODEL=your_ai_model
|
CV_AI_MODEL=your_ai_model
|
||||||
OPENROUTER_API_KEY=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
|
OPENROUTER_API_KEY=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
|
||||||
19
services/scraper-glassdoor/db/db.ts
Normal file
19
services/scraper-glassdoor/db/db.ts
Normal file
|
|
@ -0,0 +1,19 @@
|
||||||
|
import postgres from "postgres";
|
||||||
|
import dotenv from "dotenv";
|
||||||
|
|
||||||
|
if (dotenv.config({ path: ".local.env" }).error) {
|
||||||
|
console.log("No .local.env file found, trying .env");
|
||||||
|
if (dotenv.config().error) {
|
||||||
|
console.log("No .env file found, using system environment variables");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const sql = postgres({
|
||||||
|
host: process.env.DB_HOST,
|
||||||
|
port: Number(process.env.DB_PORT) || 5432,
|
||||||
|
database: process.env.DB_NAME,
|
||||||
|
username: process.env.DB_USER,
|
||||||
|
password: process.env.DB_PASSWORD,
|
||||||
|
});
|
||||||
|
|
||||||
|
export default sql;
|
||||||
|
|
@ -13,8 +13,15 @@ import {
|
||||||
createRabbitMQClient,
|
createRabbitMQClient,
|
||||||
} from "./utils/rabbitmq";
|
} from "./utils/rabbitmq";
|
||||||
import { BrowserService } from "./services/browserService";
|
import { BrowserService } from "./services/browserService";
|
||||||
|
import { saveJobs } from "./queries/save-jobs";
|
||||||
|
import { saveJobDescriptions } from "./queries/save-job-description";
|
||||||
|
|
||||||
dotenv.config();
|
if (dotenv.config({ path: ".local.env" }).error) {
|
||||||
|
console.log("No .local.env file found, trying .env");
|
||||||
|
if (dotenv.config().error) {
|
||||||
|
console.log("No .env file found, using system environment variables");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
let globalBrowser: Browser | null = null;
|
let globalBrowser: Browser | null = null;
|
||||||
let chrome: ChildProcess;
|
let chrome: ChildProcess;
|
||||||
|
|
@ -88,17 +95,11 @@ async function initializeBrowser(): Promise<{ browser: Browser; page: Page }> {
|
||||||
}
|
}
|
||||||
|
|
||||||
async function scrapeGlassdoorJobs(searchQuery: SearchQuery): Promise<void> {
|
async function scrapeGlassdoorJobs(searchQuery: SearchQuery): Promise<void> {
|
||||||
console.log(
|
|
||||||
`Starting Glassdoor scrape for: ${searchQuery.keywords} in ${searchQuery.location}`
|
|
||||||
);
|
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const { page } = await initializeBrowser();
|
const { page } = await initializeBrowser();
|
||||||
const browserService = new BrowserService(page);
|
const browserService = new BrowserService(page);
|
||||||
|
|
||||||
// Build Glassdoor search URL
|
const searchUrl = `https://www.glassdoor.com/Job/jobs.htm`;
|
||||||
const searchUrl = buildGlassdoorSearchURL(searchQuery);
|
|
||||||
console.log(`Navigating to: ${searchUrl}`);
|
|
||||||
|
|
||||||
await page.goto(searchUrl, {
|
await page.goto(searchUrl, {
|
||||||
waitUntil: ["networkidle2", "domcontentloaded"],
|
waitUntil: ["networkidle2", "domcontentloaded"],
|
||||||
|
|
@ -108,37 +109,16 @@ async function scrapeGlassdoorJobs(searchQuery: SearchQuery): Promise<void> {
|
||||||
await randomDelay(1500, 2500);
|
await randomDelay(1500, 2500);
|
||||||
|
|
||||||
await browserService.searchJobs(searchQuery.keywords, searchQuery.location);
|
await browserService.searchJobs(searchQuery.keywords, searchQuery.location);
|
||||||
const jobs = await browserService.extractJobListings();
|
const jobs = await browserService.extractJobListings(searchQuery.location);
|
||||||
|
|
||||||
console.log(`\n=== SCRAPING RESULTS ===`);
|
|
||||||
console.log(`Successfully extracted ${jobs.length} jobs from Glassdoor`);
|
|
||||||
|
|
||||||
if (jobs.length > 0) {
|
if (jobs.length > 0) {
|
||||||
console.log(`\n=== JOB LISTINGS ===`);
|
await saveJobs(jobs);
|
||||||
jobs.forEach((job, index) => {
|
await saveJobDescriptions(jobs);
|
||||||
console.log(`\n${index + 1}. ${job.title}`);
|
|
||||||
console.log(` Company: ${job.company}`);
|
|
||||||
console.log(` Location: ${job.location}`);
|
|
||||||
console.log(` Age: ${job.age}`);
|
|
||||||
console.log(` Job ID: ${job.jobId}`);
|
|
||||||
if (job.skills.length > 0) {
|
|
||||||
console.log(` Skills: ${job.skills.join(", ")}`);
|
|
||||||
}
|
|
||||||
if (job.jobLink) {
|
|
||||||
console.log(` Job Link: ${job.jobLink}`);
|
|
||||||
}
|
|
||||||
if (job.description) {
|
|
||||||
console.log(
|
|
||||||
` Description: ${job.description.substring(0, 200)}...`
|
|
||||||
);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
// Here you can process the jobs further:
|
console.log(
|
||||||
// - Save to database
|
`saved ${jobs.length} jobs from Glassdoor for query:`,
|
||||||
// - Send to another queue
|
searchQuery
|
||||||
// - Perform additional analysis
|
);
|
||||||
console.log(`\n=== END RESULTS ===`);
|
|
||||||
} else {
|
} else {
|
||||||
console.log("No jobs found for this search query");
|
console.log("No jobs found for this search query");
|
||||||
}
|
}
|
||||||
|
|
@ -148,99 +128,16 @@ async function scrapeGlassdoorJobs(searchQuery: SearchQuery): Promise<void> {
|
||||||
console.error("Error scraping Glassdoor jobs:", error);
|
console.error("Error scraping Glassdoor jobs:", error);
|
||||||
throw error;
|
throw error;
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
function getGlassdoorDomain(location: string): string {
|
if (globalBrowser) {
|
||||||
const locationLower = location.toLowerCase();
|
console.log("Closing browser on app shutdown");
|
||||||
|
await globalBrowser.close();
|
||||||
// Country-specific Glassdoor domains
|
globalBrowser = null;
|
||||||
const countryDomains: { [key: string]: string } = {
|
|
||||||
// Asia Pacific
|
|
||||||
japan: "https://www.glassdoor.com/Job", // Glassdoor Japan uses .com
|
|
||||||
singapore: "https://www.glassdoor.sg/Job",
|
|
||||||
australia: "https://www.glassdoor.com.au/Job",
|
|
||||||
"hong kong": "https://www.glassdoor.com/Job", // Uses .com with HK location
|
|
||||||
india: "https://www.glassdoor.co.in/Job",
|
|
||||||
|
|
||||||
// Europe
|
|
||||||
"united kingdom": "https://www.glassdoor.co.uk/Job",
|
|
||||||
uk: "https://www.glassdoor.co.uk/Job",
|
|
||||||
germany: "https://www.glassdoor.de/Job",
|
|
||||||
france: "https://www.glassdoor.fr/Job",
|
|
||||||
netherlands: "https://www.glassdoor.nl/Job",
|
|
||||||
belgium: "https://www.glassdoor.be/Job",
|
|
||||||
austria: "https://www.glassdoor.at/Job",
|
|
||||||
switzerland: "https://www.glassdoor.ch/Job",
|
|
||||||
ireland: "https://www.glassdoor.ie/Job",
|
|
||||||
spain: "https://www.glassdoor.es/Job",
|
|
||||||
italy: "https://www.glassdoor.it/Job",
|
|
||||||
|
|
||||||
// North America
|
|
||||||
canada: "https://www.glassdoor.ca/Job",
|
|
||||||
mexico: "https://www.glassdoor.com.mx/Job",
|
|
||||||
|
|
||||||
// South America
|
|
||||||
brazil: "https://www.glassdoor.com.br/Job",
|
|
||||||
argentina: "https://www.glassdoor.com.ar/Job",
|
|
||||||
};
|
|
||||||
|
|
||||||
// Check for country matches
|
|
||||||
for (const [country, domain] of Object.entries(countryDomains)) {
|
|
||||||
if (locationLower.includes(country)) {
|
|
||||||
return domain;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// Check for city-specific mappings
|
if (chrome) {
|
||||||
const cityMappings: { [key: string]: string } = {
|
chrome.kill("SIGINT");
|
||||||
tokyo: "https://www.glassdoor.com/Job",
|
|
||||||
osaka: "https://www.glassdoor.com/Job",
|
|
||||||
kyoto: "https://www.glassdoor.com/Job",
|
|
||||||
london: "https://www.glassdoor.co.uk/Job",
|
|
||||||
manchester: "https://www.glassdoor.co.uk/Job",
|
|
||||||
berlin: "https://www.glassdoor.de/Job",
|
|
||||||
munich: "https://www.glassdoor.de/Job",
|
|
||||||
paris: "https://www.glassdoor.fr/Job",
|
|
||||||
amsterdam: "https://www.glassdoor.nl/Job",
|
|
||||||
toronto: "https://www.glassdoor.ca/Job",
|
|
||||||
vancouver: "https://www.glassdoor.ca/Job",
|
|
||||||
sydney: "https://www.glassdoor.com.au/Job",
|
|
||||||
melbourne: "https://www.glassdoor.com.au/Job",
|
|
||||||
singapore: "https://www.glassdoor.sg/Job",
|
|
||||||
mumbai: "https://www.glassdoor.co.in/Job",
|
|
||||||
bangalore: "https://www.glassdoor.co.in/Job",
|
|
||||||
delhi: "https://www.glassdoor.co.in/Job",
|
|
||||||
};
|
|
||||||
|
|
||||||
for (const [city, domain] of Object.entries(cityMappings)) {
|
|
||||||
if (locationLower.includes(city)) {
|
|
||||||
return domain;
|
|
||||||
}
|
}
|
||||||
}
|
|
||||||
|
|
||||||
// Default to US site
|
|
||||||
return "https://www.glassdoor.com/Job";
|
|
||||||
}
|
|
||||||
|
|
||||||
function buildGlassdoorSearchURL(query: SearchQuery): string {
|
|
||||||
const baseDomain = getGlassdoorDomain(query.location);
|
|
||||||
const baseUrl = `${baseDomain}/jobs.htm`;
|
|
||||||
|
|
||||||
console.log(
|
|
||||||
`Selected Glassdoor domain for location "${query.location}": ${baseDomain}`
|
|
||||||
);
|
|
||||||
|
|
||||||
const params = new URLSearchParams();
|
|
||||||
|
|
||||||
// if (query.Keywords) {
|
|
||||||
// params.set("sc.keyword", query.Keywords);
|
|
||||||
// }
|
|
||||||
// if (query.Location) {
|
|
||||||
// params.set("locT", "C");
|
|
||||||
// params.set("locId", query.Location);
|
|
||||||
// }
|
|
||||||
|
|
||||||
return `${baseUrl}?${params.toString()}`;
|
|
||||||
}
|
}
|
||||||
|
|
||||||
let rabbitMQClient: RabbitMQClient | null = null;
|
let rabbitMQClient: RabbitMQClient | null = null;
|
||||||
|
|
|
||||||
13
services/scraper-glassdoor/package-lock.json
generated
13
services/scraper-glassdoor/package-lock.json
generated
|
|
@ -12,6 +12,7 @@
|
||||||
"@types/amqplib": "^0.10.7",
|
"@types/amqplib": "^0.10.7",
|
||||||
"amqplib": "^0.10.9",
|
"amqplib": "^0.10.9",
|
||||||
"dotenv": "^17.2.3",
|
"dotenv": "^17.2.3",
|
||||||
|
"postgres": "^3.4.7",
|
||||||
"puppeteer-core": "^24.25.0"
|
"puppeteer-core": "^24.25.0"
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
|
|
@ -701,6 +702,18 @@
|
||||||
"resolved": "https://registry.npmjs.org/pend/-/pend-1.2.0.tgz",
|
"resolved": "https://registry.npmjs.org/pend/-/pend-1.2.0.tgz",
|
||||||
"integrity": "sha512-F3asv42UuXchdzt+xXqfW1OGlVBe+mxa2mqI0pg5yAHZPvFmY3Y6drSf/GQ1A86WgWEN9Kzh/WrgKa6iGcHXLg=="
|
"integrity": "sha512-F3asv42UuXchdzt+xXqfW1OGlVBe+mxa2mqI0pg5yAHZPvFmY3Y6drSf/GQ1A86WgWEN9Kzh/WrgKa6iGcHXLg=="
|
||||||
},
|
},
|
||||||
|
"node_modules/postgres": {
|
||||||
|
"version": "3.4.7",
|
||||||
|
"resolved": "https://registry.npmjs.org/postgres/-/postgres-3.4.7.tgz",
|
||||||
|
"integrity": "sha512-Jtc2612XINuBjIl/QTWsV5UvE8UHuNblcO3vVADSrKsrc6RqGX6lOW1cEo3CM2v0XG4Nat8nI+YM7/f26VxXLw==",
|
||||||
|
"engines": {
|
||||||
|
"node": ">=12"
|
||||||
|
},
|
||||||
|
"funding": {
|
||||||
|
"type": "individual",
|
||||||
|
"url": "https://github.com/sponsors/porsager"
|
||||||
|
}
|
||||||
|
},
|
||||||
"node_modules/progress": {
|
"node_modules/progress": {
|
||||||
"version": "2.0.3",
|
"version": "2.0.3",
|
||||||
"resolved": "https://registry.npmjs.org/progress/-/progress-2.0.3.tgz",
|
"resolved": "https://registry.npmjs.org/progress/-/progress-2.0.3.tgz",
|
||||||
|
|
|
||||||
|
|
@ -16,6 +16,7 @@
|
||||||
"@types/amqplib": "^0.10.7",
|
"@types/amqplib": "^0.10.7",
|
||||||
"amqplib": "^0.10.9",
|
"amqplib": "^0.10.9",
|
||||||
"dotenv": "^17.2.3",
|
"dotenv": "^17.2.3",
|
||||||
|
"postgres": "^3.4.7",
|
||||||
"puppeteer-core": "^24.25.0"
|
"puppeteer-core": "^24.25.0"
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
|
|
|
||||||
33
services/scraper-glassdoor/queries/save-job-description.ts
Normal file
33
services/scraper-glassdoor/queries/save-job-description.ts
Normal file
|
|
@ -0,0 +1,33 @@
|
||||||
|
import sql from "../db/db";
|
||||||
|
import { Job } from "../types/job";
|
||||||
|
|
||||||
|
export const saveJobDescriptions = async (jobs: Job[]) => {
|
||||||
|
if (jobs.length === 0) {
|
||||||
|
console.log("No jobs to save descriptions for.");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const jobDescriptionsForDb = jobs.map((job) => ({
|
||||||
|
job_id: job.id,
|
||||||
|
description: job.description,
|
||||||
|
job_criteria: job.criteria,
|
||||||
|
}));
|
||||||
|
|
||||||
|
await sql`
|
||||||
|
INSERT INTO public.job_descriptions ${sql(
|
||||||
|
jobDescriptionsForDb,
|
||||||
|
"job_id",
|
||||||
|
"description",
|
||||||
|
"job_criteria"
|
||||||
|
)}
|
||||||
|
ON CONFLICT (job_id) DO UPDATE SET
|
||||||
|
description = EXCLUDED.description,
|
||||||
|
job_criteria = EXCLUDED.job_criteria,
|
||||||
|
updated_at = CURRENT_TIMESTAMP
|
||||||
|
`;
|
||||||
|
console.log(`Updated descriptions for ${jobs.length} jobs.`);
|
||||||
|
} catch (error) {
|
||||||
|
console.error("Error updating job descriptions:", error);
|
||||||
|
}
|
||||||
|
};
|
||||||
52
services/scraper-glassdoor/queries/save-jobs.ts
Normal file
52
services/scraper-glassdoor/queries/save-jobs.ts
Normal file
|
|
@ -0,0 +1,52 @@
|
||||||
|
import sql from "../db/db";
|
||||||
|
import { Job } from "../types/job";
|
||||||
|
|
||||||
|
export const saveJobs = async (jobs: Job[]) => {
|
||||||
|
if (jobs.length === 0) {
|
||||||
|
console.log("No jobs to save.");
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
try {
|
||||||
|
const jobsForDb = jobs.map((job) => ({
|
||||||
|
id: job.id,
|
||||||
|
title: job.title,
|
||||||
|
company: job.company,
|
||||||
|
location: job.location,
|
||||||
|
company_link: job.companyLink,
|
||||||
|
job_link: job.jobLink,
|
||||||
|
provider: job.provider,
|
||||||
|
status: job.status,
|
||||||
|
job_timestamp:
|
||||||
|
job.jobPostTime instanceof Date && !isNaN(job.jobPostTime.getTime())
|
||||||
|
? job.jobPostTime
|
||||||
|
: new Date(),
|
||||||
|
}));
|
||||||
|
|
||||||
|
const result = await sql`
|
||||||
|
INSERT INTO public.jobs ${sql(
|
||||||
|
jobsForDb,
|
||||||
|
"id",
|
||||||
|
"title",
|
||||||
|
"company",
|
||||||
|
"location",
|
||||||
|
"company_link",
|
||||||
|
"job_link",
|
||||||
|
"provider",
|
||||||
|
"status",
|
||||||
|
"job_timestamp"
|
||||||
|
)}
|
||||||
|
ON CONFLICT (id) DO UPDATE SET
|
||||||
|
title = EXCLUDED.title,
|
||||||
|
company = EXCLUDED.company,
|
||||||
|
company_link = EXCLUDED.company_link,
|
||||||
|
location = EXCLUDED.location,
|
||||||
|
job_link = EXCLUDED.job_link,
|
||||||
|
job_timestamp = EXCLUDED.job_timestamp
|
||||||
|
`;
|
||||||
|
|
||||||
|
console.log(`Saved ${jobs.length} jobs to the database.`);
|
||||||
|
} catch (error) {
|
||||||
|
console.error("Error saving jobs to the database:", error);
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
@ -1,18 +1,6 @@
|
||||||
import { Page } from "puppeteer-core";
|
import { Page } from "puppeteer-core";
|
||||||
import { randomDelay } from "../utils/browser";
|
import { randomDelay } from "../utils/browser";
|
||||||
|
import { Job } from "../types/job";
|
||||||
export interface JobListing {
|
|
||||||
title: string;
|
|
||||||
company: string;
|
|
||||||
location: string;
|
|
||||||
description: string;
|
|
||||||
skills: string[];
|
|
||||||
jobLink: string;
|
|
||||||
companyLink: string;
|
|
||||||
age: string;
|
|
||||||
jobId: string;
|
|
||||||
provider: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
export class BrowserService {
|
export class BrowserService {
|
||||||
private page: Page;
|
private page: Page;
|
||||||
|
|
@ -23,7 +11,6 @@ export class BrowserService {
|
||||||
|
|
||||||
async searchJobs(keywords: string, location: string) {
|
async searchJobs(keywords: string, location: string) {
|
||||||
try {
|
try {
|
||||||
// Wait for search form to load
|
|
||||||
const keywordInput = await this.page.waitForSelector(
|
const keywordInput = await this.page.waitForSelector(
|
||||||
"#searchBar-jobTitle, input[name='sc.keyword'], input[placeholder*='Job title']",
|
"#searchBar-jobTitle, input[name='sc.keyword'], input[placeholder*='Job title']",
|
||||||
{ timeout: 10000 }
|
{ timeout: 10000 }
|
||||||
|
|
@ -46,7 +33,6 @@ export class BrowserService {
|
||||||
await randomDelay(1000, 2000);
|
await randomDelay(1000, 2000);
|
||||||
await locationInput.press("Enter");
|
await locationInput.press("Enter");
|
||||||
|
|
||||||
// Wait for search results to load
|
|
||||||
await this.page.waitForNavigation({
|
await this.page.waitForNavigation({
|
||||||
waitUntil: "networkidle2",
|
waitUntil: "networkidle2",
|
||||||
timeout: 30000,
|
timeout: 30000,
|
||||||
|
|
@ -59,8 +45,7 @@ export class BrowserService {
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
async extractJobListings(): Promise<JobListing[]> {
|
async extractJobListings(location: string): Promise<Job[]> {
|
||||||
// Wait for job listings to appear
|
|
||||||
try {
|
try {
|
||||||
await this.page.waitForSelector(
|
await this.page.waitForSelector(
|
||||||
'li[data-test="jobListing"], .JobsList_jobListItem__wjTHv, [data-jobid]',
|
'li[data-test="jobListing"], .JobsList_jobListItem__wjTHv, [data-jobid]',
|
||||||
|
|
@ -71,27 +56,24 @@ export class BrowserService {
|
||||||
return [];
|
return [];
|
||||||
}
|
}
|
||||||
|
|
||||||
// Extract job data using page.evaluate
|
const jobs = await this.page.evaluate((location) => {
|
||||||
const jobs = await this.page.evaluate(() => {
|
|
||||||
// Multiple selectors to handle different Glassdoor layouts
|
|
||||||
const jobElements = document.querySelectorAll(`
|
const jobElements = document.querySelectorAll(`
|
||||||
li[data-test="jobListing"],
|
li[data-test="jobListing"],
|
||||||
.JobsList_jobListItem__wjTHv,
|
.JobsList_jobListItem__wjTHv,
|
||||||
li[data-jobid],
|
li[data-jobid],
|
||||||
.react-job-listing
|
.react-job-listing
|
||||||
`);
|
`) as NodeListOf<HTMLElement>;
|
||||||
|
|
||||||
const jobs: any[] = [];
|
const jobs: any[] = [];
|
||||||
|
console.log("Found job elements:", jobElements.length);
|
||||||
|
|
||||||
jobElements.forEach((element, index) => {
|
jobElements.forEach((element, index) => {
|
||||||
try {
|
try {
|
||||||
// Extract job ID
|
|
||||||
const jobId =
|
const jobId =
|
||||||
element.getAttribute("data-jobid") ||
|
element.getAttribute("data-jobid") ||
|
||||||
element.getAttribute("data-brandviews")?.match(/jlid=(\d+)/)?.[1] ||
|
element.getAttribute("data-brandviews")?.match(/jlid=(\d+)/)?.[1] ||
|
||||||
`job-${index + 1}`;
|
`job-${index + 1}`;
|
||||||
|
|
||||||
// Extract job title and link
|
|
||||||
const titleElement = element.querySelector(`
|
const titleElement = element.querySelector(`
|
||||||
a[data-test="job-title"],
|
a[data-test="job-title"],
|
||||||
.JobCard_jobTitle__GLyJ1,
|
.JobCard_jobTitle__GLyJ1,
|
||||||
|
|
@ -99,7 +81,6 @@ export class BrowserService {
|
||||||
a[id*="job-title"]
|
a[id*="job-title"]
|
||||||
`) as HTMLAnchorElement;
|
`) as HTMLAnchorElement;
|
||||||
|
|
||||||
// Extract company name and link
|
|
||||||
const companyElement = element.querySelector(`
|
const companyElement = element.querySelector(`
|
||||||
.EmployerProfile_compactEmployerName__9MGcV,
|
.EmployerProfile_compactEmployerName__9MGcV,
|
||||||
[data-test="employer-name"] a,
|
[data-test="employer-name"] a,
|
||||||
|
|
@ -107,57 +88,31 @@ export class BrowserService {
|
||||||
.EmployerProfile_employerNameContainer__ptolz span
|
.EmployerProfile_employerNameContainer__ptolz span
|
||||||
`) as HTMLElement;
|
`) as HTMLElement;
|
||||||
|
|
||||||
// Extract location
|
|
||||||
const locationElement = element.querySelector(`
|
|
||||||
.JobCard_location__Ds1fM,
|
|
||||||
[data-test="emp-location"],
|
|
||||||
[data-test="job-location"],
|
|
||||||
.location,
|
|
||||||
[id*="job-location"]
|
|
||||||
`) as HTMLElement;
|
|
||||||
|
|
||||||
// Extract job description
|
|
||||||
const descriptionElement = element.querySelector(`
|
const descriptionElement = element.querySelector(`
|
||||||
.JobCard_jobDescriptionSnippet__l1tnl,
|
.JobDetails_jobDescription__uW_fK,
|
||||||
[data-test="descSnippet"],
|
.JobDetails_showHidden__C_FOA,
|
||||||
.jobDescriptionSnippet
|
[class*="jobDescription"]
|
||||||
`) as HTMLElement;
|
`) as HTMLElement;
|
||||||
|
|
||||||
// Extract job age
|
const jobPostingTimeElement = element.querySelector(`
|
||||||
const ageElement = element.querySelector(`
|
|
||||||
.JobCard_listingAge__jJsuc,
|
.JobCard_listingAge__jJsuc,
|
||||||
[data-test="job-age"],
|
[data-test="job-age"],
|
||||||
.listingAge
|
.listingAge
|
||||||
`) as HTMLElement;
|
`) as HTMLElement;
|
||||||
|
|
||||||
// Extract skills if available
|
const title = titleElement?.textContent?.trim() || "";
|
||||||
const skillsText = descriptionElement?.textContent || "";
|
const company = companyElement?.textContent?.trim() || "";
|
||||||
const skillsMatch = skillsText.match(
|
|
||||||
/(?:Vaardigheden|Skills):\s*([^.]+)/i
|
|
||||||
);
|
|
||||||
const skills = skillsMatch
|
|
||||||
? skillsMatch[1]
|
|
||||||
.split(",")
|
|
||||||
.map((skill) => skill.trim())
|
|
||||||
.filter((skill) => skill.length > 0)
|
|
||||||
: [];
|
|
||||||
|
|
||||||
if (titleElement && companyElement) {
|
|
||||||
const title = titleElement.textContent?.trim() || "";
|
|
||||||
const company = companyElement.textContent?.trim() || "";
|
|
||||||
const location = locationElement?.textContent?.trim() || "";
|
|
||||||
const description = descriptionElement?.textContent?.trim() || "";
|
const description = descriptionElement?.textContent?.trim() || "";
|
||||||
const age = ageElement?.textContent?.trim() || "";
|
const jobPosting = jobPostingTimeElement?.textContent?.trim() || "";
|
||||||
|
|
||||||
// Build full URLs for links
|
const jobLink = titleElement?.href
|
||||||
const jobLink = titleElement.href
|
|
||||||
? titleElement.href.startsWith("http")
|
? titleElement.href.startsWith("http")
|
||||||
? titleElement.href
|
? titleElement.href
|
||||||
: `https://www.glassdoor.com${titleElement.href}`
|
: `https://www.glassdoor.com${titleElement.href}`
|
||||||
: "";
|
: "";
|
||||||
|
|
||||||
const companyLink =
|
const companyLink =
|
||||||
companyElement.tagName === "A"
|
companyElement?.tagName === "A"
|
||||||
? (companyElement as HTMLAnchorElement).href?.startsWith("http")
|
? (companyElement as HTMLAnchorElement).href?.startsWith("http")
|
||||||
? (companyElement as HTMLAnchorElement).href
|
? (companyElement as HTMLAnchorElement).href
|
||||||
: `https://www.glassdoor.com${
|
: `https://www.glassdoor.com${
|
||||||
|
|
@ -165,37 +120,66 @@ export class BrowserService {
|
||||||
}`
|
}`
|
||||||
: "";
|
: "";
|
||||||
|
|
||||||
|
let jobPostingDate = new Date();
|
||||||
|
|
||||||
|
try {
|
||||||
|
if (jobPosting.includes("d")) {
|
||||||
|
const match = jobPosting.match(/(\d+)\s*d/);
|
||||||
|
if (match) {
|
||||||
|
const daysAgo = parseInt(match[1], 10);
|
||||||
|
if (!isNaN(daysAgo) && daysAgo >= 0) {
|
||||||
|
jobPostingDate.setDate(jobPostingDate.getDate() - daysAgo);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else if (jobPosting.includes("h")) {
|
||||||
|
const match = jobPosting.match(/(\d+)\s*h/);
|
||||||
|
if (match) {
|
||||||
|
const hoursAgo = parseInt(match[1], 10);
|
||||||
|
if (!isNaN(hoursAgo) && hoursAgo >= 0) {
|
||||||
|
jobPostingDate.setHours(jobPostingDate.getHours() - hoursAgo);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (isNaN(jobPostingDate.getTime())) {
|
||||||
|
jobPostingDate = new Date();
|
||||||
|
}
|
||||||
|
} catch (error) {
|
||||||
|
console.warn(
|
||||||
|
`Failed to parse job posting date: "${jobPosting}", using current date`
|
||||||
|
);
|
||||||
|
jobPostingDate = new Date();
|
||||||
|
}
|
||||||
|
|
||||||
|
const parsedJobId = Number.parseInt(jobId, 10);
|
||||||
|
|
||||||
jobs.push({
|
jobs.push({
|
||||||
title,
|
title,
|
||||||
company,
|
company,
|
||||||
location,
|
location,
|
||||||
description: description.substring(0, 500), // Limit description length
|
description: description,
|
||||||
skills,
|
criteria: {},
|
||||||
jobLink,
|
jobLink,
|
||||||
companyLink,
|
companyLink,
|
||||||
age,
|
|
||||||
jobId,
|
|
||||||
provider: 2,
|
provider: 2,
|
||||||
|
id: parsedJobId,
|
||||||
|
jobPostTime: jobPostingDate.toISOString(),
|
||||||
|
status: 0,
|
||||||
});
|
});
|
||||||
}
|
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
console.error(`Error extracting job ${index}:`, error);
|
console.error(`Error extracting job ${index}:`, error);
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
|
||||||
|
console.log("Extracted jobs:", jobs.length);
|
||||||
return jobs;
|
return jobs;
|
||||||
});
|
}, location);
|
||||||
|
|
||||||
console.log(`Found ${jobs.length} jobs on Glassdoor`);
|
console.log(`Found ${jobs.length} jobs on Glassdoor`);
|
||||||
jobs.forEach((job, index) => {
|
|
||||||
console.log(
|
|
||||||
`${index + 1}. ${job.title} at ${job.company} - ${job.location}`
|
|
||||||
);
|
|
||||||
if (job.skills.length > 0) {
|
|
||||||
console.log(` Skills: ${job.skills.join(", ")}`);
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
return jobs;
|
return jobs.map((job) => ({
|
||||||
|
...job,
|
||||||
|
jobPostTime: new Date(job.jobPostTime),
|
||||||
|
}));
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
|
||||||
13
services/scraper-glassdoor/types/job.ts
Normal file
13
services/scraper-glassdoor/types/job.ts
Normal file
|
|
@ -0,0 +1,13 @@
|
||||||
|
export interface Job {
|
||||||
|
id: number;
|
||||||
|
title: string;
|
||||||
|
company: string;
|
||||||
|
location: string;
|
||||||
|
description: string;
|
||||||
|
criteria: Record<string, string>;
|
||||||
|
jobLink: string;
|
||||||
|
companyLink: string;
|
||||||
|
provider: 2;
|
||||||
|
jobPostTime: string;
|
||||||
|
status: number;
|
||||||
|
}
|
||||||
16
services/scraper-google/.env
Normal file
16
services/scraper-google/.env
Normal file
|
|
@ -0,0 +1,16 @@
|
||||||
|
DB_HOST=localhost
|
||||||
|
DB_PORT=5432
|
||||||
|
DB_USER=your_db_user
|
||||||
|
DB_PASSWORD=your_db_password
|
||||||
|
DB_NAME=linkedin_jobs
|
||||||
|
DB_SSLMODE=disable
|
||||||
|
|
||||||
|
SERVER_PORT=8080
|
||||||
|
SERVER_HOST=localhost
|
||||||
|
|
||||||
|
OPENAI_API_KEY=sk-GtdLX9YBOCsBEgBL4rBONA
|
||||||
|
OPENAI_BASE_URL=https://hubai.loe.gg/v1
|
||||||
|
OPENAI_MODEL=gpt-4o-mini
|
||||||
|
|
||||||
|
CV_AI_MODEL=your_ai_model
|
||||||
|
OPENROUTER_API_KEY=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
|
||||||
34
services/scraper-google/analyzer/get-job-description.go
Normal file
34
services/scraper-google/analyzer/get-job-description.go
Normal file
|
|
@ -0,0 +1,34 @@
|
||||||
|
package analyzer
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"os"
|
||||||
|
|
||||||
|
"github.com/jobs-scraper/internal/pkg/domain"
|
||||||
|
"github.com/jobs-scraper/internal/pkg/openai"
|
||||||
|
// "github.com/jobs-scraper/internal/pkg/openrouter"
|
||||||
|
)
|
||||||
|
|
||||||
|
func GetJobDescription(htmlContent string) (*domain.JobDescription, error) {
|
||||||
|
// apiKey := os.Getenv("OPENROUTER_API_KEY")
|
||||||
|
// model := os.Getenv("CV_AI_MODEL")
|
||||||
|
openAiapiKey := os.Getenv("OPENAI_API_KEY")
|
||||||
|
if openAiapiKey == "" {
|
||||||
|
return nil, fmt.Errorf("OPENAI_API_KEY environment variable is required")
|
||||||
|
}
|
||||||
|
|
||||||
|
openAiModel := os.Getenv("OPENAI_MODEL")
|
||||||
|
if openAiModel == "" {
|
||||||
|
openAiModel = "gpt-4o-mini"
|
||||||
|
}
|
||||||
|
|
||||||
|
openAiService := openai.NewOpenAIService(openAiapiKey, openAiModel)
|
||||||
|
|
||||||
|
// client := openrouter.NewOpenRouterService(model, apiKey)
|
||||||
|
result, err := openAiService.CreateJobDescription(htmlContent)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
return result, nil
|
||||||
|
}
|
||||||
22
services/scraper-google/go.mod
Normal file
22
services/scraper-google/go.mod
Normal file
|
|
@ -0,0 +1,22 @@
|
||||||
|
module github.com/jobs-scraper/services/scraper-google
|
||||||
|
|
||||||
|
go 1.24.0
|
||||||
|
|
||||||
|
require (
|
||||||
|
github.com/chromedp/cdproto v0.0.0-20250803210736-d308e07a266d
|
||||||
|
github.com/chromedp/chromedp v0.14.2
|
||||||
|
github.com/jobs-scraper/internal/pkg/infrastructure v0.0.0
|
||||||
|
)
|
||||||
|
|
||||||
|
require (
|
||||||
|
github.com/chromedp/sysutil v1.1.0 // indirect
|
||||||
|
github.com/go-json-experiment/json v0.0.0-20250725192818-e39067aee2d2 // indirect
|
||||||
|
github.com/gobwas/httphead v0.1.0 // indirect
|
||||||
|
github.com/gobwas/pool v0.2.1 // indirect
|
||||||
|
github.com/gobwas/ws v1.4.0 // indirect
|
||||||
|
github.com/levmv/sked v0.2.1 // indirect
|
||||||
|
github.com/rabbitmq/amqp091-go v1.10.0 // indirect
|
||||||
|
golang.org/x/sys v0.36.0 // indirect
|
||||||
|
)
|
||||||
|
|
||||||
|
replace github.com/jobs-scraper/internal/pkg/infrastructure => ../../internal/pkg/infrastructure
|
||||||
27
services/scraper-google/go.sum
Normal file
27
services/scraper-google/go.sum
Normal file
|
|
@ -0,0 +1,27 @@
|
||||||
|
github.com/chromedp/cdproto v0.0.0-20250803210736-d308e07a266d h1:ZtA1sedVbEW7EW80Iz2GR3Ye6PwbJAJXjv7D74xG6HU=
|
||||||
|
github.com/chromedp/cdproto v0.0.0-20250803210736-d308e07a266d/go.mod h1:NItd7aLkcfOA/dcMXvl8p1u+lQqioRMq/SqDp71Pb/k=
|
||||||
|
github.com/chromedp/chromedp v0.14.2 h1:r3b/WtwM50RsBZHMUm9fsNhhzRStTHrKdr2zmwbZSzM=
|
||||||
|
github.com/chromedp/chromedp v0.14.2/go.mod h1:rHzAv60xDE7VNy/MYtTUrYreSc0ujt2O1/C3bzctYBo=
|
||||||
|
github.com/chromedp/sysutil v1.1.0 h1:PUFNv5EcprjqXZD9nJb9b/c9ibAbxiYo4exNWZyipwM=
|
||||||
|
github.com/chromedp/sysutil v1.1.0/go.mod h1:WiThHUdltqCNKGc4gaU50XgYjwjYIhKWoHGPTUfWTJ8=
|
||||||
|
github.com/go-json-experiment/json v0.0.0-20250725192818-e39067aee2d2 h1:iizUGZ9pEquQS5jTGkh4AqeeHCMbfbjeb0zMt0aEFzs=
|
||||||
|
github.com/go-json-experiment/json v0.0.0-20250725192818-e39067aee2d2/go.mod h1:TiCD2a1pcmjd7YnhGH0f/zKNcCD06B029pHhzV23c2M=
|
||||||
|
github.com/gobwas/httphead v0.1.0 h1:exrUm0f4YX0L7EBwZHuCF4GDp8aJfVeBrlLQrs6NqWU=
|
||||||
|
github.com/gobwas/httphead v0.1.0/go.mod h1:O/RXo79gxV8G+RqlR/otEwx4Q36zl9rqC5u12GKvMCM=
|
||||||
|
github.com/gobwas/pool v0.2.1 h1:xfeeEhW7pwmX8nuLVlqbzVc7udMDrwetjEv+TZIz1og=
|
||||||
|
github.com/gobwas/pool v0.2.1/go.mod h1:q8bcK0KcYlCgd9e7WYLm9LpyS+YeLd8JVDW6WezmKEw=
|
||||||
|
github.com/gobwas/ws v1.4.0 h1:CTaoG1tojrh4ucGPcoJFiAQUAsEWekEWvLy7GsVNqGs=
|
||||||
|
github.com/gobwas/ws v1.4.0/go.mod h1:G3gNqMNtPppf5XUz7O4shetPpcZ1VJ7zt18dlUeakrc=
|
||||||
|
github.com/ledongthuc/pdf v0.0.0-20220302134840-0c2507a12d80 h1:6Yzfa6GP0rIo/kULo2bwGEkFvCePZ3qHDDTC3/J9Swo=
|
||||||
|
github.com/ledongthuc/pdf v0.0.0-20220302134840-0c2507a12d80/go.mod h1:imJHygn/1yfhB7XSJJKlFZKl/J+dCPAknuiaGOshXAs=
|
||||||
|
github.com/levmv/sked v0.2.1 h1:XMdLkbuajPM5LSzoBvNzalj6oYHkEfeA7/VPvkIHqs0=
|
||||||
|
github.com/levmv/sked v0.2.1/go.mod h1:w9pFUIpZLPu0Jr2+2JqJtaEIHmwXcvjPHWm6pvbqHVQ=
|
||||||
|
github.com/orisano/pixelmatch v0.0.0-20220722002657-fb0b55479cde h1:x0TT0RDC7UhAVbbWWBzr41ElhJx5tXPWkIHA2HWPRuw=
|
||||||
|
github.com/orisano/pixelmatch v0.0.0-20220722002657-fb0b55479cde/go.mod h1:nZgzbfBr3hhjoZnS66nKrHmduYNpc34ny7RK4z5/HM0=
|
||||||
|
github.com/rabbitmq/amqp091-go v1.10.0 h1:STpn5XsHlHGcecLmMFCtg7mqq0RnD+zFr4uzukfVhBw=
|
||||||
|
github.com/rabbitmq/amqp091-go v1.10.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||||
|
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
|
||||||
|
go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE=
|
||||||
|
golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||||
|
golang.org/x/sys v0.36.0 h1:KVRy2GtZBrk1cBYA7MKu5bEZFxQk4NIDV6RLVcC8o0k=
|
||||||
|
golang.org/x/sys v0.36.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
|
||||||
113
services/scraper-google/main.go
Normal file
113
services/scraper-google/main.go
Normal file
|
|
@ -0,0 +1,113 @@
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
"os"
|
||||||
|
"os/signal"
|
||||||
|
"syscall"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/jobs-scraper/internal/pkg/domain"
|
||||||
|
"github.com/jobs-scraper/libs/repo"
|
||||||
|
"github.com/jobs-scraper/services/scraper-google/analyzer"
|
||||||
|
si "github.com/jobs-scraper/services/scraper-google/setup-infrastructure"
|
||||||
|
"github.com/jobs-scraper/services/scraper-google/utils"
|
||||||
|
"github.com/levmv/sked"
|
||||||
|
)
|
||||||
|
|
||||||
|
type SearchResult struct {
|
||||||
|
Title string
|
||||||
|
Description string
|
||||||
|
Link string
|
||||||
|
CompanyName string
|
||||||
|
}
|
||||||
|
|
||||||
|
func main() {
|
||||||
|
db, err := si.SetupInfrastructure()
|
||||||
|
if err != nil {
|
||||||
|
log.Fatalf("Failed to set up infrastructure: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
jobRepo := repo.NewJobRepository(db)
|
||||||
|
jobDescRepo := repo.NewJobDescriptionRepository(db)
|
||||||
|
|
||||||
|
ctx, cancel := context.WithCancel(context.Background())
|
||||||
|
|
||||||
|
sched := sked.New(ctx)
|
||||||
|
googleLinkStream := make(chan domain.GoogleLink, 100)
|
||||||
|
// pages := []int{0, 10, 20, 30, 40, 50, 60, 70, 80, 90, 100}
|
||||||
|
pages := []int{90}
|
||||||
|
|
||||||
|
log.Println("Starting scraper with scheduler...")
|
||||||
|
|
||||||
|
for _, page := range pages {
|
||||||
|
Work(ctx)(page, googleLinkStream)
|
||||||
|
// sched.Schedule(func(ctx context.Context) {
|
||||||
|
// log.Printf("Running scrape for page offset: %d", page)
|
||||||
|
// Work(ctx)(page, googleLinkStream)
|
||||||
|
// }).Every(24 * time.Hour)
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := sched.Run(); err != nil {
|
||||||
|
log.Fatalf("Failed to start scheduler: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Process and save jobs immediately as they're scraped
|
||||||
|
go func() {
|
||||||
|
for googleLink := range googleLinkStream {
|
||||||
|
htmlRaw, url, err := utils.GetHTMLRaw(ctx, googleLink)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Error fetching HTML raw for link %s: %v", googleLink.Link, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
result, err := analyzer.GetJobDescription(htmlRaw)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Error analyzing job description for link %s: %v", googleLink.Link, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
now := time.Now()
|
||||||
|
|
||||||
|
job := domain.Job{
|
||||||
|
ID: googleLink.ID,
|
||||||
|
Title: googleLink.Title,
|
||||||
|
Company: googleLink.CompanyName,
|
||||||
|
CompanyLink: googleLink.Link,
|
||||||
|
Location: result.Criteria["location"],
|
||||||
|
JobLink: url,
|
||||||
|
Provider: domain.Google,
|
||||||
|
JobPostTime: &now,
|
||||||
|
Status: domain.JobStatusCreated,
|
||||||
|
}
|
||||||
|
|
||||||
|
jobDescription := domain.JobDescription{
|
||||||
|
JobID: googleLink.ID,
|
||||||
|
Description: result.Description,
|
||||||
|
Criteria: result.Criteria,
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := jobRepo.SaveJobs([]domain.Job{job}); err != nil {
|
||||||
|
log.Printf("Failed to save job %d: %v", googleLink.ID, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := jobDescRepo.SaveJobDescriptions([]domain.JobDescription{jobDescription}); err != nil {
|
||||||
|
log.Printf("Failed to save job description for job %d: %v", googleLink.ID, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
log.Printf("Successfully saved job: %s from %s (ID: %d)", job.Title, job.Company, job.ID)
|
||||||
|
}
|
||||||
|
}()
|
||||||
|
|
||||||
|
c := make(chan os.Signal, 1)
|
||||||
|
signal.Notify(c, os.Interrupt, syscall.SIGTERM)
|
||||||
|
<-c
|
||||||
|
cancel()
|
||||||
|
log.Println("Shutting down scraper...")
|
||||||
|
|
||||||
|
}
|
||||||
|
|
@ -0,0 +1,38 @@
|
||||||
|
package setupinfrastructure
|
||||||
|
|
||||||
|
import (
|
||||||
|
"database/sql"
|
||||||
|
"log"
|
||||||
|
|
||||||
|
"github.com/jobs-scraper/internal/pkg/infrastructure"
|
||||||
|
"github.com/joho/godotenv"
|
||||||
|
)
|
||||||
|
|
||||||
|
func SetupInfrastructure() (*sql.DB, error) {
|
||||||
|
// Load .env from the scraper-google service directory
|
||||||
|
if err := godotenv.Load(".local.env"); err != nil {
|
||||||
|
log.Printf("No .local.env file found: %v, trying .env", err)
|
||||||
|
if err := godotenv.Load(".env"); err != nil {
|
||||||
|
log.Println("No .env file found, using system environment variables")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
dbConfig := infrastructure.LoadConfigFromEnv()
|
||||||
|
|
||||||
|
db, err := infrastructure.NewConnection(dbConfig)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Fatalf("Error connecting to db: %v", err)
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
err = db.Ping()
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Fatal("Error pinging db")
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
log.Println("Successfully connected to db")
|
||||||
|
return db, err
|
||||||
|
}
|
||||||
15
services/scraper-google/utils/build-url.go
Normal file
15
services/scraper-google/utils/build-url.go
Normal file
|
|
@ -0,0 +1,15 @@
|
||||||
|
package utils
|
||||||
|
|
||||||
|
import (
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
func BuildUrl(query string, page int) string {
|
||||||
|
var url strings.Builder
|
||||||
|
url.WriteString("https://www.google.com/search?q=")
|
||||||
|
url.WriteString(query)
|
||||||
|
url.WriteString("&start=")
|
||||||
|
url.WriteString(strconv.Itoa(page))
|
||||||
|
return url.String()
|
||||||
|
}
|
||||||
24
services/scraper-google/utils/extract-company-name.go
Normal file
24
services/scraper-google/utils/extract-company-name.go
Normal file
|
|
@ -0,0 +1,24 @@
|
||||||
|
package utils
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"net/url"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
func ExtractCompanyName(urlStr string) (string, error) {
|
||||||
|
parsedURL, err := url.Parse(urlStr)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("failed to parse URL: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
path := parsedURL.Path
|
||||||
|
segments := strings.Split(strings.Trim(path, "/"), "/")
|
||||||
|
|
||||||
|
if len(segments) > 0 && segments[0] != "" {
|
||||||
|
return segments[0], nil
|
||||||
|
}
|
||||||
|
|
||||||
|
return "", fmt.Errorf("no path segments found")
|
||||||
|
}
|
||||||
30
services/scraper-google/utils/extract-job-id.go
Normal file
30
services/scraper-google/utils/extract-job-id.go
Normal file
|
|
@ -0,0 +1,30 @@
|
||||||
|
package utils
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"net/url"
|
||||||
|
"strconv"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
func ExtractJobId(urlStr string) (int64, error) {
|
||||||
|
parsedURL, err := url.Parse(urlStr)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
return 0, fmt.Errorf("failed to parse URL: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
path := parsedURL.Path
|
||||||
|
segments := strings.Split(strings.Trim(path, "/"), "/")
|
||||||
|
|
||||||
|
for i := len(segments) - 1; i >= 0; i-- {
|
||||||
|
if segments[i] != "" {
|
||||||
|
jobId, err := strconv.ParseInt(segments[i], 10, 64)
|
||||||
|
if err == nil {
|
||||||
|
return jobId, nil
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return 0, fmt.Errorf("no numeric job ID found in URL path: %s", path)
|
||||||
|
}
|
||||||
92
services/scraper-google/utils/get-html-raw.go
Normal file
92
services/scraper-google/utils/get-html-raw.go
Normal file
|
|
@ -0,0 +1,92 @@
|
||||||
|
package utils
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"fmt"
|
||||||
|
"log"
|
||||||
|
"net/url"
|
||||||
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/PuerkitoBio/goquery"
|
||||||
|
"github.com/chromedp/chromedp"
|
||||||
|
"github.com/jobs-scraper/internal/pkg/domain"
|
||||||
|
)
|
||||||
|
|
||||||
|
func GetHTMLRaw(ctx context.Context, googleLink domain.GoogleLink) (string, string, error) {
|
||||||
|
// Set up chromedp options
|
||||||
|
opts := append(chromedp.DefaultExecAllocatorOptions[:],
|
||||||
|
chromedp.Flag("headless", false),
|
||||||
|
chromedp.Flag("disable-blink-features", "AutomationControlled"),
|
||||||
|
chromedp.Flag("excludeSwitches", "enable-automation"),
|
||||||
|
chromedp.UserAgent("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"),
|
||||||
|
chromedp.WindowSize(1920, 1080),
|
||||||
|
)
|
||||||
|
|
||||||
|
allocCtx, allocCancel := chromedp.NewExecAllocator(ctx, opts...)
|
||||||
|
defer allocCancel()
|
||||||
|
|
||||||
|
browserCtx, browserCancel := chromedp.NewContext(allocCtx)
|
||||||
|
defer browserCancel()
|
||||||
|
|
||||||
|
// Add timeout for the entire operation
|
||||||
|
timeoutCtx, timeoutCancel := context.WithTimeout(browserCtx, 30*time.Second)
|
||||||
|
defer timeoutCancel()
|
||||||
|
|
||||||
|
var htmlContent string
|
||||||
|
var finalURL string
|
||||||
|
|
||||||
|
err := chromedp.Run(timeoutCtx,
|
||||||
|
chromedp.Navigate(googleLink.Link),
|
||||||
|
chromedp.WaitVisible("body", chromedp.ByQuery),
|
||||||
|
chromedp.Sleep(2*time.Second), // Wait for JS to render
|
||||||
|
chromedp.Location(&finalURL),
|
||||||
|
chromedp.OuterHTML("html", &htmlContent, chromedp.ByQuery),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to fetch job description with chromedp for link %s: %v", googleLink.Link, err)
|
||||||
|
return "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Parse the HTML with goquery
|
||||||
|
doc, err := goquery.NewDocumentFromReader(strings.NewReader(htmlContent))
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to parse HTML for link %s: %v", googleLink.Link, err)
|
||||||
|
return "", "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
// Remove unwanted elements
|
||||||
|
doc.Find("script, style, nav, footer, header, .ads, .sidebar").Remove()
|
||||||
|
|
||||||
|
log.Printf(" Received: %s from %s\n, ID: %d", googleLink.Title, googleLink.CompanyName, googleLink.ID)
|
||||||
|
|
||||||
|
// Extract text content
|
||||||
|
var content strings.Builder
|
||||||
|
doc.Find("body").Each(func(i int, s *goquery.Selection) {
|
||||||
|
content.WriteString(s.Text())
|
||||||
|
})
|
||||||
|
|
||||||
|
// Extract hostname from the final URL
|
||||||
|
hostname, err := extractHostname(finalURL)
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to extract hostname from final URL %s: %v", finalURL, err)
|
||||||
|
// Fall back to the original link if extraction fails
|
||||||
|
return cleanText(content.String()), googleLink.Link, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
return cleanText(content.String()), hostname, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func cleanText(text string) string {
|
||||||
|
text = strings.Join(strings.Fields(text), " ")
|
||||||
|
return text
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractHostname(urlStr string) (string, error) {
|
||||||
|
parsedURL, err := url.Parse(urlStr)
|
||||||
|
if err != nil {
|
||||||
|
return "", fmt.Errorf("failed to parse URL: %w", err)
|
||||||
|
}
|
||||||
|
return parsedURL.Hostname(), nil
|
||||||
|
}
|
||||||
104
services/scraper-google/work.go
Normal file
104
services/scraper-google/work.go
Normal file
|
|
@ -0,0 +1,104 @@
|
||||||
|
package main
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"log"
|
||||||
|
|
||||||
|
"github.com/chromedp/cdproto/cdp"
|
||||||
|
"github.com/chromedp/chromedp"
|
||||||
|
"github.com/jobs-scraper/internal/pkg/domain"
|
||||||
|
"github.com/jobs-scraper/services/scraper-google/utils"
|
||||||
|
)
|
||||||
|
|
||||||
|
func Work(mainCtx context.Context) func(page int, stream chan<- domain.GoogleLink) {
|
||||||
|
|
||||||
|
return func(page int, stream chan<- domain.GoogleLink) {
|
||||||
|
query := "site:boards.greenhouse.io (Software OR Backend OR Full-Stack) Engineer"
|
||||||
|
|
||||||
|
opts := append(chromedp.DefaultExecAllocatorOptions[:],
|
||||||
|
chromedp.Flag("headless", false),
|
||||||
|
chromedp.Flag("disable-blink-features", "AutomationControlled"),
|
||||||
|
chromedp.Flag("excludeSwitches", "enable-automation"),
|
||||||
|
chromedp.Flag("disable-web-security", false),
|
||||||
|
chromedp.UserAgent("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"),
|
||||||
|
chromedp.WindowSize(1920, 1080),
|
||||||
|
)
|
||||||
|
allocCtx, cancel := chromedp.NewExecAllocator(mainCtx, opts...)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
ctx, cancel := chromedp.NewContext(allocCtx)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
var nodes []*cdp.Node
|
||||||
|
|
||||||
|
// start = 0, start = 10, start = 20 and so on for offset pagination
|
||||||
|
err := chromedp.Run(ctx,
|
||||||
|
chromedp.Navigate(utils.BuildUrl(query, page)),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = chromedp.Run(ctx,
|
||||||
|
chromedp.WaitVisible("#search", chromedp.ByID),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = chromedp.Run(ctx,
|
||||||
|
chromedp.Nodes("[data-snc]", &nodes, chromedp.ByQueryAll),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Fatal(err)
|
||||||
|
}
|
||||||
|
|
||||||
|
for i, node := range nodes {
|
||||||
|
result := domain.GoogleLink{}
|
||||||
|
|
||||||
|
err = chromedp.Run(ctx,
|
||||||
|
chromedp.Text("h3", &result.Title, chromedp.ByQuery, chromedp.FromNode(node)),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to get title for node %d: %v", i, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
err = chromedp.Run(ctx,
|
||||||
|
chromedp.AttributeValue("a", "href", &result.Link, nil, chromedp.ByQuery, chromedp.FromNode(node)),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to get link for node %d: %v", i, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
result.ID, err = utils.ExtractJobId(result.Link)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to extract job ID for node %d: %v (link: %s)", i, err, result.Link)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
result.CompanyName, err = utils.ExtractCompanyName(result.Link)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to extract company name for node %d: %v (link: %s)", i, err, result.Link)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
err = chromedp.Run(ctx,
|
||||||
|
chromedp.Text("div.VwiC3b", &result.Description, chromedp.ByQuery, chromedp.FromNode(node)),
|
||||||
|
)
|
||||||
|
|
||||||
|
if err != nil {
|
||||||
|
log.Printf("Failed to get description for node %d: %v", i, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
stream <- result
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
3
services/scraper-hiringcafe/go.mod
Normal file
3
services/scraper-hiringcafe/go.mod
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
module scraper-hiringcafe
|
||||||
|
|
||||||
|
go 1.24.0
|
||||||
3
services/scraper-hiringcafe/main.go
Normal file
3
services/scraper-hiringcafe/main.go
Normal file
|
|
@ -0,0 +1,3 @@
|
||||||
|
package main
|
||||||
|
|
||||||
|
func main() {}
|
||||||
|
|
@ -9,4 +9,4 @@ SERVER_PORT=8080
|
||||||
SERVER_HOST=localhost
|
SERVER_HOST=localhost
|
||||||
|
|
||||||
CV_AI_MODEL=your_ai_model
|
CV_AI_MODEL=your_ai_model
|
||||||
OPENROUTER_API_KEY=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
|
OPENROUTER_API_KEY=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
|
||||||
Loading…
Reference in a new issue