This commit is contained in:
Elshimy Ziad Magdy Taha 2025-09-29 22:44:30 +05:00
parent c470edd187
commit 7bf35ccb6b
24 changed files with 238 additions and 66 deletions

View file

@ -91,7 +91,7 @@ jobs-scraper/
func (s *Scraper) RetryableHTTPRequest(ctx context.Context, url string) (*http.Response, error)
// Context-aware job scraping with retry support
func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params SearchQuery) ([]models.Job, error)
func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params SearchQuery) ([]domain.Job, error)
```
- **Retry Logic**: Up to 3 attempts with exponential backoff (1s → 2s → 4s → 8s)
- **Smart Error Handling**: Retries 5xx errors, fails fast on 4xx errors
@ -101,7 +101,7 @@ func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params Se
### 2. Scraper Service (`internal/services/scraper.go`)
```go
// Streams jobs with configurable search parameters
func (s *Scraper) ScrapeLinkedInJobsStreaming(ctx context.Context, numPages int, jobChan chan<- models.Job, params SearchQuery) error
func (s *Scraper) ScrapeLinkedInJobsStreaming(ctx context.Context, numPages int, jobChan chan<- domain.Job, params SearchQuery) error
```
- **Sequential Processing**: Pages scraped one by one to avoid detection
- **Rate Limited**: 2-second delays between requests
@ -124,7 +124,7 @@ func (p *JobPipeline) ProcessJobsStreaming(ctx context.Context, numPages int, jo
### 4. Job Description Workers
```go
// Each worker processes jobs with retry logic
func (p *JobPipeline) jobDescriptionWorker(ctx context.Context, jobChan <-chan models.Job, resultChan chan<- JobDescriptionResult)
func (p *JobPipeline) jobDescriptionWorker(ctx context.Context, jobChan <-chan domain.Job, resultChan chan<- JobDescriptionResult)
```
- **Retry-Enabled**: Uses `ScrapeJobDescriptionWithContext` with retry logic
- **Rate Limited**: 2-second delays per worker to avoid overwhelming servers

26
api/app.go Normal file
View file

@ -0,0 +1,26 @@
package main
import (
"database/sql"
"github.com/gorilla/mux"
"github.com/jobs-scraper/api/commands/job"
"github.com/jobs-scraper/application"
"github.com/jobs-scraper/internal/repo"
"github.com/jobs-scraper/internal/server"
)
func NewApplication(router *mux.Router, db *sql.DB) *application.Application {
s := server.NewServer(router)
jobRepo := repo.NewJobRepository(db)
return &application.Application{
Server: s,
Router: router,
DB: db,
JobCommands: &application.JobCommands{
CreateJob: job.NewCreateJobHandler(jobRepo),
},
}
}

View file

@ -0,0 +1,21 @@
package job
import (
"github.com/jobs-scraper/internal/ports"
"github.com/jobs-scraper/internal/repo"
)
type CreateJob struct {
jobRepo *repo.JobRepository
}
func NewCreateJobHandler(jobRepo *repo.JobRepository) *CreateJob {
return &CreateJob{
jobRepo: jobRepo,
}
}
func (cj *CreateJob) Handle(cmd ports.CreateJobCommand) error {
// Implementation will go here
return nil
}

5
api/main.go Normal file
View file

@ -0,0 +1,5 @@
package main
func main() {
}

20
application/app.go Normal file
View file

@ -0,0 +1,20 @@
package application
import (
"database/sql"
"github.com/gorilla/mux"
"github.com/jobs-scraper/internal/ports"
"github.com/jobs-scraper/internal/server"
)
type Application struct {
Router *mux.Router
DB *sql.DB
Server *server.AppServer
JobCommands *JobCommands
}
type JobCommands struct {
CreateJob ports.JobCommandHandler
}

View file

@ -5,7 +5,7 @@ import (
"os"
"github.com/jobs-scraper/infrastructure"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
"github.com/jobs-scraper/internal/repo"
"github.com/jobs-scraper/internal/services"
"github.com/joho/godotenv"
@ -65,7 +65,7 @@ func main() {
log.Fatalf("Failed to get job description: %v", err)
}
result, err := openRouterService.AnalyzeJobDescription(string(cv), models.JobDescription{
result, err := openRouterService.AnalyzeJobDescription(string(cv), domain.JobDescription{
JobID: job.ID,
Description: jobDescription,
Criteria: jobCriteria,

2
go.mod
View file

@ -8,9 +8,9 @@ require (
github.com/PuerkitoBio/goquery v1.10.3
github.com/eduardolat/openroutergo v0.1.0
github.com/golang-migrate/migrate/v4 v4.19.0
github.com/gorilla/mux v1.8.1
github.com/joho/godotenv v1.5.1
github.com/lib/pq v1.10.9
golang.org/x/time v0.5.0
)
require (

4
go.sum
View file

@ -35,6 +35,8 @@ github.com/gogo/protobuf v1.3.2/go.mod h1:P1XiOD3dCwIKUDQYPy72D8LYyHL2YPYrpS2s69
github.com/golang-migrate/migrate/v4 v4.19.0 h1:RcjOnCGz3Or6HQYEJ/EEVLfWnmw9KnoigPSjzhCuaSE=
github.com/golang-migrate/migrate/v4 v4.19.0/go.mod h1:9dyEcu+hO+G9hPSw8AIg50yg622pXJsoHItQnDGZkI0=
github.com/google/go-cmp v0.6.0/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY=
github.com/gorilla/mux v1.8.1 h1:TuBL49tXwgrFYWhqrNgrUNEY92u81SPhu7sTdzQEiWY=
github.com/gorilla/mux v1.8.1/go.mod h1:AKf9I4AEqPTmMytcMc0KkNouC66V3BtZ4qD5fmWSiMQ=
github.com/hashicorp/errwrap v1.0.0/go.mod h1:YH+1FKiLXxHSkmPseP+kNlulaMuP3n2brvKWEqk/Jc4=
github.com/hashicorp/errwrap v1.1.0 h1:OxrOeh75EUXMY8TBjag2fzXGZ40LB6IKw45YeGUDY2I=
github.com/hashicorp/errwrap v1.1.0/go.mod h1:YH+1FKiLXxHSkmPseP+kNlulaMuP3n2brvKWEqk/Jc4=
@ -135,8 +137,6 @@ golang.org/x/text v0.13.0/go.mod h1:TvPlkZtksWOMsz7fbANvkp4WM8x/WCo/om8BMLbz+aE=
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
golang.org/x/text v0.15.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
golang.org/x/text v0.21.0/go.mod h1:4IBbMaMmOPCJ8SecivzSH54+73PCFmPWxNTLm+vZkEQ=
golang.org/x/time v0.5.0 h1:o7cqy6amK/52YcAKIPlM3a+Fpj35zvRj2TP+e1xFSfk=
golang.org/x/time v0.5.0/go.mod h1:3BpzKBy/shNhVucY/MWOyx10tF3SFh9QdLuxbVysPQM=
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
golang.org/x/tools v0.0.0-20191119224855-298f0cb1881e/go.mod h1:b+2E5dAYhXwXZwtnZ6UAqBI28+e2cm9otk0dWdXHAEo=
golang.org/x/tools v0.1.12/go.mod h1:hNGJHUnrk76NpqgfD5Aqm5Crs+Hm0VOH/i9J2+nxYbc=

View file

@ -0,0 +1,7 @@
package domain
type JobDescription struct {
JobID int64
Description string
Criteria map[string]string
}

View file

@ -0,0 +1,6 @@
package domain
type JobWithDescription struct {
Job Job
JobDescription JobDescription
}

29
internal/domain/job.go Normal file
View file

@ -0,0 +1,29 @@
package domain
// JobProvider represents different job board providers
type JobProvider int
const (
LinkedIn JobProvider = iota
Indeed
Glassdoor
Bayt
TokyoDev
JapanDev
)
type Job struct {
ID int64
Title string
Company string
CompanyLink string
Location string
JobLink string
Provider JobProvider
}
type SearchQuery struct {
Keywords string `json:"keywords"`
Location string `json:"location"`
FWT string `json:"f_WT"` // Work type filter (1=onsite, 2=remote, 3=hybrid)
}

View file

@ -1,5 +1,7 @@
package models
import "github.com/jobs-scraper/internal/domain"
type Job struct {
ID int64
Title string
@ -7,11 +9,11 @@ type Job struct {
CompanyLink string
Location string
JobLink string
Provider domain.JobProvider
}
type SearchQuery struct {
Keywords string `json:"keywords"`
Location string `json:"location"`
FWT string `json:"f_WT"` // Work type filter (1=onsite, 2=remote, 3=hybrid)
GeoId string `json:"geoId"` // Geographic location ID
}

View file

@ -9,13 +9,13 @@ import (
"time"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
"github.com/jobs-scraper/internal/repo"
)
// JobDescriptionResult represents the result of job description scraping
type JobDescriptionResult struct {
Job models.Job
Job domain.Job
Description string
Criteria map[string]string
Error error
@ -38,12 +38,12 @@ func NewJobPipeline(scraperService *Scraper, numWorkers int, rateLimit time.Dura
}
// ProcessJobsStreaming processes jobs and job descriptions concurrently
func (p *JobPipeline) ProcessJobsStreaming(ctx context.Context, numPages int, jobRepo *repo.JobRepository, jobDescRepo *repo.JobDescriptionRepository, params models.SearchQuery) error {
func (p *JobPipeline) ProcessJobsStreaming(ctx context.Context, numPages int, jobRepo *repo.JobRepository, jobDescRepo *repo.JobDescriptionRepository, searchQuery domain.SearchQuery) error {
// Create channels for the pipeline
allJobs := make([]models.Job, 0, 100)
allJobDescriptions := make([]models.JobDescription, 0, 100)
allJobs := make([]domain.Job, 0, 100)
allJobDescriptions := make([]domain.JobDescription, 0, 100)
var jbMu sync.Mutex
jobsChan := GetJobs(ctx, p.scraperService)
jobsChan := GetJobs(ctx, p.scraperService, searchQuery)
jobWithDescriptionChan := GetJobDescription(ctx, p.scraperService, jobsChan, 3)
for jobWithDescription := range jobWithDescriptionChan {

View file

@ -5,19 +5,15 @@ import (
"log"
"sync"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
)
func GetJobs(context context.Context, scraperService *Scraper) <-chan models.Job {
jobChan := make(chan models.Job, 100)
func GetJobs(context context.Context, scraperService *Scraper, searchQuery domain.SearchQuery) <-chan domain.Job {
jobChan := make(chan domain.Job, 100)
go func() {
defer close(jobChan)
if err := scraperService.ScrapeLinkedInJobsStreaming(context, 10, jobChan, models.SearchQuery{
Keywords: "Frontend Developer",
Location: "Japan",
FWT: "2,3",
}); err != nil {
if err := scraperService.ScrapeLinkedInJobsStreaming(context, 10, jobChan, searchQuery); err != nil {
log.Printf("Error scraping jobs: %v", err)
}
}()
@ -25,8 +21,8 @@ func GetJobs(context context.Context, scraperService *Scraper) <-chan models.Job
return jobChan
}
func GetJobDescription(context context.Context, scraperService *Scraper, jobChan <-chan models.Job, numWorkers int) <-chan models.JobWithDescription {
jobDescriptionChan := make(chan models.JobWithDescription, 100)
func GetJobDescription(context context.Context, scraperService *Scraper, jobChan <-chan domain.Job, numWorkers int) <-chan domain.JobWithDescription {
jobDescriptionChan := make(chan domain.JobWithDescription, 100)
var wg sync.WaitGroup
@ -46,9 +42,9 @@ func GetJobDescription(context context.Context, scraperService *Scraper, jobChan
if jd, jc, err := scraperService.ScrapeJobDescriptionWithContext(context, job); err != nil {
log.Printf("Error scraping jobs: %v", err)
} else {
jobDescriptionChan <- models.JobWithDescription{
jobDescriptionChan <- domain.JobWithDescription{
Job: job,
JobDescription: models.JobDescription{
JobDescription: domain.JobDescription{
JobID: job.ID,
Description: jd,
Criteria: jc,

View file

@ -11,13 +11,12 @@ import (
"time"
"github.com/PuerkitoBio/goquery"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
"github.com/jobs-scraper/internal/utils"
)
type Config struct {
Timespan string // Time filter for job postings (e.g., "r86400" for last 24 hours)
Distance string // Search radius in miles (e.g., "25")
SortBy string // Sort results by (R for relevance, DD for date posted)
MaxRetries int // Maximum number of retries for failed requests
BaseDelay time.Duration // Base delay for exponential backoff
@ -59,7 +58,7 @@ func NewScraper(config Config) *Scraper {
}
// ScrapeLinkedInJobsStreaming scrapes jobs page by page and sends them to channel immediately
func (s *Scraper) ScrapeLinkedInJobsStreaming(ctx context.Context, numPages int, jobChan chan<- models.Job, params models.SearchQuery) error {
func (s *Scraper) ScrapeLinkedInJobsStreaming(ctx context.Context, numPages int, jobChan chan<- domain.Job, params domain.SearchQuery) error {
// Process pages sequentially to send jobs immediately
for i := range numPages {
select {
@ -84,8 +83,8 @@ func (s *Scraper) ScrapeLinkedInJobsStreaming(ctx context.Context, numPages int,
return nil
}
func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params models.SearchQuery) ([]models.Job, error) {
jobs := make([]models.Job, 0, 10)
func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params domain.SearchQuery) ([]domain.Job, error) {
jobs := make([]domain.Job, 0, 10)
url := s.buildSearchURL(params, page)
retryableRequest := utils.NewRetryableHTTPRequest(utils.RetryConfig{
@ -108,12 +107,12 @@ func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params mo
}
doc.Find("li > div.base-card").Each(func(i int, s *goquery.Selection) {
job := models.Job{}
job := domain.Job{}
title := s.Find("[class*=_title]").Text()
job.Title = strings.TrimSpace(title)
job.Company = strings.TrimSpace(s.Find(".hidden-nested-link").Text())
job.CompanyLink = strings.TrimSpace(s.Find(".hidden-nested-link").AttrOr("href", ""))
job.Location = strings.TrimSpace(s.Find(".job-search-card__location").Text())
job.Location = strings.TrimSpace(params.Location)
job.JobLink = strings.TrimSpace(s.Find("a.base-card__full-link").AttrOr("href", ""))
jobId, err := extractJobIDFromURL(job.JobLink)
if err != nil {
@ -125,6 +124,7 @@ func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params mo
fmt.Printf("Error extracting job ID from URL %s: %v\n", job.JobLink, err)
}
job.ID = jobIdInt
job.Provider = 0
if job.Title != "" && job.Company != "" && job.Location != "" && job.JobLink != "" {
jobs = append(jobs, job)
@ -134,7 +134,7 @@ func (s *Scraper) ScrapeJobsWithContext(ctx context.Context, page int, params mo
return jobs, nil
}
func (s *Scraper) ScrapeJobDescriptionWithContext(ctx context.Context, job models.Job) (string, map[string]string, error) {
func (s *Scraper) ScrapeJobDescriptionWithContext(ctx context.Context, job domain.Job) (string, map[string]string, error) {
var jobDescription string
url := s.buildJobDescriptionSearchURL(job.JobLink)
@ -166,23 +166,20 @@ func (s *Scraper) ScrapeJobDescriptionWithContext(ctx context.Context, job model
return strings.TrimSpace(jobDescription), jobCriteria, nil
}
func (s *Scraper) buildSearchURL(query models.SearchQuery, page int) string {
func (s *Scraper) buildSearchURL(query domain.SearchQuery, page int) string {
baseURL := "https://www.linkedin.com/jobs-guest/jobs/api/seeMoreJobPostings/search"
params := url.Values{}
// Required search parameters
params.Set("keywords", url.QueryEscape(query.Keywords))
params.Set("location", url.QueryEscape(query.Location))
// Set keywords and location WITHOUT pre-escaping
params.Set("keywords", query.Keywords)
params.Set("location", query.Location)
// Time filter
// params.Set("f_TPR", s.config.Timespan)
// Work type filter
// Work type filter (e.g., "2,3" for remote + hybrid)
if query.FWT != "" {
params.Set("f_WT", query.FWT)
params.Set("f_WT", query.FWT) // "2,3" is valid; url.Values will encode comma if needed, but LinkedIn accepts raw comma
}
// Pagination
// Pagination: LinkedIn uses 0, 25, 50, ...
params.Set("start", strconv.Itoa(25*page))
return fmt.Sprintf("%s?%s", baseURL, params.Encode())

View file

@ -0,0 +1,16 @@
package ports
import "github.com/jobs-scraper/internal/domain"
// CreateJobCommand represents the command to create a job
type CreateJobCommand struct {
Location string
Keywords string
FWT string
Provider domain.JobProvider
}
// JobCommandHandler defines the interface for handling job commands
type JobCommandHandler interface {
Handle(cmd CreateJobCommand) error
}

View file

@ -6,7 +6,7 @@ import (
"fmt"
"strings"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
)
type JobDescriptionRepository struct {
@ -23,7 +23,7 @@ func NewJobDescriptionRepository(db *sql.DB) *JobDescriptionRepository {
return &JobDescriptionRepository{db: db}
}
func (r *JobDescriptionRepository) SaveJobDescriptions(jobDescriptions []models.JobDescription) error {
func (r *JobDescriptionRepository) SaveJobDescriptions(jobDescriptions []domain.JobDescription) error {
if len(jobDescriptions) == 0 {
return nil
}

View file

@ -4,7 +4,7 @@ import (
"database/sql"
"fmt"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
)
type JobRepository struct {
@ -15,18 +15,18 @@ func NewJobRepository(db *sql.DB) *JobRepository {
return &JobRepository{db: db}
}
func (r *JobRepository) SaveJobs(jobs []models.Job) error {
func (r *JobRepository) SaveJobs(jobs []domain.Job) error {
if len(jobs) == 0 {
return nil
}
// Deduplicate jobs by ID to avoid duplicate errors
jobMap := make(map[int64]models.Job)
jobMap := make(map[int64]domain.Job)
for _, job := range jobs {
jobMap[job.ID] = job
}
uniqueJobs := make([]models.Job, 0, len(jobMap))
uniqueJobs := make([]domain.Job, 0, len(jobMap))
for _, job := range jobMap {
uniqueJobs = append(uniqueJobs, job)
}
@ -67,16 +67,16 @@ func (r *JobRepository) SaveJobs(jobs []models.Job) error {
return nil
}
func (r *JobRepository) GetAllJobs() ([]models.Job, error) {
func (r *JobRepository) GetAllJobs() ([]domain.Job, error) {
rows, err := r.db.Query("SELECT id, title, company, company_link, location, job_link FROM jobs")
if err != nil {
return nil, fmt.Errorf("error querying jobs: %v", err)
}
defer rows.Close()
var jobs []models.Job
var jobs []domain.Job
for rows.Next() {
var job models.Job
var job domain.Job
if err := rows.Scan(&job.ID, &job.Title, &job.Company, &job.CompanyLink, &job.Location, &job.JobLink); err != nil {
return nil, fmt.Errorf("error scanning job row: %v", err)
}
@ -89,8 +89,8 @@ func (r *JobRepository) GetAllJobs() ([]models.Job, error) {
return jobs, nil
}
func (r *JobRepository) GetJobByID(id int) (*models.Job, error) {
var job models.Job
func (r *JobRepository) GetJobByID(id int) (*domain.Job, error) {
var job domain.Job
sqlStatement := `
SELECT id, title, company, company_link, location, job_link

40
internal/server/server.go Normal file
View file

@ -0,0 +1,40 @@
package server
import (
"context"
http "net/http"
"time"
"github.com/gorilla/mux"
)
type AppServer struct {
router *mux.Router
httpServer *http.Server
}
func NewServer(router *mux.Router) *AppServer {
port := "8080"
httpServer := &http.Server{
Addr: ":" + port,
Handler: router,
ReadTimeout: 15 * time.Second,
WriteTimeout: 15 * time.Second,
IdleTimeout: 60 * time.Second,
}
server := &AppServer{
router: router,
httpServer: httpServer,
}
return server
}
func (s *AppServer) Start() error {
return s.httpServer.ListenAndServe()
}
func (s *AppServer) Stop(ctx context.Context) {
s.httpServer.Shutdown(ctx)
}

View file

@ -6,7 +6,7 @@ import (
"log"
"github.com/eduardolat/openroutergo"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
)
// JobAnalysisResult represents the structured response from job analysis
@ -42,7 +42,7 @@ func NewOpenRouterService(model string, apiKey string) OpenRouterService {
}
}
func (s *OpenRouterService) AnalyzeJobDescription(cv string, jobDesc models.JobDescription) (*JobAnalysisResult, error) {
func (s *OpenRouterService) AnalyzeJobDescription(cv string, jobDesc domain.JobDescription) (*JobAnalysisResult, error) {
client, err := openroutergo.
NewClient().
WithAPIKey(s.apiKey).
@ -110,5 +110,5 @@ func (s *OpenRouterService) AnalyzeJobDescription(cv string, jobDesc models.JobD
}
// func (s *OpenRouterService) CreateCV(cv string, jobDesc models.JobDescription) (*JobAnalysisResult, error) {
// func (s *OpenRouterService) CreateCV(cv string, jobDesc domain.JobDescription) (*JobAnalysisResult, error) {
// }

View file

@ -66,7 +66,13 @@ func (s *RetryableHTTPRequestImpl) RetryableHTTPRequest(ctx context.Context, url
// max attempts for now = 3
if attempt < s.config.MaxRetries {
delay := time.Second * 2
// Increment delay by 2 seconds for each attempt
delay := time.Second * time.Duration(2*(attempt+1)) // 2s, 4s, 6s, ...
// Cap the delay to MaxDelay if set
if s.config.MaxDelay > 0 && delay > s.config.MaxDelay {
delay = s.config.MaxDelay
}
fmt.Printf("Retrying in %v... (attempt %d/%d)\n", delay, attempt+1, s.config.MaxRetries)

View file

@ -0,0 +1 @@
ALTER TABLE jobs DROP COLUMN IF EXISTS provider;

View file

@ -0,0 +1 @@
ALTER TABLE jobs ADD COLUMN IF NOT EXISTS provider INTEGER DEFAULT 0;

View file

@ -6,7 +6,7 @@ import (
"time"
"github.com/jobs-scraper/infrastructure"
"github.com/jobs-scraper/internal/models"
"github.com/jobs-scraper/internal/domain"
"github.com/jobs-scraper/internal/pipeline"
"github.com/jobs-scraper/internal/repo"
"github.com/joho/godotenv"
@ -40,7 +40,6 @@ func main() {
}
scraper := pipeline.NewScraper(pipeline.Config{
Distance: "25",
SortBy: "R",
MaxRetries: 3,
BaseDelay: 1 * time.Second,
@ -55,9 +54,9 @@ func main() {
ctx := context.Background()
searchParams := models.SearchQuery{
Keywords: "Frontend Developer",
Location: "Japan",
searchParams := domain.SearchQuery{
Keywords: "Javascript",
Location: "US",
FWT: "2,3",
}