fix(scraper): add greenhouse/lever board fetching to crawler, proxy support for datacenter IPs, and ziprecruiter fallback
This commit is contained in:
parent
fb3d886527
commit
2d2aab2999
5 changed files with 139 additions and 11 deletions
|
|
@ -15,5 +15,7 @@ NEXTAUTH_SECRET=replace_with_a_random_32_char_secret_string
|
|||
NEXTAUTH_URL=http://localhost:3000
|
||||
NODE_ENV=production
|
||||
|
||||
# Scraper Interval
|
||||
# Scraper Interval & Optional Proxy Configuration
|
||||
SCRAPE_INTERVAL_MINUTES=30
|
||||
# SOCKS5_PROXY=socks5://username:password@host:port
|
||||
# PROXY_URL=http://username:password@host:port
|
||||
|
|
|
|||
|
|
@ -45,6 +45,8 @@ services:
|
|||
environment:
|
||||
DATABASE_URL: "postgresql://postgres:postgres@db:5432/jobsboard?schema=public"
|
||||
SCRAPE_INTERVAL_MINUTES: "30"
|
||||
SOCKS5_PROXY: "${SOCKS5_PROXY:-}"
|
||||
PROXY_URL: "${PROXY_URL:-}"
|
||||
depends_on:
|
||||
db:
|
||||
condition: service_healthy
|
||||
|
|
|
|||
|
|
@ -4,3 +4,4 @@ beautifulsoup4>=4.12.3
|
|||
requests>=2.31.0
|
||||
apscheduler>=3.10.4
|
||||
python-dotenv>=1.0.1
|
||||
requests[socks]>=2.31.0
|
||||
|
|
|
|||
|
|
@ -1,11 +1,19 @@
|
|||
import os
|
||||
import time
|
||||
import pandas as pd
|
||||
from typing import List, Dict, Any
|
||||
|
||||
def get_jobspy_proxies():
|
||||
proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
|
||||
if proxy:
|
||||
return [proxy]
|
||||
return None
|
||||
|
||||
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||
"""
|
||||
Executes JobSpy searches across nationwide US hubs and remote roles
|
||||
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
||||
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
|
||||
"""
|
||||
collected_jobs = []
|
||||
|
||||
|
|
@ -15,6 +23,16 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|||
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
||||
return []
|
||||
|
||||
proxies = get_jobspy_proxies()
|
||||
if proxies:
|
||||
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
|
||||
|
||||
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
|
||||
# If proxies are configured, enable zip_recruiter and glassdoor
|
||||
sites = ["indeed"]
|
||||
if proxies:
|
||||
sites.extend(["zip_recruiter", "glassdoor"])
|
||||
|
||||
# Target key employment hubs across US states
|
||||
us_regions = [
|
||||
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
||||
|
|
@ -35,19 +53,20 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|||
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
||||
]
|
||||
|
||||
print("[JobSpy] Starting Nationwide Regional Scrapes...")
|
||||
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
||||
for loc, queries in us_regions:
|
||||
for query in queries:
|
||||
try:
|
||||
print(f"[JobSpy] Searching {loc}: '{query}'")
|
||||
jobs_df = scrape_jobs(
|
||||
site_name=["indeed", "zip_recruiter"],
|
||||
site_name=sites,
|
||||
search_term=query,
|
||||
location=loc,
|
||||
results_wanted=35,
|
||||
hours_old=72,
|
||||
country_indeed="USA",
|
||||
is_remote=False
|
||||
is_remote=False,
|
||||
proxies=proxies
|
||||
)
|
||||
|
||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||
|
|
@ -81,12 +100,13 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|||
try:
|
||||
print(f"[JobSpy] Searching US Remote: '{query}'")
|
||||
jobs_df = scrape_jobs(
|
||||
site_name=["indeed", "zip_recruiter"],
|
||||
site_name=sites,
|
||||
search_term=query,
|
||||
results_wanted=35,
|
||||
hours_old=72,
|
||||
country_indeed="USA",
|
||||
is_remote=True
|
||||
is_remote=True,
|
||||
proxies=proxies
|
||||
)
|
||||
|
||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import os
|
||||
import requests
|
||||
import re
|
||||
from urllib.parse import urlparse, urljoin
|
||||
|
|
@ -45,12 +46,22 @@ class SmartCareersCrawler:
|
|||
"""
|
||||
Crawls company domains, finds their /careers or /jobs pages,
|
||||
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
|
||||
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
|
||||
"""
|
||||
def __init__(self, headers: Optional[Dict[str, str]] = None):
|
||||
self.headers = headers or {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
||||
}
|
||||
|
||||
# Configure proxy if supplied in environment
|
||||
proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
|
||||
self.proxies = {}
|
||||
if proxy_url:
|
||||
self.proxies = {
|
||||
"http": proxy_url,
|
||||
"https": proxy_url
|
||||
}
|
||||
|
||||
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
||||
parsed = urlparse(url)
|
||||
|
|
@ -96,7 +107,7 @@ class SmartCareersCrawler:
|
|||
for path in COMMON_CAREER_PATHS:
|
||||
test_url = f"{base_url}{path}"
|
||||
try:
|
||||
r = requests.get(test_url, headers=self.headers, timeout=5, allow_redirects=True)
|
||||
r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
||||
final_url = r.url
|
||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
||||
if provider and slug:
|
||||
|
|
@ -105,20 +116,20 @@ class SmartCareersCrawler:
|
|||
continue
|
||||
|
||||
try:
|
||||
r = requests.get(base_url, headers=self.headers, timeout=5, allow_redirects=True)
|
||||
r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
||||
if r.status_code == 200:
|
||||
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
||||
if provider and slug:
|
||||
return provider, slug, r.url
|
||||
|
||||
links = re.findall(r'href=[\'"]([^\'"]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\'"]*)[\'"]', r.text, re.IGNORECASE)
|
||||
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
|
||||
for link in links[:5]:
|
||||
target = link if link.startswith("http") else urljoin(base_url, link)
|
||||
provider, slug = self.detect_ats_from_url_or_html(target)
|
||||
if provider and slug:
|
||||
return provider, slug, target
|
||||
try:
|
||||
cr = requests.get(target, headers=self.headers, timeout=5, allow_redirects=True)
|
||||
cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
||||
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
||||
if provider and slug:
|
||||
return provider, slug, cr.url
|
||||
|
|
@ -129,11 +140,103 @@ class SmartCareersCrawler:
|
|||
|
||||
return None, None, None
|
||||
|
||||
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
||||
collected = []
|
||||
try:
|
||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
||||
if r.status_code != 200:
|
||||
return collected
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("absolute_url", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
location_obj = j.get("location", {})
|
||||
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(location_name, title)
|
||||
departments = j.get("departments", [])
|
||||
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||||
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
desc_clean = clean_html_text(j.get("content", "") or "")
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "greenhouse"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}")
|
||||
return collected
|
||||
|
||||
def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
||||
collected = []
|
||||
try:
|
||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
||||
if r.status_code != 200:
|
||||
return collected
|
||||
jobs = r.json()
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("text", ""))
|
||||
job_url = j.get("hostedUrl", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
categories = j.get("categories", {})
|
||||
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
||||
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
|
||||
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
||||
desc_clean = clean_html_text(description_plain)
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "lever"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Lever Warning] {company_name} ({slug}) error: {e}")
|
||||
return collected
|
||||
|
||||
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
||||
collected = []
|
||||
try:
|
||||
r = requests.get(url, headers=self.headers, timeout=8)
|
||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
||||
if r.status_code != 200:
|
||||
return collected
|
||||
data = r.json()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue