fix(scraper): add greenhouse/lever board fetching to crawler, proxy support for datacenter IPs, and ziprecruiter fallback

This commit is contained in:
JobsBoard Deployer 2026-09-05 12:33:57 -04:00
parent fb3d886527
commit 2d2aab2999
5 changed files with 139 additions and 11 deletions

View file

@ -15,5 +15,7 @@ NEXTAUTH_SECRET=replace_with_a_random_32_char_secret_string
NEXTAUTH_URL=http://localhost:3000
NODE_ENV=production
# Scraper Interval
# Scraper Interval & Optional Proxy Configuration
SCRAPE_INTERVAL_MINUTES=30
# SOCKS5_PROXY=socks5://username:password@host:port
# PROXY_URL=http://username:password@host:port

View file

@ -45,6 +45,8 @@ services:
environment:
DATABASE_URL: "postgresql://postgres:postgres@db:5432/jobsboard?schema=public"
SCRAPE_INTERVAL_MINUTES: "30"
SOCKS5_PROXY: "${SOCKS5_PROXY:-}"
PROXY_URL: "${PROXY_URL:-}"
depends_on:
db:
condition: service_healthy

View file

@ -4,3 +4,4 @@ beautifulsoup4>=4.12.3
requests>=2.31.0
apscheduler>=3.10.4
python-dotenv>=1.0.1
requests[socks]>=2.31.0

View file

@ -1,11 +1,19 @@
import os
import time
import pandas as pd
from typing import List, Dict, Any
def get_jobspy_proxies():
proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
if proxy:
return [proxy]
return None
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"""
Executes JobSpy searches across nationwide US hubs and remote roles
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
"""
collected_jobs = []
@ -15,6 +23,16 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
return []
proxies = get_jobspy_proxies()
if proxies:
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
# If proxies are configured, enable zip_recruiter and glassdoor
sites = ["indeed"]
if proxies:
sites.extend(["zip_recruiter", "glassdoor"])
# Target key employment hubs across US states
us_regions = [
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
@ -35,19 +53,20 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
]
print("[JobSpy] Starting Nationwide Regional Scrapes...")
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
for loc, queries in us_regions:
for query in queries:
try:
print(f"[JobSpy] Searching {loc}: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed", "zip_recruiter"],
site_name=sites,
search_term=query,
location=loc,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=False
is_remote=False,
proxies=proxies
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
@ -81,12 +100,13 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
try:
print(f"[JobSpy] Searching US Remote: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed", "zip_recruiter"],
site_name=sites,
search_term=query,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=True
is_remote=True,
proxies=proxies
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:

View file

@ -1,3 +1,4 @@
import os
import requests
import re
from urllib.parse import urlparse, urljoin
@ -45,12 +46,22 @@ class SmartCareersCrawler:
"""
Crawls company domains, finds their /careers or /jobs pages,
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
"""
def __init__(self, headers: Optional[Dict[str, str]] = None):
self.headers = headers or {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
}
# Configure proxy if supplied in environment
proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
self.proxies = {}
if proxy_url:
self.proxies = {
"http": proxy_url,
"https": proxy_url
}
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
parsed = urlparse(url)
@ -96,7 +107,7 @@ class SmartCareersCrawler:
for path in COMMON_CAREER_PATHS:
test_url = f"{base_url}{path}"
try:
r = requests.get(test_url, headers=self.headers, timeout=5, allow_redirects=True)
r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
final_url = r.url
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
if provider and slug:
@ -105,20 +116,20 @@ class SmartCareersCrawler:
continue
try:
r = requests.get(base_url, headers=self.headers, timeout=5, allow_redirects=True)
r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
if r.status_code == 200:
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
if provider and slug:
return provider, slug, r.url
links = re.findall(r'href=[\'"]([^\'"]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\'"]*)[\'"]', r.text, re.IGNORECASE)
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
for link in links[:5]:
target = link if link.startswith("http") else urljoin(base_url, link)
provider, slug = self.detect_ats_from_url_or_html(target)
if provider and slug:
return provider, slug, target
try:
cr = requests.get(target, headers=self.headers, timeout=5, allow_redirects=True)
cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
if provider and slug:
return provider, slug, cr.url
@ -129,11 +140,103 @@ class SmartCareersCrawler:
return None, None, None
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
collected = []
try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
if r.status_code != 200:
return collected
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
desc_clean = clean_html_text(j.get("content", "") or "")
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}")
return collected
def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
collected = []
try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
if r.status_code != 200:
return collected
jobs = r.json()
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} ({slug}) error: {e}")
return collected
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
collected = []
try:
r = requests.get(url, headers=self.headers, timeout=8)
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
if r.status_code != 200:
return collected
data = r.json()