fix(scraper): add greenhouse/lever board fetching to crawler, proxy support for datacenter IPs, and ziprecruiter fallback
This commit is contained in:
parent
fb3d886527
commit
2d2aab2999
5 changed files with 139 additions and 11 deletions
|
|
@ -15,5 +15,7 @@ NEXTAUTH_SECRET=replace_with_a_random_32_char_secret_string
|
||||||
NEXTAUTH_URL=http://localhost:3000
|
NEXTAUTH_URL=http://localhost:3000
|
||||||
NODE_ENV=production
|
NODE_ENV=production
|
||||||
|
|
||||||
# Scraper Interval
|
# Scraper Interval & Optional Proxy Configuration
|
||||||
SCRAPE_INTERVAL_MINUTES=30
|
SCRAPE_INTERVAL_MINUTES=30
|
||||||
|
# SOCKS5_PROXY=socks5://username:password@host:port
|
||||||
|
# PROXY_URL=http://username:password@host:port
|
||||||
|
|
|
||||||
|
|
@ -45,6 +45,8 @@ services:
|
||||||
environment:
|
environment:
|
||||||
DATABASE_URL: "postgresql://postgres:postgres@db:5432/jobsboard?schema=public"
|
DATABASE_URL: "postgresql://postgres:postgres@db:5432/jobsboard?schema=public"
|
||||||
SCRAPE_INTERVAL_MINUTES: "30"
|
SCRAPE_INTERVAL_MINUTES: "30"
|
||||||
|
SOCKS5_PROXY: "${SOCKS5_PROXY:-}"
|
||||||
|
PROXY_URL: "${PROXY_URL:-}"
|
||||||
depends_on:
|
depends_on:
|
||||||
db:
|
db:
|
||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
|
|
||||||
|
|
@ -4,3 +4,4 @@ beautifulsoup4>=4.12.3
|
||||||
requests>=2.31.0
|
requests>=2.31.0
|
||||||
apscheduler>=3.10.4
|
apscheduler>=3.10.4
|
||||||
python-dotenv>=1.0.1
|
python-dotenv>=1.0.1
|
||||||
|
requests[socks]>=2.31.0
|
||||||
|
|
|
||||||
|
|
@ -1,11 +1,19 @@
|
||||||
|
import os
|
||||||
import time
|
import time
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
from typing import List, Dict, Any
|
from typing import List, Dict, Any
|
||||||
|
|
||||||
|
def get_jobspy_proxies():
|
||||||
|
proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
|
||||||
|
if proxy:
|
||||||
|
return [proxy]
|
||||||
|
return None
|
||||||
|
|
||||||
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
"""
|
"""
|
||||||
Executes JobSpy searches across nationwide US hubs and remote roles
|
Executes JobSpy searches across nationwide US hubs and remote roles
|
||||||
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
||||||
|
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
|
||||||
"""
|
"""
|
||||||
collected_jobs = []
|
collected_jobs = []
|
||||||
|
|
||||||
|
|
@ -15,6 +23,16 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
||||||
return []
|
return []
|
||||||
|
|
||||||
|
proxies = get_jobspy_proxies()
|
||||||
|
if proxies:
|
||||||
|
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
|
||||||
|
|
||||||
|
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
|
||||||
|
# If proxies are configured, enable zip_recruiter and glassdoor
|
||||||
|
sites = ["indeed"]
|
||||||
|
if proxies:
|
||||||
|
sites.extend(["zip_recruiter", "glassdoor"])
|
||||||
|
|
||||||
# Target key employment hubs across US states
|
# Target key employment hubs across US states
|
||||||
us_regions = [
|
us_regions = [
|
||||||
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
||||||
|
|
@ -35,19 +53,20 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
||||||
]
|
]
|
||||||
|
|
||||||
print("[JobSpy] Starting Nationwide Regional Scrapes...")
|
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
||||||
for loc, queries in us_regions:
|
for loc, queries in us_regions:
|
||||||
for query in queries:
|
for query in queries:
|
||||||
try:
|
try:
|
||||||
print(f"[JobSpy] Searching {loc}: '{query}'")
|
print(f"[JobSpy] Searching {loc}: '{query}'")
|
||||||
jobs_df = scrape_jobs(
|
jobs_df = scrape_jobs(
|
||||||
site_name=["indeed", "zip_recruiter"],
|
site_name=sites,
|
||||||
search_term=query,
|
search_term=query,
|
||||||
location=loc,
|
location=loc,
|
||||||
results_wanted=35,
|
results_wanted=35,
|
||||||
hours_old=72,
|
hours_old=72,
|
||||||
country_indeed="USA",
|
country_indeed="USA",
|
||||||
is_remote=False
|
is_remote=False,
|
||||||
|
proxies=proxies
|
||||||
)
|
)
|
||||||
|
|
||||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||||
|
|
@ -81,12 +100,13 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
try:
|
try:
|
||||||
print(f"[JobSpy] Searching US Remote: '{query}'")
|
print(f"[JobSpy] Searching US Remote: '{query}'")
|
||||||
jobs_df = scrape_jobs(
|
jobs_df = scrape_jobs(
|
||||||
site_name=["indeed", "zip_recruiter"],
|
site_name=sites,
|
||||||
search_term=query,
|
search_term=query,
|
||||||
results_wanted=35,
|
results_wanted=35,
|
||||||
hours_old=72,
|
hours_old=72,
|
||||||
country_indeed="USA",
|
country_indeed="USA",
|
||||||
is_remote=True
|
is_remote=True,
|
||||||
|
proxies=proxies
|
||||||
)
|
)
|
||||||
|
|
||||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||||
|
|
|
||||||
|
|
@ -1,3 +1,4 @@
|
||||||
|
import os
|
||||||
import requests
|
import requests
|
||||||
import re
|
import re
|
||||||
from urllib.parse import urlparse, urljoin
|
from urllib.parse import urlparse, urljoin
|
||||||
|
|
@ -45,12 +46,22 @@ class SmartCareersCrawler:
|
||||||
"""
|
"""
|
||||||
Crawls company domains, finds their /careers or /jobs pages,
|
Crawls company domains, finds their /careers or /jobs pages,
|
||||||
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
|
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
|
||||||
|
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
|
||||||
"""
|
"""
|
||||||
def __init__(self, headers: Optional[Dict[str, str]] = None):
|
def __init__(self, headers: Optional[Dict[str, str]] = None):
|
||||||
self.headers = headers or {
|
self.headers = headers or {
|
||||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# Configure proxy if supplied in environment
|
||||||
|
proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
|
||||||
|
self.proxies = {}
|
||||||
|
if proxy_url:
|
||||||
|
self.proxies = {
|
||||||
|
"http": proxy_url,
|
||||||
|
"https": proxy_url
|
||||||
|
}
|
||||||
|
|
||||||
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
||||||
parsed = urlparse(url)
|
parsed = urlparse(url)
|
||||||
|
|
@ -96,7 +107,7 @@ class SmartCareersCrawler:
|
||||||
for path in COMMON_CAREER_PATHS:
|
for path in COMMON_CAREER_PATHS:
|
||||||
test_url = f"{base_url}{path}"
|
test_url = f"{base_url}{path}"
|
||||||
try:
|
try:
|
||||||
r = requests.get(test_url, headers=self.headers, timeout=5, allow_redirects=True)
|
r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
||||||
final_url = r.url
|
final_url = r.url
|
||||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
|
|
@ -105,20 +116,20 @@ class SmartCareersCrawler:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
try:
|
try:
|
||||||
r = requests.get(base_url, headers=self.headers, timeout=5, allow_redirects=True)
|
r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
||||||
if r.status_code == 200:
|
if r.status_code == 200:
|
||||||
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, r.url
|
return provider, slug, r.url
|
||||||
|
|
||||||
links = re.findall(r'href=[\'"]([^\'"]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\'"]*)[\'"]', r.text, re.IGNORECASE)
|
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
|
||||||
for link in links[:5]:
|
for link in links[:5]:
|
||||||
target = link if link.startswith("http") else urljoin(base_url, link)
|
target = link if link.startswith("http") else urljoin(base_url, link)
|
||||||
provider, slug = self.detect_ats_from_url_or_html(target)
|
provider, slug = self.detect_ats_from_url_or_html(target)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, target
|
return provider, slug, target
|
||||||
try:
|
try:
|
||||||
cr = requests.get(target, headers=self.headers, timeout=5, allow_redirects=True)
|
cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
||||||
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, cr.url
|
return provider, slug, cr.url
|
||||||
|
|
@ -129,11 +140,103 @@ class SmartCareersCrawler:
|
||||||
|
|
||||||
return None, None, None
|
return None, None, None
|
||||||
|
|
||||||
|
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||||
|
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
||||||
|
collected = []
|
||||||
|
try:
|
||||||
|
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
||||||
|
if r.status_code != 200:
|
||||||
|
return collected
|
||||||
|
data = r.json()
|
||||||
|
jobs = data.get("jobs", [])
|
||||||
|
for j in jobs:
|
||||||
|
title = clean_html_text(j.get("title", ""))
|
||||||
|
job_url = j.get("absolute_url", "")
|
||||||
|
if not title or not job_url:
|
||||||
|
continue
|
||||||
|
|
||||||
|
location_obj = j.get("location", {})
|
||||||
|
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
||||||
|
|
||||||
|
if not is_valid_us_location(location_name, title):
|
||||||
|
continue
|
||||||
|
|
||||||
|
is_remote = parse_is_us_remote(location_name, title)
|
||||||
|
departments = j.get("departments", [])
|
||||||
|
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||||||
|
|
||||||
|
dept = determine_department(title, dept_text)
|
||||||
|
exp_level = determine_experience_level(title)
|
||||||
|
desc_clean = clean_html_text(j.get("content", "") or "")
|
||||||
|
|
||||||
|
collected.append({
|
||||||
|
"title": title,
|
||||||
|
"company": company_name,
|
||||||
|
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||||
|
"is_remote": is_remote,
|
||||||
|
"department": dept,
|
||||||
|
"experience_level": exp_level,
|
||||||
|
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
|
||||||
|
"salary_min": None,
|
||||||
|
"salary_max": None,
|
||||||
|
"job_url": job_url,
|
||||||
|
"source": "greenhouse"
|
||||||
|
})
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}")
|
||||||
|
return collected
|
||||||
|
|
||||||
|
def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||||
|
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
||||||
|
collected = []
|
||||||
|
try:
|
||||||
|
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
||||||
|
if r.status_code != 200:
|
||||||
|
return collected
|
||||||
|
jobs = r.json()
|
||||||
|
for j in jobs:
|
||||||
|
title = clean_html_text(j.get("text", ""))
|
||||||
|
job_url = j.get("hostedUrl", "")
|
||||||
|
if not title or not job_url:
|
||||||
|
continue
|
||||||
|
|
||||||
|
categories = j.get("categories", {})
|
||||||
|
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
||||||
|
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
||||||
|
|
||||||
|
if not is_valid_us_location(location_name, title):
|
||||||
|
continue
|
||||||
|
|
||||||
|
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
|
||||||
|
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
||||||
|
dept = determine_department(title, dept_text)
|
||||||
|
exp_level = determine_experience_level(title)
|
||||||
|
|
||||||
|
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
||||||
|
desc_clean = clean_html_text(description_plain)
|
||||||
|
|
||||||
|
collected.append({
|
||||||
|
"title": title,
|
||||||
|
"company": company_name,
|
||||||
|
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||||
|
"is_remote": is_remote,
|
||||||
|
"department": dept,
|
||||||
|
"experience_level": exp_level,
|
||||||
|
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
||||||
|
"salary_min": None,
|
||||||
|
"salary_max": None,
|
||||||
|
"job_url": job_url,
|
||||||
|
"source": "lever"
|
||||||
|
})
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[Lever Warning] {company_name} ({slug}) error: {e}")
|
||||||
|
return collected
|
||||||
|
|
||||||
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, timeout=8)
|
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
data = r.json()
|
data = r.json()
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue