diff --git a/.env.example b/.env.example index 41aca99..c8e8bec 100644 --- a/.env.example +++ b/.env.example @@ -15,5 +15,7 @@ NEXTAUTH_SECRET=replace_with_a_random_32_char_secret_string NEXTAUTH_URL=http://localhost:3000 NODE_ENV=production -# Scraper Interval +# Scraper Interval & Optional Proxy Configuration SCRAPE_INTERVAL_MINUTES=30 +# SOCKS5_PROXY=socks5://username:password@host:port +# PROXY_URL=http://username:password@host:port diff --git a/docker-compose.yml b/docker-compose.yml index 7105e11..686ef36 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -45,6 +45,8 @@ services: environment: DATABASE_URL: "postgresql://postgres:postgres@db:5432/jobsboard?schema=public" SCRAPE_INTERVAL_MINUTES: "30" + SOCKS5_PROXY: "${SOCKS5_PROXY:-}" + PROXY_URL: "${PROXY_URL:-}" depends_on: db: condition: service_healthy diff --git a/scraper/requirements.txt b/scraper/requirements.txt index 0930c12..f9590a8 100644 --- a/scraper/requirements.txt +++ b/scraper/requirements.txt @@ -4,3 +4,4 @@ beautifulsoup4>=4.12.3 requests>=2.31.0 apscheduler>=3.10.4 python-dotenv>=1.0.1 +requests[socks]>=2.31.0 diff --git a/scraper/scrapers/jobspy_runner.py b/scraper/scrapers/jobspy_runner.py index 23647e1..62de2bc 100644 --- a/scraper/scrapers/jobspy_runner.py +++ b/scraper/scrapers/jobspy_runner.py @@ -1,11 +1,19 @@ +import os import time import pandas as pd from typing import List, Dict, Any +def get_jobspy_proxies(): + proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or "" + if proxy: + return [proxy] + return None + def run_jobspy_scrapes() -> List[Dict[Any, Any]]: """ Executes JobSpy searches across nationwide US hubs and remote roles covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc. + Gracefully handles Cloudflare/datacenter blocks and proxy rotation. """ collected_jobs = [] @@ -15,6 +23,16 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.") return [] + proxies = get_jobspy_proxies() + if proxies: + print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}") + + # Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs) + # If proxies are configured, enable zip_recruiter and glassdoor + sites = ["indeed"] + if proxies: + sites.extend(["zip_recruiter", "glassdoor"]) + # Target key employment hubs across US states us_regions = [ ("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]), @@ -35,19 +53,20 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: "DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer" ] - print("[JobSpy] Starting Nationwide Regional Scrapes...") + print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...") for loc, queries in us_regions: for query in queries: try: print(f"[JobSpy] Searching {loc}: '{query}'") jobs_df = scrape_jobs( - site_name=["indeed", "zip_recruiter"], + site_name=sites, search_term=query, location=loc, results_wanted=35, hours_old=72, country_indeed="USA", - is_remote=False + is_remote=False, + proxies=proxies ) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: @@ -81,12 +100,13 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: try: print(f"[JobSpy] Searching US Remote: '{query}'") jobs_df = scrape_jobs( - site_name=["indeed", "zip_recruiter"], + site_name=sites, search_term=query, results_wanted=35, hours_old=72, country_indeed="USA", - is_remote=True + is_remote=True, + proxies=proxies ) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: diff --git a/scraper/scrapers/smart_careers_crawler.py b/scraper/scrapers/smart_careers_crawler.py index 828495a..67f7e43 100644 --- a/scraper/scrapers/smart_careers_crawler.py +++ b/scraper/scrapers/smart_careers_crawler.py @@ -1,3 +1,4 @@ +import os import requests import re from urllib.parse import urlparse, urljoin @@ -45,12 +46,22 @@ class SmartCareersCrawler: """ Crawls company domains, finds their /careers or /jobs pages, and automatically detects and ingests from Greenhouse, Lever, or Ashby. + Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY. """ def __init__(self, headers: Optional[Dict[str, str]] = None): self.headers = headers or { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" } + + # Configure proxy if supplied in environment + proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or "" + self.proxies = {} + if proxy_url: + self.proxies = { + "http": proxy_url, + "https": proxy_url + } def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]: parsed = urlparse(url) @@ -96,7 +107,7 @@ class SmartCareersCrawler: for path in COMMON_CAREER_PATHS: test_url = f"{base_url}{path}" try: - r = requests.get(test_url, headers=self.headers, timeout=5, allow_redirects=True) + r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) final_url = r.url provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "") if provider and slug: @@ -105,20 +116,20 @@ class SmartCareersCrawler: continue try: - r = requests.get(base_url, headers=self.headers, timeout=5, allow_redirects=True) + r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) if r.status_code == 200: provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) if provider and slug: return provider, slug, r.url - links = re.findall(r'href=[\'"]([^\'"]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\'"]*)[\'"]', r.text, re.IGNORECASE) + links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) for link in links[:5]: target = link if link.startswith("http") else urljoin(base_url, link) provider, slug = self.detect_ats_from_url_or_html(target) if provider and slug: return provider, slug, target try: - cr = requests.get(target, headers=self.headers, timeout=5, allow_redirects=True) + cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) if provider and slug: return provider, slug, cr.url @@ -129,11 +140,103 @@ class SmartCareersCrawler: return None, None, None + def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: + url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true" + collected = [] + try: + r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) + if r.status_code != 200: + return collected + data = r.json() + jobs = data.get("jobs", []) + for j in jobs: + title = clean_html_text(j.get("title", "")) + job_url = j.get("absolute_url", "") + if not title or not job_url: + continue + + location_obj = j.get("location", {}) + location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) + + if not is_valid_us_location(location_name, title): + continue + + is_remote = parse_is_us_remote(location_name, title) + departments = j.get("departments", []) + dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" + + dept = determine_department(title, dept_text) + exp_level = determine_experience_level(title) + desc_clean = clean_html_text(j.get("content", "") or "") + + collected.append({ + "title": title, + "company": company_name, + "location": location_name if location_name != "Remote" else "Remote, USA", + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.", + "salary_min": None, + "salary_max": None, + "job_url": job_url, + "source": "greenhouse" + }) + except Exception as e: + print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}") + return collected + + def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: + url = f"https://api.lever.co/v0/postings/{slug}?mode=json" + collected = [] + try: + r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) + if r.status_code != 200: + return collected + jobs = r.json() + for j in jobs: + title = clean_html_text(j.get("text", "")) + job_url = j.get("hostedUrl", "") + if not title or not job_url: + continue + + categories = j.get("categories", {}) + location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA" + workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else "" + + if not is_valid_us_location(location_name, title): + continue + + is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) + dept_text = categories.get("department", "") if isinstance(categories, dict) else "" + dept = determine_department(title, dept_text) + exp_level = determine_experience_level(title) + + description_plain = j.get("descriptionPlain", "") or j.get("description", "") + desc_clean = clean_html_text(description_plain) + + collected.append({ + "title": title, + "company": company_name, + "location": location_name if location_name != "Remote" else "Remote, USA", + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.", + "salary_min": None, + "salary_max": None, + "job_url": job_url, + "source": "lever" + }) + except Exception as e: + print(f"[Lever Warning] {company_name} ({slug}) error: {e}") + return collected + def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}" collected = [] try: - r = requests.get(url, headers=self.headers, timeout=8) + r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) if r.status_code != 200: return collected data = r.json()