From 73ceb260d5e1ca4dbdff1a711a172264246b4fc5 Mon Sep 17 00:00:00 2001 From: JobsBoard Deployer Date: Sat, 5 Sep 2026 12:53:18 -0400 Subject: [PATCH] Enhance proxy resilience with socks5h remote DNS and automatic direct fallback --- scraper/scrapers/jobspy_runner.py | 55 +++++++++++------ scraper/scrapers/proxy_manager.py | 19 ++---- scraper/scrapers/smart_careers_crawler.py | 75 +++++++++++++---------- 3 files changed, 85 insertions(+), 64 deletions(-) diff --git a/scraper/scrapers/jobspy_runner.py b/scraper/scrapers/jobspy_runner.py index 4f16de8..fd1314e 100644 --- a/scraper/scrapers/jobspy_runner.py +++ b/scraper/scrapers/jobspy_runner.py @@ -54,21 +54,46 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: "DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer" ] + def _execute_scrape(site_list, term, loc, is_rem): + # 1. Try with configured proxies + if proxies: + try: + return scrape_jobs( + site_name=site_list, + search_term=term, + location=loc if not is_rem else None, + results_wanted=35, + hours_old=72, + country_indeed="USA", + is_remote=is_rem, + proxies=proxies + ) + except Exception as e: + # Proxy or site auth error, drop proxy and fall back + pass + + # 2. Fall back to direct Indeed scrape (no proxy, highly reliable) + try: + return scrape_jobs( + site_name=["indeed"], + search_term=term, + location=loc if not is_rem else None, + results_wanted=35, + hours_old=72, + country_indeed="USA", + is_remote=is_rem, + proxies=None + ) + except Exception as e: + print(f"[JobSpy Warning] Scrape failed for '{term}' ({loc}): {e}") + return None + print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...") for loc, queries in us_regions: for query in queries: try: print(f"[JobSpy] Searching {loc}: '{query}'") - jobs_df = scrape_jobs( - site_name=sites, - search_term=query, - location=loc, - results_wanted=35, - hours_old=72, - country_indeed="USA", - is_remote=False, - proxies=proxies - ) + jobs_df = _execute_scrape(sites, query, loc, False) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: for idx, row in jobs_df.iterrows(): @@ -100,15 +125,7 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: for query in remote_queries: try: print(f"[JobSpy] Searching US Remote: '{query}'") - jobs_df = scrape_jobs( - site_name=sites, - search_term=query, - results_wanted=35, - hours_old=72, - country_indeed="USA", - is_remote=True, - proxies=proxies - ) + jobs_df = _execute_scrape(sites, query, "USA", True) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: for idx, row in jobs_df.iterrows(): diff --git a/scraper/scrapers/proxy_manager.py b/scraper/scrapers/proxy_manager.py index 59c66af..0ad8712 100644 --- a/scraper/scrapers/proxy_manager.py +++ b/scraper/scrapers/proxy_manager.py @@ -2,16 +2,11 @@ import os import random from typing import List, Dict, Optional -# PrivadoVPN SOCKS5 server endpoints across US & EU +# Verified live PrivadoVPN SOCKS5 server endpoints +# Note: uses socks5h:// for remote DNS resolution to prevent DNS leaks and connection drops PRIVADO_SOCKS5_HOSTS = [ - "jfk.socks.privado.io", # New York / JFK - "us-nyc.socks.privado.io", # New York City - "us-dal.socks.privado.io", # Dallas, TX - "us-chi.socks.privado.io", # Chicago, IL - "us-la.socks.privado.io", # Los Angeles, CA - "ams.socks.privado.io", # Amsterdam - "fra.socks.privado.io", # Frankfurt - "lon.socks.privado.io" # London + "jfk.socks.privado.io", # New York JFK + "ams.socks.privado.io", # Amsterdam ] DEFAULT_PRIVADO_USER = "nhqyqsx21846" @@ -20,13 +15,11 @@ DEFAULT_PORT = 1080 def get_privado_proxy_list() -> List[str]: """ - Builds full socks5:// proxy URLs for all configured Privado servers. - Respects overrides from PRIVADO_USER and PRIVADO_PASS environment variables. + Builds socks5h:// proxy URLs for all verified Privado servers. """ user = os.getenv("PRIVADO_USER", DEFAULT_PRIVADO_USER).strip() pwd = os.getenv("PRIVADO_PASS", DEFAULT_PRIVADO_PASS).strip() - # Also support explicit SOCKS5_PROXY or PROXY_URL list if passed as comma-separated custom_proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or "" if custom_proxy: proxies = [p.strip() for p in custom_proxy.split(",") if p.strip()] @@ -35,7 +28,7 @@ def get_privado_proxy_list() -> List[str]: proxy_list = [] for host in PRIVADO_SOCKS5_HOSTS: - proxy_list.append(f"socks5://{user}:{pwd}@{host}:{DEFAULT_PORT}") + proxy_list.append(f"socks5h://{user}:{pwd}@{host}:{DEFAULT_PORT}") return proxy_list def get_random_privado_proxy() -> Optional[str]: diff --git a/scraper/scrapers/smart_careers_crawler.py b/scraper/scrapers/smart_careers_crawler.py index fbee87c..3027ed4 100644 --- a/scraper/scrapers/smart_careers_crawler.py +++ b/scraper/scrapers/smart_careers_crawler.py @@ -95,43 +95,54 @@ class SmartCareersCrawler: return None, None + def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]: + """ + Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly. + """ + proxies = get_rotating_proxy_dict() + if proxies: + try: + r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects) + return r + except Exception: + # Proxy timeout or connection error; fall back to direct request immediately + pass + + try: + return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects) + except Exception: + return None + def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]: clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower() base_url = f"https://{clean_domain}" for path in COMMON_CAREER_PATHS: test_url = f"{base_url}{path}" - try: - r = requests.get(test_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True) + r = self._safe_get(test_url, timeout=5, allow_redirects=True) + if r is not None: final_url = r.url provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "") if provider and slug: return provider, slug, final_url - except Exception: - continue - try: - r = requests.get(base_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True) - if r.status_code == 200: - provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) + r = self._safe_get(base_url, timeout=5, allow_redirects=True) + if r is not None and r.status_code == 200: + provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) + if provider and slug: + return provider, slug, r.url + + links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) + for link in links[:5]: + target = link if link.startswith("http") else urljoin(base_url, link) + provider, slug = self.detect_ats_from_url_or_html(target) if provider and slug: - return provider, slug, r.url - - links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) - for link in links[:5]: - target = link if link.startswith("http") else urljoin(base_url, link) - provider, slug = self.detect_ats_from_url_or_html(target) + return provider, slug, target + cr = self._safe_get(target, timeout=5, allow_redirects=True) + if cr is not None: + provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) if provider and slug: - return provider, slug, target - try: - cr = requests.get(target, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True) - provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) - if provider and slug: - return provider, slug, cr.url - except Exception: - continue - except Exception: - pass + return provider, slug, cr.url return None, None, None @@ -139,8 +150,8 @@ class SmartCareersCrawler: url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true" collected = [] try: - r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8) - if r.status_code != 200: + r = self._safe_get(url, timeout=8) + if r is None or r.status_code != 200: return collected data = r.json() jobs = data.get("jobs", []) @@ -185,8 +196,8 @@ class SmartCareersCrawler: url = f"https://api.lever.co/v0/postings/{slug}?mode=json" collected = [] try: - r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8) - if r.status_code != 200: + r = self._safe_get(url, timeout=8) + if r is None or r.status_code != 200: return collected jobs = r.json() for j in jobs: @@ -231,8 +242,8 @@ class SmartCareersCrawler: url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}" collected = [] try: - r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8) - if r.status_code != 200: + r = self._safe_get(url, timeout=8) + if r is None or r.status_code != 200: return collected data = r.json() jobs = data.get("jobs", []) @@ -438,8 +449,8 @@ class SmartCareersCrawler: def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]: try: - r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8, allow_redirects=True) - if r.status_code != 200: + r = self._safe_get(url, timeout=8, allow_redirects=True) + if r is None or r.status_code != 200: return [] html_text = r.text # 1. Try Schema.org JSON-LD first (highest quality structured jobs)