Enhance proxy resilience with socks5h remote DNS and automatic direct fallback
This commit is contained in:
parent
1a914df12c
commit
73ceb260d5
3 changed files with 85 additions and 64 deletions
|
|
@ -54,21 +54,46 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
||||||
]
|
]
|
||||||
|
|
||||||
|
def _execute_scrape(site_list, term, loc, is_rem):
|
||||||
|
# 1. Try with configured proxies
|
||||||
|
if proxies:
|
||||||
|
try:
|
||||||
|
return scrape_jobs(
|
||||||
|
site_name=site_list,
|
||||||
|
search_term=term,
|
||||||
|
location=loc if not is_rem else None,
|
||||||
|
results_wanted=35,
|
||||||
|
hours_old=72,
|
||||||
|
country_indeed="USA",
|
||||||
|
is_remote=is_rem,
|
||||||
|
proxies=proxies
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
# Proxy or site auth error, drop proxy and fall back
|
||||||
|
pass
|
||||||
|
|
||||||
|
# 2. Fall back to direct Indeed scrape (no proxy, highly reliable)
|
||||||
|
try:
|
||||||
|
return scrape_jobs(
|
||||||
|
site_name=["indeed"],
|
||||||
|
search_term=term,
|
||||||
|
location=loc if not is_rem else None,
|
||||||
|
results_wanted=35,
|
||||||
|
hours_old=72,
|
||||||
|
country_indeed="USA",
|
||||||
|
is_remote=is_rem,
|
||||||
|
proxies=None
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[JobSpy Warning] Scrape failed for '{term}' ({loc}): {e}")
|
||||||
|
return None
|
||||||
|
|
||||||
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
||||||
for loc, queries in us_regions:
|
for loc, queries in us_regions:
|
||||||
for query in queries:
|
for query in queries:
|
||||||
try:
|
try:
|
||||||
print(f"[JobSpy] Searching {loc}: '{query}'")
|
print(f"[JobSpy] Searching {loc}: '{query}'")
|
||||||
jobs_df = scrape_jobs(
|
jobs_df = _execute_scrape(sites, query, loc, False)
|
||||||
site_name=sites,
|
|
||||||
search_term=query,
|
|
||||||
location=loc,
|
|
||||||
results_wanted=35,
|
|
||||||
hours_old=72,
|
|
||||||
country_indeed="USA",
|
|
||||||
is_remote=False,
|
|
||||||
proxies=proxies
|
|
||||||
)
|
|
||||||
|
|
||||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||||
for idx, row in jobs_df.iterrows():
|
for idx, row in jobs_df.iterrows():
|
||||||
|
|
@ -100,15 +125,7 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
for query in remote_queries:
|
for query in remote_queries:
|
||||||
try:
|
try:
|
||||||
print(f"[JobSpy] Searching US Remote: '{query}'")
|
print(f"[JobSpy] Searching US Remote: '{query}'")
|
||||||
jobs_df = scrape_jobs(
|
jobs_df = _execute_scrape(sites, query, "USA", True)
|
||||||
site_name=sites,
|
|
||||||
search_term=query,
|
|
||||||
results_wanted=35,
|
|
||||||
hours_old=72,
|
|
||||||
country_indeed="USA",
|
|
||||||
is_remote=True,
|
|
||||||
proxies=proxies
|
|
||||||
)
|
|
||||||
|
|
||||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||||
for idx, row in jobs_df.iterrows():
|
for idx, row in jobs_df.iterrows():
|
||||||
|
|
|
||||||
|
|
@ -2,16 +2,11 @@ import os
|
||||||
import random
|
import random
|
||||||
from typing import List, Dict, Optional
|
from typing import List, Dict, Optional
|
||||||
|
|
||||||
# PrivadoVPN SOCKS5 server endpoints across US & EU
|
# Verified live PrivadoVPN SOCKS5 server endpoints
|
||||||
|
# Note: uses socks5h:// for remote DNS resolution to prevent DNS leaks and connection drops
|
||||||
PRIVADO_SOCKS5_HOSTS = [
|
PRIVADO_SOCKS5_HOSTS = [
|
||||||
"jfk.socks.privado.io", # New York / JFK
|
"jfk.socks.privado.io", # New York JFK
|
||||||
"us-nyc.socks.privado.io", # New York City
|
|
||||||
"us-dal.socks.privado.io", # Dallas, TX
|
|
||||||
"us-chi.socks.privado.io", # Chicago, IL
|
|
||||||
"us-la.socks.privado.io", # Los Angeles, CA
|
|
||||||
"ams.socks.privado.io", # Amsterdam
|
"ams.socks.privado.io", # Amsterdam
|
||||||
"fra.socks.privado.io", # Frankfurt
|
|
||||||
"lon.socks.privado.io" # London
|
|
||||||
]
|
]
|
||||||
|
|
||||||
DEFAULT_PRIVADO_USER = "nhqyqsx21846"
|
DEFAULT_PRIVADO_USER = "nhqyqsx21846"
|
||||||
|
|
@ -20,13 +15,11 @@ DEFAULT_PORT = 1080
|
||||||
|
|
||||||
def get_privado_proxy_list() -> List[str]:
|
def get_privado_proxy_list() -> List[str]:
|
||||||
"""
|
"""
|
||||||
Builds full socks5:// proxy URLs for all configured Privado servers.
|
Builds socks5h:// proxy URLs for all verified Privado servers.
|
||||||
Respects overrides from PRIVADO_USER and PRIVADO_PASS environment variables.
|
|
||||||
"""
|
"""
|
||||||
user = os.getenv("PRIVADO_USER", DEFAULT_PRIVADO_USER).strip()
|
user = os.getenv("PRIVADO_USER", DEFAULT_PRIVADO_USER).strip()
|
||||||
pwd = os.getenv("PRIVADO_PASS", DEFAULT_PRIVADO_PASS).strip()
|
pwd = os.getenv("PRIVADO_PASS", DEFAULT_PRIVADO_PASS).strip()
|
||||||
|
|
||||||
# Also support explicit SOCKS5_PROXY or PROXY_URL list if passed as comma-separated
|
|
||||||
custom_proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or ""
|
custom_proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or ""
|
||||||
if custom_proxy:
|
if custom_proxy:
|
||||||
proxies = [p.strip() for p in custom_proxy.split(",") if p.strip()]
|
proxies = [p.strip() for p in custom_proxy.split(",") if p.strip()]
|
||||||
|
|
@ -35,7 +28,7 @@ def get_privado_proxy_list() -> List[str]:
|
||||||
|
|
||||||
proxy_list = []
|
proxy_list = []
|
||||||
for host in PRIVADO_SOCKS5_HOSTS:
|
for host in PRIVADO_SOCKS5_HOSTS:
|
||||||
proxy_list.append(f"socks5://{user}:{pwd}@{host}:{DEFAULT_PORT}")
|
proxy_list.append(f"socks5h://{user}:{pwd}@{host}:{DEFAULT_PORT}")
|
||||||
return proxy_list
|
return proxy_list
|
||||||
|
|
||||||
def get_random_privado_proxy() -> Optional[str]:
|
def get_random_privado_proxy() -> Optional[str]:
|
||||||
|
|
|
||||||
|
|
@ -95,24 +95,39 @@ class SmartCareersCrawler:
|
||||||
|
|
||||||
return None, None
|
return None, None
|
||||||
|
|
||||||
|
def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]:
|
||||||
|
"""
|
||||||
|
Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly.
|
||||||
|
"""
|
||||||
|
proxies = get_rotating_proxy_dict()
|
||||||
|
if proxies:
|
||||||
|
try:
|
||||||
|
r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
|
||||||
|
return r
|
||||||
|
except Exception:
|
||||||
|
# Proxy timeout or connection error; fall back to direct request immediately
|
||||||
|
pass
|
||||||
|
|
||||||
|
try:
|
||||||
|
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
||||||
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
|
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
|
||||||
base_url = f"https://{clean_domain}"
|
base_url = f"https://{clean_domain}"
|
||||||
|
|
||||||
for path in COMMON_CAREER_PATHS:
|
for path in COMMON_CAREER_PATHS:
|
||||||
test_url = f"{base_url}{path}"
|
test_url = f"{base_url}{path}"
|
||||||
try:
|
r = self._safe_get(test_url, timeout=5, allow_redirects=True)
|
||||||
r = requests.get(test_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
|
if r is not None:
|
||||||
final_url = r.url
|
final_url = r.url
|
||||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, final_url
|
return provider, slug, final_url
|
||||||
except Exception:
|
|
||||||
continue
|
|
||||||
|
|
||||||
try:
|
r = self._safe_get(base_url, timeout=5, allow_redirects=True)
|
||||||
r = requests.get(base_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
|
if r is not None and r.status_code == 200:
|
||||||
if r.status_code == 200:
|
|
||||||
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, r.url
|
return provider, slug, r.url
|
||||||
|
|
@ -123,15 +138,11 @@ class SmartCareersCrawler:
|
||||||
provider, slug = self.detect_ats_from_url_or_html(target)
|
provider, slug = self.detect_ats_from_url_or_html(target)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, target
|
return provider, slug, target
|
||||||
try:
|
cr = self._safe_get(target, timeout=5, allow_redirects=True)
|
||||||
cr = requests.get(target, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
|
if cr is not None:
|
||||||
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, cr.url
|
return provider, slug, cr.url
|
||||||
except Exception:
|
|
||||||
continue
|
|
||||||
except Exception:
|
|
||||||
pass
|
|
||||||
|
|
||||||
return None, None, None
|
return None, None, None
|
||||||
|
|
||||||
|
|
@ -139,8 +150,8 @@ class SmartCareersCrawler:
|
||||||
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
|
r = self._safe_get(url, timeout=8)
|
||||||
if r.status_code != 200:
|
if r is None or r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
data = r.json()
|
data = r.json()
|
||||||
jobs = data.get("jobs", [])
|
jobs = data.get("jobs", [])
|
||||||
|
|
@ -185,8 +196,8 @@ class SmartCareersCrawler:
|
||||||
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
|
r = self._safe_get(url, timeout=8)
|
||||||
if r.status_code != 200:
|
if r is None or r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
jobs = r.json()
|
jobs = r.json()
|
||||||
for j in jobs:
|
for j in jobs:
|
||||||
|
|
@ -231,8 +242,8 @@ class SmartCareersCrawler:
|
||||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
|
r = self._safe_get(url, timeout=8)
|
||||||
if r.status_code != 200:
|
if r is None or r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
data = r.json()
|
data = r.json()
|
||||||
jobs = data.get("jobs", [])
|
jobs = data.get("jobs", [])
|
||||||
|
|
@ -438,8 +449,8 @@ class SmartCareersCrawler:
|
||||||
|
|
||||||
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8, allow_redirects=True)
|
r = self._safe_get(url, timeout=8, allow_redirects=True)
|
||||||
if r.status_code != 200:
|
if r is None or r.status_code != 200:
|
||||||
return []
|
return []
|
||||||
html_text = r.text
|
html_text = r.text
|
||||||
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
|
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue