feat(scraper): integrate PrivadoVPN rotating SOCKS5 proxy pool across US and EU endpoints

This commit is contained in:
JobsBoard Deployer 2026-09-05 12:42:22 -04:00
parent f22b826193
commit 8ecd99ea9f
4 changed files with 67 additions and 18 deletions

View file

@ -47,6 +47,8 @@ services:
SCRAPE_INTERVAL_MINUTES: "30" SCRAPE_INTERVAL_MINUTES: "30"
SOCKS5_PROXY: "${SOCKS5_PROXY:-}" SOCKS5_PROXY: "${SOCKS5_PROXY:-}"
PROXY_URL: "${PROXY_URL:-}" PROXY_URL: "${PROXY_URL:-}"
PRIVADO_USER: "${PRIVADO_USER:-}"
PRIVADO_PASS: "${PRIVADO_PASS:-}"
depends_on: depends_on:
db: db:
condition: service_healthy condition: service_healthy

View file

@ -1,12 +1,13 @@
from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
import os import os
import time import time
import pandas as pd import pandas as pd
from typing import List, Dict, Any from typing import List, Dict, Any
def get_jobspy_proxies(): def get_jobspy_proxies():
proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or "" proxies = get_privado_proxy_list()
if proxy: if proxies:
return [proxy] return proxies
return None return None
def run_jobspy_scrapes() -> List[Dict[Any, Any]]: def run_jobspy_scrapes() -> List[Dict[Any, Any]]:

View file

@ -0,0 +1,51 @@
import os
import random
from typing import List, Dict, Optional
# PrivadoVPN SOCKS5 server endpoints across US & EU
PRIVADO_SOCKS5_HOSTS = [
"jfk.socks.privado.io", # New York / JFK
"us-nyc.socks.privado.io", # New York City
"us-dal.socks.privado.io", # Dallas, TX
"us-chi.socks.privado.io", # Chicago, IL
"us-la.socks.privado.io", # Los Angeles, CA
"ams.socks.privado.io", # Amsterdam
"fra.socks.privado.io", # Frankfurt
"lon.socks.privado.io" # London
]
DEFAULT_PRIVADO_USER = "nhqyqsx21846"
DEFAULT_PRIVADO_PASS = "umjy1xyucnch"
DEFAULT_PORT = 1080
def get_privado_proxy_list() -> List[str]:
"""
Builds full socks5:// proxy URLs for all configured Privado servers.
Respects overrides from PRIVADO_USER and PRIVADO_PASS environment variables.
"""
user = os.getenv("PRIVADO_USER", DEFAULT_PRIVADO_USER).strip()
pwd = os.getenv("PRIVADO_PASS", DEFAULT_PRIVADO_PASS).strip()
# Also support explicit SOCKS5_PROXY or PROXY_URL list if passed as comma-separated
custom_proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or ""
if custom_proxy:
proxies = [p.strip() for p in custom_proxy.split(",") if p.strip()]
if proxies:
return proxies
proxy_list = []
for host in PRIVADO_SOCKS5_HOSTS:
proxy_list.append(f"socks5://{user}:{pwd}@{host}:{DEFAULT_PORT}")
return proxy_list
def get_random_privado_proxy() -> Optional[str]:
proxies = get_privado_proxy_list()
if proxies:
return random.choice(proxies)
return None
def get_rotating_proxy_dict() -> Dict[str, str]:
proxy = get_random_privado_proxy()
if proxy:
return {"http": proxy, "https": proxy}
return {}

View file

@ -1,3 +1,4 @@
from scrapers.proxy_manager import get_rotating_proxy_dict
import os import os
import requests import requests
import re import re
@ -54,14 +55,8 @@ class SmartCareersCrawler:
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
} }
# Configure proxy if supplied in environment # Configure rotating SOCKS5 proxies
proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or "" self.proxies = get_rotating_proxy_dict()
self.proxies = {}
if proxy_url:
self.proxies = {
"http": proxy_url,
"https": proxy_url
}
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]: def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
parsed = urlparse(url) parsed = urlparse(url)
@ -107,7 +102,7 @@ class SmartCareersCrawler:
for path in COMMON_CAREER_PATHS: for path in COMMON_CAREER_PATHS:
test_url = f"{base_url}{path}" test_url = f"{base_url}{path}"
try: try:
r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) r = requests.get(test_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
final_url = r.url final_url = r.url
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "") provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
if provider and slug: if provider and slug:
@ -116,7 +111,7 @@ class SmartCareersCrawler:
continue continue
try: try:
r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) r = requests.get(base_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
if r.status_code == 200: if r.status_code == 200:
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
if provider and slug: if provider and slug:
@ -129,7 +124,7 @@ class SmartCareersCrawler:
if provider and slug: if provider and slug:
return provider, slug, target return provider, slug, target
try: try:
cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) cr = requests.get(target, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
if provider and slug: if provider and slug:
return provider, slug, cr.url return provider, slug, cr.url
@ -144,7 +139,7 @@ class SmartCareersCrawler:
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true" url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
collected = [] collected = []
try: try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
if r.status_code != 200: if r.status_code != 200:
return collected return collected
data = r.json() data = r.json()
@ -190,7 +185,7 @@ class SmartCareersCrawler:
url = f"https://api.lever.co/v0/postings/{slug}?mode=json" url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
collected = [] collected = []
try: try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
if r.status_code != 200: if r.status_code != 200:
return collected return collected
jobs = r.json() jobs = r.json()
@ -236,7 +231,7 @@ class SmartCareersCrawler:
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}" url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
collected = [] collected = []
try: try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
if r.status_code != 200: if r.status_code != 200:
return collected return collected
data = r.json() data = r.json()
@ -443,7 +438,7 @@ class SmartCareersCrawler:
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]: def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
try: try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8, allow_redirects=True) r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8, allow_redirects=True)
if r.status_code != 200: if r.status_code != 200:
return [] return []
html_text = r.text html_text = r.text