feat(scraper): integrate PrivadoVPN rotating SOCKS5 proxy pool across US and EU endpoints
This commit is contained in:
parent
f22b826193
commit
8ecd99ea9f
4 changed files with 67 additions and 18 deletions
|
|
@ -47,6 +47,8 @@ services:
|
||||||
SCRAPE_INTERVAL_MINUTES: "30"
|
SCRAPE_INTERVAL_MINUTES: "30"
|
||||||
SOCKS5_PROXY: "${SOCKS5_PROXY:-}"
|
SOCKS5_PROXY: "${SOCKS5_PROXY:-}"
|
||||||
PROXY_URL: "${PROXY_URL:-}"
|
PROXY_URL: "${PROXY_URL:-}"
|
||||||
|
PRIVADO_USER: "${PRIVADO_USER:-}"
|
||||||
|
PRIVADO_PASS: "${PRIVADO_PASS:-}"
|
||||||
depends_on:
|
depends_on:
|
||||||
db:
|
db:
|
||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
|
|
|
||||||
|
|
@ -1,12 +1,13 @@
|
||||||
|
from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
|
||||||
import os
|
import os
|
||||||
import time
|
import time
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
from typing import List, Dict, Any
|
from typing import List, Dict, Any
|
||||||
|
|
||||||
def get_jobspy_proxies():
|
def get_jobspy_proxies():
|
||||||
proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
|
proxies = get_privado_proxy_list()
|
||||||
if proxy:
|
if proxies:
|
||||||
return [proxy]
|
return proxies
|
||||||
return None
|
return None
|
||||||
|
|
||||||
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
|
|
|
||||||
51
scraper/scrapers/proxy_manager.py
Normal file
51
scraper/scrapers/proxy_manager.py
Normal file
|
|
@ -0,0 +1,51 @@
|
||||||
|
import os
|
||||||
|
import random
|
||||||
|
from typing import List, Dict, Optional
|
||||||
|
|
||||||
|
# PrivadoVPN SOCKS5 server endpoints across US & EU
|
||||||
|
PRIVADO_SOCKS5_HOSTS = [
|
||||||
|
"jfk.socks.privado.io", # New York / JFK
|
||||||
|
"us-nyc.socks.privado.io", # New York City
|
||||||
|
"us-dal.socks.privado.io", # Dallas, TX
|
||||||
|
"us-chi.socks.privado.io", # Chicago, IL
|
||||||
|
"us-la.socks.privado.io", # Los Angeles, CA
|
||||||
|
"ams.socks.privado.io", # Amsterdam
|
||||||
|
"fra.socks.privado.io", # Frankfurt
|
||||||
|
"lon.socks.privado.io" # London
|
||||||
|
]
|
||||||
|
|
||||||
|
DEFAULT_PRIVADO_USER = "nhqyqsx21846"
|
||||||
|
DEFAULT_PRIVADO_PASS = "umjy1xyucnch"
|
||||||
|
DEFAULT_PORT = 1080
|
||||||
|
|
||||||
|
def get_privado_proxy_list() -> List[str]:
|
||||||
|
"""
|
||||||
|
Builds full socks5:// proxy URLs for all configured Privado servers.
|
||||||
|
Respects overrides from PRIVADO_USER and PRIVADO_PASS environment variables.
|
||||||
|
"""
|
||||||
|
user = os.getenv("PRIVADO_USER", DEFAULT_PRIVADO_USER).strip()
|
||||||
|
pwd = os.getenv("PRIVADO_PASS", DEFAULT_PRIVADO_PASS).strip()
|
||||||
|
|
||||||
|
# Also support explicit SOCKS5_PROXY or PROXY_URL list if passed as comma-separated
|
||||||
|
custom_proxy = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or ""
|
||||||
|
if custom_proxy:
|
||||||
|
proxies = [p.strip() for p in custom_proxy.split(",") if p.strip()]
|
||||||
|
if proxies:
|
||||||
|
return proxies
|
||||||
|
|
||||||
|
proxy_list = []
|
||||||
|
for host in PRIVADO_SOCKS5_HOSTS:
|
||||||
|
proxy_list.append(f"socks5://{user}:{pwd}@{host}:{DEFAULT_PORT}")
|
||||||
|
return proxy_list
|
||||||
|
|
||||||
|
def get_random_privado_proxy() -> Optional[str]:
|
||||||
|
proxies = get_privado_proxy_list()
|
||||||
|
if proxies:
|
||||||
|
return random.choice(proxies)
|
||||||
|
return None
|
||||||
|
|
||||||
|
def get_rotating_proxy_dict() -> Dict[str, str]:
|
||||||
|
proxy = get_random_privado_proxy()
|
||||||
|
if proxy:
|
||||||
|
return {"http": proxy, "https": proxy}
|
||||||
|
return {}
|
||||||
|
|
@ -1,3 +1,4 @@
|
||||||
|
from scrapers.proxy_manager import get_rotating_proxy_dict
|
||||||
import os
|
import os
|
||||||
import requests
|
import requests
|
||||||
import re
|
import re
|
||||||
|
|
@ -54,14 +55,8 @@ class SmartCareersCrawler:
|
||||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
||||||
}
|
}
|
||||||
|
|
||||||
# Configure proxy if supplied in environment
|
# Configure rotating SOCKS5 proxies
|
||||||
proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or ""
|
self.proxies = get_rotating_proxy_dict()
|
||||||
self.proxies = {}
|
|
||||||
if proxy_url:
|
|
||||||
self.proxies = {
|
|
||||||
"http": proxy_url,
|
|
||||||
"https": proxy_url
|
|
||||||
}
|
|
||||||
|
|
||||||
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
||||||
parsed = urlparse(url)
|
parsed = urlparse(url)
|
||||||
|
|
@ -107,7 +102,7 @@ class SmartCareersCrawler:
|
||||||
for path in COMMON_CAREER_PATHS:
|
for path in COMMON_CAREER_PATHS:
|
||||||
test_url = f"{base_url}{path}"
|
test_url = f"{base_url}{path}"
|
||||||
try:
|
try:
|
||||||
r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
r = requests.get(test_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
|
||||||
final_url = r.url
|
final_url = r.url
|
||||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
|
|
@ -116,7 +111,7 @@ class SmartCareersCrawler:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
try:
|
try:
|
||||||
r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
r = requests.get(base_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
|
||||||
if r.status_code == 200:
|
if r.status_code == 200:
|
||||||
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
|
|
@ -129,7 +124,7 @@ class SmartCareersCrawler:
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, target
|
return provider, slug, target
|
||||||
try:
|
try:
|
||||||
cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True)
|
cr = requests.get(target, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True)
|
||||||
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
return provider, slug, cr.url
|
return provider, slug, cr.url
|
||||||
|
|
@ -144,7 +139,7 @@ class SmartCareersCrawler:
|
||||||
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
data = r.json()
|
data = r.json()
|
||||||
|
|
@ -190,7 +185,7 @@ class SmartCareersCrawler:
|
||||||
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
jobs = r.json()
|
jobs = r.json()
|
||||||
|
|
@ -236,7 +231,7 @@ class SmartCareersCrawler:
|
||||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
||||||
collected = []
|
collected = []
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8)
|
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
return collected
|
return collected
|
||||||
data = r.json()
|
data = r.json()
|
||||||
|
|
@ -443,7 +438,7 @@ class SmartCareersCrawler:
|
||||||
|
|
||||||
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||||
try:
|
try:
|
||||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8, allow_redirects=True)
|
r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8, allow_redirects=True)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
return []
|
return []
|
||||||
html_text = r.text
|
html_text = r.text
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue