JB/scraper/scrapers/jobspy_runner.py

136 lines
6.2 KiB
Python

from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
import os
import time
import pandas as pd
from typing import List, Dict, Any
def get_jobspy_proxies():
proxies = get_privado_proxy_list()
if proxies:
return proxies
return None
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"""
Executes JobSpy searches across nationwide US hubs and remote roles
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
"""
collected_jobs = []
try:
from jobspy import scrape_jobs
except ImportError:
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
return []
proxies = get_jobspy_proxies()
if proxies:
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
# If proxies are configured, enable zip_recruiter and glassdoor
sites = ["indeed"]
if proxies:
sites.extend(["zip_recruiter", "glassdoor"])
# Target key employment hubs across US states
us_regions = [
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
]
# Nationwide Remote queries
remote_queries = [
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
"Customer Support", "Administrative Assistant", "IT Support Specialist",
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
]
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
for loc, queries in us_regions:
for query in queries:
try:
print(f"[JobSpy] Searching {loc}: '{query}'")
jobs_df = scrape_jobs(
site_name=sites,
search_term=query,
location=loc,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=False,
proxies=proxies
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
loc_val = str(row.get("location", loc) or loc)
if loc_val == "nan":
loc_val = loc
collected_jobs.append({
"title": str(row.get("title", "Untitled")),
"company": str(row.get("company", "Unknown")),
"location": loc_val,
"is_remote": False,
"description": str(row.get("description", "") or "No description provided."),
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(1.5)
except Exception as e:
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
for query in remote_queries:
try:
print(f"[JobSpy] Searching US Remote: '{query}'")
jobs_df = scrape_jobs(
site_name=sites,
search_term=query,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=True,
proxies=proxies
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
collected_jobs.append({
"title": str(row.get("title", "Untitled")),
"company": str(row.get("company", "Unknown")),
"location": "Remote, USA",
"is_remote": True,
"description": str(row.get("description", "") or "No description provided."),
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(1.5)
except Exception as e:
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
return collected_jobs