JB/scraper/scrapers/jobspy_runner.py

226 lines
12 KiB
Python

from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
import os
import time
import pandas as pd
from typing import List, Dict, Any
from scrapers.ats_ingestion import determine_experience_level, determine_department, parse_is_us_remote
def get_jobspy_proxies():
proxies = get_privado_proxy_list()
if proxies:
return proxies
return None
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"""
Executes JobSpy searches across nationwide US hubs and remote roles
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
"""
collected_jobs = []
try:
from jobspy import scrape_jobs
except ImportError:
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
return []
proxies = get_jobspy_proxies()
if proxies:
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
# If proxies are configured, enable zip_recruiter and glassdoor
sites = ["indeed"]
if proxies:
sites.extend(["zip_recruiter", "glassdoor"])
# Target key employment hubs across US states covering all industries
us_regions = [
("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]),
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]),
("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]),
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]),
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]),
("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]),
("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]),
("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]),
("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"])
]
# Nationwide Remote queries across ALL professional disciplines and seniority levels
remote_queries = [
# IT & Systems Administration (Entry Level & Support Focus)
"Entry Level IT Support", "Remote Help Desk Tier 1", "Junior IT Specialist", "Technical Support Representative",
"Junior Systems Administrator", "Remote Desktop Support Technician", "Service Desk Analyst", "IT Support Specialist",
"Systems Administrator", "Network Support Technician", "Cloud Support Associate", "Senior Systems Administrator",
# Healthcare & Medical (Entry through Senior)
"Medical Biller", "Telehealth Care Coordinator", "Remote Medical Records Clerk", "Telehealth Nurse",
"Clinical Research Coordinator", "Healthcare Recruiter", "Healthcare Data Analyst",
# Finance, Accounting & Legal
"Junior Staff Accountant", "Remote Bookkeeper", "Accounts Payable Specialist", "Staff Accountant",
"Financial Analyst", "Junior Legal Assistant", "Paralegal", "Compliance Specialist", "Underwriter",
"Senior Financial Analyst",
# Sales, Marketing & Customer Support (Entry through Senior)
"Remote Customer Support", "Customer Support Representative", "Customer Success Specialist",
"Sales Development Representative", "Account Executive", "Digital Marketing Specialist", "Content Writer",
"Senior Account Executive",
# Human Resources & Operations
"HR Assistant", "People Operations Coordinator", "Junior Recruiter", "HR Generalist",
"Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Senior HR Manager",
# Art, Design & Creative
"Junior Graphic Designer", "UI UX Designer", "Graphic Designer", "Product Designer",
"Video Editor", "Motion Graphics Animator", "Instructional Designer",
# Logistics, Supply Chain & Purchasing
"Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist", "Freight Broker Associate",
# Software & Engineering (Entry through Senior)
"Junior Software Engineer", "Entry Level Developer", "Software Engineer", "Frontend Developer",
"Backend Developer", "Full Stack Developer", "Data Analyst", "Data Engineer", "DevOps Engineer",
"Senior Software Engineer"
]
def _execute_scrape(site_list, term, loc, is_rem):
# 1. Try with configured proxies
if proxies:
try:
return scrape_jobs(
site_name=site_list,
search_term=term,
location=loc if not is_rem else None,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=is_rem,
proxies=proxies
)
except Exception as e:
# Proxy or site auth error, drop proxy and fall back
pass
# 2. Fall back to direct Indeed scrape (no proxy, highly reliable)
try:
return scrape_jobs(
site_name=["indeed"],
search_term=term,
location=loc if not is_rem else None,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=is_rem,
proxies=None
)
except Exception as e:
print(f"[JobSpy Warning] Scrape failed for '{term}' ({loc}): {e}")
return None
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
for loc, queries in us_regions:
for query in queries:
try:
print(f"[JobSpy] Searching {loc}: '{query}'")
jobs_df = _execute_scrape(sites, query, loc, False)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
loc_val = str(row.get("location", loc) or loc)
if loc_val == "nan":
loc_val = loc
title_str = str(row.get("title", "Untitled"))
desc_str = str(row.get("description", "") or "No description provided.")
# Even in regional searches, verify if JobSpy returned a true remote role
raw_is_remote = row.get("is_remote")
is_remote_flag = bool(raw_is_remote) if pd.notnull(raw_is_remote) and raw_is_remote not in ["nan", "None", ""] else False
is_remote = is_remote_flag or parse_is_us_remote(loc_val, title_str, desc_str)
# But strictly verify negative indicators
if is_remote and not parse_is_us_remote(loc_val, title_str, desc_str):
is_remote = False
collected_jobs.append({
"title": title_str,
"company": str(row.get("company", "Unknown")),
"location": "Remote, USA" if is_remote and ("remote" in loc_val.lower() or loc_val == loc) else loc_val,
"is_remote": is_remote,
"department": determine_department(title_str, ""),
"experience_level": determine_experience_level(title_str),
"description": desc_str,
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(1.5)
except Exception as e:
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
for query in remote_queries:
try:
print(f"[JobSpy] Searching US Remote: '{query}'")
jobs_df = _execute_scrape(sites, query, "USA", True)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
title_str = str(row.get("title", "Untitled"))
desc_str = str(row.get("description", "") or "No description provided.")
loc_val = str(row.get("location", "") or "").strip()
if loc_val in ["nan", "None"]:
loc_val = ""
# Verify remote status strictly
raw_is_remote = row.get("is_remote")
is_remote_flag = bool(raw_is_remote) if pd.notnull(raw_is_remote) and raw_is_remote not in ["nan", "None", ""] else True
# Pass through our strict parse_is_us_remote validator
header_loc = loc_val or "Remote, USA"
is_truly_remote = parse_is_us_remote(header_loc, title_str, desc_str)
# If the returned location is clearly a physical location (e.g. "Sacramento, CA") and not remote
if not is_truly_remote and not is_remote_flag:
resolved_loc = loc_val or "United States"
is_remote = False
elif not is_truly_remote and loc_val and ("remote" not in loc_val.lower()):
resolved_loc = loc_val
is_remote = False
else:
resolved_loc = loc_val if ("remote" in loc_val.lower()) else ("Remote, USA" if not loc_val else f"{loc_val} (Remote)")
is_remote = is_truly_remote
collected_jobs.append({
"title": title_str,
"company": str(row.get("company", "Unknown")),
"location": resolved_loc,
"is_remote": is_remote,
"department": determine_department(title_str, ""),
"experience_level": determine_experience_level(title_str),
"description": desc_str,
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(1.5)
except Exception as e:
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
return collected_jobs