226 lines
12 KiB
Python
226 lines
12 KiB
Python
from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
|
|
import os
|
|
import time
|
|
import pandas as pd
|
|
from typing import List, Dict, Any
|
|
from scrapers.ats_ingestion import determine_experience_level, determine_department, parse_is_us_remote
|
|
|
|
def get_jobspy_proxies():
|
|
proxies = get_privado_proxy_list()
|
|
if proxies:
|
|
return proxies
|
|
return None
|
|
|
|
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|
"""
|
|
Executes JobSpy searches across nationwide US hubs and remote roles
|
|
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
|
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
|
|
"""
|
|
collected_jobs = []
|
|
|
|
try:
|
|
from jobspy import scrape_jobs
|
|
except ImportError:
|
|
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
|
return []
|
|
|
|
proxies = get_jobspy_proxies()
|
|
if proxies:
|
|
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
|
|
|
|
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
|
|
# If proxies are configured, enable zip_recruiter and glassdoor
|
|
sites = ["indeed"]
|
|
if proxies:
|
|
sites.extend(["zip_recruiter", "glassdoor"])
|
|
|
|
# Target key employment hubs across US states covering all industries
|
|
us_regions = [
|
|
("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]),
|
|
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]),
|
|
("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]),
|
|
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]),
|
|
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]),
|
|
("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]),
|
|
("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]),
|
|
("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]),
|
|
("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"])
|
|
]
|
|
|
|
# Nationwide Remote queries across ALL professional disciplines and seniority levels
|
|
remote_queries = [
|
|
# IT & Systems Administration (Entry Level & Support Focus)
|
|
"Entry Level IT Support", "Remote Help Desk Tier 1", "Junior IT Specialist", "Technical Support Representative",
|
|
"Junior Systems Administrator", "Remote Desktop Support Technician", "Service Desk Analyst", "IT Support Specialist",
|
|
"Systems Administrator", "Network Support Technician", "Cloud Support Associate", "Senior Systems Administrator",
|
|
|
|
# Healthcare & Medical (Entry through Senior)
|
|
"Medical Biller", "Telehealth Care Coordinator", "Remote Medical Records Clerk", "Telehealth Nurse",
|
|
"Clinical Research Coordinator", "Healthcare Recruiter", "Healthcare Data Analyst",
|
|
|
|
# Finance, Accounting & Legal
|
|
"Junior Staff Accountant", "Remote Bookkeeper", "Accounts Payable Specialist", "Staff Accountant",
|
|
"Financial Analyst", "Junior Legal Assistant", "Paralegal", "Compliance Specialist", "Underwriter",
|
|
"Senior Financial Analyst",
|
|
|
|
# Sales, Marketing & Customer Support (Entry through Senior)
|
|
"Remote Customer Support", "Customer Support Representative", "Customer Success Specialist",
|
|
"Sales Development Representative", "Account Executive", "Digital Marketing Specialist", "Content Writer",
|
|
"Senior Account Executive",
|
|
|
|
# Human Resources & Operations
|
|
"HR Assistant", "People Operations Coordinator", "Junior Recruiter", "HR Generalist",
|
|
"Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Senior HR Manager",
|
|
|
|
# Art, Design & Creative
|
|
"Junior Graphic Designer", "UI UX Designer", "Graphic Designer", "Product Designer",
|
|
"Video Editor", "Motion Graphics Animator", "Instructional Designer",
|
|
|
|
# Logistics, Supply Chain & Purchasing
|
|
"Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist", "Freight Broker Associate",
|
|
|
|
# Software & Engineering (Entry through Senior)
|
|
"Junior Software Engineer", "Entry Level Developer", "Software Engineer", "Frontend Developer",
|
|
"Backend Developer", "Full Stack Developer", "Data Analyst", "Data Engineer", "DevOps Engineer",
|
|
"Senior Software Engineer"
|
|
]
|
|
|
|
def _execute_scrape(site_list, term, loc, is_rem):
|
|
# 1. Try with configured proxies
|
|
if proxies:
|
|
try:
|
|
return scrape_jobs(
|
|
site_name=site_list,
|
|
search_term=term,
|
|
location=loc if not is_rem else None,
|
|
results_wanted=35,
|
|
hours_old=72,
|
|
country_indeed="USA",
|
|
is_remote=is_rem,
|
|
proxies=proxies
|
|
)
|
|
except Exception as e:
|
|
# Proxy or site auth error, drop proxy and fall back
|
|
pass
|
|
|
|
# 2. Fall back to direct Indeed scrape (no proxy, highly reliable)
|
|
try:
|
|
return scrape_jobs(
|
|
site_name=["indeed"],
|
|
search_term=term,
|
|
location=loc if not is_rem else None,
|
|
results_wanted=35,
|
|
hours_old=72,
|
|
country_indeed="USA",
|
|
is_remote=is_rem,
|
|
proxies=None
|
|
)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Scrape failed for '{term}' ({loc}): {e}")
|
|
return None
|
|
|
|
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
|
for loc, queries in us_regions:
|
|
for query in queries:
|
|
try:
|
|
print(f"[JobSpy] Searching {loc}: '{query}'")
|
|
jobs_df = _execute_scrape(sites, query, loc, False)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
loc_val = str(row.get("location", loc) or loc)
|
|
if loc_val == "nan":
|
|
loc_val = loc
|
|
|
|
title_str = str(row.get("title", "Untitled"))
|
|
desc_str = str(row.get("description", "") or "No description provided.")
|
|
|
|
# Even in regional searches, verify if JobSpy returned a true remote role
|
|
raw_is_remote = row.get("is_remote")
|
|
is_remote_flag = bool(raw_is_remote) if pd.notnull(raw_is_remote) and raw_is_remote not in ["nan", "None", ""] else False
|
|
is_remote = is_remote_flag or parse_is_us_remote(loc_val, title_str, desc_str)
|
|
|
|
# But strictly verify negative indicators
|
|
if is_remote and not parse_is_us_remote(loc_val, title_str, desc_str):
|
|
is_remote = False
|
|
|
|
collected_jobs.append({
|
|
"title": title_str,
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": "Remote, USA" if is_remote and ("remote" in loc_val.lower() or loc_val == loc) else loc_val,
|
|
"is_remote": is_remote,
|
|
"department": determine_department(title_str, ""),
|
|
"experience_level": determine_experience_level(title_str),
|
|
"description": desc_str,
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(1.5)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
|
|
|
|
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
|
|
for query in remote_queries:
|
|
try:
|
|
print(f"[JobSpy] Searching US Remote: '{query}'")
|
|
jobs_df = _execute_scrape(sites, query, "USA", True)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
title_str = str(row.get("title", "Untitled"))
|
|
desc_str = str(row.get("description", "") or "No description provided.")
|
|
loc_val = str(row.get("location", "") or "").strip()
|
|
if loc_val in ["nan", "None"]:
|
|
loc_val = ""
|
|
|
|
# Verify remote status strictly
|
|
raw_is_remote = row.get("is_remote")
|
|
is_remote_flag = bool(raw_is_remote) if pd.notnull(raw_is_remote) and raw_is_remote not in ["nan", "None", ""] else True
|
|
|
|
# Pass through our strict parse_is_us_remote validator
|
|
header_loc = loc_val or "Remote, USA"
|
|
is_truly_remote = parse_is_us_remote(header_loc, title_str, desc_str)
|
|
|
|
# If the returned location is clearly a physical location (e.g. "Sacramento, CA") and not remote
|
|
if not is_truly_remote and not is_remote_flag:
|
|
resolved_loc = loc_val or "United States"
|
|
is_remote = False
|
|
elif not is_truly_remote and loc_val and ("remote" not in loc_val.lower()):
|
|
resolved_loc = loc_val
|
|
is_remote = False
|
|
else:
|
|
resolved_loc = loc_val if ("remote" in loc_val.lower()) else ("Remote, USA" if not loc_val else f"{loc_val} (Remote)")
|
|
is_remote = is_truly_remote
|
|
|
|
collected_jobs.append({
|
|
"title": title_str,
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": resolved_loc,
|
|
"is_remote": is_remote,
|
|
"department": determine_department(title_str, ""),
|
|
"experience_level": determine_experience_level(title_str),
|
|
"description": desc_str,
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(1.5)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
|
|
|
|
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
|
|
return collected_jobs
|