175 lines
8.8 KiB
Python
175 lines
8.8 KiB
Python
from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
|
|
import os
|
|
import time
|
|
import pandas as pd
|
|
from typing import List, Dict, Any
|
|
from scrapers.ats_ingestion import determine_experience_level, determine_department
|
|
|
|
def get_jobspy_proxies():
|
|
proxies = get_privado_proxy_list()
|
|
if proxies:
|
|
return proxies
|
|
return None
|
|
|
|
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|
"""
|
|
Executes JobSpy searches across nationwide US hubs and remote roles
|
|
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
|
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
|
|
"""
|
|
collected_jobs = []
|
|
|
|
try:
|
|
from jobspy import scrape_jobs
|
|
except ImportError:
|
|
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
|
return []
|
|
|
|
proxies = get_jobspy_proxies()
|
|
if proxies:
|
|
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
|
|
|
|
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
|
|
# If proxies are configured, enable zip_recruiter and glassdoor
|
|
sites = ["indeed"]
|
|
if proxies:
|
|
sites.extend(["zip_recruiter", "glassdoor"])
|
|
|
|
# Target key employment hubs across US states covering all industries
|
|
us_regions = [
|
|
("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]),
|
|
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]),
|
|
("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]),
|
|
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]),
|
|
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]),
|
|
("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]),
|
|
("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]),
|
|
("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]),
|
|
("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"])
|
|
]
|
|
|
|
# Nationwide Remote queries across ALL professional disciplines
|
|
remote_queries = [
|
|
# Healthcare & Medical
|
|
"Medical Biller", "Telehealth Nurse", "Clinical Research Coordinator", "Healthcare Recruiter",
|
|
# Finance, Accounting & Legal
|
|
"Staff Accountant", "Financial Analyst", "Bookkeeper", "Paralegal", "Compliance Specialist", "Underwriter",
|
|
# Sales, Marketing & Customer Support
|
|
"Customer Support Representative", "Customer Success Manager", "Account Executive", "Digital Marketing Specialist", "Content Writer",
|
|
# Human Resources & Operations
|
|
"HR Generalist", "Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Project Coordinator",
|
|
# Art, Design & Creative
|
|
"Graphic Designer", "UX Designer", "Video Editor", "Instructional Designer",
|
|
# Logistics, Supply Chain & Purchasing
|
|
"Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist",
|
|
# IT, Systems Administration & Technical Support (Entry Level & Support Focus)
|
|
"Entry Level IT Support", "Remote Help Desk Tier 1", "Junior IT Specialist", "Technical Support Representative",
|
|
"Junior Systems Administrator", "Remote Desktop Support Technician", "Service Desk Analyst", "IT Support Specialist",
|
|
"Systems Administrator", "Network Support Technician", "Cloud Support Associate",
|
|
# Software & Engineering
|
|
"Junior Software Engineer", "Software Engineer", "Entry Level Developer", "Data Analyst", "DevOps Engineer"
|
|
]
|
|
|
|
def _execute_scrape(site_list, term, loc, is_rem):
|
|
# 1. Try with configured proxies
|
|
if proxies:
|
|
try:
|
|
return scrape_jobs(
|
|
site_name=site_list,
|
|
search_term=term,
|
|
location=loc if not is_rem else None,
|
|
results_wanted=35,
|
|
hours_old=72,
|
|
country_indeed="USA",
|
|
is_remote=is_rem,
|
|
proxies=proxies
|
|
)
|
|
except Exception as e:
|
|
# Proxy or site auth error, drop proxy and fall back
|
|
pass
|
|
|
|
# 2. Fall back to direct Indeed scrape (no proxy, highly reliable)
|
|
try:
|
|
return scrape_jobs(
|
|
site_name=["indeed"],
|
|
search_term=term,
|
|
location=loc if not is_rem else None,
|
|
results_wanted=35,
|
|
hours_old=72,
|
|
country_indeed="USA",
|
|
is_remote=is_rem,
|
|
proxies=None
|
|
)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Scrape failed for '{term}' ({loc}): {e}")
|
|
return None
|
|
|
|
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
|
for loc, queries in us_regions:
|
|
for query in queries:
|
|
try:
|
|
print(f"[JobSpy] Searching {loc}: '{query}'")
|
|
jobs_df = _execute_scrape(sites, query, loc, False)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
loc_val = str(row.get("location", loc) or loc)
|
|
if loc_val == "nan":
|
|
loc_val = loc
|
|
|
|
title_str = str(row.get("title", "Untitled"))
|
|
collected_jobs.append({
|
|
"title": title_str,
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": loc_val,
|
|
"is_remote": False,
|
|
"department": determine_department(title_str, ""),
|
|
"experience_level": determine_experience_level(title_str),
|
|
"description": str(row.get("description", "") or "No description provided."),
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(1.5)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
|
|
|
|
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
|
|
for query in remote_queries:
|
|
try:
|
|
print(f"[JobSpy] Searching US Remote: '{query}'")
|
|
jobs_df = _execute_scrape(sites, query, "USA", True)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
title_str = str(row.get("title", "Untitled"))
|
|
collected_jobs.append({
|
|
"title": title_str,
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": "Remote, USA",
|
|
"is_remote": True,
|
|
"department": determine_department(title_str, ""),
|
|
"experience_level": determine_experience_level(title_str),
|
|
"description": str(row.get("description", "") or "No description provided."),
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(1.5)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
|
|
|
|
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
|
|
return collected_jobs
|