136 lines
6.2 KiB
Python
136 lines
6.2 KiB
Python
from scrapers.proxy_manager import get_privado_proxy_list, get_random_privado_proxy
|
|
import os
|
|
import time
|
|
import pandas as pd
|
|
from typing import List, Dict, Any
|
|
|
|
def get_jobspy_proxies():
|
|
proxies = get_privado_proxy_list()
|
|
if proxies:
|
|
return proxies
|
|
return None
|
|
|
|
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|
"""
|
|
Executes JobSpy searches across nationwide US hubs and remote roles
|
|
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
|
Gracefully handles Cloudflare/datacenter blocks and proxy rotation.
|
|
"""
|
|
collected_jobs = []
|
|
|
|
try:
|
|
from jobspy import scrape_jobs
|
|
except ImportError:
|
|
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
|
return []
|
|
|
|
proxies = get_jobspy_proxies()
|
|
if proxies:
|
|
print(f"[JobSpy] Using configured proxy for JobSpy requests: {proxies[0].split('@')[-1]}")
|
|
|
|
# Prioritize Indeed (reliable without Cloudflare Captcha compared to ZipRecruiter on server IPs)
|
|
# If proxies are configured, enable zip_recruiter and glassdoor
|
|
sites = ["indeed"]
|
|
if proxies:
|
|
sites.extend(["zip_recruiter", "glassdoor"])
|
|
|
|
# Target key employment hubs across US states
|
|
us_regions = [
|
|
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
|
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
|
|
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
|
|
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
|
|
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
|
|
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
|
|
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
|
|
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
|
|
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
|
|
]
|
|
|
|
# Nationwide Remote queries
|
|
remote_queries = [
|
|
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
|
|
"Customer Support", "Administrative Assistant", "IT Support Specialist",
|
|
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
|
]
|
|
|
|
print(f"[JobSpy] Starting Nationwide Regional Scrapes using sites: {sites}...")
|
|
for loc, queries in us_regions:
|
|
for query in queries:
|
|
try:
|
|
print(f"[JobSpy] Searching {loc}: '{query}'")
|
|
jobs_df = scrape_jobs(
|
|
site_name=sites,
|
|
search_term=query,
|
|
location=loc,
|
|
results_wanted=35,
|
|
hours_old=72,
|
|
country_indeed="USA",
|
|
is_remote=False,
|
|
proxies=proxies
|
|
)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
loc_val = str(row.get("location", loc) or loc)
|
|
if loc_val == "nan":
|
|
loc_val = loc
|
|
|
|
collected_jobs.append({
|
|
"title": str(row.get("title", "Untitled")),
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": loc_val,
|
|
"is_remote": False,
|
|
"description": str(row.get("description", "") or "No description provided."),
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(1.5)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
|
|
|
|
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
|
|
for query in remote_queries:
|
|
try:
|
|
print(f"[JobSpy] Searching US Remote: '{query}'")
|
|
jobs_df = scrape_jobs(
|
|
site_name=sites,
|
|
search_term=query,
|
|
results_wanted=35,
|
|
hours_old=72,
|
|
country_indeed="USA",
|
|
is_remote=True,
|
|
proxies=proxies
|
|
)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
collected_jobs.append({
|
|
"title": str(row.get("title", "Untitled")),
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": "Remote, USA",
|
|
"is_remote": True,
|
|
"description": str(row.get("description", "") or "No description provided."),
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(1.5)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
|
|
|
|
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
|
|
return collected_jobs
|