JB/scraper/scrapers/jobspy_runner.py

102 lines
4.5 KiB
Python

import time
import pandas as pd
from typing import List, Dict, Any
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"""
Executes JobSpy searches for CT local jobs across all industries
and nationwide remote roles across major job categories.
"""
collected_jobs = []
try:
from jobspy import scrape_jobs
except ImportError:
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
return []
# CT local queries (all industries / general)
ct_queries = [
"Customer Support", "Administrative", "Healthcare", "Education",
"Operations", "Retail", "Finance", "Software", "Engineering", "General"
]
# Remote queries
remote_queries = [
"Customer Support", "Administrative Assistant", "IT Support",
"Data Analyst", "Project Manager", "Software Engineer", "Marketing"
]
print("[JobSpy] Starting CT Local Scrapes...")
for query in ct_queries:
try:
print(f"[JobSpy] Searching CT Local: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed", "zip_recruiter"],
search_term=query,
location="Connecticut",
results_wanted=15,
hours_old=72,
country_indeed='USA',
is_remote=False
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
collected_jobs.append({
"title": str(row.get("title", "Untitled")),
"company": str(row.get("company", "Unknown")),
"location": str(row.get("location", "Connecticut, USA")),
"is_remote": False,
"description": str(row.get("description", "") or "No description provided."),
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(2) # Graceful delay between queries
except Exception as e:
print(f"[JobSpy Warning] CT search '{query}' failed: {e}")
print("[JobSpy] Starting Remote Scrapes...")
for query in remote_queries:
try:
print(f"[JobSpy] Searching Remote: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed", "zip_recruiter"],
search_term=query,
results_wanted=15,
hours_old=72,
country_indeed='USA',
is_remote=True
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
collected_jobs.append({
"title": str(row.get("title", "Untitled")),
"company": str(row.get("company", "Unknown")),
"location": "Remote, USA",
"is_remote": True,
"description": str(row.get("description", "") or "No description provided."),
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(2)
except Exception as e:
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
return collected_jobs