102 lines
4.5 KiB
Python
102 lines
4.5 KiB
Python
import time
|
|
import pandas as pd
|
|
from typing import List, Dict, Any
|
|
|
|
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|
"""
|
|
Executes JobSpy searches for CT local jobs across all industries
|
|
and nationwide remote roles across major job categories.
|
|
"""
|
|
collected_jobs = []
|
|
|
|
try:
|
|
from jobspy import scrape_jobs
|
|
except ImportError:
|
|
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
|
return []
|
|
|
|
# CT local queries (all industries / general)
|
|
ct_queries = [
|
|
"Customer Support", "Administrative", "Healthcare", "Education",
|
|
"Operations", "Retail", "Finance", "Software", "Engineering", "General"
|
|
]
|
|
|
|
# Remote queries
|
|
remote_queries = [
|
|
"Customer Support", "Administrative Assistant", "IT Support",
|
|
"Data Analyst", "Project Manager", "Software Engineer", "Marketing"
|
|
]
|
|
|
|
print("[JobSpy] Starting CT Local Scrapes...")
|
|
for query in ct_queries:
|
|
try:
|
|
print(f"[JobSpy] Searching CT Local: '{query}'")
|
|
jobs_df = scrape_jobs(
|
|
site_name=["indeed", "zip_recruiter"],
|
|
search_term=query,
|
|
location="Connecticut",
|
|
results_wanted=15,
|
|
hours_old=72,
|
|
country_indeed='USA',
|
|
is_remote=False
|
|
)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
collected_jobs.append({
|
|
"title": str(row.get("title", "Untitled")),
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": str(row.get("location", "Connecticut, USA")),
|
|
"is_remote": False,
|
|
"description": str(row.get("description", "") or "No description provided."),
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(2) # Graceful delay between queries
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] CT search '{query}' failed: {e}")
|
|
|
|
print("[JobSpy] Starting Remote Scrapes...")
|
|
for query in remote_queries:
|
|
try:
|
|
print(f"[JobSpy] Searching Remote: '{query}'")
|
|
jobs_df = scrape_jobs(
|
|
site_name=["indeed", "zip_recruiter"],
|
|
search_term=query,
|
|
results_wanted=15,
|
|
hours_old=72,
|
|
country_indeed='USA',
|
|
is_remote=True
|
|
)
|
|
|
|
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
|
for idx, row in jobs_df.iterrows():
|
|
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
|
if not job_url or job_url == "nan":
|
|
continue
|
|
|
|
collected_jobs.append({
|
|
"title": str(row.get("title", "Untitled")),
|
|
"company": str(row.get("company", "Unknown")),
|
|
"location": "Remote, USA",
|
|
"is_remote": True,
|
|
"description": str(row.get("description", "") or "No description provided."),
|
|
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
|
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
|
"job_url": job_url,
|
|
"source": str(row.get("site", "jobspy")),
|
|
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
|
})
|
|
time.sleep(2)
|
|
except Exception as e:
|
|
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
|
|
|
|
print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}")
|
|
return collected_jobs
|