import time import pandas as pd from typing import List, Dict, Any def run_jobspy_scrapes() -> List[Dict[Any, Any]]: """ Executes JobSpy searches for CT local jobs across all industries and nationwide remote roles across major job categories. """ collected_jobs = [] try: from jobspy import scrape_jobs except ImportError: print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.") return [] # CT local queries (all industries / general) ct_queries = [ "Customer Support", "Administrative", "Healthcare", "Education", "Operations", "Retail", "Finance", "Software", "Engineering", "General" ] # Remote queries remote_queries = [ "Customer Support", "Administrative Assistant", "IT Support", "Data Analyst", "Project Manager", "Software Engineer", "Marketing" ] print("[JobSpy] Starting CT Local Scrapes...") for query in ct_queries: try: print(f"[JobSpy] Searching CT Local: '{query}'") jobs_df = scrape_jobs( site_name=["indeed", "zip_recruiter"], search_term=query, location="Connecticut", results_wanted=15, hours_old=72, country_indeed='USA', is_remote=False ) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: for idx, row in jobs_df.iterrows(): job_url = str(row.get("job_url", "") or row.get("job_url_direct", "")) if not job_url or job_url == "nan": continue collected_jobs.append({ "title": str(row.get("title", "Untitled")), "company": str(row.get("company", "Unknown")), "location": str(row.get("location", "Connecticut, USA")), "is_remote": False, "description": str(row.get("description", "") or "No description provided."), "salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None, "salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None, "job_url": job_url, "source": str(row.get("site", "jobspy")), "date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None }) time.sleep(2) # Graceful delay between queries except Exception as e: print(f"[JobSpy Warning] CT search '{query}' failed: {e}") print("[JobSpy] Starting Remote Scrapes...") for query in remote_queries: try: print(f"[JobSpy] Searching Remote: '{query}'") jobs_df = scrape_jobs( site_name=["indeed", "zip_recruiter"], search_term=query, results_wanted=15, hours_old=72, country_indeed='USA', is_remote=True ) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: for idx, row in jobs_df.iterrows(): job_url = str(row.get("job_url", "") or row.get("job_url_direct", "")) if not job_url or job_url == "nan": continue collected_jobs.append({ "title": str(row.get("title", "Untitled")), "company": str(row.get("company", "Unknown")), "location": "Remote, USA", "is_remote": True, "description": str(row.get("description", "") or "No description provided."), "salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None, "salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None, "job_url": job_url, "source": str(row.get("site", "jobspy")), "date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None }) time.sleep(2) except Exception as e: print(f"[JobSpy Warning] Remote search '{query}' failed: {e}") print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}") return collected_jobs