import time import pandas as pd from typing import List, Dict, Any def run_jobspy_scrapes() -> List[Dict[Any, Any]]: """ Executes JobSpy searches across nationwide US hubs and remote roles covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc. """ collected_jobs = [] try: from jobspy import scrape_jobs except ImportError: print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.") return [] # Target key employment hubs across US states us_regions = [ ("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]), ("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]), ("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]), ("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]), ("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]), ("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]), ("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]), ("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]), ("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"]) ] # Nationwide Remote queries remote_queries = [ "Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager", "Customer Support", "Administrative Assistant", "IT Support Specialist", "DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer" ] print("[JobSpy] Starting Nationwide Regional Scrapes...") for loc, queries in us_regions: for query in queries: try: print(f"[JobSpy] Searching {loc}: '{query}'") jobs_df = scrape_jobs( site_name=["indeed", "zip_recruiter"], search_term=query, location=loc, results_wanted=35, hours_old=72, country_indeed="USA", is_remote=False ) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: for idx, row in jobs_df.iterrows(): job_url = str(row.get("job_url", "") or row.get("job_url_direct", "")) if not job_url or job_url == "nan": continue loc_val = str(row.get("location", loc) or loc) if loc_val == "nan": loc_val = loc collected_jobs.append({ "title": str(row.get("title", "Untitled")), "company": str(row.get("company", "Unknown")), "location": loc_val, "is_remote": False, "description": str(row.get("description", "") or "No description provided."), "salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None, "salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None, "job_url": job_url, "source": str(row.get("site", "jobspy")), "date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None }) time.sleep(1.5) except Exception as e: print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}") print("[JobSpy] Starting US Nationwide Remote Scrapes...") for query in remote_queries: try: print(f"[JobSpy] Searching US Remote: '{query}'") jobs_df = scrape_jobs( site_name=["indeed", "zip_recruiter"], search_term=query, results_wanted=35, hours_old=72, country_indeed="USA", is_remote=True ) if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty: for idx, row in jobs_df.iterrows(): job_url = str(row.get("job_url", "") or row.get("job_url_direct", "")) if not job_url or job_url == "nan": continue collected_jobs.append({ "title": str(row.get("title", "Untitled")), "company": str(row.get("company", "Unknown")), "location": "Remote, USA", "is_remote": True, "description": str(row.get("description", "") or "No description provided."), "salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None, "salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None, "job_url": job_url, "source": str(row.get("site", "jobspy")), "date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None }) time.sleep(1.5) except Exception as e: print(f"[JobSpy Warning] Remote search '{query}' failed: {e}") print(f"[JobSpy] Finished. Total jobs parsed: {len(collected_jobs)}") return collected_jobs