import os import time import sys from scrapers.ats_ingestion import run_ats_direct_ingestion from scrapers.art_and_design_ingestion import run_art_and_design_ingestion from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion from scrapers.major_ct_employers import run_major_ct_employers_scrape from scrapers.jobaps_ct import run_ct_jobaps_scrape from scrapers.jobspy_runner import run_jobspy_scrapes from scrapers.smart_careers_crawler import SmartCareersCrawler from db import upsert_jobs # Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial DOMAINS_TO_PROBE = [ # Top Tech & AI ("anthropic.com", "Anthropic"), ("openai.com", "OpenAI"), ("stripe.com", "Stripe"), ("ramp.com", "Ramp"), ("brex.com", "Brex"), ("plaid.com", "Plaid"), ("figma.com", "Figma"), ("linear.app", "Linear"), ("notion.so", "Notion"), ("cursor.com", "Cursor"), ("superhuman.com", "Superhuman"), ("vercel.com", "Vercel"), ("supabase.com", "Supabase"), ("datadoghq.com", "Datadog"), ("cloudflare.com", "Cloudflare"), ("posthog.com", "PostHog"), ("sentry.io", "Sentry"), ("resend.com", "Resend"), ("retool.com", "Retool"), ("duolingo.com", "Duolingo"), ("canva.com", "Canva"), ("roblox.com", "Roblox"), ("epicgames.com", "Epic Games"), ("flexport.com", "Flexport"), # Enterprise, Retail, Logistics & Healthcare ("target.com", "Target"), ("homedepot.com", "The Home Depot"), ("costco.com", "Costco Wholesale"), ("cvshealth.com", "CVS Health"), ("unitedhealthgroup.com", "UnitedHealth Group"), ("cigna.com", "The Cigna Group"), ("travelers.com", "Travelers Insurance"), ("hartfordhealthcare.org", "Hartford HealthCare"), ("yalehealth.org", "Yale Health"), ("lockheedmartin.com", "Lockheed Martin"), ("boeing.com", "Boeing"), ("caterpillar.com", "Caterpillar"), ("deere.com", "John Deere"), ("fedex.com", "FedEx"), ("ups.com", "UPS") ] def run_smart_crawler_scrapes(): print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...") crawler = SmartCareersCrawler() discovered_jobs = [] for domain, company_name in DOMAINS_TO_PROBE: try: provider, slug, careers_url = crawler.probe_domain_for_careers(domain) if provider and slug: print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'") if provider == "greenhouse": jobs = crawler.fetch_greenhouse_board(slug, company_name) discovered_jobs.extend(jobs) elif provider == "lever": jobs = crawler.fetch_lever_board(slug, company_name) discovered_jobs.extend(jobs) elif provider == "ashby": jobs = crawler.fetch_ashby_board(slug, company_name) discovered_jobs.extend(jobs) elif careers_url: print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}") jobs = crawler.scrape_native_career_page(careers_url, company_name) discovered_jobs.extend(jobs) except Exception as e: print(f"[Smart Crawler Warning] Probing {domain} failed: {e}") print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.") return discovered_jobs def execute_all_scrapes(): print("\n==============================================") print("Starting Nationwide Multi-Source Ingestion Pipeline...") print("==============================================") all_jobs = [] # 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby) try: crawler_jobs = run_smart_crawler_scrapes() all_jobs.extend(crawler_jobs) except Exception as e: print(f"[Error] Smart Careers Crawler error: {e}") # 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped) try: ats_jobs = run_ats_direct_ingestion() all_jobs.extend(ats_jobs) except Exception as e: print(f"[Error] ATS Ingestion error: {e}") # 3. Dedicated Art, Creative, Gaming & Design Ingestion try: art_jobs = run_art_and_design_ingestion() all_jobs.extend(art_jobs) except Exception as e: print(f"[Error] Art & Design Ingestion error: {e}") # 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech try: exp_jobs = run_expanded_categories_ingestion() all_jobs.extend(exp_jobs) except Exception as e: print(f"[Error] Expanded categories error: {e}") # 5. Major Regional & Enterprise Employers try: emp_jobs = run_major_ct_employers_scrape() all_jobs.extend(emp_jobs) except Exception as e: print(f"[Error] Major Employers error: {e}") # 6. Public Sector Portals try: ct_jobs = run_ct_jobaps_scrape() all_jobs.extend(ct_jobs) except Exception as e: print(f"[Error] JobAps execution error: {e}") # 7. JobSpy Nationwide US Metros & Remote Broad Searches try: jobspy_jobs = run_jobspy_scrapes() all_jobs.extend(jobspy_jobs) except Exception as e: print(f"[Error] JobSpy execution error: {e}") print(f"\n==============================================") print(f"Total job postings ingested: {len(all_jobs)}") print("==============================================") if all_jobs: count = upsert_jobs(all_jobs) print(f"✅ Successfully written/updated {count} jobs into database.") else: print("No job records were collected in this run.") if __name__ == "__main__": if len(sys.argv) > 1 and sys.argv[1] == "--once": execute_all_scrapes() sys.exit(0) try: from apscheduler.schedulers.blocking import BlockingScheduler interval = int(os.getenv("SCRAPE_INTERVAL_MINUTES", "30")) print(f"Scraper worker starting. Scheduled to run every {interval} minutes.") execute_all_scrapes() scheduler = BlockingScheduler() scheduler.add_job(execute_all_scrapes, 'interval', minutes=interval) scheduler.start() except ImportError: print("[Warning] APScheduler not found. Executing single scrape run.") execute_all_scrapes() except (KeyboardInterrupt, SystemExit): print("Scraper worker stopped.")