import os import time import sys from scrapers.ats_ingestion import run_ats_direct_ingestion from scrapers.art_and_design_ingestion import run_art_and_design_ingestion from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion from scrapers.major_ct_employers import run_major_ct_employers_scrape from scrapers.jobaps_ct import run_ct_jobaps_scrape from scrapers.jobspy_runner import run_jobspy_scrapes from scrapers.smart_careers_crawler import SmartCareersCrawler from db import upsert_jobs # Targeted domains for smart careers auto-discovery across US DOMAINS_TO_PROBE = [ ("anthropic.com", "Anthropic"), ("openai.com", "OpenAI"), ("stripe.com", "Stripe"), ("ramp.com", "Ramp"), ("brex.com", "Brex"), ("plaid.com", "Plaid"), ("figma.com", "Figma"), ("linear.app", "Linear"), ("notion.so", "Notion"), ("cursor.com", "Cursor"), ("superhuman.com", "Superhuman"), ("vercel.com", "Vercel"), ("supabase.com", "Supabase"), ("datadoghq.com", "Datadog"), ("cloudflare.com", "Cloudflare"), ("posthog.com", "PostHog"), ("sentry.io", "Sentry"), ("resend.com", "Resend"), ("retool.com", "Retool"), ("duolingo.com", "Duolingo"), ("canva.com", "Canva"), ("roblox.com", "Roblox"), ("epicgames.com", "Epic Games"), ("flexport.com", "Flexport") ] def run_smart_crawler_scrapes(): print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...") crawler = SmartCareersCrawler() discovered_jobs = [] for domain, company_name in DOMAINS_TO_PROBE: try: provider, slug, careers_url = crawler.probe_domain_for_careers(domain) if provider and slug: print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'") if provider == "greenhouse": jobs = crawler.fetch_greenhouse_board(slug, company_name) discovered_jobs.extend(jobs) elif provider == "lever": jobs = crawler.fetch_lever_board(slug, company_name) discovered_jobs.extend(jobs) elif provider == "ashby": jobs = crawler.fetch_ashby_board(slug, company_name) discovered_jobs.extend(jobs) except Exception as e: print(f"[Smart Crawler Warning] Probing {domain} failed: {e}") print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.") return discovered_jobs def execute_all_scrapes(): print("\n==============================================") print("Starting Nationwide Multi-Source Ingestion Pipeline...") print("==============================================") all_jobs = [] # 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby) try: crawler_jobs = run_smart_crawler_scrapes() all_jobs.extend(crawler_jobs) except Exception as e: print(f"[Error] Smart Careers Crawler error: {e}") # 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped) try: ats_jobs = run_ats_direct_ingestion() all_jobs.extend(ats_jobs) except Exception as e: print(f"[Error] ATS Ingestion error: {e}") # 3. Dedicated Art, Creative, Gaming & Design Ingestion try: art_jobs = run_art_and_design_ingestion() all_jobs.extend(art_jobs) except Exception as e: print(f"[Error] Art & Design Ingestion error: {e}") # 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech try: exp_jobs = run_expanded_categories_ingestion() all_jobs.extend(exp_jobs) except Exception as e: print(f"[Error] Expanded categories error: {e}") # 5. Major CT & Regional Enterprise Employers try: emp_jobs = run_major_ct_employers_scrape() all_jobs.extend(emp_jobs) except Exception as e: print(f"[Error] Major Employers error: {e}") # 6. CT State JobAps Government & Public Portal try: ct_jobs = run_ct_jobaps_scrape() all_jobs.extend(ct_jobs) except Exception as e: print(f"[Error] CT JobAps execution error: {e}") # 7. JobSpy Nationwide US Metros & Remote Broad Searches try: jobspy_jobs = run_jobspy_scrapes() all_jobs.extend(jobspy_jobs) except Exception as e: print(f"[Error] JobSpy execution error: {e}") print(f"\n==============================================") print(f"Total job postings ingested: {len(all_jobs)}") print("==============================================") if all_jobs: count = upsert_jobs(all_jobs) print(f"✅ Successfully written/updated {count} jobs into database.") else: print("No job records were collected in this run.") if __name__ == "__main__": if len(sys.argv) > 1 and sys.argv[1] == "--once": execute_all_scrapes() sys.exit(0) try: from apscheduler.schedulers.blocking import BlockingScheduler interval = int(os.getenv("SCRAPE_INTERVAL_MINUTES", "30")) print(f"Scraper worker starting. Scheduled to run every {interval} minutes.") execute_all_scrapes() scheduler = BlockingScheduler() scheduler.add_job(execute_all_scrapes, 'interval', minutes=interval) scheduler.start() except ImportError: print("[Warning] APScheduler not found. Executing single scrape run.") execute_all_scrapes() except (KeyboardInterrupt, SystemExit): print("Scraper worker stopped.")