JB/scraper/main.py

151 lines
5.5 KiB
Python

import os
import time
import sys
from scrapers.ats_ingestion import run_ats_direct_ingestion
from scrapers.art_and_design_ingestion import run_art_and_design_ingestion
from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion
from scrapers.major_ct_employers import run_major_ct_employers_scrape
from scrapers.jobaps_ct import run_ct_jobaps_scrape
from scrapers.jobspy_runner import run_jobspy_scrapes
from scrapers.smart_careers_crawler import SmartCareersCrawler
from db import upsert_jobs
# Targeted domains for smart careers auto-discovery across US
DOMAINS_TO_PROBE = [
("anthropic.com", "Anthropic"),
("openai.com", "OpenAI"),
("stripe.com", "Stripe"),
("ramp.com", "Ramp"),
("brex.com", "Brex"),
("plaid.com", "Plaid"),
("figma.com", "Figma"),
("linear.app", "Linear"),
("notion.so", "Notion"),
("cursor.com", "Cursor"),
("superhuman.com", "Superhuman"),
("vercel.com", "Vercel"),
("supabase.com", "Supabase"),
("datadoghq.com", "Datadog"),
("cloudflare.com", "Cloudflare"),
("posthog.com", "PostHog"),
("sentry.io", "Sentry"),
("resend.com", "Resend"),
("retool.com", "Retool"),
("duolingo.com", "Duolingo"),
("canva.com", "Canva"),
("roblox.com", "Roblox"),
("epicgames.com", "Epic Games"),
("flexport.com", "Flexport")
]
def run_smart_crawler_scrapes():
print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...")
crawler = SmartCareersCrawler()
discovered_jobs = []
for domain, company_name in DOMAINS_TO_PROBE:
try:
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
if provider and slug:
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
if provider == "greenhouse":
jobs = crawler.fetch_greenhouse_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "lever":
jobs = crawler.fetch_lever_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "ashby":
jobs = crawler.fetch_ashby_board(slug, company_name)
discovered_jobs.extend(jobs)
except Exception as e:
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.")
return discovered_jobs
def execute_all_scrapes():
print("\n==============================================")
print("Starting Nationwide Multi-Source Ingestion Pipeline...")
print("==============================================")
all_jobs = []
# 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby)
try:
crawler_jobs = run_smart_crawler_scrapes()
all_jobs.extend(crawler_jobs)
except Exception as e:
print(f"[Error] Smart Careers Crawler error: {e}")
# 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped)
try:
ats_jobs = run_ats_direct_ingestion()
all_jobs.extend(ats_jobs)
except Exception as e:
print(f"[Error] ATS Ingestion error: {e}")
# 3. Dedicated Art, Creative, Gaming & Design Ingestion
try:
art_jobs = run_art_and_design_ingestion()
all_jobs.extend(art_jobs)
except Exception as e:
print(f"[Error] Art & Design Ingestion error: {e}")
# 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech
try:
exp_jobs = run_expanded_categories_ingestion()
all_jobs.extend(exp_jobs)
except Exception as e:
print(f"[Error] Expanded categories error: {e}")
# 5. Major CT & Regional Enterprise Employers
try:
emp_jobs = run_major_ct_employers_scrape()
all_jobs.extend(emp_jobs)
except Exception as e:
print(f"[Error] Major Employers error: {e}")
# 6. CT State JobAps Government & Public Portal
try:
ct_jobs = run_ct_jobaps_scrape()
all_jobs.extend(ct_jobs)
except Exception as e:
print(f"[Error] CT JobAps execution error: {e}")
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
try:
jobspy_jobs = run_jobspy_scrapes()
all_jobs.extend(jobspy_jobs)
except Exception as e:
print(f"[Error] JobSpy execution error: {e}")
print(f"\n==============================================")
print(f"Total job postings ingested: {len(all_jobs)}")
print("==============================================")
if all_jobs:
count = upsert_jobs(all_jobs)
print(f"✅ Successfully written/updated {count} jobs into database.")
else:
print("No job records were collected in this run.")
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == "--once":
execute_all_scrapes()
sys.exit(0)
try:
from apscheduler.schedulers.blocking import BlockingScheduler
interval = int(os.getenv("SCRAPE_INTERVAL_MINUTES", "30"))
print(f"Scraper worker starting. Scheduled to run every {interval} minutes.")
execute_all_scrapes()
scheduler = BlockingScheduler()
scheduler.add_job(execute_all_scrapes, 'interval', minutes=interval)
scheduler.start()
except ImportError:
print("[Warning] APScheduler not found. Executing single scrape run.")
execute_all_scrapes()
except (KeyboardInterrupt, SystemExit):
print("Scraper worker stopped.")