JB/scraper/main.py

187 lines
6.9 KiB
Python

import os
import time
import sys
import concurrent.futures
from scrapers.ats_ingestion import run_ats_direct_ingestion
from scrapers.art_and_design_ingestion import run_art_and_design_ingestion
from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion
from scrapers.major_ct_employers import run_major_ct_employers_scrape
from scrapers.jobaps_ct import run_ct_jobaps_scrape
from scrapers.jobspy_runner import run_jobspy_scrapes
from scrapers.smart_careers_crawler import SmartCareersCrawler
from db import upsert_jobs
# Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial
DOMAINS_TO_PROBE = [
# Top Tech & AI
("anthropic.com", "Anthropic"),
("openai.com", "OpenAI"),
("stripe.com", "Stripe"),
("ramp.com", "Ramp"),
("brex.com", "Brex"),
("plaid.com", "Plaid"),
("figma.com", "Figma"),
("linear.app", "Linear"),
("notion.so", "Notion"),
("cursor.com", "Cursor"),
("superhuman.com", "Superhuman"),
("vercel.com", "Vercel"),
("supabase.com", "Supabase"),
("datadoghq.com", "Datadog"),
("cloudflare.com", "Cloudflare"),
("posthog.com", "PostHog"),
("sentry.io", "Sentry"),
("resend.com", "Resend"),
("retool.com", "Retool"),
("duolingo.com", "Duolingo"),
("canva.com", "Canva"),
("roblox.com", "Roblox"),
("epicgames.com", "Epic Games"),
("flexport.com", "Flexport"),
# Enterprise, Retail, Logistics & Healthcare
("target.com", "Target"),
("homedepot.com", "The Home Depot"),
("costco.com", "Costco Wholesale"),
("cvshealth.com", "CVS Health"),
("unitedhealthgroup.com", "UnitedHealth Group"),
("cigna.com", "The Cigna Group"),
("travelers.com", "Travelers Insurance"),
("hartfordhealthcare.org", "Hartford HealthCare"),
("yalehealth.org", "Yale Health"),
("labcorp.com", "Labcorp"),
("questdiagnostics.com", "Quest Diagnostics"),
("lockheedmartin.com", "Lockheed Martin"),
("boeing.com", "Boeing"),
("caterpillar.com", "Caterpillar"),
("deere.com", "John Deere"),
("fedex.com", "FedEx"),
("ups.com", "UPS"),
("dhl.com", "DHL Express"),
("marriott.com", "Marriott International"),
("hilton.com", "Hilton"),
("starbucks.com", "Starbucks")
]
def run_smart_crawler_scrapes():
print("[Smart Crawler] Universal Web Crawler scanning domains concurrently...")
crawler = SmartCareersCrawler()
discovered_jobs = []
def probe_and_collect(item):
domain, company_name = item
jobs = []
try:
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
if provider and slug:
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
if provider == "greenhouse":
jobs = crawler.fetch_greenhouse_board(slug, company_name)
elif provider == "lever":
jobs = crawler.fetch_lever_board(slug, company_name)
elif provider == "ashby":
jobs = crawler.fetch_ashby_board(slug, company_name)
elif provider == "workday":
jobs = crawler.fetch_workday_board(slug, company_name)
elif careers_url:
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
jobs = crawler.scrape_native_career_page(careers_url, company_name)
except Exception as e:
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
return jobs
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
results = executor.map(probe_and_collect, DOMAINS_TO_PROBE)
for job_batch in results:
if job_batch:
discovered_jobs.extend(job_batch)
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
return discovered_jobs
def execute_all_scrapes():
print("\n==============================================")
print("Starting Nationwide Multi-Source Ingestion Pipeline...")
print("==============================================")
all_jobs = []
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby/Workday)
try:
crawler_jobs = run_smart_crawler_scrapes()
all_jobs.extend(crawler_jobs)
except Exception as e:
print(f"[Error] Smart Careers Crawler error: {e}")
# 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped)
try:
ats_jobs = run_ats_direct_ingestion()
all_jobs.extend(ats_jobs)
except Exception as e:
print(f"[Error] ATS Ingestion error: {e}")
# 3. Dedicated Art, Creative, Gaming & Design Ingestion
try:
art_jobs = run_art_and_design_ingestion()
all_jobs.extend(art_jobs)
except Exception as e:
print(f"[Error] Art & Design Ingestion error: {e}")
# 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech
try:
exp_jobs = run_expanded_categories_ingestion()
all_jobs.extend(exp_jobs)
except Exception as e:
print(f"[Error] Expanded categories error: {e}")
# 5. Major Regional & Enterprise Employers
try:
emp_jobs = run_major_ct_employers_scrape()
all_jobs.extend(emp_jobs)
except Exception as e:
print(f"[Error] Major Employers error: {e}")
# 6. Public Sector Portals
try:
ct_jobs = run_ct_jobaps_scrape()
all_jobs.extend(ct_jobs)
except Exception as e:
print(f"[Error] JobAps execution error: {e}")
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
try:
jobspy_jobs = run_jobspy_scrapes()
all_jobs.extend(jobspy_jobs)
except Exception as e:
print(f"[Error] JobSpy execution error: {e}")
print(f"\n==============================================")
print(f"Total job postings ingested: {len(all_jobs)}")
print("==============================================")
if all_jobs:
count = upsert_jobs(all_jobs)
print(f"[DB Ingestion] Successfully written/updated {count} jobs into database.")
else:
print("No job records were collected in this run.")
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == "--once":
execute_all_scrapes()
sys.exit(0)
try:
from apscheduler.schedulers.blocking import BlockingScheduler
interval = int(os.getenv("SCRAPE_INTERVAL_MINUTES", "30"))
print(f"Scraper worker starting. Scheduled to run every {interval} minutes.")
execute_all_scrapes()
scheduler = BlockingScheduler()
scheduler.add_job(execute_all_scrapes, 'interval', minutes=interval)
scheduler.start()
except ImportError:
print("[Warning] APScheduler not found. Executing single scrape run.")
execute_all_scrapes()
except (KeyboardInterrupt, SystemExit):
print("Scraper worker stopped.")