187 lines
6.9 KiB
Python
187 lines
6.9 KiB
Python
import os
|
|
import time
|
|
import sys
|
|
import concurrent.futures
|
|
from scrapers.ats_ingestion import run_ats_direct_ingestion
|
|
from scrapers.art_and_design_ingestion import run_art_and_design_ingestion
|
|
from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion
|
|
from scrapers.major_ct_employers import run_major_ct_employers_scrape
|
|
from scrapers.jobaps_ct import run_ct_jobaps_scrape
|
|
from scrapers.jobspy_runner import run_jobspy_scrapes
|
|
from scrapers.smart_careers_crawler import SmartCareersCrawler
|
|
from db import upsert_jobs
|
|
|
|
# Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial
|
|
DOMAINS_TO_PROBE = [
|
|
# Top Tech & AI
|
|
("anthropic.com", "Anthropic"),
|
|
("openai.com", "OpenAI"),
|
|
("stripe.com", "Stripe"),
|
|
("ramp.com", "Ramp"),
|
|
("brex.com", "Brex"),
|
|
("plaid.com", "Plaid"),
|
|
("figma.com", "Figma"),
|
|
("linear.app", "Linear"),
|
|
("notion.so", "Notion"),
|
|
("cursor.com", "Cursor"),
|
|
("superhuman.com", "Superhuman"),
|
|
("vercel.com", "Vercel"),
|
|
("supabase.com", "Supabase"),
|
|
("datadoghq.com", "Datadog"),
|
|
("cloudflare.com", "Cloudflare"),
|
|
("posthog.com", "PostHog"),
|
|
("sentry.io", "Sentry"),
|
|
("resend.com", "Resend"),
|
|
("retool.com", "Retool"),
|
|
("duolingo.com", "Duolingo"),
|
|
("canva.com", "Canva"),
|
|
("roblox.com", "Roblox"),
|
|
("epicgames.com", "Epic Games"),
|
|
("flexport.com", "Flexport"),
|
|
|
|
# Enterprise, Retail, Logistics & Healthcare
|
|
("target.com", "Target"),
|
|
("homedepot.com", "The Home Depot"),
|
|
("costco.com", "Costco Wholesale"),
|
|
("cvshealth.com", "CVS Health"),
|
|
("unitedhealthgroup.com", "UnitedHealth Group"),
|
|
("cigna.com", "The Cigna Group"),
|
|
("travelers.com", "Travelers Insurance"),
|
|
("hartfordhealthcare.org", "Hartford HealthCare"),
|
|
("yalehealth.org", "Yale Health"),
|
|
("labcorp.com", "Labcorp"),
|
|
("questdiagnostics.com", "Quest Diagnostics"),
|
|
("lockheedmartin.com", "Lockheed Martin"),
|
|
("boeing.com", "Boeing"),
|
|
("caterpillar.com", "Caterpillar"),
|
|
("deere.com", "John Deere"),
|
|
("fedex.com", "FedEx"),
|
|
("ups.com", "UPS"),
|
|
("dhl.com", "DHL Express"),
|
|
("marriott.com", "Marriott International"),
|
|
("hilton.com", "Hilton"),
|
|
("starbucks.com", "Starbucks")
|
|
]
|
|
|
|
def run_smart_crawler_scrapes():
|
|
print("[Smart Crawler] Universal Web Crawler scanning domains concurrently...")
|
|
crawler = SmartCareersCrawler()
|
|
discovered_jobs = []
|
|
|
|
def probe_and_collect(item):
|
|
domain, company_name = item
|
|
jobs = []
|
|
try:
|
|
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
|
|
if provider and slug:
|
|
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
|
|
if provider == "greenhouse":
|
|
jobs = crawler.fetch_greenhouse_board(slug, company_name)
|
|
elif provider == "lever":
|
|
jobs = crawler.fetch_lever_board(slug, company_name)
|
|
elif provider == "ashby":
|
|
jobs = crawler.fetch_ashby_board(slug, company_name)
|
|
elif provider == "workday":
|
|
jobs = crawler.fetch_workday_board(slug, company_name)
|
|
elif careers_url:
|
|
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
|
|
jobs = crawler.scrape_native_career_page(careers_url, company_name)
|
|
except Exception as e:
|
|
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
|
|
return jobs
|
|
|
|
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
|
|
results = executor.map(probe_and_collect, DOMAINS_TO_PROBE)
|
|
for job_batch in results:
|
|
if job_batch:
|
|
discovered_jobs.extend(job_batch)
|
|
|
|
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
|
|
return discovered_jobs
|
|
|
|
def execute_all_scrapes():
|
|
print("\n==============================================")
|
|
print("Starting Nationwide Multi-Source Ingestion Pipeline...")
|
|
print("==============================================")
|
|
|
|
all_jobs = []
|
|
|
|
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby/Workday)
|
|
try:
|
|
crawler_jobs = run_smart_crawler_scrapes()
|
|
all_jobs.extend(crawler_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] Smart Careers Crawler error: {e}")
|
|
|
|
# 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped)
|
|
try:
|
|
ats_jobs = run_ats_direct_ingestion()
|
|
all_jobs.extend(ats_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] ATS Ingestion error: {e}")
|
|
|
|
# 3. Dedicated Art, Creative, Gaming & Design Ingestion
|
|
try:
|
|
art_jobs = run_art_and_design_ingestion()
|
|
all_jobs.extend(art_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] Art & Design Ingestion error: {e}")
|
|
|
|
# 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech
|
|
try:
|
|
exp_jobs = run_expanded_categories_ingestion()
|
|
all_jobs.extend(exp_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] Expanded categories error: {e}")
|
|
|
|
# 5. Major Regional & Enterprise Employers
|
|
try:
|
|
emp_jobs = run_major_ct_employers_scrape()
|
|
all_jobs.extend(emp_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] Major Employers error: {e}")
|
|
|
|
# 6. Public Sector Portals
|
|
try:
|
|
ct_jobs = run_ct_jobaps_scrape()
|
|
all_jobs.extend(ct_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] JobAps execution error: {e}")
|
|
|
|
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
|
|
try:
|
|
jobspy_jobs = run_jobspy_scrapes()
|
|
all_jobs.extend(jobspy_jobs)
|
|
except Exception as e:
|
|
print(f"[Error] JobSpy execution error: {e}")
|
|
|
|
print(f"\n==============================================")
|
|
print(f"Total job postings ingested: {len(all_jobs)}")
|
|
print("==============================================")
|
|
|
|
if all_jobs:
|
|
count = upsert_jobs(all_jobs)
|
|
print(f"[DB Ingestion] Successfully written/updated {count} jobs into database.")
|
|
else:
|
|
print("No job records were collected in this run.")
|
|
|
|
if __name__ == "__main__":
|
|
if len(sys.argv) > 1 and sys.argv[1] == "--once":
|
|
execute_all_scrapes()
|
|
sys.exit(0)
|
|
|
|
try:
|
|
from apscheduler.schedulers.blocking import BlockingScheduler
|
|
interval = int(os.getenv("SCRAPE_INTERVAL_MINUTES", "30"))
|
|
print(f"Scraper worker starting. Scheduled to run every {interval} minutes.")
|
|
|
|
execute_all_scrapes()
|
|
|
|
scheduler = BlockingScheduler()
|
|
scheduler.add_job(execute_all_scrapes, 'interval', minutes=interval)
|
|
scheduler.start()
|
|
except ImportError:
|
|
print("[Warning] APScheduler not found. Executing single scrape run.")
|
|
execute_all_scrapes()
|
|
except (KeyboardInterrupt, SystemExit):
|
|
print("Scraper worker stopped.")
|