JB/scraper/main.py

90 lines
3.2 KiB
Python

import os
import time
import sys
from scrapers.ats_ingestion import run_ats_direct_ingestion
from scrapers.art_and_design_ingestion import run_art_and_design_ingestion
from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion
from scrapers.major_ct_employers import run_major_ct_employers_scrape
from scrapers.jobaps_ct import run_ct_jobaps_scrape
from scrapers.jobspy_runner import run_jobspy_scrapes
from db import upsert_jobs
def execute_all_scrapes():
print("\n==============================================")
print("Starting CareerHound-Class Multi-Source Ingestion...")
print("==============================================")
all_jobs = []
# 1. Expanded Legal, Education, Trades, Logistics, HR, Biotech
try:
exp_jobs = run_expanded_categories_ingestion()
all_jobs.extend(exp_jobs)
except Exception as e:
print(f"[Error] Expanded categories error: {e}")
# 2. Dedicated Art, Creative, Gaming & Design Ingestion
try:
art_jobs = run_art_and_design_ingestion()
all_jobs.extend(art_jobs)
except Exception as e:
print(f"[Error] Art & Design Ingestion error: {e}")
# 3. Direct Public ATS Board Ingestion (Greenhouse, Lever)
try:
ats_jobs = run_ats_direct_ingestion()
all_jobs.extend(ats_jobs)
except Exception as e:
print(f"[Error] ATS Ingestion error: {e}")
# 4. Major CT Enterprise & Healthcare Employers
try:
emp_jobs = run_major_ct_employers_scrape()
all_jobs.extend(emp_jobs)
except Exception as e:
print(f"[Error] Major CT Employers error: {e}")
# 5. CT State JobAps Government & Public Portal
try:
ct_jobs = run_ct_jobaps_scrape()
all_jobs.extend(ct_jobs)
except Exception as e:
print(f"[Error] CT JobAps execution error: {e}")
# 6. JobSpy Local & Remote Broad Searches
try:
jobspy_jobs = run_jobspy_scrapes()
all_jobs.extend(jobspy_jobs)
except Exception as e:
print(f"[Error] JobSpy execution error: {e}")
print(f"\n==============================================")
print(f"Total job postings ingested: {len(all_jobs)}")
print("==============================================")
if all_jobs:
count = upsert_jobs(all_jobs)
print(f"✅ Successfully written/updated {count} jobs into database.")
else:
print("No job records were collected in this run.")
if __name__ == "__main__":
if len(sys.argv) > 1 and sys.argv[1] == "--once":
execute_all_scrapes()
sys.exit(0)
try:
from apscheduler.schedulers.blocking import BlockingScheduler
interval = int(os.getenv("SCRAPE_INTERVAL_MINUTES", "30"))
print(f"Scraper worker starting. Scheduled to run every {interval} minutes.")
execute_all_scrapes()
scheduler = BlockingScheduler()
scheduler.add_job(execute_all_scrapes, 'interval', minutes=interval)
scheduler.start()
except ImportError:
print("[Warning] APScheduler not found. Executing single scrape run.")
execute_all_scrapes()
except (KeyboardInterrupt, SystemExit):
print("Scraper worker stopped.")