JB/scraper/scrapers/ats_ingestion.py

291 lines
12 KiB
Python

import requests
import html
import re
from typing import List, Dict, Any
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
GREENHOUSE_BOARDS = [
# Payments, Banking & Fintech
("stripe", "Stripe"),
("ramp", "Ramp"),
("brex", "Brex"),
("plaid", "Plaid"),
("chime", "Chime"),
("robinhood", "Robinhood"),
("coinbase", "Coinbase"),
("square", "Block (Square)"),
("toast", "Toast"),
("klarna", "Klarna"),
("marqeta", "Marqeta"),
# AI & Frontier Tech
("scaleai", "Scale AI"),
("huggingface", "Hugging Face"),
("cohere", "Cohere"),
("perplexity", "Perplexity AI"),
("midjourney", "Midjourney"),
("stabilityai", "Stability AI"),
# Legal, Compliance & RegTech
("ironclad", "Ironclad (LegalTech)"),
("relativity", "Relativity (Legal Solutions)"),
# Developer Tools & Cloud Infrastructure
("datadog", "Datadog"),
("cloudflare", "Cloudflare"),
("figma", "Figma"),
("posthog", "PostHog"),
("linear", "Linear"),
("supabase", "Supabase"),
("databricks", "Databricks"),
("snowflake", "Snowflake"),
("mongodb", "MongoDB"),
("elastic", "Elastic"),
("hashicorp", "HashiCorp"),
("zapier", "Zapier"),
("retool", "Retool"),
("sentry", "Sentry"),
("pinecone", "Pinecone"),
("github", "GitHub"),
("gitlab", "GitLab"),
("launchdarkly", "LaunchDarkly"),
("snyk", "Snyk"),
("sourcegraph", "Sourcegraph"),
# Education Tech & Academia
("coursera", "Coursera"),
("duolingo", "Duolingo"),
("quizlet", "Quizlet"),
("guild", "Guild Education"),
# Construction, Real Estate & Logistics
("flexport", "Flexport (Logistics)"),
("samsara", "Samsara (IoT & Transport)"),
("procore", "Procore (Construction Software)"),
# Consumer, Retail & SaaS
("doordash", "DoorDash"),
("uber", "Uber"),
("airbnb", "Airbnb"),
("pinterest", "Pinterest"),
("reddit", "Reddit"),
("zoom", "Zoom"),
("twilio", "Twilio"),
("asana", "Asana"),
("notion", "Notion"),
("canva", "Canva"),
("hubspot", "HubSpot"),
("zendesk", "Zendesk"),
("okta", "Okta"),
("atlassian", "Atlassian"),
("crowdstrike", "CrowdStrike"),
("sentinelone", "SentinelOne"),
# Healthcare, Biotech & Science
("oscarhealth", "Oscar Health"),
("ro", "Ro Health"),
("tempus", "Tempus Labs"),
("guardanthealth", "Guardant Health"),
("flatiron", "Flatiron Health"),
("moderna", "Moderna")
]
LEVER_BOARDS = [
("vercel", "Vercel"),
("spotify", "Spotify"),
("netflix", "Netflix"),
("palantir", "Palantir"),
("anthropic", "Anthropic"),
("discord", "Discord"),
("snap", "Snapchat"),
("figma", "Figma"),
("resend", "Resend"),
("modal", "Modal Labs"),
("sentry", "Sentry")
]
NON_US_KEYWORDS = [
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
"ireland", "dublin", "switzerland", "zurich"
]
def is_valid_us_location(location_name: str, title: str) -> bool:
loc_lower = location_name.lower()
title_lower = title.lower()
for non_us in NON_US_KEYWORDS:
if non_us in loc_lower or non_us in title_lower:
return False
return True
def parse_is_us_remote(location_name: str, title: str) -> bool:
loc_lower = location_name.lower()
title_lower = title.lower()
if not is_valid_us_location(location_name, title):
return False
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"])
return is_remote_mention and (has_us_indicator or "remote" in loc_lower)
def clean_html_text(raw: str) -> str:
if not raw:
return ""
text = html.unescape(raw)
text = html.unescape(text)
text = re.sub(r'<[^>]+>', ' ', text)
text = re.sub(r'\s+', ' ', text).strip()
return text
def determine_department(title: str, dept_name: str = "") -> str:
combined = f"{title} {dept_name}".lower()
if any(k in combined for k in ["data", "analytics", "analyst", "machine learning", "ai ", "bi ", "business intelligence"]):
return "Data, AI & Analytics"
if any(k in combined for k in ["it support", "help desk", "helpdesk", "sysadmin", "systems admin", "network engineer", "desktop support", "it specialist", "infrastructure", "active directory"]):
return "IT & Systems Administration"
if any(k in combined for k in ["art", "artist", "designer", "design", "illustrator", "animation", "animator", "ui/ux", "ux", "ui ", "graphic", "motion", "3d", "2d", "concept art"]):
return "Art, Design & Creative"
if any(k in combined for k in ["software", "engineer", "frontend", "backend", "fullstack", "full stack", "developer", "infra", "devops", "cloud", "mobile", "ios", "android"]):
return "Software & Engineering"
if any(k in combined for k in ["nurse", "medical", "clinical", "doctor", "health", "biotech"]):
return "Healthcare & Medical"
if any(k in combined for k in ["finance", "accounting", "treasury", "tax", "payroll", "audit", "legal", "counsel", "paralegal", "compliance"]):
return "Finance, Accounting & Legal"
if any(k in combined for k in ["sales", "marketing", "growth", "business development", "content", "product manager", "product owner"]):
return "Sales, Marketing & Product"
if any(k in combined for k in ["operations", "office", "admin", "recruiter", "people", "hr", "human resources", "workplace", "talent acquisition"]):
return "Human Resources & Operations"
if any(k in combined for k in ["construction", "electrician", "plumber", "contractor", "real estate", "property manager", "supply chain", "logistics", "warehouse"]):
return "Trades, Construction & Logistics"
if any(k in combined for k in ["state of", "teacher", "professor", "education", "curriculum", "academic"]):
return "Government & Education"
return "Other"
def determine_experience_level(title: str) -> str:
t = title.lower()
if any(k in t for k in ["chief", "vp", "vice president", "head of", "director"]):
return "Executive"
if any(k in t for k in ["lead", "staff", "principal", "manager", "architect"]):
return "Lead / Staff"
if any(k in t for k in ["senior", "sr.", "sr "]):
return "Senior"
if any(k in t for k in ["junior", "jr.", "entry", "associate", "intern", "trainee"]):
return "Entry Level"
return "Mid-Level"
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
collected = []
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
print("[ATS Direct] Fetching 90+ Greenhouse public API boards (US Remote Filtered)...")
for board_slug, company_name in GREENHOUSE_BOARDS:
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs[:30]:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply direct on company board.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} failed: {e}")
print("[ATS Direct] Fetching Lever public API boards...")
for board_slug, company_name in LEVER_BOARDS:
try:
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
jobs = r.json()
for j in jobs[:30]:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} failed: {e}")
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
return collected