JB/scraper/scrapers/ats_ingestion.py

460 lines
19 KiB
Python
Raw Blame History

This file contains invisible Unicode characters

This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import requests
import html
import re
import concurrent.futures
from typing import List, Dict, Any
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
GREENHOUSE_BOARDS = [
# Payments, Banking & Fintech
("stripe", "Stripe"),
("ramp", "Ramp"),
("brex", "Brex"),
("plaid", "Plaid"),
("chime", "Chime"),
("robinhood", "Robinhood"),
("coinbase", "Coinbase"),
("square", "Block (Square)"),
("toast", "Toast"),
("klarna", "Klarna"),
("marqeta", "Marqeta"),
# AI & Frontier Tech
("scaleai", "Scale AI"),
("huggingface", "Hugging Face"),
("cohere", "Cohere"),
("perplexity", "Perplexity AI"),
("midjourney", "Midjourney"),
("stabilityai", "Stability AI"),
# Legal, Compliance & RegTech
("ironclad", "Ironclad (LegalTech)"),
("relativity", "Relativity (Legal Solutions)"),
# Developer Tools & Cloud Infrastructure
("datadog", "Datadog"),
("cloudflare", "Cloudflare"),
("figma", "Figma"),
("posthog", "PostHog"),
("linear", "Linear"),
("supabase", "Supabase"),
("databricks", "Databricks"),
("snowflake", "Snowflake"),
("mongodb", "MongoDB"),
("elastic", "Elastic"),
("hashicorp", "HashiCorp"),
("zapier", "Zapier"),
("retool", "Retool"),
("sentry", "Sentry"),
("pinecone", "Pinecone"),
("github", "GitHub"),
("gitlab", "GitLab"),
("launchdarkly", "LaunchDarkly"),
("snyk", "Snyk"),
("sourcegraph", "Sourcegraph"),
# Education Tech & Academia
("coursera", "Coursera"),
("duolingo", "Duolingo"),
("quizlet", "Quizlet"),
("guild", "Guild Education"),
# Construction, Real Estate & Logistics
("flexport", "Flexport (Logistics)"),
("samsara", "Samsara (IoT & Transport)"),
("procore", "Procore (Construction Software)"),
# Consumer, Retail & SaaS
("doordash", "DoorDash"),
("uber", "Uber"),
("airbnb", "Airbnb"),
("pinterest", "Pinterest"),
("reddit", "Reddit"),
("zoom", "Zoom"),
("twilio", "Twilio"),
("asana", "Asana"),
("notion", "Notion"),
("canva", "Canva"),
("hubspot", "HubSpot"),
("zendesk", "Zendesk"),
("okta", "Okta"),
("atlassian", "Atlassian"),
("crowdstrike", "CrowdStrike"),
("sentinelone", "SentinelOne"),
# Healthcare, Biotech & Science
("oscarhealth", "Oscar Health"),
("ro", "Ro Health"),
("tempus", "Tempus Labs"),
("guardanthealth", "Guardant Health"),
("flatiron", "Flatiron Health"),
("moderna", "Moderna"),
("cityblock", "Cityblock Health"),
("hims", "Hims & Hers Health"),
("springhealth", "Spring Health"),
("headway", "Headway"),
("talkspace", "Talkspace"),
("carbonhealth", "Carbon Health"),
("omadahealth", "Omada Health"),
("mavenclinic", "Maven Clinic"),
("goodrx", "GoodRx"),
("color", "Color Health"),
("invitae", "Invitae"),
# Logistics, Supply Chain & Industrial
("deliverr", "Deliverr (Logistics)"),
("convoy", "Convoy (Freight & Logistics)"),
("fulfill", "Fulfill.com"),
("shipbob", "ShipBob"),
("flockfreight", "Flock Freight"),
# Real Estate, Property & Construction
("compass", "Compass Real Estate"),
("opendoor", "Opendoor"),
("redfin", "Redfin"),
("cbre", "CBRE"),
# Hospitality, Food & Travel
("sweetgreen", "Sweetgreen"),
("instacart", "Instacart"),
("grubhub", "Grubhub"),
("goldbelly", "Goldbelly"),
# Professional Services, Accounting & Legal
("pilot", "Pilot (Bookkeeping & Tax)"),
("bench", "Bench Accounting"),
("brex", "Brex"),
("gusto", "Gusto (Payroll & HR)"),
("rippling", "Rippling (HR & Workforce)")
]
LEVER_BOARDS = [
("vercel", "Vercel"),
("spotify", "Spotify"),
("netflix", "Netflix"),
("palantir", "Palantir"),
("anthropic", "Anthropic"),
("discord", "Discord"),
("snap", "Snapchat"),
("figma", "Figma"),
("resend", "Resend"),
("modal", "Modal Labs"),
("sentry", "Sentry")
]
ASHBY_BOARDS = [
("ramp", "Ramp"),
("openai", "OpenAI"),
("anthropic", "Anthropic"),
("linear", "Linear"),
("cursor", "Cursor (Anysphere)"),
("replit", "Replit"),
("dust", "Dust"),
("ironclad", "Ironclad"),
("deel", "Deel"),
("superhuman", "Superhuman"),
("notion", "Notion"),
("retell", "Retell AI"),
("postman", "Postman"),
("vapi", "Vapi"),
("browserbase", "Browserbase"),
("tavus", "Tavus"),
("pave", "Pave"),
("cohere", "Cohere")
]
NON_US_REGEX = re.compile(
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
re.IGNORECASE
)
def is_valid_us_location(location_name: str, title: str = "") -> bool:
combined = f"{location_name} {title}"
return not bool(NON_US_REGEX.search(combined))
NEGATIVE_REMOTE_REGEX = re.compile(
r'\b(?:not\s+remote|non-remote|no\s+remote|on-site|onsite|in-office|in\s+office|hybrid|office\s+only|relocation\s+required|must\s+report\s+to\s+office)\b',
re.IGNORECASE
)
POSITIVE_REMOTE_REGEX = re.compile(
r'\b(?:100%\s+remote|fully\s+remote|remote\s+only|strictly\s+remote|anywhere\s+in\s+(?:the\s+)?(?:us|usa|united states))\b',
re.IGNORECASE
)
def parse_is_us_remote(location_name: str, title: str = "", description: str = "") -> bool:
loc = (location_name or "").strip()
tit = (title or "").strip()
desc = (description or "").strip()
if not is_valid_us_location(loc, tit):
return False
loc_lower = loc.lower()
tit_lower = tit.lower()
combined_header = f"{loc_lower} {tit_lower}"
# Negative indicator overrides on location/title unless explicitly 100% / fully remote
if NEGATIVE_REMOTE_REGEX.search(combined_header):
if not POSITIVE_REMOTE_REGEX.search(combined_header):
return False
# Check first 1500 chars of description for explicit on-site or hybrid mandate
if desc:
desc_start = desc[:1500].lower()
if re.search(r'\b(?:this\s+position\s+is\s+not\s+remote|not\s+a\s+remote\s+position|must\s+be\s+willing\s+to\s+work\s+on-site|requires\s+working\s+on-site|on-site\s+attendance\s+is\s+required|hybrid\s+work\s+schedule|in-person\s+attendance\s+required|must\s+commute\s+to\s+the\s+office)\b', desc_start):
return False
is_remote_mention = bool(re.search(r'\b(?:remote|telecommute|work\s+from\s+home|virtual|anywhere)\b', combined_header))
has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"])
return is_remote_mention and (has_us_indicator or "remote" in loc_lower or "anywhere" in loc_lower)
def clean_html_text(raw: str) -> str:
if not raw:
return ""
text = html.unescape(raw)
text = html.unescape(text)
text = re.sub(r'<[^>]+>', ' ', text)
text = re.sub(r'\s+', ' ', text).strip()
return text
def determine_department(title: str, dept_name: str = "") -> str:
combined = f"{title} {dept_name}".lower()
# Legal & Compliance check FIRST so titles like "Associate General Counsel, Infrastructure" or "Legal Counsel" don't become IT
if any(k in combined for k in ["counsel", "attorney", "lawyer", "legal", "paralegal", "compliance officer", "regulatory compliance", "litigation"]):
return "Finance, Accounting & Legal"
if any(k in combined for k in ["data", "analytics", "analyst", "machine learning", "ai ", "bi ", "business intelligence"]):
return "Data, AI & Analytics"
if any(k in combined for k in [
"it support", "help desk", "helpdesk", "sysadmin", "systems admin", "network engineer",
"desktop support", "it specialist", "it technician", "tech support", "service desk",
"desktop technician", "tier 1", "tier 2", "active directory", "systems administrator",
"cloud support", "support specialist"
]):
return "IT & Systems Administration"
# Infrastructure is IT if not already legal or software
if "infrastructure" in combined:
return "IT & Systems Administration"
if any(k in combined for k in ["art", "artist", "designer", "design", "illustrator", "animation", "animator", "ui/ux", "ux", "ui ", "graphic", "motion", "3d", "2d", "concept art"]):
return "Art, Design & Creative"
if any(k in combined for k in ["software", "engineer", "frontend", "backend", "fullstack", "full stack", "developer", "infra", "devops", "cloud", "mobile", "ios", "android"]):
return "Software & Engineering"
if any(k in combined for k in ["nurse", "medical", "clinical", "doctor", "health", "biotech"]):
return "Healthcare & Medical"
if any(k in combined for k in ["finance", "accounting", "treasury", "tax", "payroll", "audit", "underwriter"]):
return "Finance, Accounting & Legal"
if any(k in combined for k in ["sales", "marketing", "growth", "business development", "content", "product manager", "product owner"]):
return "Sales, Marketing & Product"
if any(k in combined for k in ["operations", "office", "admin", "recruiter", "people", "hr", "human resources", "workplace", "talent acquisition"]):
return "Human Resources & Operations"
if any(k in combined for k in ["construction", "electrician", "plumber", "contractor", "real estate", "property manager", "supply chain", "logistics", "warehouse"]):
return "Trades, Construction & Logistics"
if any(k in combined for k in ["state of", "teacher", "professor", "education", "curriculum", "academic"]):
return "Government & Education"
return "Other"
def determine_experience_level(title: str) -> str:
t = title.lower()
# Executive & Leadership titles (including Associate General Counsel / General Counsel)
if any(k in t for k in ["general counsel", "chief", "vp", "vice president", "head of", "director", "officer"]):
return "Executive"
if any(k in t for k in ["lead", "staff", "principal", "manager", "architect", "counsel"]):
return "Lead / Staff"
if any(k in t for k in ["senior", "sr.", "sr "]):
return "Senior"
if any(k in t for k in ["junior", "jr.", "entry", "intern", "internship", "trainee", "apprentice", "help desk", "tier 1", "entry-level"]):
return "Entry Level"
if "associate" in t and not any(k in t for k in ["senior", "lead", "counsel", "director", "manager", "vp"]):
return "Entry Level"
return "Mid-Level"
def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
if not is_valid_us_location(location_name, title):
continue
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
is_remote = parse_is_us_remote(location_name, title, desc_clean)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} failed: {e}")
return results
def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
jobs = r.json()
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title, desc_clean)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} failed: {e}")
return results
def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
is_remote = parse_is_us_remote(location_name, title, desc_clean) if not bool(j.get("isRemote", False)) else parse_is_us_remote(location_name, title, desc_clean) or ("remote" in location_name.lower())
if bool(j.get("isRemote", False)) and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()):
is_remote = True
else:
is_remote = parse_is_us_remote(location_name, title, desc_clean)
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} failed: {e}")
return results
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
collected = []
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS)
for res in gh_results:
collected.extend(res)
print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS)
for res in lever_results:
collected.extend(res)
print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS)
for res in ashby_results:
collected.extend(res)
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
return collected