import requests
import html
import re
import concurrent.futures
from typing import List, Dict, Any
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
GREENHOUSE_BOARDS = [
# Payments, Banking & Fintech
("stripe", "Stripe"),
("ramp", "Ramp"),
("brex", "Brex"),
("plaid", "Plaid"),
("chime", "Chime"),
("robinhood", "Robinhood"),
("coinbase", "Coinbase"),
("square", "Block (Square)"),
("toast", "Toast"),
("klarna", "Klarna"),
("marqeta", "Marqeta"),
# AI & Frontier Tech
("scaleai", "Scale AI"),
("huggingface", "Hugging Face"),
("cohere", "Cohere"),
("perplexity", "Perplexity AI"),
("midjourney", "Midjourney"),
("stabilityai", "Stability AI"),
# Legal, Compliance & RegTech
("ironclad", "Ironclad (LegalTech)"),
("relativity", "Relativity (Legal Solutions)"),
# Developer Tools & Cloud Infrastructure
("datadog", "Datadog"),
("cloudflare", "Cloudflare"),
("figma", "Figma"),
("posthog", "PostHog"),
("linear", "Linear"),
("supabase", "Supabase"),
("databricks", "Databricks"),
("snowflake", "Snowflake"),
("mongodb", "MongoDB"),
("elastic", "Elastic"),
("hashicorp", "HashiCorp"),
("zapier", "Zapier"),
("retool", "Retool"),
("sentry", "Sentry"),
("pinecone", "Pinecone"),
("github", "GitHub"),
("gitlab", "GitLab"),
("launchdarkly", "LaunchDarkly"),
("snyk", "Snyk"),
("sourcegraph", "Sourcegraph"),
# Education Tech & Academia
("coursera", "Coursera"),
("duolingo", "Duolingo"),
("quizlet", "Quizlet"),
("guild", "Guild Education"),
# Construction, Real Estate & Logistics
("flexport", "Flexport (Logistics)"),
("samsara", "Samsara (IoT & Transport)"),
("procore", "Procore (Construction Software)"),
# Consumer, Retail & SaaS
("doordash", "DoorDash"),
("uber", "Uber"),
("airbnb", "Airbnb"),
("pinterest", "Pinterest"),
("reddit", "Reddit"),
("zoom", "Zoom"),
("twilio", "Twilio"),
("asana", "Asana"),
("notion", "Notion"),
("canva", "Canva"),
("hubspot", "HubSpot"),
("zendesk", "Zendesk"),
("okta", "Okta"),
("atlassian", "Atlassian"),
("crowdstrike", "CrowdStrike"),
("sentinelone", "SentinelOne"),
# Healthcare, Biotech & Science
("oscarhealth", "Oscar Health"),
("ro", "Ro Health"),
("tempus", "Tempus Labs"),
("guardanthealth", "Guardant Health"),
("flatiron", "Flatiron Health"),
("moderna", "Moderna"),
("cityblock", "Cityblock Health"),
("hims", "Hims & Hers Health"),
("springhealth", "Spring Health"),
("headway", "Headway"),
("talkspace", "Talkspace"),
("carbonhealth", "Carbon Health"),
("omadahealth", "Omada Health"),
("mavenclinic", "Maven Clinic"),
("goodrx", "GoodRx"),
("color", "Color Health"),
("invitae", "Invitae"),
# Logistics, Supply Chain & Industrial
("deliverr", "Deliverr (Logistics)"),
("convoy", "Convoy (Freight & Logistics)"),
("fulfill", "Fulfill.com"),
("shipbob", "ShipBob"),
("flockfreight", "Flock Freight"),
# Real Estate, Property & Construction
("compass", "Compass Real Estate"),
("opendoor", "Opendoor"),
("redfin", "Redfin"),
("cbre", "CBRE"),
# Hospitality, Food & Travel
("sweetgreen", "Sweetgreen"),
("instacart", "Instacart"),
("grubhub", "Grubhub"),
("goldbelly", "Goldbelly"),
# Professional Services, Accounting & Legal
("pilot", "Pilot (Bookkeeping & Tax)"),
("bench", "Bench Accounting"),
("brex", "Brex"),
("gusto", "Gusto (Payroll & HR)"),
("rippling", "Rippling (HR & Workforce)")
]
LEVER_BOARDS = [
("vercel", "Vercel"),
("spotify", "Spotify"),
("netflix", "Netflix"),
("palantir", "Palantir"),
("anthropic", "Anthropic"),
("discord", "Discord"),
("snap", "Snapchat"),
("figma", "Figma"),
("resend", "Resend"),
("modal", "Modal Labs"),
("sentry", "Sentry")
]
ASHBY_BOARDS = [
("ramp", "Ramp"),
("openai", "OpenAI"),
("anthropic", "Anthropic"),
("linear", "Linear"),
("cursor", "Cursor (Anysphere)"),
("replit", "Replit"),
("dust", "Dust"),
("ironclad", "Ironclad"),
("deel", "Deel"),
("superhuman", "Superhuman"),
("notion", "Notion"),
("retell", "Retell AI"),
("postman", "Postman"),
("vapi", "Vapi"),
("browserbase", "Browserbase"),
("tavus", "Tavus"),
("pave", "Pave"),
("cohere", "Cohere")
]
NON_US_REGEX = re.compile(
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
re.IGNORECASE
)
def is_valid_us_location(location_name: str, title: str = "") -> bool:
combined = f"{location_name} {title}"
return not bool(NON_US_REGEX.search(combined))
def parse_is_us_remote(location_name: str, title: str) -> bool:
loc_lower = location_name.lower()
title_lower = title.lower()
if not is_valid_us_location(location_name, title):
return False
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"])
return is_remote_mention and (has_us_indicator or "remote" in loc_lower)
def clean_html_text(raw: str) -> str:
if not raw:
return ""
text = html.unescape(raw)
text = html.unescape(text)
text = re.sub(r'<[^>]+>', ' ', text)
text = re.sub(r'\s+', ' ', text).strip()
return text
def determine_department(title: str, dept_name: str = "") -> str:
combined = f"{title} {dept_name}".lower()
# Legal & Compliance check FIRST so titles like "Associate General Counsel, Infrastructure" or "Legal Counsel" don't become IT
if any(k in combined for k in ["counsel", "attorney", "lawyer", "legal", "paralegal", "compliance officer", "regulatory compliance", "litigation"]):
return "Finance, Accounting & Legal"
if any(k in combined for k in ["data", "analytics", "analyst", "machine learning", "ai ", "bi ", "business intelligence"]):
return "Data, AI & Analytics"
if any(k in combined for k in [
"it support", "help desk", "helpdesk", "sysadmin", "systems admin", "network engineer",
"desktop support", "it specialist", "it technician", "tech support", "service desk",
"desktop technician", "tier 1", "tier 2", "active directory", "systems administrator",
"cloud support", "support specialist"
]):
return "IT & Systems Administration"
# Infrastructure is IT if not already legal or software
if "infrastructure" in combined:
return "IT & Systems Administration"
if any(k in combined for k in ["art", "artist", "designer", "design", "illustrator", "animation", "animator", "ui/ux", "ux", "ui ", "graphic", "motion", "3d", "2d", "concept art"]):
return "Art, Design & Creative"
if any(k in combined for k in ["software", "engineer", "frontend", "backend", "fullstack", "full stack", "developer", "infra", "devops", "cloud", "mobile", "ios", "android"]):
return "Software & Engineering"
if any(k in combined for k in ["nurse", "medical", "clinical", "doctor", "health", "biotech"]):
return "Healthcare & Medical"
if any(k in combined for k in ["finance", "accounting", "treasury", "tax", "payroll", "audit", "underwriter"]):
return "Finance, Accounting & Legal"
if any(k in combined for k in ["sales", "marketing", "growth", "business development", "content", "product manager", "product owner"]):
return "Sales, Marketing & Product"
if any(k in combined for k in ["operations", "office", "admin", "recruiter", "people", "hr", "human resources", "workplace", "talent acquisition"]):
return "Human Resources & Operations"
if any(k in combined for k in ["construction", "electrician", "plumber", "contractor", "real estate", "property manager", "supply chain", "logistics", "warehouse"]):
return "Trades, Construction & Logistics"
if any(k in combined for k in ["state of", "teacher", "professor", "education", "curriculum", "academic"]):
return "Government & Education"
return "Other"
def determine_experience_level(title: str) -> str:
t = title.lower()
# Executive & Leadership titles (including Associate General Counsel / General Counsel)
if any(k in t for k in ["general counsel", "chief", "vp", "vice president", "head of", "director", "officer"]):
return "Executive"
if any(k in t for k in ["lead", "staff", "principal", "manager", "architect", "counsel"]):
return "Lead / Staff"
if any(k in t for k in ["senior", "sr.", "sr "]):
return "Senior"
if any(k in t for k in ["junior", "jr.", "entry", "intern", "internship", "trainee", "apprentice", "help desk", "tier 1", "entry-level"]):
return "Entry Level"
if "associate" in t and not any(k in t for k in ["senior", "lead", "counsel", "director", "manager", "vp"]):
return "Entry Level"
return "Mid-Level"
def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} failed: {e}")
return results
def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
jobs = r.json()
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} failed: {e}")
return results
def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
dept_text = j.get("department", "")
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} failed: {e}")
return results
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
collected = []
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS)
for res in gh_results:
collected.extend(res)
print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS)
for res in lever_results:
collected.extend(res)
print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS)
for res in ashby_results:
collected.extend(res)
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
return collected