460 lines
19 KiB
Python
460 lines
19 KiB
Python
import requests
|
||
import html
|
||
import re
|
||
import concurrent.futures
|
||
from typing import List, Dict, Any
|
||
|
||
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
|
||
GREENHOUSE_BOARDS = [
|
||
# Payments, Banking & Fintech
|
||
("stripe", "Stripe"),
|
||
("ramp", "Ramp"),
|
||
("brex", "Brex"),
|
||
("plaid", "Plaid"),
|
||
("chime", "Chime"),
|
||
("robinhood", "Robinhood"),
|
||
("coinbase", "Coinbase"),
|
||
("square", "Block (Square)"),
|
||
("toast", "Toast"),
|
||
("klarna", "Klarna"),
|
||
("marqeta", "Marqeta"),
|
||
|
||
# AI & Frontier Tech
|
||
("scaleai", "Scale AI"),
|
||
("huggingface", "Hugging Face"),
|
||
("cohere", "Cohere"),
|
||
("perplexity", "Perplexity AI"),
|
||
("midjourney", "Midjourney"),
|
||
("stabilityai", "Stability AI"),
|
||
|
||
# Legal, Compliance & RegTech
|
||
("ironclad", "Ironclad (LegalTech)"),
|
||
("relativity", "Relativity (Legal Solutions)"),
|
||
|
||
# Developer Tools & Cloud Infrastructure
|
||
("datadog", "Datadog"),
|
||
("cloudflare", "Cloudflare"),
|
||
("figma", "Figma"),
|
||
("posthog", "PostHog"),
|
||
("linear", "Linear"),
|
||
("supabase", "Supabase"),
|
||
("databricks", "Databricks"),
|
||
("snowflake", "Snowflake"),
|
||
("mongodb", "MongoDB"),
|
||
("elastic", "Elastic"),
|
||
("hashicorp", "HashiCorp"),
|
||
("zapier", "Zapier"),
|
||
("retool", "Retool"),
|
||
("sentry", "Sentry"),
|
||
("pinecone", "Pinecone"),
|
||
("github", "GitHub"),
|
||
("gitlab", "GitLab"),
|
||
("launchdarkly", "LaunchDarkly"),
|
||
("snyk", "Snyk"),
|
||
("sourcegraph", "Sourcegraph"),
|
||
|
||
# Education Tech & Academia
|
||
("coursera", "Coursera"),
|
||
("duolingo", "Duolingo"),
|
||
("quizlet", "Quizlet"),
|
||
("guild", "Guild Education"),
|
||
|
||
# Construction, Real Estate & Logistics
|
||
("flexport", "Flexport (Logistics)"),
|
||
("samsara", "Samsara (IoT & Transport)"),
|
||
("procore", "Procore (Construction Software)"),
|
||
|
||
# Consumer, Retail & SaaS
|
||
("doordash", "DoorDash"),
|
||
("uber", "Uber"),
|
||
("airbnb", "Airbnb"),
|
||
("pinterest", "Pinterest"),
|
||
("reddit", "Reddit"),
|
||
("zoom", "Zoom"),
|
||
("twilio", "Twilio"),
|
||
("asana", "Asana"),
|
||
("notion", "Notion"),
|
||
("canva", "Canva"),
|
||
("hubspot", "HubSpot"),
|
||
("zendesk", "Zendesk"),
|
||
("okta", "Okta"),
|
||
("atlassian", "Atlassian"),
|
||
("crowdstrike", "CrowdStrike"),
|
||
("sentinelone", "SentinelOne"),
|
||
|
||
# Healthcare, Biotech & Science
|
||
("oscarhealth", "Oscar Health"),
|
||
("ro", "Ro Health"),
|
||
("tempus", "Tempus Labs"),
|
||
("guardanthealth", "Guardant Health"),
|
||
("flatiron", "Flatiron Health"),
|
||
("moderna", "Moderna"),
|
||
("cityblock", "Cityblock Health"),
|
||
("hims", "Hims & Hers Health"),
|
||
("springhealth", "Spring Health"),
|
||
("headway", "Headway"),
|
||
("talkspace", "Talkspace"),
|
||
("carbonhealth", "Carbon Health"),
|
||
("omadahealth", "Omada Health"),
|
||
("mavenclinic", "Maven Clinic"),
|
||
("goodrx", "GoodRx"),
|
||
("color", "Color Health"),
|
||
("invitae", "Invitae"),
|
||
|
||
# Logistics, Supply Chain & Industrial
|
||
("deliverr", "Deliverr (Logistics)"),
|
||
("convoy", "Convoy (Freight & Logistics)"),
|
||
("fulfill", "Fulfill.com"),
|
||
("shipbob", "ShipBob"),
|
||
("flockfreight", "Flock Freight"),
|
||
|
||
# Real Estate, Property & Construction
|
||
("compass", "Compass Real Estate"),
|
||
("opendoor", "Opendoor"),
|
||
("redfin", "Redfin"),
|
||
("cbre", "CBRE"),
|
||
|
||
# Hospitality, Food & Travel
|
||
("sweetgreen", "Sweetgreen"),
|
||
("instacart", "Instacart"),
|
||
("grubhub", "Grubhub"),
|
||
("goldbelly", "Goldbelly"),
|
||
|
||
# Professional Services, Accounting & Legal
|
||
("pilot", "Pilot (Bookkeeping & Tax)"),
|
||
("bench", "Bench Accounting"),
|
||
("brex", "Brex"),
|
||
("gusto", "Gusto (Payroll & HR)"),
|
||
("rippling", "Rippling (HR & Workforce)")
|
||
]
|
||
|
||
LEVER_BOARDS = [
|
||
("vercel", "Vercel"),
|
||
("spotify", "Spotify"),
|
||
("netflix", "Netflix"),
|
||
("palantir", "Palantir"),
|
||
("anthropic", "Anthropic"),
|
||
("discord", "Discord"),
|
||
("snap", "Snapchat"),
|
||
("figma", "Figma"),
|
||
("resend", "Resend"),
|
||
("modal", "Modal Labs"),
|
||
("sentry", "Sentry")
|
||
]
|
||
|
||
ASHBY_BOARDS = [
|
||
("ramp", "Ramp"),
|
||
("openai", "OpenAI"),
|
||
("anthropic", "Anthropic"),
|
||
("linear", "Linear"),
|
||
("cursor", "Cursor (Anysphere)"),
|
||
("replit", "Replit"),
|
||
("dust", "Dust"),
|
||
("ironclad", "Ironclad"),
|
||
("deel", "Deel"),
|
||
("superhuman", "Superhuman"),
|
||
("notion", "Notion"),
|
||
("retell", "Retell AI"),
|
||
("postman", "Postman"),
|
||
("vapi", "Vapi"),
|
||
("browserbase", "Browserbase"),
|
||
("tavus", "Tavus"),
|
||
("pave", "Pave"),
|
||
("cohere", "Cohere")
|
||
]
|
||
|
||
NON_US_REGEX = re.compile(
|
||
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
|
||
re.IGNORECASE
|
||
)
|
||
|
||
def is_valid_us_location(location_name: str, title: str = "") -> bool:
|
||
combined = f"{location_name} {title}"
|
||
return not bool(NON_US_REGEX.search(combined))
|
||
|
||
NEGATIVE_REMOTE_REGEX = re.compile(
|
||
r'\b(?:not\s+remote|non-remote|no\s+remote|on-site|onsite|in-office|in\s+office|hybrid|office\s+only|relocation\s+required|must\s+report\s+to\s+office)\b',
|
||
re.IGNORECASE
|
||
)
|
||
|
||
POSITIVE_REMOTE_REGEX = re.compile(
|
||
r'\b(?:100%\s+remote|fully\s+remote|remote\s+only|strictly\s+remote|anywhere\s+in\s+(?:the\s+)?(?:us|usa|united states))\b',
|
||
re.IGNORECASE
|
||
)
|
||
|
||
def parse_is_us_remote(location_name: str, title: str = "", description: str = "") -> bool:
|
||
loc = (location_name or "").strip()
|
||
tit = (title or "").strip()
|
||
desc = (description or "").strip()
|
||
|
||
if not is_valid_us_location(loc, tit):
|
||
return False
|
||
|
||
loc_lower = loc.lower()
|
||
tit_lower = tit.lower()
|
||
combined_header = f"{loc_lower} {tit_lower}"
|
||
|
||
# Negative indicator overrides on location/title unless explicitly 100% / fully remote
|
||
if NEGATIVE_REMOTE_REGEX.search(combined_header):
|
||
if not POSITIVE_REMOTE_REGEX.search(combined_header):
|
||
return False
|
||
|
||
# Check first 1500 chars of description for explicit on-site or hybrid mandate
|
||
if desc:
|
||
desc_start = desc[:1500].lower()
|
||
if re.search(r'\b(?:this\s+position\s+is\s+not\s+remote|not\s+a\s+remote\s+position|must\s+be\s+willing\s+to\s+work\s+on-site|requires\s+working\s+on-site|on-site\s+attendance\s+is\s+required|hybrid\s+work\s+schedule|in-person\s+attendance\s+required|must\s+commute\s+to\s+the\s+office)\b', desc_start):
|
||
return False
|
||
|
||
is_remote_mention = bool(re.search(r'\b(?:remote|telecommute|work\s+from\s+home|virtual|anywhere)\b', combined_header))
|
||
has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"])
|
||
|
||
return is_remote_mention and (has_us_indicator or "remote" in loc_lower or "anywhere" in loc_lower)
|
||
|
||
def clean_html_text(raw: str) -> str:
|
||
if not raw:
|
||
return ""
|
||
text = html.unescape(raw)
|
||
text = html.unescape(text)
|
||
text = re.sub(r'<[^>]+>', ' ', text)
|
||
text = re.sub(r'\s+', ' ', text).strip()
|
||
return text
|
||
|
||
def determine_department(title: str, dept_name: str = "") -> str:
|
||
combined = f"{title} {dept_name}".lower()
|
||
|
||
# Legal & Compliance check FIRST so titles like "Associate General Counsel, Infrastructure" or "Legal Counsel" don't become IT
|
||
if any(k in combined for k in ["counsel", "attorney", "lawyer", "legal", "paralegal", "compliance officer", "regulatory compliance", "litigation"]):
|
||
return "Finance, Accounting & Legal"
|
||
|
||
if any(k in combined for k in ["data", "analytics", "analyst", "machine learning", "ai ", "bi ", "business intelligence"]):
|
||
return "Data, AI & Analytics"
|
||
|
||
if any(k in combined for k in [
|
||
"it support", "help desk", "helpdesk", "sysadmin", "systems admin", "network engineer",
|
||
"desktop support", "it specialist", "it technician", "tech support", "service desk",
|
||
"desktop technician", "tier 1", "tier 2", "active directory", "systems administrator",
|
||
"cloud support", "support specialist"
|
||
]):
|
||
return "IT & Systems Administration"
|
||
|
||
# Infrastructure is IT if not already legal or software
|
||
if "infrastructure" in combined:
|
||
return "IT & Systems Administration"
|
||
|
||
if any(k in combined for k in ["art", "artist", "designer", "design", "illustrator", "animation", "animator", "ui/ux", "ux", "ui ", "graphic", "motion", "3d", "2d", "concept art"]):
|
||
return "Art, Design & Creative"
|
||
|
||
if any(k in combined for k in ["software", "engineer", "frontend", "backend", "fullstack", "full stack", "developer", "infra", "devops", "cloud", "mobile", "ios", "android"]):
|
||
return "Software & Engineering"
|
||
|
||
if any(k in combined for k in ["nurse", "medical", "clinical", "doctor", "health", "biotech"]):
|
||
return "Healthcare & Medical"
|
||
|
||
if any(k in combined for k in ["finance", "accounting", "treasury", "tax", "payroll", "audit", "underwriter"]):
|
||
return "Finance, Accounting & Legal"
|
||
|
||
if any(k in combined for k in ["sales", "marketing", "growth", "business development", "content", "product manager", "product owner"]):
|
||
return "Sales, Marketing & Product"
|
||
|
||
if any(k in combined for k in ["operations", "office", "admin", "recruiter", "people", "hr", "human resources", "workplace", "talent acquisition"]):
|
||
return "Human Resources & Operations"
|
||
|
||
if any(k in combined for k in ["construction", "electrician", "plumber", "contractor", "real estate", "property manager", "supply chain", "logistics", "warehouse"]):
|
||
return "Trades, Construction & Logistics"
|
||
|
||
if any(k in combined for k in ["state of", "teacher", "professor", "education", "curriculum", "academic"]):
|
||
return "Government & Education"
|
||
|
||
return "Other"
|
||
|
||
def determine_experience_level(title: str) -> str:
|
||
t = title.lower()
|
||
# Executive & Leadership titles (including Associate General Counsel / General Counsel)
|
||
if any(k in t for k in ["general counsel", "chief", "vp", "vice president", "head of", "director", "officer"]):
|
||
return "Executive"
|
||
if any(k in t for k in ["lead", "staff", "principal", "manager", "architect", "counsel"]):
|
||
return "Lead / Staff"
|
||
if any(k in t for k in ["senior", "sr.", "sr "]):
|
||
return "Senior"
|
||
if any(k in t for k in ["junior", "jr.", "entry", "intern", "internship", "trainee", "apprentice", "help desk", "tier 1", "entry-level"]):
|
||
return "Entry Level"
|
||
if "associate" in t and not any(k in t for k in ["senior", "lead", "counsel", "director", "manager", "vp"]):
|
||
return "Entry Level"
|
||
return "Mid-Level"
|
||
|
||
def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||
board_slug, company_name = item
|
||
results = []
|
||
try:
|
||
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
|
||
r = requests.get(url, headers=headers, timeout=6)
|
||
if r.status_code == 200:
|
||
data = r.json()
|
||
jobs = data.get("jobs", [])
|
||
for j in jobs:
|
||
title = clean_html_text(j.get("title", ""))
|
||
job_url = j.get("absolute_url", "")
|
||
if not title or not job_url:
|
||
continue
|
||
|
||
location_obj = j.get("location", {})
|
||
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
||
|
||
if not is_valid_us_location(location_name, title):
|
||
continue
|
||
|
||
content_raw = j.get("content", "") or ""
|
||
desc_clean = clean_html_text(content_raw)
|
||
is_remote = parse_is_us_remote(location_name, title, desc_clean)
|
||
departments = j.get("departments", [])
|
||
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||
|
||
dept = determine_department(title, dept_text)
|
||
exp_level = determine_experience_level(title)
|
||
|
||
results.append({
|
||
"title": title,
|
||
"company": company_name,
|
||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||
"is_remote": is_remote,
|
||
"department": dept,
|
||
"experience_level": exp_level,
|
||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
|
||
"salary_min": None,
|
||
"salary_max": None,
|
||
"job_url": job_url,
|
||
"source": "greenhouse"
|
||
})
|
||
except Exception as e:
|
||
print(f"[Greenhouse Warning] {company_name} failed: {e}")
|
||
return results
|
||
|
||
def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||
board_slug, company_name = item
|
||
results = []
|
||
try:
|
||
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
|
||
r = requests.get(url, headers=headers, timeout=6)
|
||
if r.status_code == 200:
|
||
jobs = r.json()
|
||
for j in jobs:
|
||
title = clean_html_text(j.get("text", ""))
|
||
job_url = j.get("hostedUrl", "")
|
||
if not title or not job_url:
|
||
continue
|
||
|
||
categories = j.get("categories", {})
|
||
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
||
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
||
|
||
if not is_valid_us_location(location_name, title):
|
||
continue
|
||
|
||
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
||
desc_clean = clean_html_text(description_plain)
|
||
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title, desc_clean)
|
||
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
||
dept = determine_department(title, dept_text)
|
||
exp_level = determine_experience_level(title)
|
||
|
||
results.append({
|
||
"title": title,
|
||
"company": company_name,
|
||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||
"is_remote": is_remote,
|
||
"department": dept,
|
||
"experience_level": exp_level,
|
||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
||
"salary_min": None,
|
||
"salary_max": None,
|
||
"job_url": job_url,
|
||
"source": "lever"
|
||
})
|
||
except Exception as e:
|
||
print(f"[Lever Warning] {company_name} failed: {e}")
|
||
return results
|
||
|
||
def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||
board_slug, company_name = item
|
||
results = []
|
||
try:
|
||
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
|
||
r = requests.get(url, headers=headers, timeout=6)
|
||
if r.status_code == 200:
|
||
data = r.json()
|
||
jobs = data.get("jobs", [])
|
||
for j in jobs:
|
||
title = clean_html_text(j.get("title", ""))
|
||
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
|
||
if not title or not job_url:
|
||
continue
|
||
|
||
location_name = j.get("location", "Remote, USA") or "Remote, USA"
|
||
if not is_valid_us_location(location_name, title):
|
||
continue
|
||
|
||
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
|
||
is_remote = parse_is_us_remote(location_name, title, desc_clean) if not bool(j.get("isRemote", False)) else parse_is_us_remote(location_name, title, desc_clean) or ("remote" in location_name.lower())
|
||
if bool(j.get("isRemote", False)) and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()):
|
||
is_remote = True
|
||
else:
|
||
is_remote = parse_is_us_remote(location_name, title, desc_clean)
|
||
|
||
salary_min = None
|
||
salary_max = None
|
||
comp = j.get("compensation")
|
||
if isinstance(comp, dict):
|
||
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
|
||
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
|
||
if comp_min:
|
||
try:
|
||
salary_min = float(comp_min)
|
||
except (ValueError, TypeError):
|
||
pass
|
||
if comp_max:
|
||
try:
|
||
salary_max = float(comp_max)
|
||
except (ValueError, TypeError):
|
||
pass
|
||
|
||
results.append({
|
||
"title": title,
|
||
"company": company_name,
|
||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||
"is_remote": is_remote,
|
||
"department": dept,
|
||
"experience_level": exp_level,
|
||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
|
||
"salary_min": salary_min,
|
||
"salary_max": salary_max,
|
||
"job_url": job_url,
|
||
"source": "ashby"
|
||
})
|
||
except Exception as e:
|
||
print(f"[Ashby Warning] {company_name} failed: {e}")
|
||
return results
|
||
|
||
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
|
||
collected = []
|
||
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
|
||
|
||
print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...")
|
||
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
|
||
gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS)
|
||
for res in gh_results:
|
||
collected.extend(res)
|
||
|
||
print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...")
|
||
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
|
||
lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS)
|
||
for res in lever_results:
|
||
collected.extend(res)
|
||
|
||
print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...")
|
||
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
|
||
ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS)
|
||
for res in ashby_results:
|
||
collected.extend(res)
|
||
|
||
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
|
||
return collected
|