291 lines
12 KiB
Python
291 lines
12 KiB
Python
import requests
|
|
import html
|
|
import re
|
|
from typing import List, Dict, Any
|
|
|
|
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
|
|
GREENHOUSE_BOARDS = [
|
|
# Payments, Banking & Fintech
|
|
("stripe", "Stripe"),
|
|
("ramp", "Ramp"),
|
|
("brex", "Brex"),
|
|
("plaid", "Plaid"),
|
|
("chime", "Chime"),
|
|
("robinhood", "Robinhood"),
|
|
("coinbase", "Coinbase"),
|
|
("square", "Block (Square)"),
|
|
("toast", "Toast"),
|
|
("klarna", "Klarna"),
|
|
("marqeta", "Marqeta"),
|
|
|
|
# AI & Frontier Tech
|
|
("scaleai", "Scale AI"),
|
|
("huggingface", "Hugging Face"),
|
|
("cohere", "Cohere"),
|
|
("perplexity", "Perplexity AI"),
|
|
("midjourney", "Midjourney"),
|
|
("stabilityai", "Stability AI"),
|
|
|
|
# Legal, Compliance & RegTech
|
|
("ironclad", "Ironclad (LegalTech)"),
|
|
("relativity", "Relativity (Legal Solutions)"),
|
|
|
|
# Developer Tools & Cloud Infrastructure
|
|
("datadog", "Datadog"),
|
|
("cloudflare", "Cloudflare"),
|
|
("figma", "Figma"),
|
|
("posthog", "PostHog"),
|
|
("linear", "Linear"),
|
|
("supabase", "Supabase"),
|
|
("databricks", "Databricks"),
|
|
("snowflake", "Snowflake"),
|
|
("mongodb", "MongoDB"),
|
|
("elastic", "Elastic"),
|
|
("hashicorp", "HashiCorp"),
|
|
("zapier", "Zapier"),
|
|
("retool", "Retool"),
|
|
("sentry", "Sentry"),
|
|
("pinecone", "Pinecone"),
|
|
("github", "GitHub"),
|
|
("gitlab", "GitLab"),
|
|
("launchdarkly", "LaunchDarkly"),
|
|
("snyk", "Snyk"),
|
|
("sourcegraph", "Sourcegraph"),
|
|
|
|
# Education Tech & Academia
|
|
("coursera", "Coursera"),
|
|
("duolingo", "Duolingo"),
|
|
("quizlet", "Quizlet"),
|
|
("guild", "Guild Education"),
|
|
|
|
# Construction, Real Estate & Logistics
|
|
("flexport", "Flexport (Logistics)"),
|
|
("samsara", "Samsara (IoT & Transport)"),
|
|
("procore", "Procore (Construction Software)"),
|
|
|
|
# Consumer, Retail & SaaS
|
|
("doordash", "DoorDash"),
|
|
("uber", "Uber"),
|
|
("airbnb", "Airbnb"),
|
|
("pinterest", "Pinterest"),
|
|
("reddit", "Reddit"),
|
|
("zoom", "Zoom"),
|
|
("twilio", "Twilio"),
|
|
("asana", "Asana"),
|
|
("notion", "Notion"),
|
|
("canva", "Canva"),
|
|
("hubspot", "HubSpot"),
|
|
("zendesk", "Zendesk"),
|
|
("okta", "Okta"),
|
|
("atlassian", "Atlassian"),
|
|
("crowdstrike", "CrowdStrike"),
|
|
("sentinelone", "SentinelOne"),
|
|
|
|
# Healthcare, Biotech & Science
|
|
("oscarhealth", "Oscar Health"),
|
|
("ro", "Ro Health"),
|
|
("tempus", "Tempus Labs"),
|
|
("guardanthealth", "Guardant Health"),
|
|
("flatiron", "Flatiron Health"),
|
|
("moderna", "Moderna")
|
|
]
|
|
|
|
LEVER_BOARDS = [
|
|
("vercel", "Vercel"),
|
|
("spotify", "Spotify"),
|
|
("netflix", "Netflix"),
|
|
("palantir", "Palantir"),
|
|
("anthropic", "Anthropic"),
|
|
("discord", "Discord"),
|
|
("snap", "Snapchat"),
|
|
("figma", "Figma"),
|
|
("resend", "Resend"),
|
|
("modal", "Modal Labs"),
|
|
("sentry", "Sentry")
|
|
]
|
|
|
|
NON_US_KEYWORDS = [
|
|
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
|
|
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
|
|
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
|
|
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
|
|
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
|
|
"ireland", "dublin", "switzerland", "zurich"
|
|
]
|
|
|
|
def is_valid_us_location(location_name: str, title: str) -> bool:
|
|
loc_lower = location_name.lower()
|
|
title_lower = title.lower()
|
|
|
|
for non_us in NON_US_KEYWORDS:
|
|
if non_us in loc_lower or non_us in title_lower:
|
|
return False
|
|
|
|
return True
|
|
|
|
def parse_is_us_remote(location_name: str, title: str) -> bool:
|
|
loc_lower = location_name.lower()
|
|
title_lower = title.lower()
|
|
|
|
if not is_valid_us_location(location_name, title):
|
|
return False
|
|
|
|
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
|
|
has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"])
|
|
|
|
return is_remote_mention and (has_us_indicator or "remote" in loc_lower)
|
|
|
|
def clean_html_text(raw: str) -> str:
|
|
if not raw:
|
|
return ""
|
|
text = html.unescape(raw)
|
|
text = html.unescape(text)
|
|
text = re.sub(r'<[^>]+>', ' ', text)
|
|
text = re.sub(r'\s+', ' ', text).strip()
|
|
return text
|
|
|
|
def determine_department(title: str, dept_name: str = "") -> str:
|
|
combined = f"{title} {dept_name}".lower()
|
|
|
|
if any(k in combined for k in ["data", "analytics", "analyst", "machine learning", "ai ", "bi ", "business intelligence"]):
|
|
return "Data, AI & Analytics"
|
|
|
|
if any(k in combined for k in ["it support", "help desk", "helpdesk", "sysadmin", "systems admin", "network engineer", "desktop support", "it specialist", "infrastructure", "active directory"]):
|
|
return "IT & Systems Administration"
|
|
|
|
if any(k in combined for k in ["art", "artist", "designer", "design", "illustrator", "animation", "animator", "ui/ux", "ux", "ui ", "graphic", "motion", "3d", "2d", "concept art"]):
|
|
return "Art, Design & Creative"
|
|
|
|
if any(k in combined for k in ["software", "engineer", "frontend", "backend", "fullstack", "full stack", "developer", "infra", "devops", "cloud", "mobile", "ios", "android"]):
|
|
return "Software & Engineering"
|
|
|
|
if any(k in combined for k in ["nurse", "medical", "clinical", "doctor", "health", "biotech"]):
|
|
return "Healthcare & Medical"
|
|
|
|
if any(k in combined for k in ["finance", "accounting", "treasury", "tax", "payroll", "audit", "legal", "counsel", "paralegal", "compliance"]):
|
|
return "Finance, Accounting & Legal"
|
|
|
|
if any(k in combined for k in ["sales", "marketing", "growth", "business development", "content", "product manager", "product owner"]):
|
|
return "Sales, Marketing & Product"
|
|
|
|
if any(k in combined for k in ["operations", "office", "admin", "recruiter", "people", "hr", "human resources", "workplace", "talent acquisition"]):
|
|
return "Human Resources & Operations"
|
|
|
|
if any(k in combined for k in ["construction", "electrician", "plumber", "contractor", "real estate", "property manager", "supply chain", "logistics", "warehouse"]):
|
|
return "Trades, Construction & Logistics"
|
|
|
|
if any(k in combined for k in ["state of", "teacher", "professor", "education", "curriculum", "academic"]):
|
|
return "Government & Education"
|
|
|
|
return "Other"
|
|
|
|
def determine_experience_level(title: str) -> str:
|
|
t = title.lower()
|
|
if any(k in t for k in ["chief", "vp", "vice president", "head of", "director"]):
|
|
return "Executive"
|
|
if any(k in t for k in ["lead", "staff", "principal", "manager", "architect"]):
|
|
return "Lead / Staff"
|
|
if any(k in t for k in ["senior", "sr.", "sr "]):
|
|
return "Senior"
|
|
if any(k in t for k in ["junior", "jr.", "entry", "associate", "intern", "trainee"]):
|
|
return "Entry Level"
|
|
return "Mid-Level"
|
|
|
|
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
|
|
collected = []
|
|
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
|
|
|
|
print("[ATS Direct] Fetching 90+ Greenhouse public API boards (US Remote Filtered)...")
|
|
for board_slug, company_name in GREENHOUSE_BOARDS:
|
|
try:
|
|
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
|
|
r = requests.get(url, headers=headers, timeout=6)
|
|
if r.status_code == 200:
|
|
data = r.json()
|
|
jobs = data.get("jobs", [])
|
|
|
|
for j in jobs[:30]:
|
|
title = clean_html_text(j.get("title", ""))
|
|
job_url = j.get("absolute_url", "")
|
|
if not title or not job_url:
|
|
continue
|
|
|
|
location_obj = j.get("location", {})
|
|
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
|
|
|
if not is_valid_us_location(location_name, title):
|
|
continue
|
|
|
|
is_remote = parse_is_us_remote(location_name, title)
|
|
departments = j.get("departments", [])
|
|
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
|
|
|
dept = determine_department(title, dept_text)
|
|
exp_level = determine_experience_level(title)
|
|
|
|
content_raw = j.get("content", "") or ""
|
|
desc_clean = clean_html_text(content_raw)
|
|
|
|
collected.append({
|
|
"title": title,
|
|
"company": company_name,
|
|
"location": location_name if location_name != "Remote" else "Remote, USA",
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply direct on company board.",
|
|
"salary_min": None,
|
|
"salary_max": None,
|
|
"job_url": job_url,
|
|
"source": "greenhouse"
|
|
})
|
|
except Exception as e:
|
|
print(f"[Greenhouse Warning] {company_name} failed: {e}")
|
|
|
|
print("[ATS Direct] Fetching Lever public API boards...")
|
|
for board_slug, company_name in LEVER_BOARDS:
|
|
try:
|
|
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
|
|
r = requests.get(url, headers=headers, timeout=6)
|
|
if r.status_code == 200:
|
|
jobs = r.json()
|
|
for j in jobs[:30]:
|
|
title = clean_html_text(j.get("text", ""))
|
|
job_url = j.get("hostedUrl", "")
|
|
if not title or not job_url:
|
|
continue
|
|
|
|
categories = j.get("categories", {})
|
|
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
|
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
|
|
|
if not is_valid_us_location(location_name, title):
|
|
continue
|
|
|
|
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
|
|
|
|
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
|
dept = determine_department(title, dept_text)
|
|
exp_level = determine_experience_level(title)
|
|
|
|
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
|
desc_clean = clean_html_text(description_plain)
|
|
|
|
collected.append({
|
|
"title": title,
|
|
"company": company_name,
|
|
"location": location_name if location_name != "Remote" else "Remote, USA",
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
|
"salary_min": None,
|
|
"salary_max": None,
|
|
"job_url": job_url,
|
|
"source": "lever"
|
|
})
|
|
except Exception as e:
|
|
print(f"[Lever Warning] {company_name} failed: {e}")
|
|
|
|
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
|
|
return collected
|