import requests import html import re import concurrent.futures from typing import List, Dict, Any # Expanded 90+ top company Greenhouse & Lever boards across all sectors GREENHOUSE_BOARDS = [ # Payments, Banking & Fintech ("stripe", "Stripe"), ("ramp", "Ramp"), ("brex", "Brex"), ("plaid", "Plaid"), ("chime", "Chime"), ("robinhood", "Robinhood"), ("coinbase", "Coinbase"), ("square", "Block (Square)"), ("toast", "Toast"), ("klarna", "Klarna"), ("marqeta", "Marqeta"), # AI & Frontier Tech ("scaleai", "Scale AI"), ("huggingface", "Hugging Face"), ("cohere", "Cohere"), ("perplexity", "Perplexity AI"), ("midjourney", "Midjourney"), ("stabilityai", "Stability AI"), # Legal, Compliance & RegTech ("ironclad", "Ironclad (LegalTech)"), ("relativity", "Relativity (Legal Solutions)"), # Developer Tools & Cloud Infrastructure ("datadog", "Datadog"), ("cloudflare", "Cloudflare"), ("figma", "Figma"), ("posthog", "PostHog"), ("linear", "Linear"), ("supabase", "Supabase"), ("databricks", "Databricks"), ("snowflake", "Snowflake"), ("mongodb", "MongoDB"), ("elastic", "Elastic"), ("hashicorp", "HashiCorp"), ("zapier", "Zapier"), ("retool", "Retool"), ("sentry", "Sentry"), ("pinecone", "Pinecone"), ("github", "GitHub"), ("gitlab", "GitLab"), ("launchdarkly", "LaunchDarkly"), ("snyk", "Snyk"), ("sourcegraph", "Sourcegraph"), # Education Tech & Academia ("coursera", "Coursera"), ("duolingo", "Duolingo"), ("quizlet", "Quizlet"), ("guild", "Guild Education"), # Construction, Real Estate & Logistics ("flexport", "Flexport (Logistics)"), ("samsara", "Samsara (IoT & Transport)"), ("procore", "Procore (Construction Software)"), # Consumer, Retail & SaaS ("doordash", "DoorDash"), ("uber", "Uber"), ("airbnb", "Airbnb"), ("pinterest", "Pinterest"), ("reddit", "Reddit"), ("zoom", "Zoom"), ("twilio", "Twilio"), ("asana", "Asana"), ("notion", "Notion"), ("canva", "Canva"), ("hubspot", "HubSpot"), ("zendesk", "Zendesk"), ("okta", "Okta"), ("atlassian", "Atlassian"), ("crowdstrike", "CrowdStrike"), ("sentinelone", "SentinelOne"), # Healthcare, Biotech & Science ("oscarhealth", "Oscar Health"), ("ro", "Ro Health"), ("tempus", "Tempus Labs"), ("guardanthealth", "Guardant Health"), ("flatiron", "Flatiron Health"), ("moderna", "Moderna"), ("cityblock", "Cityblock Health"), ("hims", "Hims & Hers Health"), ("springhealth", "Spring Health"), ("headway", "Headway"), ("talkspace", "Talkspace"), ("carbonhealth", "Carbon Health"), ("omadahealth", "Omada Health"), ("mavenclinic", "Maven Clinic"), ("goodrx", "GoodRx"), ("color", "Color Health"), ("invitae", "Invitae"), # Logistics, Supply Chain & Industrial ("deliverr", "Deliverr (Logistics)"), ("convoy", "Convoy (Freight & Logistics)"), ("fulfill", "Fulfill.com"), ("shipbob", "ShipBob"), ("flockfreight", "Flock Freight"), # Real Estate, Property & Construction ("compass", "Compass Real Estate"), ("opendoor", "Opendoor"), ("redfin", "Redfin"), ("cbre", "CBRE"), # Hospitality, Food & Travel ("sweetgreen", "Sweetgreen"), ("instacart", "Instacart"), ("grubhub", "Grubhub"), ("goldbelly", "Goldbelly"), # Professional Services, Accounting & Legal ("pilot", "Pilot (Bookkeeping & Tax)"), ("bench", "Bench Accounting"), ("brex", "Brex"), ("gusto", "Gusto (Payroll & HR)"), ("rippling", "Rippling (HR & Workforce)") ] LEVER_BOARDS = [ ("vercel", "Vercel"), ("spotify", "Spotify"), ("netflix", "Netflix"), ("palantir", "Palantir"), ("anthropic", "Anthropic"), ("discord", "Discord"), ("snap", "Snapchat"), ("figma", "Figma"), ("resend", "Resend"), ("modal", "Modal Labs"), ("sentry", "Sentry") ] ASHBY_BOARDS = [ ("ramp", "Ramp"), ("openai", "OpenAI"), ("anthropic", "Anthropic"), ("linear", "Linear"), ("cursor", "Cursor (Anysphere)"), ("replit", "Replit"), ("dust", "Dust"), ("ironclad", "Ironclad"), ("deel", "Deel"), ("superhuman", "Superhuman"), ("notion", "Notion"), ("retell", "Retell AI"), ("postman", "Postman"), ("vapi", "Vapi"), ("browserbase", "Browserbase"), ("tavus", "Tavus"), ("pave", "Pave"), ("cohere", "Cohere") ] NON_US_REGEX = re.compile( r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)', re.IGNORECASE ) def is_valid_us_location(location_name: str, title: str = "") -> bool: combined = f"{location_name} {title}" return not bool(NON_US_REGEX.search(combined)) def parse_is_us_remote(location_name: str, title: str) -> bool: loc_lower = location_name.lower() title_lower = title.lower() if not is_valid_us_location(location_name, title): return False is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"]) return is_remote_mention and (has_us_indicator or "remote" in loc_lower) def clean_html_text(raw: str) -> str: if not raw: return "" text = html.unescape(raw) text = html.unescape(text) text = re.sub(r'<[^>]+>', ' ', text) text = re.sub(r'\s+', ' ', text).strip() return text def determine_department(title: str, dept_name: str = "") -> str: combined = f"{title} {dept_name}".lower() if any(k in combined for k in ["data", "analytics", "analyst", "machine learning", "ai ", "bi ", "business intelligence"]): return "Data, AI & Analytics" if any(k in combined for k in ["it support", "help desk", "helpdesk", "sysadmin", "systems admin", "network engineer", "desktop support", "it specialist", "infrastructure", "active directory"]): return "IT & Systems Administration" if any(k in combined for k in ["art", "artist", "designer", "design", "illustrator", "animation", "animator", "ui/ux", "ux", "ui ", "graphic", "motion", "3d", "2d", "concept art"]): return "Art, Design & Creative" if any(k in combined for k in ["software", "engineer", "frontend", "backend", "fullstack", "full stack", "developer", "infra", "devops", "cloud", "mobile", "ios", "android"]): return "Software & Engineering" if any(k in combined for k in ["nurse", "medical", "clinical", "doctor", "health", "biotech"]): return "Healthcare & Medical" if any(k in combined for k in ["finance", "accounting", "treasury", "tax", "payroll", "audit", "legal", "counsel", "paralegal", "compliance"]): return "Finance, Accounting & Legal" if any(k in combined for k in ["sales", "marketing", "growth", "business development", "content", "product manager", "product owner"]): return "Sales, Marketing & Product" if any(k in combined for k in ["operations", "office", "admin", "recruiter", "people", "hr", "human resources", "workplace", "talent acquisition"]): return "Human Resources & Operations" if any(k in combined for k in ["construction", "electrician", "plumber", "contractor", "real estate", "property manager", "supply chain", "logistics", "warehouse"]): return "Trades, Construction & Logistics" if any(k in combined for k in ["state of", "teacher", "professor", "education", "curriculum", "academic"]): return "Government & Education" return "Other" def determine_experience_level(title: str) -> str: t = title.lower() if any(k in t for k in ["chief", "vp", "vice president", "head of", "director"]): return "Executive" if any(k in t for k in ["lead", "staff", "principal", "manager", "architect"]): return "Lead / Staff" if any(k in t for k in ["senior", "sr.", "sr "]): return "Senior" if any(k in t for k in ["junior", "jr.", "entry", "associate", "intern", "trainee"]): return "Entry Level" return "Mid-Level" def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: board_slug, company_name = item results = [] try: url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true" r = requests.get(url, headers=headers, timeout=6) if r.status_code == 200: data = r.json() jobs = data.get("jobs", []) for j in jobs: title = clean_html_text(j.get("title", "")) job_url = j.get("absolute_url", "") if not title or not job_url: continue location_obj = j.get("location", {}) location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) if not is_valid_us_location(location_name, title): continue is_remote = parse_is_us_remote(location_name, title) departments = j.get("departments", []) dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) content_raw = j.get("content", "") or "" desc_clean = clean_html_text(content_raw) results.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.", "salary_min": None, "salary_max": None, "job_url": job_url, "source": "greenhouse" }) except Exception as e: print(f"[Greenhouse Warning] {company_name} failed: {e}") return results def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: board_slug, company_name = item results = [] try: url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json" r = requests.get(url, headers=headers, timeout=6) if r.status_code == 200: jobs = r.json() for j in jobs: title = clean_html_text(j.get("text", "")) job_url = j.get("hostedUrl", "") if not title or not job_url: continue categories = j.get("categories", {}) location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA" workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else "" if not is_valid_us_location(location_name, title): continue is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) dept_text = categories.get("department", "") if isinstance(categories, dict) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) description_plain = j.get("descriptionPlain", "") or j.get("description", "") desc_clean = clean_html_text(description_plain) results.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.", "salary_min": None, "salary_max": None, "job_url": job_url, "source": "lever" }) except Exception as e: print(f"[Lever Warning] {company_name} failed: {e}") return results def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: board_slug, company_name = item results = [] try: url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}" r = requests.get(url, headers=headers, timeout=6) if r.status_code == 200: data = r.json() jobs = data.get("jobs", []) for j in jobs: title = clean_html_text(j.get("title", "")) job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}" if not title or not job_url: continue location_name = j.get("location", "Remote, USA") or "Remote, USA" if not is_valid_us_location(location_name, title): continue is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) dept_text = j.get("department", "") dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") salary_min = None salary_max = None comp = j.get("compensation") if isinstance(comp, dict): comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min") comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max") if comp_min: try: salary_min = float(comp_min) except (ValueError, TypeError): pass if comp_max: try: salary_max = float(comp_max) except (ValueError, TypeError): pass results.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.", "salary_min": salary_min, "salary_max": salary_max, "job_url": job_url, "source": "ashby" }) except Exception as e: print(f"[Ashby Warning] {company_name} failed: {e}") return results def run_ats_direct_ingestion() -> List[Dict[Any, Any]]: collected = [] headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...") with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor: gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS) for res in gh_results: collected.extend(res) print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...") with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor: lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS) for res in lever_results: collected.extend(res) print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...") with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor: ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS) for res in ashby_results: collected.extend(res) print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}") return collected