import os import requests import re from urllib.parse import urlparse, urljoin from typing import Optional, Tuple, Dict, Any, List from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level COMMON_CAREER_PATHS = [ "/careers", "/career", "/jobs", "/job", "/work-with-us", "/join-us", "/join", "/open-positions", "/vacancies" ] NON_US_KEYWORDS = [ "london", "uk", "united kingdom", "england", "germany", "berlin", "munich", "france", "paris", "canada", "toronto", "vancouver", "montreal", "india", "bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney", "melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands", "emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona", "ireland", "dublin", "switzerland", "zurich" ] def is_valid_us_location(location_name: str, title: str = "") -> bool: loc_lower = (location_name or "").lower() title_lower = (title or "").lower() for non_us in NON_US_KEYWORDS: if non_us in loc_lower or non_us in title_lower: return False return True def parse_is_us_remote(location_name: str, title: str = "") -> bool: loc_lower = (location_name or "").lower() title_lower = (title or "").lower() if not is_valid_us_location(location_name, title): return False is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower return bool(is_remote_mention) class SmartCareersCrawler: """ Crawls company domains, finds their /careers or /jobs pages, and automatically detects and ingests from Greenhouse, Lever, or Ashby. Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY. """ def __init__(self, headers: Optional[Dict[str, str]] = None): self.headers = headers or { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" } # Configure proxy if supplied in environment proxy_url = os.getenv("SOCKS5_PROXY") or os.getenv("PROXY_URL") or os.getenv("HTTP_PROXY") or "" self.proxies = {} if proxy_url: self.proxies = { "http": proxy_url, "https": proxy_url } def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]: parsed = urlparse(url) host = parsed.netloc.lower() path = parsed.path.strip("/") if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host: parts = [p for p in path.split("/") if p and p != "embed"] if parts: return "greenhouse", parts[0] if "jobs.lever.co" in host: parts = [p for p in path.split("/") if p] if parts: return "lever", parts[0] if "jobs.ashbyhq.com" in host: parts = [p for p in path.split("/") if p] if parts: return "ashby", parts[0] if not html_text: return None, None gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if gh_match: return "greenhouse", gh_match.group(1) lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if lever_match: return "lever", lever_match.group(1) ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if ashby_match: return "ashby", ashby_match.group(1) return None, None def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]: clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower() base_url = f"https://{clean_domain}" for path in COMMON_CAREER_PATHS: test_url = f"{base_url}{path}" try: r = requests.get(test_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) final_url = r.url provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "") if provider and slug: return provider, slug, final_url except Exception: continue try: r = requests.get(base_url, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) if r.status_code == 200: provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) if provider and slug: return provider, slug, r.url links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) for link in links[:5]: target = link if link.startswith("http") else urljoin(base_url, link) provider, slug = self.detect_ats_from_url_or_html(target) if provider and slug: return provider, slug, target try: cr = requests.get(target, headers=self.headers, proxies=self.proxies, timeout=5, allow_redirects=True) provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) if provider and slug: return provider, slug, cr.url except Exception: continue except Exception: pass return None, None, None def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true" collected = [] try: r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) if r.status_code != 200: return collected data = r.json() jobs = data.get("jobs", []) for j in jobs: title = clean_html_text(j.get("title", "")) job_url = j.get("absolute_url", "") if not title or not job_url: continue location_obj = j.get("location", {}) location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) if not is_valid_us_location(location_name, title): continue is_remote = parse_is_us_remote(location_name, title) departments = j.get("departments", []) dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) desc_clean = clean_html_text(j.get("content", "") or "") collected.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.", "salary_min": None, "salary_max": None, "job_url": job_url, "source": "greenhouse" }) except Exception as e: print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}") return collected def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://api.lever.co/v0/postings/{slug}?mode=json" collected = [] try: r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) if r.status_code != 200: return collected jobs = r.json() for j in jobs: title = clean_html_text(j.get("text", "")) job_url = j.get("hostedUrl", "") if not title or not job_url: continue categories = j.get("categories", {}) location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA" workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else "" if not is_valid_us_location(location_name, title): continue is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) dept_text = categories.get("department", "") if isinstance(categories, dict) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) description_plain = j.get("descriptionPlain", "") or j.get("description", "") desc_clean = clean_html_text(description_plain) collected.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.", "salary_min": None, "salary_max": None, "job_url": job_url, "source": "lever" }) except Exception as e: print(f"[Lever Warning] {company_name} ({slug}) error: {e}") return collected def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}" collected = [] try: r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8) if r.status_code != 200: return collected data = r.json() jobs = data.get("jobs", []) for j in jobs: title = clean_html_text(j.get("title", "")) job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}" if not title or not job_url: continue location_name = j.get("location", "Remote, USA") or "Remote, USA" if not is_valid_us_location(location_name, title): continue is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) dept_text = j.get("department", "") dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") salary_min = None salary_max = None comp = j.get("compensation") if isinstance(comp, dict): comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min") comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max") if comp_min: try: salary_min = float(comp_min) except (ValueError, TypeError): pass if comp_max: try: salary_max = float(comp_max) except (ValueError, TypeError): pass collected.append({ "title": title, "company": company_name, "location": location_name, "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.", "salary_min": salary_min, "salary_max": salary_max, "job_url": job_url, "source": "ashby" }) except Exception as e: print(f"[Ashby Warning] {company_name} ({slug}) error: {e}") return collected