from scrapers.proxy_manager import get_rotating_proxy_dict import os import requests import re from urllib.parse import urlparse, urljoin from typing import Optional, Tuple, Dict, Any, List from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level COMMON_CAREER_PATHS = [ "/careers", "/career", "/jobs", "/job", "/work-with-us", "/join-us", "/join", "/open-positions", "/vacancies" ] NON_US_KEYWORDS = [ "london", "uk", "united kingdom", "england", "germany", "berlin", "munich", "france", "paris", "canada", "toronto", "vancouver", "montreal", "india", "bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney", "melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands", "emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona", "ireland", "dublin", "switzerland", "zurich" ] def is_valid_us_location(location_name: str, title: str = "") -> bool: loc_lower = (location_name or "").lower() title_lower = (title or "").lower() for non_us in NON_US_KEYWORDS: if non_us in loc_lower or non_us in title_lower: return False return True def parse_is_us_remote(location_name: str, title: str = "") -> bool: loc_lower = (location_name or "").lower() title_lower = (title or "").lower() if not is_valid_us_location(location_name, title): return False is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower return bool(is_remote_mention) class SmartCareersCrawler: """ Crawls company domains, finds their /careers or /jobs pages, and automatically detects and ingests from Greenhouse, Lever, or Ashby. Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY. """ def __init__(self, headers: Optional[Dict[str, str]] = None): self.headers = headers or { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" } # Configure rotating SOCKS5 proxies self.proxies = get_rotating_proxy_dict() def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]: parsed = urlparse(url) host = parsed.netloc.lower() path = parsed.path.strip("/") if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host: parts = [p for p in path.split("/") if p and p != "embed"] if parts: return "greenhouse", parts[0] if "jobs.lever.co" in host: parts = [p for p in path.split("/") if p] if parts: return "lever", parts[0] if "jobs.ashbyhq.com" in host: parts = [p for p in path.split("/") if p] if parts: return "ashby", parts[0] if not html_text: return None, None gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if gh_match: return "greenhouse", gh_match.group(1) lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if lever_match: return "lever", lever_match.group(1) ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if ashby_match: return "ashby", ashby_match.group(1) return None, None def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]: clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower() base_url = f"https://{clean_domain}" for path in COMMON_CAREER_PATHS: test_url = f"{base_url}{path}" try: r = requests.get(test_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True) final_url = r.url provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "") if provider and slug: return provider, slug, final_url except Exception: continue try: r = requests.get(base_url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True) if r.status_code == 200: provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) if provider and slug: return provider, slug, r.url links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) for link in links[:5]: target = link if link.startswith("http") else urljoin(base_url, link) provider, slug = self.detect_ats_from_url_or_html(target) if provider and slug: return provider, slug, target try: cr = requests.get(target, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=5, allow_redirects=True) provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) if provider and slug: return provider, slug, cr.url except Exception: continue except Exception: pass return None, None, None def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true" collected = [] try: r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8) if r.status_code != 200: return collected data = r.json() jobs = data.get("jobs", []) for j in jobs: title = clean_html_text(j.get("title", "")) job_url = j.get("absolute_url", "") if not title or not job_url: continue location_obj = j.get("location", {}) location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) if not is_valid_us_location(location_name, title): continue is_remote = parse_is_us_remote(location_name, title) departments = j.get("departments", []) dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) desc_clean = clean_html_text(j.get("content", "") or "") collected.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.", "salary_min": None, "salary_max": None, "job_url": job_url, "source": "greenhouse" }) except Exception as e: print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}") return collected def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://api.lever.co/v0/postings/{slug}?mode=json" collected = [] try: r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8) if r.status_code != 200: return collected jobs = r.json() for j in jobs: title = clean_html_text(j.get("text", "")) job_url = j.get("hostedUrl", "") if not title or not job_url: continue categories = j.get("categories", {}) location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA" workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else "" if not is_valid_us_location(location_name, title): continue is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) dept_text = categories.get("department", "") if isinstance(categories, dict) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) description_plain = j.get("descriptionPlain", "") or j.get("description", "") desc_clean = clean_html_text(description_plain) collected.append({ "title": title, "company": company_name, "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.", "salary_min": None, "salary_max": None, "job_url": job_url, "source": "lever" }) except Exception as e: print(f"[Lever Warning] {company_name} ({slug}) error: {e}") return collected def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}" collected = [] try: r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8) if r.status_code != 200: return collected data = r.json() jobs = data.get("jobs", []) for j in jobs: title = clean_html_text(j.get("title", "")) job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}" if not title or not job_url: continue location_name = j.get("location", "Remote, USA") or "Remote, USA" if not is_valid_us_location(location_name, title): continue is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) dept_text = j.get("department", "") dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") salary_min = None salary_max = None comp = j.get("compensation") if isinstance(comp, dict): comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min") comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max") if comp_min: try: salary_min = float(comp_min) except (ValueError, TypeError): pass if comp_max: try: salary_max = float(comp_max) except (ValueError, TypeError): pass collected.append({ "title": title, "company": company_name, "location": location_name, "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.", "salary_min": salary_min, "salary_max": salary_max, "job_url": job_url, "source": "ashby" }) except Exception as e: print(f"[Ashby Warning] {company_name} ({slug}) error: {e}") return collected def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]: import json from bs4 import BeautifulSoup jobs = [] if not html_text: return jobs soup = BeautifulSoup(html_text, "html.parser") scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I)) for s in scripts: try: data = json.loads(s.string or "{}") except Exception: continue items = data if isinstance(data, list) else [data] for item in items: if not isinstance(item, dict): continue graph = item.get("@graph", []) candidates = graph if isinstance(graph, list) and graph else [item] for c in candidates: if not isinstance(c, dict): continue item_type = str(c.get("@type", "")).lower() if "jobposting" in item_type: title = clean_html_text(c.get("title", "")) if not title: continue hiring_org = c.get("hiringOrganization", {}) company = default_company if isinstance(hiring_org, dict): company = hiring_org.get("name") or default_company elif isinstance(hiring_org, str): company = hiring_org job_loc = c.get("jobLocation", {}) location_name = "Remote, USA" if isinstance(job_loc, dict): addr = job_loc.get("address", {}) if isinstance(addr, dict): city = addr.get("addressLocality", "") region = addr.get("addressRegion", "") country = addr.get("addressCountry", "") parts = [p for p in [city, region, country] if p] location_name = ", ".join(parts) or "USA" elif isinstance(addr, str): location_name = addr work_location_type = str(c.get("jobLocationType", "")).upper() is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower() if not is_valid_us_location(location_name, title): continue desc_raw = c.get("description", "") desc_clean = clean_html_text(desc_raw) salary_min = None salary_max = None base_salary = c.get("baseSalary", {}) if isinstance(base_salary, dict): val = base_salary.get("value", {}) if isinstance(val, dict): s_min = val.get("minValue") or val.get("value") s_max = val.get("maxValue") or val.get("value") if s_min: try: salary_min = float(s_min) except: pass if s_max: try: salary_max = float(s_max) except: pass job_url = c.get("url") or page_url if job_url.startswith("/"): job_url = urljoin(page_url, job_url) dept = determine_department(title, "") exp_level = determine_experience_level(title) jobs.append({ "title": title, "company": company, "location": location_name, "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.", "salary_min": salary_min, "salary_max": salary_max, "job_url": job_url, "source": "web_direct" }) return jobs def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]: from bs4 import BeautifulSoup jobs = [] if not html_text: return jobs soup = BeautifulSoup(html_text, "html.parser") anchors = soup.find_all("a", href=True) seen_links = set() for a in anchors: href = a.get("href", "").strip() text = clean_html_text(a.get_text(strip=True)) if not href or not text or len(text) < 4 or len(text) > 90: continue is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="]) if not is_job_href: continue full_url = href if href.startswith("http") else urljoin(page_url, href) if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"): continue seen_links.add(full_url) lower_text = text.lower() if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]): continue dept = determine_department(text, "") exp_level = determine_experience_level(text) is_remote = "remote" in lower_text jobs.append({ "title": text, "company": default_company, "location": "Remote, USA" if is_remote else "United States", "is_remote": is_remote, "department": dept, "experience_level": exp_level, "description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}", "salary_min": None, "salary_max": None, "job_url": full_url, "source": "web_direct" }) if len(jobs) >= 50: break return jobs def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]: try: r = requests.get(url, headers=self.headers, proxies=get_rotating_proxy_dict(), timeout=8, allow_redirects=True) if r.status_code != 200: return [] html_text = r.text # 1. Try Schema.org JSON-LD first (highest quality structured jobs) jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company) if jobs: print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}") return jobs # 2. Try HTML job links jobs = self.extract_html_job_links(html_text, r.url, default_company) if jobs: print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}") return jobs except Exception as e: print(f"[Smart Crawler Warning] Scraping {url} failed: {e}") return []