diff --git a/scraper/main.py b/scraper/main.py index fc6d278..5cbb34c 100644 --- a/scraper/main.py +++ b/scraper/main.py @@ -1,6 +1,7 @@ import os import time import sys +import concurrent.futures from scrapers.ats_ingestion import run_ats_direct_ingestion from scrapers.art_and_design_ingestion import run_art_and_design_ingestion from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion @@ -57,30 +58,37 @@ DOMAINS_TO_PROBE = [ ] def run_smart_crawler_scrapes(): - print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...") + print("[Smart Crawler] Universal Web Crawler scanning domains concurrently...") crawler = SmartCareersCrawler() discovered_jobs = [] - for domain, company_name in DOMAINS_TO_PROBE: + def probe_and_collect(item): + domain, company_name = item + jobs = [] try: provider, slug, careers_url = crawler.probe_domain_for_careers(domain) if provider and slug: print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'") if provider == "greenhouse": jobs = crawler.fetch_greenhouse_board(slug, company_name) - discovered_jobs.extend(jobs) elif provider == "lever": jobs = crawler.fetch_lever_board(slug, company_name) - discovered_jobs.extend(jobs) elif provider == "ashby": jobs = crawler.fetch_ashby_board(slug, company_name) - discovered_jobs.extend(jobs) + elif provider == "workday": + jobs = crawler.fetch_workday_board(slug, company_name) elif careers_url: print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}") jobs = crawler.scrape_native_career_page(careers_url, company_name) - discovered_jobs.extend(jobs) except Exception as e: print(f"[Smart Crawler Warning] Probing {domain} failed: {e}") + return jobs + + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor: + results = executor.map(probe_and_collect, DOMAINS_TO_PROBE) + for job_batch in results: + if job_batch: + discovered_jobs.extend(job_batch) print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.") return discovered_jobs @@ -92,7 +100,7 @@ def execute_all_scrapes(): all_jobs = [] - # 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby) + # 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby/Workday) try: crawler_jobs = run_smart_crawler_scrapes() all_jobs.extend(crawler_jobs) @@ -147,7 +155,7 @@ def execute_all_scrapes(): if all_jobs: count = upsert_jobs(all_jobs) - print(f"✅ Successfully written/updated {count} jobs into database.") + print(f"[DB Ingestion] Successfully written/updated {count} jobs into database.") else: print("No job records were collected in this run.") diff --git a/scraper/scrapers/art_and_design_ingestion.py b/scraper/scrapers/art_and_design_ingestion.py index 71ba5a3..98365d1 100644 --- a/scraper/scrapers/art_and_design_ingestion.py +++ b/scraper/scrapers/art_and_design_ingestion.py @@ -1,6 +1,7 @@ import requests import html import re +import concurrent.futures from typing import List, Dict, Any from scrapers.ats_ingestion import clean_html_text, determine_experience_level @@ -60,7 +61,7 @@ CURATED_ART_DESIGN_JOBS = [ "is_remote": True, "department": "Art & Design", "experience_level": "Mid-Level", - "description": "Create high-fidelity 3D environment assets, digital matte paintings, lighting setups, and visual concept art for Unreal Engine real-time virtual production.", + "description": "Create photorealistic 3D environments, textures, lighting, and concept digital paintings using Unreal Engine 5, Blender, Maya, and Substance Painter.", "salary_min": 105000, "salary_max": 145000, "job_url": "https://www.epicgames.com/careers/3d-environment-artist-remote", @@ -143,54 +144,59 @@ def is_art_and_design_job(title: str, dept_text: str = "") -> bool: ] return any(k in combined for k in creative_keywords) +def fetch_art_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: + board_slug, company_name = item + results = [] + try: + url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true" + r = requests.get(url, headers=headers, timeout=6) + if r.status_code == 200: + data = r.json() + jobs = data.get("jobs", []) + for j in jobs: + title = clean_html_text(j.get("title", "")) + job_url = j.get("absolute_url", "") + if not title or not job_url: + continue + + departments = j.get("departments", []) + dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" + + if is_art_and_design_job(title, dept_text): + location_obj = j.get("location", {}) + location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj) + is_remote = "remote" in location_name.lower() or "remote" in title.lower() + + exp_level = determine_experience_level(title) + content_raw = j.get("content", "") or "" + desc_clean = clean_html_text(content_raw) + + results.append({ + "title": title, + "company": company_name, + "location": location_name, + "is_remote": is_remote, + "department": "Art & Design", + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Art & Design position at {company_name}.", + "salary_min": None, + "salary_max": None, + "job_url": job_url, + "source": "greenhouse" + }) + except Exception as e: + print(f"[Art & Design Warning] {company_name} failed: {e}") + return results + def run_art_and_design_ingestion() -> List[Dict[Any, Any]]: collected = [] headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} - print("[Art & Design Ingestion] Ingesting dedicated Art, Creative, Gaming & Animation company boards...") - - # 1. Fetch from Greenhouse creative boards - for board_slug, company_name in ART_DESIGN_GREENHOUSE_BOARDS: - try: - url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true" - r = requests.get(url, headers=headers, timeout=6) - if r.status_code == 200: - data = r.json() - jobs = data.get("jobs", []) - print(f"[Art & Design Greenhouse] {company_name}: {len(jobs)} total board postings.") - for j in jobs: - title = clean_html_text(j.get("title", "")) - job_url = j.get("absolute_url", "") - if not title or not job_url: - continue - - departments = j.get("departments", []) - dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" - - if is_art_and_design_job(title, dept_text): - location_obj = j.get("location", {}) - location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj) - is_remote = "remote" in location_name.lower() or "remote" in title.lower() - - exp_level = determine_experience_level(title) - content_raw = j.get("content", "") or "" - desc_clean = clean_html_text(content_raw) - - collected.append({ - "title": title, - "company": company_name, - "location": location_name, - "is_remote": is_remote, - "department": "Art & Design", - "experience_level": exp_level, - "description": desc_clean[:2500] or f"Art & Design position at {company_name}.", - "salary_min": None, - "salary_max": None, - "job_url": job_url, - "source": "greenhouse" - }) - except Exception as e: - print(f"[Art & Design Warning] {company_name} failed: {e}") + print("[Art & Design Ingestion] Concurrently fetching dedicated Art & Creative boards...") + with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor: + results = executor.map(lambda item: fetch_art_greenhouse_board(item, headers), ART_DESIGN_GREENHOUSE_BOARDS) + for batch in results: + collected.extend(batch) # 2. Add curated CT & Remote Art/Design/Media jobs collected.extend(CURATED_ART_DESIGN_JOBS) diff --git a/scraper/scrapers/ats_ingestion.py b/scraper/scrapers/ats_ingestion.py index 2b7448b..a3a86f2 100644 --- a/scraper/scrapers/ats_ingestion.py +++ b/scraper/scrapers/ats_ingestion.py @@ -1,6 +1,7 @@ import requests import html import re +import concurrent.futures from typing import List, Dict, Any # Expanded 90+ top company Greenhouse & Lever boards across all sectors @@ -125,24 +126,14 @@ ASHBY_BOARDS = [ ("cohere", "Cohere") ] -NON_US_KEYWORDS = [ - "london", "uk", "united kingdom", "england", "germany", "berlin", "munich", - "france", "paris", "canada", "toronto", "vancouver", "montreal", "india", - "bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney", - "melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands", - "emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona", - "ireland", "dublin", "switzerland", "zurich" -] +NON_US_REGEX = re.compile( + r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)', + re.IGNORECASE +) -def is_valid_us_location(location_name: str, title: str) -> bool: - loc_lower = location_name.lower() - title_lower = title.lower() - - for non_us in NON_US_KEYWORDS: - if non_us in loc_lower or non_us in title_lower: - return False - - return True +def is_valid_us_location(location_name: str, title: str = "") -> bool: + combined = f"{location_name} {title}" + return not bool(NON_US_REGEX.search(combined)) def parse_is_us_remote(location_name: str, title: str) -> bool: loc_lower = location_name.lower() @@ -212,159 +203,181 @@ def determine_experience_level(title: str) -> str: return "Entry Level" return "Mid-Level" +def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: + board_slug, company_name = item + results = [] + try: + url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true" + r = requests.get(url, headers=headers, timeout=6) + if r.status_code == 200: + data = r.json() + jobs = data.get("jobs", []) + for j in jobs: + title = clean_html_text(j.get("title", "")) + job_url = j.get("absolute_url", "") + if not title or not job_url: + continue + + location_obj = j.get("location", {}) + location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) + + if not is_valid_us_location(location_name, title): + continue + + is_remote = parse_is_us_remote(location_name, title) + departments = j.get("departments", []) + dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" + + dept = determine_department(title, dept_text) + exp_level = determine_experience_level(title) + + content_raw = j.get("content", "") or "" + desc_clean = clean_html_text(content_raw) + + results.append({ + "title": title, + "company": company_name, + "location": location_name if location_name != "Remote" else "Remote, USA", + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.", + "salary_min": None, + "salary_max": None, + "job_url": job_url, + "source": "greenhouse" + }) + except Exception as e: + print(f"[Greenhouse Warning] {company_name} failed: {e}") + return results + +def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: + board_slug, company_name = item + results = [] + try: + url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json" + r = requests.get(url, headers=headers, timeout=6) + if r.status_code == 200: + jobs = r.json() + for j in jobs: + title = clean_html_text(j.get("text", "")) + job_url = j.get("hostedUrl", "") + if not title or not job_url: + continue + + categories = j.get("categories", {}) + location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA" + workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else "" + + if not is_valid_us_location(location_name, title): + continue + + is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) + dept_text = categories.get("department", "") if isinstance(categories, dict) else "" + dept = determine_department(title, dept_text) + exp_level = determine_experience_level(title) + + description_plain = j.get("descriptionPlain", "") or j.get("description", "") + desc_clean = clean_html_text(description_plain) + + results.append({ + "title": title, + "company": company_name, + "location": location_name if location_name != "Remote" else "Remote, USA", + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.", + "salary_min": None, + "salary_max": None, + "job_url": job_url, + "source": "lever" + }) + except Exception as e: + print(f"[Lever Warning] {company_name} failed: {e}") + return results + +def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]: + board_slug, company_name = item + results = [] + try: + url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}" + r = requests.get(url, headers=headers, timeout=6) + if r.status_code == 200: + data = r.json() + jobs = data.get("jobs", []) + for j in jobs: + title = clean_html_text(j.get("title", "")) + job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}" + if not title or not job_url: + continue + + location_name = j.get("location", "Remote, USA") or "Remote, USA" + if not is_valid_us_location(location_name, title): + continue + + is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) + dept_text = j.get("department", "") + dept = determine_department(title, dept_text) + exp_level = determine_experience_level(title) + + desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") + + salary_min = None + salary_max = None + comp = j.get("compensation") + if isinstance(comp, dict): + comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min") + comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max") + if comp_min: + try: + salary_min = float(comp_min) + except (ValueError, TypeError): + pass + if comp_max: + try: + salary_max = float(comp_max) + except (ValueError, TypeError): + pass + + results.append({ + "title": title, + "company": company_name, + "location": location_name if location_name != "Remote" else "Remote, USA", + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.", + "salary_min": salary_min, + "salary_max": salary_max, + "job_url": job_url, + "source": "ashby" + }) + except Exception as e: + print(f"[Ashby Warning] {company_name} failed: {e}") + return results + def run_ats_direct_ingestion() -> List[Dict[Any, Any]]: collected = [] headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} - print("[ATS Direct] Fetching 90+ Greenhouse public API boards (US Remote Filtered)...") - for board_slug, company_name in GREENHOUSE_BOARDS: - try: - url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true" - r = requests.get(url, headers=headers, timeout=6) - if r.status_code == 200: - data = r.json() - jobs = data.get("jobs", []) - - for j in jobs: - title = clean_html_text(j.get("title", "")) - job_url = j.get("absolute_url", "") - if not title or not job_url: - continue + print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...") + with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor: + gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS) + for res in gh_results: + collected.extend(res) - location_obj = j.get("location", {}) - location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) + print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...") + with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor: + lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS) + for res in lever_results: + collected.extend(res) - if not is_valid_us_location(location_name, title): - continue - - is_remote = parse_is_us_remote(location_name, title) - departments = j.get("departments", []) - dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" - - dept = determine_department(title, dept_text) - exp_level = determine_experience_level(title) - - content_raw = j.get("content", "") or "" - desc_clean = clean_html_text(content_raw) - - collected.append({ - "title": title, - "company": company_name, - "location": location_name if location_name != "Remote" else "Remote, USA", - "is_remote": is_remote, - "department": dept, - "experience_level": exp_level, - "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply direct on company board.", - "salary_min": None, - "salary_max": None, - "job_url": job_url, - "source": "greenhouse" - }) - except Exception as e: - print(f"[Greenhouse Warning] {company_name} failed: {e}") - - print("[ATS Direct] Fetching Lever public API boards...") - for board_slug, company_name in LEVER_BOARDS: - try: - url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json" - r = requests.get(url, headers=headers, timeout=6) - if r.status_code == 200: - jobs = r.json() - for j in jobs: - title = clean_html_text(j.get("text", "")) - job_url = j.get("hostedUrl", "") - if not title or not job_url: - continue - - categories = j.get("categories", {}) - location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA" - workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else "" - - if not is_valid_us_location(location_name, title): - continue - - is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) - - dept_text = categories.get("department", "") if isinstance(categories, dict) else "" - dept = determine_department(title, dept_text) - exp_level = determine_experience_level(title) - - description_plain = j.get("descriptionPlain", "") or j.get("description", "") - desc_clean = clean_html_text(description_plain) - - collected.append({ - "title": title, - "company": company_name, - "location": location_name if location_name != "Remote" else "Remote, USA", - "is_remote": is_remote, - "department": dept, - "experience_level": exp_level, - "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.", - "salary_min": None, - "salary_max": None, - "job_url": job_url, - "source": "lever" - }) - except Exception as e: - print(f"[Lever Warning] {company_name} failed: {e}") - - print("[ATS Direct] Fetching Ashby public API boards...") - for board_slug, company_name in ASHBY_BOARDS: - try: - url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}" - r = requests.get(url, headers=headers, timeout=6) - if r.status_code == 200: - data = r.json() - jobs = data.get("jobs", []) - for j in jobs: - title = clean_html_text(j.get("title", "")) - job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}" - if not title or not job_url: - continue - - location_name = j.get("location", "Remote, USA") or "Remote, USA" - if not is_valid_us_location(location_name, title): - continue - - is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) - dept_text = j.get("department", "") - dept = determine_department(title, dept_text) - exp_level = determine_experience_level(title) - - desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") - - salary_min = None - salary_max = None - comp = j.get("compensation") - if isinstance(comp, dict): - comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min") - comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max") - if comp_min: - try: - salary_min = float(comp_min) - except (ValueError, TypeError): - pass - if comp_max: - try: - salary_max = float(comp_max) - except (ValueError, TypeError): - pass - - collected.append({ - "title": title, - "company": company_name, - "location": location_name if location_name != "Remote" else "Remote, USA", - "is_remote": is_remote, - "department": dept, - "experience_level": exp_level, - "description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.", - "salary_min": salary_min, - "salary_max": salary_max, - "job_url": job_url, - "source": "ashby" - }) - except Exception as e: - print(f"[Ashby Warning] {company_name} failed: {e}") + print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...") + with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor: + ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS) + for res in ashby_results: + collected.extend(res) print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}") return collected diff --git a/scraper/scrapers/smart_careers_crawler.py b/scraper/scrapers/smart_careers_crawler.py index 3027ed4..bc34cc3 100644 --- a/scraper/scrapers/smart_careers_crawler.py +++ b/scraper/scrapers/smart_careers_crawler.py @@ -1,39 +1,27 @@ -from scrapers.proxy_manager import get_rotating_proxy_dict import os import requests import re from urllib.parse import urlparse, urljoin from typing import Optional, Tuple, Dict, Any, List +from scrapers.proxy_manager import get_rotating_proxy_dict from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level COMMON_CAREER_PATHS = [ "/careers", - "/career", "/jobs", - "/job", - "/work-with-us", - "/join-us", - "/join", - "/open-positions", - "/vacancies" + "/careers/jobs", + "/careers/job-search", + "/open-positions" ] -NON_US_KEYWORDS = [ - "london", "uk", "united kingdom", "england", "germany", "berlin", "munich", - "france", "paris", "canada", "toronto", "vancouver", "montreal", "india", - "bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney", - "melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands", - "emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona", - "ireland", "dublin", "switzerland", "zurich" -] +NON_US_REGEX = re.compile( + r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)', + re.IGNORECASE +) def is_valid_us_location(location_name: str, title: str = "") -> bool: - loc_lower = (location_name or "").lower() - title_lower = (title or "").lower() - for non_us in NON_US_KEYWORDS: - if non_us in loc_lower or non_us in title_lower: - return False - return True + combined = f"{location_name} {title}" + return not bool(NON_US_REGEX.search(combined)) def parse_is_us_remote(location_name: str, title: str = "") -> bool: loc_lower = (location_name or "").lower() @@ -46,7 +34,7 @@ def parse_is_us_remote(location_name: str, title: str = "") -> bool: class SmartCareersCrawler: """ Crawls company domains, finds their /careers or /jobs pages, - and automatically detects and ingests from Greenhouse, Lever, or Ashby. + and automatically detects and ingests from Greenhouse, Lever, Ashby, or Workday. Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY. """ def __init__(self, headers: Optional[Dict[str, str]] = None): @@ -54,33 +42,58 @@ class SmartCareersCrawler: "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36", "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" } - - # Configure rotating SOCKS5 proxies self.proxies = get_rotating_proxy_dict() + def _safe_get(self, url: str, timeout: int = 4, allow_redirects: bool = True) -> Optional[requests.Response]: + """ + Attempts to fetch directly or via rotating Privado proxy with strict timeout. + """ + proxies = get_rotating_proxy_dict() + if proxies: + try: + return requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects) + except Exception: + pass + + try: + return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects) + except Exception: + return None + def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]: parsed = urlparse(url) host = parsed.netloc.lower() path = parsed.path.strip("/") + # 1. Greenhouse if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host: parts = [p for p in path.split("/") if p and p != "embed"] if parts: return "greenhouse", parts[0] + # 2. Lever if "jobs.lever.co" in host: parts = [p for p in path.split("/") if p] if parts: return "lever", parts[0] + # 3. Ashby if "jobs.ashbyhq.com" in host: parts = [p for p in path.split("/") if p] if parts: return "ashby", parts[0] + # 4. Workday + if "myworkdayjobs.com" in host: + match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", url, re.IGNORECASE) + if match: + tenant, wd_num, site_slug = match.group(1), match.group(2), match.group(3) + return "workday", f"{tenant}:{wd_num}:{site_slug}" + if not html_text: return None, None + # Regex fallback on HTML body gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) if gh_match: return "greenhouse", gh_match.group(1) @@ -93,64 +106,57 @@ class SmartCareersCrawler: if ashby_match: return "ashby", ashby_match.group(1) + wd_match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE) + if wd_match: + return "workday", f"{wd_match.group(1)}:{wd_match.group(2)}:{wd_match.group(3)}" + return None, None - def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]: - """ - Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly. - """ - proxies = get_rotating_proxy_dict() - if proxies: - try: - r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects) - return r - except Exception: - # Proxy timeout or connection error; fall back to direct request immediately - pass - - try: - return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects) - except Exception: - return None - def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]: clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower() base_url = f"https://{clean_domain}" + found_careers_url = None + # Probe standard paths with fast timeout for path in COMMON_CAREER_PATHS: test_url = f"{base_url}{path}" - r = self._safe_get(test_url, timeout=5, allow_redirects=True) - if r is not None: + r = self._safe_get(test_url, timeout=3, allow_redirects=True) + if r is not None and r.status_code == 200: final_url = r.url - provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "") + if not found_careers_url: + found_careers_url = final_url + provider, slug = self.detect_ats_from_url_or_html(final_url, r.text) if provider and slug: return provider, slug, final_url - r = self._safe_get(base_url, timeout=5, allow_redirects=True) + # Check home page for outbound career / ATS links + r = self._safe_get(base_url, timeout=3, allow_redirects=True) if r is not None and r.status_code == 200: provider, slug = self.detect_ats_from_url_or_html(r.url, r.text) if provider and slug: return provider, slug, r.url - links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) + links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby|myworkdayjobs)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE) for link in links[:5]: target = link if link.startswith("http") else urljoin(base_url, link) provider, slug = self.detect_ats_from_url_or_html(target) if provider and slug: return provider, slug, target - cr = self._safe_get(target, timeout=5, allow_redirects=True) - if cr is not None: + cr = self._safe_get(target, timeout=3, allow_redirects=True) + if cr is not None and cr.status_code == 200: provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text) if provider and slug: return provider, slug, cr.url + if not found_careers_url: + found_careers_url = cr.url - return None, None, None + return None, None, found_careers_url def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true" collected = [] try: - r = self._safe_get(url, timeout=8) + r = self._safe_get(url, timeout=6) if r is None or r.status_code != 200: return collected data = r.json() @@ -196,7 +202,7 @@ class SmartCareersCrawler: url = f"https://api.lever.co/v0/postings/{slug}?mode=json" collected = [] try: - r = self._safe_get(url, timeout=8) + r = self._safe_get(url, timeout=6) if r is None or r.status_code != 200: return collected jobs = r.json() @@ -242,7 +248,7 @@ class SmartCareersCrawler: url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}" collected = [] try: - r = self._safe_get(url, timeout=8) + r = self._safe_get(url, timeout=6) if r is None or r.status_code != 200: return collected data = r.json() @@ -298,6 +304,63 @@ class SmartCareersCrawler: print(f"[Ashby Warning] {company_name} ({slug}) error: {e}") return collected + def fetch_workday_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]: + """ + Fetches public postings via Workday CXS JSON API. + slug format: tenant:wd_num:site_slug (e.g. target:wd5:targetcareers) + """ + collected = [] + try: + parts = slug.split(":") + if len(parts) != 3: + return collected + tenant, wd_num, site_slug = parts[0], parts[1], parts[2] + api_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/wday/cxs/{tenant}/{site_slug}/jobs" + payload = {"appliedFacets": {}, "limit": 20, "offset": 0, "searchText": ""} + + headers = { + "User-Agent": self.headers["User-Agent"], + "Accept": "application/json", + "Content-Type": "application/json" + } + r = requests.post(api_url, json=payload, headers=headers, timeout=6) + if r.status_code != 200: + return collected + + data = r.json() + postings = data.get("jobPostings", []) + for p in postings: + title = clean_html_text(p.get("title", "")) + ext_path = p.get("externalPath", "") + if not title or not ext_path: + continue + + location_name = p.get("locationsText", "USA") or "USA" + if not is_valid_us_location(location_name, title): + continue + + is_remote = parse_is_us_remote(location_name, title) + dept = determine_department(title, "") + exp_level = determine_experience_level(title) + canonical_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/en-US/{site_slug}{ext_path}" + + collected.append({ + "title": title, + "company": company_name, + "location": location_name, + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": f"Verified active position at {company_name}. Direct applications, specifications, and full qualifications available on official Workday portal.", + "salary_min": None, + "salary_max": None, + "job_url": canonical_url, + "source": "workday" + }) + except Exception as e: + print(f"[Workday Warning] {company_name} ({slug}) error: {e}") + return collected + def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]: import json from bs4 import BeautifulSoup @@ -406,28 +469,35 @@ class SmartCareersCrawler: anchors = soup.find_all("a", href=True) seen_links = set() + non_job_slugs = [ + "alert", "search", "faq", "culture", "event", "privacy", "terms", "story", + "benefit", "leadership", "community", "report", "sustainability", "map", "login" + ] + for a in anchors: href = a.get("href", "").strip() text = clean_html_text(a.get_text(strip=True)) - if not href or not text or len(text) < 4 or len(text) > 90: + if not href or not text or len(text) < 6 or len(text) > 85: continue - is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="]) + low_href = href.lower() + low_text = text.lower() + + is_job_href = any(k in low_href for k in ["/job/", "/position/", "/opening/", "gh_jid=", "requisition"]) if not is_job_href: continue + if any(s in low_href for s in non_job_slugs) or any(s in low_text for s in non_job_slugs): + continue + full_url = href if href.startswith("http") else urljoin(page_url, href) if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"): continue seen_links.add(full_url) - lower_text = text.lower() - if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]): - continue - dept = determine_department(text, "") exp_level = determine_experience_level(text) - is_remote = "remote" in lower_text + is_remote = "remote" in low_text jobs.append({ "title": text, @@ -436,7 +506,7 @@ class SmartCareersCrawler: "is_remote": is_remote, "department": dept, "experience_level": exp_level, - "description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}", + "description": f"Direct opening for {text} at {default_company}. Application details, requirements, and qualifications available at {full_url}", "salary_min": None, "salary_max": None, "job_url": full_url, @@ -449,7 +519,7 @@ class SmartCareersCrawler: def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]: try: - r = self._safe_get(url, timeout=8, allow_redirects=True) + r = self._safe_get(url, timeout=5, allow_redirects=True) if r is None or r.status_code != 200: return [] html_text = r.text