Comprehensive scraper overhaul: concurrent domain crawler, Workday CXS integration, parallel ATS ingestion, and word-boundary location filtering

This commit is contained in:
JobsBoard Deployer 2026-09-05 13:11:23 -04:00
parent 73ceb260d5
commit 2356a68411
4 changed files with 377 additions and 280 deletions

View file

@ -1,6 +1,7 @@
import os
import time
import sys
import concurrent.futures
from scrapers.ats_ingestion import run_ats_direct_ingestion
from scrapers.art_and_design_ingestion import run_art_and_design_ingestion
from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion
@ -57,30 +58,37 @@ DOMAINS_TO_PROBE = [
]
def run_smart_crawler_scrapes():
print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...")
print("[Smart Crawler] Universal Web Crawler scanning domains concurrently...")
crawler = SmartCareersCrawler()
discovered_jobs = []
for domain, company_name in DOMAINS_TO_PROBE:
def probe_and_collect(item):
domain, company_name = item
jobs = []
try:
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
if provider and slug:
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
if provider == "greenhouse":
jobs = crawler.fetch_greenhouse_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "lever":
jobs = crawler.fetch_lever_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "ashby":
jobs = crawler.fetch_ashby_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "workday":
jobs = crawler.fetch_workday_board(slug, company_name)
elif careers_url:
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
jobs = crawler.scrape_native_career_page(careers_url, company_name)
discovered_jobs.extend(jobs)
except Exception as e:
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
return jobs
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
results = executor.map(probe_and_collect, DOMAINS_TO_PROBE)
for job_batch in results:
if job_batch:
discovered_jobs.extend(job_batch)
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
return discovered_jobs
@ -92,7 +100,7 @@ def execute_all_scrapes():
all_jobs = []
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby)
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby/Workday)
try:
crawler_jobs = run_smart_crawler_scrapes()
all_jobs.extend(crawler_jobs)
@ -147,7 +155,7 @@ def execute_all_scrapes():
if all_jobs:
count = upsert_jobs(all_jobs)
print(f"✅ Successfully written/updated {count} jobs into database.")
print(f"[DB Ingestion] Successfully written/updated {count} jobs into database.")
else:
print("No job records were collected in this run.")

View file

@ -1,6 +1,7 @@
import requests
import html
import re
import concurrent.futures
from typing import List, Dict, Any
from scrapers.ats_ingestion import clean_html_text, determine_experience_level
@ -60,7 +61,7 @@ CURATED_ART_DESIGN_JOBS = [
"is_remote": True,
"department": "Art & Design",
"experience_level": "Mid-Level",
"description": "Create high-fidelity 3D environment assets, digital matte paintings, lighting setups, and visual concept art for Unreal Engine real-time virtual production.",
"description": "Create photorealistic 3D environments, textures, lighting, and concept digital paintings using Unreal Engine 5, Blender, Maya, and Substance Painter.",
"salary_min": 105000,
"salary_max": 145000,
"job_url": "https://www.epicgames.com/careers/3d-environment-artist-remote",
@ -143,54 +144,59 @@ def is_art_and_design_job(title: str, dept_text: str = "") -> bool:
]
return any(k in combined for k in creative_keywords)
def fetch_art_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
if is_art_and_design_job(title, dept_text):
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj)
is_remote = "remote" in location_name.lower() or "remote" in title.lower()
exp_level = determine_experience_level(title)
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
results.append({
"title": title,
"company": company_name,
"location": location_name,
"is_remote": is_remote,
"department": "Art & Design",
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Art & Design position at {company_name}.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Art & Design Warning] {company_name} failed: {e}")
return results
def run_art_and_design_ingestion() -> List[Dict[Any, Any]]:
collected = []
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
print("[Art & Design Ingestion] Ingesting dedicated Art, Creative, Gaming & Animation company boards...")
# 1. Fetch from Greenhouse creative boards
for board_slug, company_name in ART_DESIGN_GREENHOUSE_BOARDS:
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
print(f"[Art & Design Greenhouse] {company_name}: {len(jobs)} total board postings.")
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
if is_art_and_design_job(title, dept_text):
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj)
is_remote = "remote" in location_name.lower() or "remote" in title.lower()
exp_level = determine_experience_level(title)
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
collected.append({
"title": title,
"company": company_name,
"location": location_name,
"is_remote": is_remote,
"department": "Art & Design",
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Art & Design position at {company_name}.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Art & Design Warning] {company_name} failed: {e}")
print("[Art & Design Ingestion] Concurrently fetching dedicated Art & Creative boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
results = executor.map(lambda item: fetch_art_greenhouse_board(item, headers), ART_DESIGN_GREENHOUSE_BOARDS)
for batch in results:
collected.extend(batch)
# 2. Add curated CT & Remote Art/Design/Media jobs
collected.extend(CURATED_ART_DESIGN_JOBS)

View file

@ -1,6 +1,7 @@
import requests
import html
import re
import concurrent.futures
from typing import List, Dict, Any
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
@ -125,24 +126,14 @@ ASHBY_BOARDS = [
("cohere", "Cohere")
]
NON_US_KEYWORDS = [
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
"ireland", "dublin", "switzerland", "zurich"
]
NON_US_REGEX = re.compile(
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
re.IGNORECASE
)
def is_valid_us_location(location_name: str, title: str) -> bool:
loc_lower = location_name.lower()
title_lower = title.lower()
for non_us in NON_US_KEYWORDS:
if non_us in loc_lower or non_us in title_lower:
return False
return True
def is_valid_us_location(location_name: str, title: str = "") -> bool:
combined = f"{location_name} {title}"
return not bool(NON_US_REGEX.search(combined))
def parse_is_us_remote(location_name: str, title: str) -> bool:
loc_lower = location_name.lower()
@ -212,159 +203,181 @@ def determine_experience_level(title: str) -> str:
return "Entry Level"
return "Mid-Level"
def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} failed: {e}")
return results
def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
jobs = r.json()
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} failed: {e}")
return results
def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
board_slug, company_name = item
results = []
try:
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
dept_text = j.get("department", "")
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
results.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} failed: {e}")
return results
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
collected = []
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
print("[ATS Direct] Fetching 90+ Greenhouse public API boards (US Remote Filtered)...")
for board_slug, company_name in GREENHOUSE_BOARDS:
try:
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS)
for res in gh_results:
collected.extend(res)
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS)
for res in lever_results:
collected.extend(res)
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
content_raw = j.get("content", "") or ""
desc_clean = clean_html_text(content_raw)
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply direct on company board.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} failed: {e}")
print("[ATS Direct] Fetching Lever public API boards...")
for board_slug, company_name in LEVER_BOARDS:
try:
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
jobs = r.json()
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} failed: {e}")
print("[ATS Direct] Fetching Ashby public API boards...")
for board_slug, company_name in ASHBY_BOARDS:
try:
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
dept_text = j.get("department", "")
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} failed: {e}")
print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...")
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS)
for res in ashby_results:
collected.extend(res)
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
return collected

View file

@ -1,39 +1,27 @@
from scrapers.proxy_manager import get_rotating_proxy_dict
import os
import requests
import re
from urllib.parse import urlparse, urljoin
from typing import Optional, Tuple, Dict, Any, List
from scrapers.proxy_manager import get_rotating_proxy_dict
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level
COMMON_CAREER_PATHS = [
"/careers",
"/career",
"/jobs",
"/job",
"/work-with-us",
"/join-us",
"/join",
"/open-positions",
"/vacancies"
"/careers/jobs",
"/careers/job-search",
"/open-positions"
]
NON_US_KEYWORDS = [
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
"ireland", "dublin", "switzerland", "zurich"
]
NON_US_REGEX = re.compile(
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
re.IGNORECASE
)
def is_valid_us_location(location_name: str, title: str = "") -> bool:
loc_lower = (location_name or "").lower()
title_lower = (title or "").lower()
for non_us in NON_US_KEYWORDS:
if non_us in loc_lower or non_us in title_lower:
return False
return True
combined = f"{location_name} {title}"
return not bool(NON_US_REGEX.search(combined))
def parse_is_us_remote(location_name: str, title: str = "") -> bool:
loc_lower = (location_name or "").lower()
@ -46,7 +34,7 @@ def parse_is_us_remote(location_name: str, title: str = "") -> bool:
class SmartCareersCrawler:
"""
Crawls company domains, finds their /careers or /jobs pages,
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
and automatically detects and ingests from Greenhouse, Lever, Ashby, or Workday.
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
"""
def __init__(self, headers: Optional[Dict[str, str]] = None):
@ -54,33 +42,58 @@ class SmartCareersCrawler:
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
}
# Configure rotating SOCKS5 proxies
self.proxies = get_rotating_proxy_dict()
def _safe_get(self, url: str, timeout: int = 4, allow_redirects: bool = True) -> Optional[requests.Response]:
"""
Attempts to fetch directly or via rotating Privado proxy with strict timeout.
"""
proxies = get_rotating_proxy_dict()
if proxies:
try:
return requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
except Exception:
pass
try:
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
except Exception:
return None
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
parsed = urlparse(url)
host = parsed.netloc.lower()
path = parsed.path.strip("/")
# 1. Greenhouse
if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
parts = [p for p in path.split("/") if p and p != "embed"]
if parts:
return "greenhouse", parts[0]
# 2. Lever
if "jobs.lever.co" in host:
parts = [p for p in path.split("/") if p]
if parts:
return "lever", parts[0]
# 3. Ashby
if "jobs.ashbyhq.com" in host:
parts = [p for p in path.split("/") if p]
if parts:
return "ashby", parts[0]
# 4. Workday
if "myworkdayjobs.com" in host:
match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", url, re.IGNORECASE)
if match:
tenant, wd_num, site_slug = match.group(1), match.group(2), match.group(3)
return "workday", f"{tenant}:{wd_num}:{site_slug}"
if not html_text:
return None, None
# Regex fallback on HTML body
gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if gh_match:
return "greenhouse", gh_match.group(1)
@ -93,64 +106,57 @@ class SmartCareersCrawler:
if ashby_match:
return "ashby", ashby_match.group(1)
wd_match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if wd_match:
return "workday", f"{wd_match.group(1)}:{wd_match.group(2)}:{wd_match.group(3)}"
return None, None
def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]:
"""
Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly.
"""
proxies = get_rotating_proxy_dict()
if proxies:
try:
r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
return r
except Exception:
# Proxy timeout or connection error; fall back to direct request immediately
pass
try:
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
except Exception:
return None
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
base_url = f"https://{clean_domain}"
found_careers_url = None
# Probe standard paths with fast timeout
for path in COMMON_CAREER_PATHS:
test_url = f"{base_url}{path}"
r = self._safe_get(test_url, timeout=5, allow_redirects=True)
if r is not None:
r = self._safe_get(test_url, timeout=3, allow_redirects=True)
if r is not None and r.status_code == 200:
final_url = r.url
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
if not found_careers_url:
found_careers_url = final_url
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text)
if provider and slug:
return provider, slug, final_url
r = self._safe_get(base_url, timeout=5, allow_redirects=True)
# Check home page for outbound career / ATS links
r = self._safe_get(base_url, timeout=3, allow_redirects=True)
if r is not None and r.status_code == 200:
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
if provider and slug:
return provider, slug, r.url
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby|myworkdayjobs)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
for link in links[:5]:
target = link if link.startswith("http") else urljoin(base_url, link)
provider, slug = self.detect_ats_from_url_or_html(target)
if provider and slug:
return provider, slug, target
cr = self._safe_get(target, timeout=5, allow_redirects=True)
if cr is not None:
cr = self._safe_get(target, timeout=3, allow_redirects=True)
if cr is not None and cr.status_code == 200:
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
if provider and slug:
return provider, slug, cr.url
if not found_careers_url:
found_careers_url = cr.url
return None, None, None
return None, None, found_careers_url
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
collected = []
try:
r = self._safe_get(url, timeout=8)
r = self._safe_get(url, timeout=6)
if r is None or r.status_code != 200:
return collected
data = r.json()
@ -196,7 +202,7 @@ class SmartCareersCrawler:
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
collected = []
try:
r = self._safe_get(url, timeout=8)
r = self._safe_get(url, timeout=6)
if r is None or r.status_code != 200:
return collected
jobs = r.json()
@ -242,7 +248,7 @@ class SmartCareersCrawler:
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
collected = []
try:
r = self._safe_get(url, timeout=8)
r = self._safe_get(url, timeout=6)
if r is None or r.status_code != 200:
return collected
data = r.json()
@ -298,6 +304,63 @@ class SmartCareersCrawler:
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
return collected
def fetch_workday_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
"""
Fetches public postings via Workday CXS JSON API.
slug format: tenant:wd_num:site_slug (e.g. target:wd5:targetcareers)
"""
collected = []
try:
parts = slug.split(":")
if len(parts) != 3:
return collected
tenant, wd_num, site_slug = parts[0], parts[1], parts[2]
api_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/wday/cxs/{tenant}/{site_slug}/jobs"
payload = {"appliedFacets": {}, "limit": 20, "offset": 0, "searchText": ""}
headers = {
"User-Agent": self.headers["User-Agent"],
"Accept": "application/json",
"Content-Type": "application/json"
}
r = requests.post(api_url, json=payload, headers=headers, timeout=6)
if r.status_code != 200:
return collected
data = r.json()
postings = data.get("jobPostings", [])
for p in postings:
title = clean_html_text(p.get("title", ""))
ext_path = p.get("externalPath", "")
if not title or not ext_path:
continue
location_name = p.get("locationsText", "USA") or "USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
dept = determine_department(title, "")
exp_level = determine_experience_level(title)
canonical_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/en-US/{site_slug}{ext_path}"
collected.append({
"title": title,
"company": company_name,
"location": location_name,
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": f"Verified active position at {company_name}. Direct applications, specifications, and full qualifications available on official Workday portal.",
"salary_min": None,
"salary_max": None,
"job_url": canonical_url,
"source": "workday"
})
except Exception as e:
print(f"[Workday Warning] {company_name} ({slug}) error: {e}")
return collected
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
import json
from bs4 import BeautifulSoup
@ -406,28 +469,35 @@ class SmartCareersCrawler:
anchors = soup.find_all("a", href=True)
seen_links = set()
non_job_slugs = [
"alert", "search", "faq", "culture", "event", "privacy", "terms", "story",
"benefit", "leadership", "community", "report", "sustainability", "map", "login"
]
for a in anchors:
href = a.get("href", "").strip()
text = clean_html_text(a.get_text(strip=True))
if not href or not text or len(text) < 4 or len(text) > 90:
if not href or not text or len(text) < 6 or len(text) > 85:
continue
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
low_href = href.lower()
low_text = text.lower()
is_job_href = any(k in low_href for k in ["/job/", "/position/", "/opening/", "gh_jid=", "requisition"])
if not is_job_href:
continue
if any(s in low_href for s in non_job_slugs) or any(s in low_text for s in non_job_slugs):
continue
full_url = href if href.startswith("http") else urljoin(page_url, href)
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
continue
seen_links.add(full_url)
lower_text = text.lower()
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
continue
dept = determine_department(text, "")
exp_level = determine_experience_level(text)
is_remote = "remote" in lower_text
is_remote = "remote" in low_text
jobs.append({
"title": text,
@ -436,7 +506,7 @@ class SmartCareersCrawler:
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
"description": f"Direct opening for {text} at {default_company}. Application details, requirements, and qualifications available at {full_url}",
"salary_min": None,
"salary_max": None,
"job_url": full_url,
@ -449,7 +519,7 @@ class SmartCareersCrawler:
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
try:
r = self._safe_get(url, timeout=8, allow_redirects=True)
r = self._safe_get(url, timeout=5, allow_redirects=True)
if r is None or r.status_code != 200:
return []
html_text = r.text