Comprehensive scraper overhaul: concurrent domain crawler, Workday CXS integration, parallel ATS ingestion, and word-boundary location filtering
This commit is contained in:
parent
73ceb260d5
commit
2356a68411
4 changed files with 377 additions and 280 deletions
|
|
@ -1,6 +1,7 @@
|
|||
import os
|
||||
import time
|
||||
import sys
|
||||
import concurrent.futures
|
||||
from scrapers.ats_ingestion import run_ats_direct_ingestion
|
||||
from scrapers.art_and_design_ingestion import run_art_and_design_ingestion
|
||||
from scrapers.expanded_categories_ingestion import run_expanded_categories_ingestion
|
||||
|
|
@ -57,30 +58,37 @@ DOMAINS_TO_PROBE = [
|
|||
]
|
||||
|
||||
def run_smart_crawler_scrapes():
|
||||
print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...")
|
||||
print("[Smart Crawler] Universal Web Crawler scanning domains concurrently...")
|
||||
crawler = SmartCareersCrawler()
|
||||
discovered_jobs = []
|
||||
|
||||
for domain, company_name in DOMAINS_TO_PROBE:
|
||||
def probe_and_collect(item):
|
||||
domain, company_name = item
|
||||
jobs = []
|
||||
try:
|
||||
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
|
||||
if provider and slug:
|
||||
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
|
||||
if provider == "greenhouse":
|
||||
jobs = crawler.fetch_greenhouse_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
elif provider == "lever":
|
||||
jobs = crawler.fetch_lever_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
elif provider == "ashby":
|
||||
jobs = crawler.fetch_ashby_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
elif provider == "workday":
|
||||
jobs = crawler.fetch_workday_board(slug, company_name)
|
||||
elif careers_url:
|
||||
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
|
||||
jobs = crawler.scrape_native_career_page(careers_url, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
except Exception as e:
|
||||
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
|
||||
return jobs
|
||||
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
|
||||
results = executor.map(probe_and_collect, DOMAINS_TO_PROBE)
|
||||
for job_batch in results:
|
||||
if job_batch:
|
||||
discovered_jobs.extend(job_batch)
|
||||
|
||||
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
|
||||
return discovered_jobs
|
||||
|
|
@ -92,7 +100,7 @@ def execute_all_scrapes():
|
|||
|
||||
all_jobs = []
|
||||
|
||||
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby)
|
||||
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby/Workday)
|
||||
try:
|
||||
crawler_jobs = run_smart_crawler_scrapes()
|
||||
all_jobs.extend(crawler_jobs)
|
||||
|
|
@ -147,7 +155,7 @@ def execute_all_scrapes():
|
|||
|
||||
if all_jobs:
|
||||
count = upsert_jobs(all_jobs)
|
||||
print(f"✅ Successfully written/updated {count} jobs into database.")
|
||||
print(f"[DB Ingestion] Successfully written/updated {count} jobs into database.")
|
||||
else:
|
||||
print("No job records were collected in this run.")
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import requests
|
||||
import html
|
||||
import re
|
||||
import concurrent.futures
|
||||
from typing import List, Dict, Any
|
||||
from scrapers.ats_ingestion import clean_html_text, determine_experience_level
|
||||
|
||||
|
|
@ -60,7 +61,7 @@ CURATED_ART_DESIGN_JOBS = [
|
|||
"is_remote": True,
|
||||
"department": "Art & Design",
|
||||
"experience_level": "Mid-Level",
|
||||
"description": "Create high-fidelity 3D environment assets, digital matte paintings, lighting setups, and visual concept art for Unreal Engine real-time virtual production.",
|
||||
"description": "Create photorealistic 3D environments, textures, lighting, and concept digital paintings using Unreal Engine 5, Blender, Maya, and Substance Painter.",
|
||||
"salary_min": 105000,
|
||||
"salary_max": 145000,
|
||||
"job_url": "https://www.epicgames.com/careers/3d-environment-artist-remote",
|
||||
|
|
@ -143,54 +144,59 @@ def is_art_and_design_job(title: str, dept_text: str = "") -> bool:
|
|||
]
|
||||
return any(k in combined for k in creative_keywords)
|
||||
|
||||
def fetch_art_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||||
board_slug, company_name = item
|
||||
results = []
|
||||
try:
|
||||
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("absolute_url", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
departments = j.get("departments", [])
|
||||
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||||
|
||||
if is_art_and_design_job(title, dept_text):
|
||||
location_obj = j.get("location", {})
|
||||
location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj)
|
||||
is_remote = "remote" in location_name.lower() or "remote" in title.lower()
|
||||
|
||||
exp_level = determine_experience_level(title)
|
||||
content_raw = j.get("content", "") or ""
|
||||
desc_clean = clean_html_text(content_raw)
|
||||
|
||||
results.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name,
|
||||
"is_remote": is_remote,
|
||||
"department": "Art & Design",
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Art & Design position at {company_name}.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "greenhouse"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Art & Design Warning] {company_name} failed: {e}")
|
||||
return results
|
||||
|
||||
def run_art_and_design_ingestion() -> List[Dict[Any, Any]]:
|
||||
collected = []
|
||||
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
|
||||
|
||||
print("[Art & Design Ingestion] Ingesting dedicated Art, Creative, Gaming & Animation company boards...")
|
||||
|
||||
# 1. Fetch from Greenhouse creative boards
|
||||
for board_slug, company_name in ART_DESIGN_GREENHOUSE_BOARDS:
|
||||
try:
|
||||
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
print(f"[Art & Design Greenhouse] {company_name}: {len(jobs)} total board postings.")
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("absolute_url", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
departments = j.get("departments", [])
|
||||
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||||
|
||||
if is_art_and_design_job(title, dept_text):
|
||||
location_obj = j.get("location", {})
|
||||
location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj)
|
||||
is_remote = "remote" in location_name.lower() or "remote" in title.lower()
|
||||
|
||||
exp_level = determine_experience_level(title)
|
||||
content_raw = j.get("content", "") or ""
|
||||
desc_clean = clean_html_text(content_raw)
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name,
|
||||
"is_remote": is_remote,
|
||||
"department": "Art & Design",
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Art & Design position at {company_name}.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "greenhouse"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Art & Design Warning] {company_name} failed: {e}")
|
||||
print("[Art & Design Ingestion] Concurrently fetching dedicated Art & Creative boards...")
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=8) as executor:
|
||||
results = executor.map(lambda item: fetch_art_greenhouse_board(item, headers), ART_DESIGN_GREENHOUSE_BOARDS)
|
||||
for batch in results:
|
||||
collected.extend(batch)
|
||||
|
||||
# 2. Add curated CT & Remote Art/Design/Media jobs
|
||||
collected.extend(CURATED_ART_DESIGN_JOBS)
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import requests
|
||||
import html
|
||||
import re
|
||||
import concurrent.futures
|
||||
from typing import List, Dict, Any
|
||||
|
||||
# Expanded 90+ top company Greenhouse & Lever boards across all sectors
|
||||
|
|
@ -125,24 +126,14 @@ ASHBY_BOARDS = [
|
|||
("cohere", "Cohere")
|
||||
]
|
||||
|
||||
NON_US_KEYWORDS = [
|
||||
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
|
||||
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
|
||||
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
|
||||
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
|
||||
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
|
||||
"ireland", "dublin", "switzerland", "zurich"
|
||||
]
|
||||
NON_US_REGEX = re.compile(
|
||||
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
|
||||
re.IGNORECASE
|
||||
)
|
||||
|
||||
def is_valid_us_location(location_name: str, title: str) -> bool:
|
||||
loc_lower = location_name.lower()
|
||||
title_lower = title.lower()
|
||||
|
||||
for non_us in NON_US_KEYWORDS:
|
||||
if non_us in loc_lower or non_us in title_lower:
|
||||
return False
|
||||
|
||||
return True
|
||||
def is_valid_us_location(location_name: str, title: str = "") -> bool:
|
||||
combined = f"{location_name} {title}"
|
||||
return not bool(NON_US_REGEX.search(combined))
|
||||
|
||||
def parse_is_us_remote(location_name: str, title: str) -> bool:
|
||||
loc_lower = location_name.lower()
|
||||
|
|
@ -212,159 +203,181 @@ def determine_experience_level(title: str) -> str:
|
|||
return "Entry Level"
|
||||
return "Mid-Level"
|
||||
|
||||
def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||||
board_slug, company_name = item
|
||||
results = []
|
||||
try:
|
||||
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("absolute_url", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
location_obj = j.get("location", {})
|
||||
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(location_name, title)
|
||||
departments = j.get("departments", [])
|
||||
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||||
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
content_raw = j.get("content", "") or ""
|
||||
desc_clean = clean_html_text(content_raw)
|
||||
|
||||
results.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "greenhouse"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Greenhouse Warning] {company_name} failed: {e}")
|
||||
return results
|
||||
|
||||
def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||||
board_slug, company_name = item
|
||||
results = []
|
||||
try:
|
||||
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
jobs = r.json()
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("text", ""))
|
||||
job_url = j.get("hostedUrl", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
categories = j.get("categories", {})
|
||||
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
||||
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
|
||||
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
||||
desc_clean = clean_html_text(description_plain)
|
||||
|
||||
results.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "lever"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Lever Warning] {company_name} failed: {e}")
|
||||
return results
|
||||
|
||||
def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]]:
|
||||
board_slug, company_name = item
|
||||
results = []
|
||||
try:
|
||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
location_name = j.get("location", "Remote, USA") or "Remote, USA"
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
|
||||
dept_text = j.get("department", "")
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
|
||||
|
||||
salary_min = None
|
||||
salary_max = None
|
||||
comp = j.get("compensation")
|
||||
if isinstance(comp, dict):
|
||||
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
|
||||
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
|
||||
if comp_min:
|
||||
try:
|
||||
salary_min = float(comp_min)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if comp_max:
|
||||
try:
|
||||
salary_max = float(comp_max)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
results.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
|
||||
"salary_min": salary_min,
|
||||
"salary_max": salary_max,
|
||||
"job_url": job_url,
|
||||
"source": "ashby"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Ashby Warning] {company_name} failed: {e}")
|
||||
return results
|
||||
|
||||
def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
|
||||
collected = []
|
||||
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
|
||||
|
||||
print("[ATS Direct] Fetching 90+ Greenhouse public API boards (US Remote Filtered)...")
|
||||
for board_slug, company_name in GREENHOUSE_BOARDS:
|
||||
try:
|
||||
url = f"https://boards-api.greenhouse.io/v1/boards/{board_slug}/jobs?content=true"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("absolute_url", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
print(f"[ATS Direct] Concurrently fetching {len(GREENHOUSE_BOARDS)} Greenhouse boards...")
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
|
||||
gh_results = executor.map(lambda item: fetch_single_greenhouse_board(item, headers), GREENHOUSE_BOARDS)
|
||||
for res in gh_results:
|
||||
collected.extend(res)
|
||||
|
||||
location_obj = j.get("location", {})
|
||||
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
||||
print(f"[ATS Direct] Concurrently fetching {len(LEVER_BOARDS)} Lever boards...")
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
|
||||
lever_results = executor.map(lambda item: fetch_single_lever_board(item, headers), LEVER_BOARDS)
|
||||
for res in lever_results:
|
||||
collected.extend(res)
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(location_name, title)
|
||||
departments = j.get("departments", [])
|
||||
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
||||
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
content_raw = j.get("content", "") or ""
|
||||
desc_clean = clean_html_text(content_raw)
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply direct on company board.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "greenhouse"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Greenhouse Warning] {company_name} failed: {e}")
|
||||
|
||||
print("[ATS Direct] Fetching Lever public API boards...")
|
||||
for board_slug, company_name in LEVER_BOARDS:
|
||||
try:
|
||||
url = f"https://api.lever.co/v0/postings/{board_slug}?mode=json"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
jobs = r.json()
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("text", ""))
|
||||
job_url = j.get("hostedUrl", "")
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
categories = j.get("categories", {})
|
||||
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
||||
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
|
||||
|
||||
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
||||
desc_clean = clean_html_text(description_plain)
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": job_url,
|
||||
"source": "lever"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Lever Warning] {company_name} failed: {e}")
|
||||
|
||||
print("[ATS Direct] Fetching Ashby public API boards...")
|
||||
for board_slug, company_name in ASHBY_BOARDS:
|
||||
try:
|
||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
location_name = j.get("location", "Remote, USA") or "Remote, USA"
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
|
||||
dept_text = j.get("department", "")
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
|
||||
|
||||
salary_min = None
|
||||
salary_max = None
|
||||
comp = j.get("compensation")
|
||||
if isinstance(comp, dict):
|
||||
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
|
||||
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
|
||||
if comp_min:
|
||||
try:
|
||||
salary_min = float(comp_min)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if comp_max:
|
||||
try:
|
||||
salary_max = float(comp_max)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
|
||||
"salary_min": salary_min,
|
||||
"salary_max": salary_max,
|
||||
"job_url": job_url,
|
||||
"source": "ashby"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Ashby Warning] {company_name} failed: {e}")
|
||||
print(f"[ATS Direct] Concurrently fetching {len(ASHBY_BOARDS)} Ashby boards...")
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=10) as executor:
|
||||
ashby_results = executor.map(lambda item: fetch_single_ashby_board(item, headers), ASHBY_BOARDS)
|
||||
for res in ashby_results:
|
||||
collected.extend(res)
|
||||
|
||||
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
|
||||
return collected
|
||||
|
|
|
|||
|
|
@ -1,39 +1,27 @@
|
|||
from scrapers.proxy_manager import get_rotating_proxy_dict
|
||||
import os
|
||||
import requests
|
||||
import re
|
||||
from urllib.parse import urlparse, urljoin
|
||||
from typing import Optional, Tuple, Dict, Any, List
|
||||
from scrapers.proxy_manager import get_rotating_proxy_dict
|
||||
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level
|
||||
|
||||
COMMON_CAREER_PATHS = [
|
||||
"/careers",
|
||||
"/career",
|
||||
"/jobs",
|
||||
"/job",
|
||||
"/work-with-us",
|
||||
"/join-us",
|
||||
"/join",
|
||||
"/open-positions",
|
||||
"/vacancies"
|
||||
"/careers/jobs",
|
||||
"/careers/job-search",
|
||||
"/open-positions"
|
||||
]
|
||||
|
||||
NON_US_KEYWORDS = [
|
||||
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
|
||||
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
|
||||
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
|
||||
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
|
||||
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
|
||||
"ireland", "dublin", "switzerland", "zurich"
|
||||
]
|
||||
NON_US_REGEX = re.compile(
|
||||
r'(?:london|uk|united kingdom|england|germany|berlin|munich|france|paris|canada|toronto|vancouver|montreal|india|bengaluru|bangalore|delhi|singapore|australia|sydney|melbourne|tokyo|japan|brazil|sao paulo|amsterdam|netherlands|emea|apac|latam|poland|warsaw|romania|spain|madrid|barcelona|ireland|dublin|switzerland|zurich)',
|
||||
re.IGNORECASE
|
||||
)
|
||||
|
||||
def is_valid_us_location(location_name: str, title: str = "") -> bool:
|
||||
loc_lower = (location_name or "").lower()
|
||||
title_lower = (title or "").lower()
|
||||
for non_us in NON_US_KEYWORDS:
|
||||
if non_us in loc_lower or non_us in title_lower:
|
||||
return False
|
||||
return True
|
||||
combined = f"{location_name} {title}"
|
||||
return not bool(NON_US_REGEX.search(combined))
|
||||
|
||||
def parse_is_us_remote(location_name: str, title: str = "") -> bool:
|
||||
loc_lower = (location_name or "").lower()
|
||||
|
|
@ -46,7 +34,7 @@ def parse_is_us_remote(location_name: str, title: str = "") -> bool:
|
|||
class SmartCareersCrawler:
|
||||
"""
|
||||
Crawls company domains, finds their /careers or /jobs pages,
|
||||
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
|
||||
and automatically detects and ingests from Greenhouse, Lever, Ashby, or Workday.
|
||||
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
|
||||
"""
|
||||
def __init__(self, headers: Optional[Dict[str, str]] = None):
|
||||
|
|
@ -54,33 +42,58 @@ class SmartCareersCrawler:
|
|||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
||||
}
|
||||
|
||||
# Configure rotating SOCKS5 proxies
|
||||
self.proxies = get_rotating_proxy_dict()
|
||||
|
||||
def _safe_get(self, url: str, timeout: int = 4, allow_redirects: bool = True) -> Optional[requests.Response]:
|
||||
"""
|
||||
Attempts to fetch directly or via rotating Privado proxy with strict timeout.
|
||||
"""
|
||||
proxies = get_rotating_proxy_dict()
|
||||
if proxies:
|
||||
try:
|
||||
return requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
try:
|
||||
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
||||
parsed = urlparse(url)
|
||||
host = parsed.netloc.lower()
|
||||
path = parsed.path.strip("/")
|
||||
|
||||
# 1. Greenhouse
|
||||
if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
|
||||
parts = [p for p in path.split("/") if p and p != "embed"]
|
||||
if parts:
|
||||
return "greenhouse", parts[0]
|
||||
|
||||
# 2. Lever
|
||||
if "jobs.lever.co" in host:
|
||||
parts = [p for p in path.split("/") if p]
|
||||
if parts:
|
||||
return "lever", parts[0]
|
||||
|
||||
# 3. Ashby
|
||||
if "jobs.ashbyhq.com" in host:
|
||||
parts = [p for p in path.split("/") if p]
|
||||
if parts:
|
||||
return "ashby", parts[0]
|
||||
|
||||
# 4. Workday
|
||||
if "myworkdayjobs.com" in host:
|
||||
match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", url, re.IGNORECASE)
|
||||
if match:
|
||||
tenant, wd_num, site_slug = match.group(1), match.group(2), match.group(3)
|
||||
return "workday", f"{tenant}:{wd_num}:{site_slug}"
|
||||
|
||||
if not html_text:
|
||||
return None, None
|
||||
|
||||
# Regex fallback on HTML body
|
||||
gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
||||
if gh_match:
|
||||
return "greenhouse", gh_match.group(1)
|
||||
|
|
@ -93,64 +106,57 @@ class SmartCareersCrawler:
|
|||
if ashby_match:
|
||||
return "ashby", ashby_match.group(1)
|
||||
|
||||
wd_match = re.search(r"([a-zA-Z0-9_\-]+)\.(wd[0-9]+)\.myworkdayjobs\.com\/(?:en-US\/)?([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
||||
if wd_match:
|
||||
return "workday", f"{wd_match.group(1)}:{wd_match.group(2)}:{wd_match.group(3)}"
|
||||
|
||||
return None, None
|
||||
|
||||
def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]:
|
||||
"""
|
||||
Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly.
|
||||
"""
|
||||
proxies = get_rotating_proxy_dict()
|
||||
if proxies:
|
||||
try:
|
||||
r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
|
||||
return r
|
||||
except Exception:
|
||||
# Proxy timeout or connection error; fall back to direct request immediately
|
||||
pass
|
||||
|
||||
try:
|
||||
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
||||
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
|
||||
base_url = f"https://{clean_domain}"
|
||||
found_careers_url = None
|
||||
|
||||
# Probe standard paths with fast timeout
|
||||
for path in COMMON_CAREER_PATHS:
|
||||
test_url = f"{base_url}{path}"
|
||||
r = self._safe_get(test_url, timeout=5, allow_redirects=True)
|
||||
if r is not None:
|
||||
r = self._safe_get(test_url, timeout=3, allow_redirects=True)
|
||||
if r is not None and r.status_code == 200:
|
||||
final_url = r.url
|
||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
||||
if not found_careers_url:
|
||||
found_careers_url = final_url
|
||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text)
|
||||
if provider and slug:
|
||||
return provider, slug, final_url
|
||||
|
||||
r = self._safe_get(base_url, timeout=5, allow_redirects=True)
|
||||
# Check home page for outbound career / ATS links
|
||||
r = self._safe_get(base_url, timeout=3, allow_redirects=True)
|
||||
if r is not None and r.status_code == 200:
|
||||
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
||||
if provider and slug:
|
||||
return provider, slug, r.url
|
||||
|
||||
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
|
||||
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby|myworkdayjobs)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
|
||||
for link in links[:5]:
|
||||
target = link if link.startswith("http") else urljoin(base_url, link)
|
||||
provider, slug = self.detect_ats_from_url_or_html(target)
|
||||
if provider and slug:
|
||||
return provider, slug, target
|
||||
cr = self._safe_get(target, timeout=5, allow_redirects=True)
|
||||
if cr is not None:
|
||||
cr = self._safe_get(target, timeout=3, allow_redirects=True)
|
||||
if cr is not None and cr.status_code == 200:
|
||||
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
||||
if provider and slug:
|
||||
return provider, slug, cr.url
|
||||
if not found_careers_url:
|
||||
found_careers_url = cr.url
|
||||
|
||||
return None, None, None
|
||||
return None, None, found_careers_url
|
||||
|
||||
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
||||
collected = []
|
||||
try:
|
||||
r = self._safe_get(url, timeout=8)
|
||||
r = self._safe_get(url, timeout=6)
|
||||
if r is None or r.status_code != 200:
|
||||
return collected
|
||||
data = r.json()
|
||||
|
|
@ -196,7 +202,7 @@ class SmartCareersCrawler:
|
|||
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
||||
collected = []
|
||||
try:
|
||||
r = self._safe_get(url, timeout=8)
|
||||
r = self._safe_get(url, timeout=6)
|
||||
if r is None or r.status_code != 200:
|
||||
return collected
|
||||
jobs = r.json()
|
||||
|
|
@ -242,7 +248,7 @@ class SmartCareersCrawler:
|
|||
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
||||
collected = []
|
||||
try:
|
||||
r = self._safe_get(url, timeout=8)
|
||||
r = self._safe_get(url, timeout=6)
|
||||
if r is None or r.status_code != 200:
|
||||
return collected
|
||||
data = r.json()
|
||||
|
|
@ -298,6 +304,63 @@ class SmartCareersCrawler:
|
|||
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
|
||||
return collected
|
||||
|
||||
def fetch_workday_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||
"""
|
||||
Fetches public postings via Workday CXS JSON API.
|
||||
slug format: tenant:wd_num:site_slug (e.g. target:wd5:targetcareers)
|
||||
"""
|
||||
collected = []
|
||||
try:
|
||||
parts = slug.split(":")
|
||||
if len(parts) != 3:
|
||||
return collected
|
||||
tenant, wd_num, site_slug = parts[0], parts[1], parts[2]
|
||||
api_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/wday/cxs/{tenant}/{site_slug}/jobs"
|
||||
payload = {"appliedFacets": {}, "limit": 20, "offset": 0, "searchText": ""}
|
||||
|
||||
headers = {
|
||||
"User-Agent": self.headers["User-Agent"],
|
||||
"Accept": "application/json",
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
r = requests.post(api_url, json=payload, headers=headers, timeout=6)
|
||||
if r.status_code != 200:
|
||||
return collected
|
||||
|
||||
data = r.json()
|
||||
postings = data.get("jobPostings", [])
|
||||
for p in postings:
|
||||
title = clean_html_text(p.get("title", ""))
|
||||
ext_path = p.get("externalPath", "")
|
||||
if not title or not ext_path:
|
||||
continue
|
||||
|
||||
location_name = p.get("locationsText", "USA") or "USA"
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = parse_is_us_remote(location_name, title)
|
||||
dept = determine_department(title, "")
|
||||
exp_level = determine_experience_level(title)
|
||||
canonical_url = f"https://{tenant}.{wd_num}.myworkdayjobs.com/en-US/{site_slug}{ext_path}"
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name,
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": f"Verified active position at {company_name}. Direct applications, specifications, and full qualifications available on official Workday portal.",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": canonical_url,
|
||||
"source": "workday"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Workday Warning] {company_name} ({slug}) error: {e}")
|
||||
return collected
|
||||
|
||||
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||
import json
|
||||
from bs4 import BeautifulSoup
|
||||
|
|
@ -406,28 +469,35 @@ class SmartCareersCrawler:
|
|||
anchors = soup.find_all("a", href=True)
|
||||
seen_links = set()
|
||||
|
||||
non_job_slugs = [
|
||||
"alert", "search", "faq", "culture", "event", "privacy", "terms", "story",
|
||||
"benefit", "leadership", "community", "report", "sustainability", "map", "login"
|
||||
]
|
||||
|
||||
for a in anchors:
|
||||
href = a.get("href", "").strip()
|
||||
text = clean_html_text(a.get_text(strip=True))
|
||||
if not href or not text or len(text) < 4 or len(text) > 90:
|
||||
if not href or not text or len(text) < 6 or len(text) > 85:
|
||||
continue
|
||||
|
||||
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
|
||||
low_href = href.lower()
|
||||
low_text = text.lower()
|
||||
|
||||
is_job_href = any(k in low_href for k in ["/job/", "/position/", "/opening/", "gh_jid=", "requisition"])
|
||||
if not is_job_href:
|
||||
continue
|
||||
|
||||
if any(s in low_href for s in non_job_slugs) or any(s in low_text for s in non_job_slugs):
|
||||
continue
|
||||
|
||||
full_url = href if href.startswith("http") else urljoin(page_url, href)
|
||||
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
|
||||
continue
|
||||
seen_links.add(full_url)
|
||||
|
||||
lower_text = text.lower()
|
||||
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
|
||||
continue
|
||||
|
||||
dept = determine_department(text, "")
|
||||
exp_level = determine_experience_level(text)
|
||||
is_remote = "remote" in lower_text
|
||||
is_remote = "remote" in low_text
|
||||
|
||||
jobs.append({
|
||||
"title": text,
|
||||
|
|
@ -436,7 +506,7 @@ class SmartCareersCrawler:
|
|||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
|
||||
"description": f"Direct opening for {text} at {default_company}. Application details, requirements, and qualifications available at {full_url}",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": full_url,
|
||||
|
|
@ -449,7 +519,7 @@ class SmartCareersCrawler:
|
|||
|
||||
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||
try:
|
||||
r = self._safe_get(url, timeout=8, allow_redirects=True)
|
||||
r = self._safe_get(url, timeout=5, allow_redirects=True)
|
||||
if r is None or r.status_code != 200:
|
||||
return []
|
||||
html_text = r.text
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue