JB/scraper/scrapers/smart_careers_crawler.py

468 lines
20 KiB
Python

from scrapers.proxy_manager import get_rotating_proxy_dict
import os
import requests
import re
from urllib.parse import urlparse, urljoin
from typing import Optional, Tuple, Dict, Any, List
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level
COMMON_CAREER_PATHS = [
"/careers",
"/career",
"/jobs",
"/job",
"/work-with-us",
"/join-us",
"/join",
"/open-positions",
"/vacancies"
]
NON_US_KEYWORDS = [
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
"ireland", "dublin", "switzerland", "zurich"
]
def is_valid_us_location(location_name: str, title: str = "") -> bool:
loc_lower = (location_name or "").lower()
title_lower = (title or "").lower()
for non_us in NON_US_KEYWORDS:
if non_us in loc_lower or non_us in title_lower:
return False
return True
def parse_is_us_remote(location_name: str, title: str = "") -> bool:
loc_lower = (location_name or "").lower()
title_lower = (title or "").lower()
if not is_valid_us_location(location_name, title):
return False
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
return bool(is_remote_mention)
class SmartCareersCrawler:
"""
Crawls company domains, finds their /careers or /jobs pages,
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
"""
def __init__(self, headers: Optional[Dict[str, str]] = None):
self.headers = headers or {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
}
# Configure rotating SOCKS5 proxies
self.proxies = get_rotating_proxy_dict()
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
parsed = urlparse(url)
host = parsed.netloc.lower()
path = parsed.path.strip("/")
if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
parts = [p for p in path.split("/") if p and p != "embed"]
if parts:
return "greenhouse", parts[0]
if "jobs.lever.co" in host:
parts = [p for p in path.split("/") if p]
if parts:
return "lever", parts[0]
if "jobs.ashbyhq.com" in host:
parts = [p for p in path.split("/") if p]
if parts:
return "ashby", parts[0]
if not html_text:
return None, None
gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if gh_match:
return "greenhouse", gh_match.group(1)
lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if lever_match:
return "lever", lever_match.group(1)
ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if ashby_match:
return "ashby", ashby_match.group(1)
return None, None
def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]:
"""
Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly.
"""
proxies = get_rotating_proxy_dict()
if proxies:
try:
r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
return r
except Exception:
# Proxy timeout or connection error; fall back to direct request immediately
pass
try:
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
except Exception:
return None
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
base_url = f"https://{clean_domain}"
for path in COMMON_CAREER_PATHS:
test_url = f"{base_url}{path}"
r = self._safe_get(test_url, timeout=5, allow_redirects=True)
if r is not None:
final_url = r.url
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
if provider and slug:
return provider, slug, final_url
r = self._safe_get(base_url, timeout=5, allow_redirects=True)
if r is not None and r.status_code == 200:
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
if provider and slug:
return provider, slug, r.url
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
for link in links[:5]:
target = link if link.startswith("http") else urljoin(base_url, link)
provider, slug = self.detect_ats_from_url_or_html(target)
if provider and slug:
return provider, slug, target
cr = self._safe_get(target, timeout=5, allow_redirects=True)
if cr is not None:
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
if provider and slug:
return provider, slug, cr.url
return None, None, None
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
collected = []
try:
r = self._safe_get(url, timeout=8)
if r is None or r.status_code != 200:
return collected
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
continue
location_obj = j.get("location", {})
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(location_name, title)
departments = j.get("departments", [])
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
desc_clean = clean_html_text(j.get("content", "") or "")
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "greenhouse"
})
except Exception as e:
print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}")
return collected
def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
collected = []
try:
r = self._safe_get(url, timeout=8)
if r is None or r.status_code != 200:
return collected
jobs = r.json()
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
continue
categories = j.get("categories", {})
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
if not is_valid_us_location(location_name, title):
continue
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
desc_clean = clean_html_text(description_plain)
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
"salary_min": None,
"salary_max": None,
"job_url": job_url,
"source": "lever"
})
except Exception as e:
print(f"[Lever Warning] {company_name} ({slug}) error: {e}")
return collected
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
collected = []
try:
r = self._safe_get(url, timeout=8)
if r is None or r.status_code != 200:
return collected
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
dept_text = j.get("department", "")
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
collected.append({
"title": title,
"company": company_name,
"location": location_name,
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
return collected
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
import json
from bs4 import BeautifulSoup
jobs = []
if not html_text:
return jobs
soup = BeautifulSoup(html_text, "html.parser")
scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I))
for s in scripts:
try:
data = json.loads(s.string or "{}")
except Exception:
continue
items = data if isinstance(data, list) else [data]
for item in items:
if not isinstance(item, dict):
continue
graph = item.get("@graph", [])
candidates = graph if isinstance(graph, list) and graph else [item]
for c in candidates:
if not isinstance(c, dict):
continue
item_type = str(c.get("@type", "")).lower()
if "jobposting" in item_type:
title = clean_html_text(c.get("title", ""))
if not title:
continue
hiring_org = c.get("hiringOrganization", {})
company = default_company
if isinstance(hiring_org, dict):
company = hiring_org.get("name") or default_company
elif isinstance(hiring_org, str):
company = hiring_org
job_loc = c.get("jobLocation", {})
location_name = "Remote, USA"
if isinstance(job_loc, dict):
addr = job_loc.get("address", {})
if isinstance(addr, dict):
city = addr.get("addressLocality", "")
region = addr.get("addressRegion", "")
country = addr.get("addressCountry", "")
parts = [p for p in [city, region, country] if p]
location_name = ", ".join(parts) or "USA"
elif isinstance(addr, str):
location_name = addr
work_location_type = str(c.get("jobLocationType", "")).upper()
is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower()
if not is_valid_us_location(location_name, title):
continue
desc_raw = c.get("description", "")
desc_clean = clean_html_text(desc_raw)
salary_min = None
salary_max = None
base_salary = c.get("baseSalary", {})
if isinstance(base_salary, dict):
val = base_salary.get("value", {})
if isinstance(val, dict):
s_min = val.get("minValue") or val.get("value")
s_max = val.get("maxValue") or val.get("value")
if s_min:
try: salary_min = float(s_min)
except: pass
if s_max:
try: salary_max = float(s_max)
except: pass
job_url = c.get("url") or page_url
if job_url.startswith("/"):
job_url = urljoin(page_url, job_url)
dept = determine_department(title, "")
exp_level = determine_experience_level(title)
jobs.append({
"title": title,
"company": company,
"location": location_name,
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "web_direct"
})
return jobs
def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
from bs4 import BeautifulSoup
jobs = []
if not html_text:
return jobs
soup = BeautifulSoup(html_text, "html.parser")
anchors = soup.find_all("a", href=True)
seen_links = set()
for a in anchors:
href = a.get("href", "").strip()
text = clean_html_text(a.get_text(strip=True))
if not href or not text or len(text) < 4 or len(text) > 90:
continue
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
if not is_job_href:
continue
full_url = href if href.startswith("http") else urljoin(page_url, href)
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
continue
seen_links.add(full_url)
lower_text = text.lower()
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
continue
dept = determine_department(text, "")
exp_level = determine_experience_level(text)
is_remote = "remote" in lower_text
jobs.append({
"title": text,
"company": default_company,
"location": "Remote, USA" if is_remote else "United States",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
"salary_min": None,
"salary_max": None,
"job_url": full_url,
"source": "web_direct"
})
if len(jobs) >= 50:
break
return jobs
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
try:
r = self._safe_get(url, timeout=8, allow_redirects=True)
if r is None or r.status_code != 200:
return []
html_text = r.text
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company)
if jobs:
print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}")
return jobs
# 2. Try HTML job links
jobs = self.extract_html_job_links(html_text, r.url, default_company)
if jobs:
print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}")
return jobs
except Exception as e:
print(f"[Smart Crawler Warning] Scraping {url} failed: {e}")
return []