468 lines
20 KiB
Python
468 lines
20 KiB
Python
from scrapers.proxy_manager import get_rotating_proxy_dict
|
|
import os
|
|
import requests
|
|
import re
|
|
from urllib.parse import urlparse, urljoin
|
|
from typing import Optional, Tuple, Dict, Any, List
|
|
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level
|
|
|
|
COMMON_CAREER_PATHS = [
|
|
"/careers",
|
|
"/career",
|
|
"/jobs",
|
|
"/job",
|
|
"/work-with-us",
|
|
"/join-us",
|
|
"/join",
|
|
"/open-positions",
|
|
"/vacancies"
|
|
]
|
|
|
|
NON_US_KEYWORDS = [
|
|
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
|
|
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
|
|
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
|
|
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
|
|
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
|
|
"ireland", "dublin", "switzerland", "zurich"
|
|
]
|
|
|
|
def is_valid_us_location(location_name: str, title: str = "") -> bool:
|
|
loc_lower = (location_name or "").lower()
|
|
title_lower = (title or "").lower()
|
|
for non_us in NON_US_KEYWORDS:
|
|
if non_us in loc_lower or non_us in title_lower:
|
|
return False
|
|
return True
|
|
|
|
def parse_is_us_remote(location_name: str, title: str = "") -> bool:
|
|
loc_lower = (location_name or "").lower()
|
|
title_lower = (title or "").lower()
|
|
if not is_valid_us_location(location_name, title):
|
|
return False
|
|
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
|
|
return bool(is_remote_mention)
|
|
|
|
class SmartCareersCrawler:
|
|
"""
|
|
Crawls company domains, finds their /careers or /jobs pages,
|
|
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
|
|
Supports optional SOCKS5 / HTTP proxies via PROXY_URL or SOCKS5_PROXY.
|
|
"""
|
|
def __init__(self, headers: Optional[Dict[str, str]] = None):
|
|
self.headers = headers or {
|
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
|
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
|
}
|
|
|
|
# Configure rotating SOCKS5 proxies
|
|
self.proxies = get_rotating_proxy_dict()
|
|
|
|
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
|
parsed = urlparse(url)
|
|
host = parsed.netloc.lower()
|
|
path = parsed.path.strip("/")
|
|
|
|
if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
|
|
parts = [p for p in path.split("/") if p and p != "embed"]
|
|
if parts:
|
|
return "greenhouse", parts[0]
|
|
|
|
if "jobs.lever.co" in host:
|
|
parts = [p for p in path.split("/") if p]
|
|
if parts:
|
|
return "lever", parts[0]
|
|
|
|
if "jobs.ashbyhq.com" in host:
|
|
parts = [p for p in path.split("/") if p]
|
|
if parts:
|
|
return "ashby", parts[0]
|
|
|
|
if not html_text:
|
|
return None, None
|
|
|
|
gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
|
if gh_match:
|
|
return "greenhouse", gh_match.group(1)
|
|
|
|
lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
|
if lever_match:
|
|
return "lever", lever_match.group(1)
|
|
|
|
ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
|
if ashby_match:
|
|
return "ashby", ashby_match.group(1)
|
|
|
|
return None, None
|
|
|
|
def _safe_get(self, url: str, timeout: int = 8, allow_redirects: bool = True) -> Optional[requests.Response]:
|
|
"""
|
|
Attempts to fetch via rotating Privado proxy. If proxy fails, drops proxy and fetches directly.
|
|
"""
|
|
proxies = get_rotating_proxy_dict()
|
|
if proxies:
|
|
try:
|
|
r = requests.get(url, headers=self.headers, proxies=proxies, timeout=timeout, allow_redirects=allow_redirects)
|
|
return r
|
|
except Exception:
|
|
# Proxy timeout or connection error; fall back to direct request immediately
|
|
pass
|
|
|
|
try:
|
|
return requests.get(url, headers=self.headers, timeout=timeout, allow_redirects=allow_redirects)
|
|
except Exception:
|
|
return None
|
|
|
|
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
|
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
|
|
base_url = f"https://{clean_domain}"
|
|
|
|
for path in COMMON_CAREER_PATHS:
|
|
test_url = f"{base_url}{path}"
|
|
r = self._safe_get(test_url, timeout=5, allow_redirects=True)
|
|
if r is not None:
|
|
final_url = r.url
|
|
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
|
if provider and slug:
|
|
return provider, slug, final_url
|
|
|
|
r = self._safe_get(base_url, timeout=5, allow_redirects=True)
|
|
if r is not None and r.status_code == 200:
|
|
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
|
if provider and slug:
|
|
return provider, slug, r.url
|
|
|
|
links = re.findall(r'href=[\x27\x22]([^\x27\x22]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^\x27\x22]*)[\x27\x22]', r.text, re.IGNORECASE)
|
|
for link in links[:5]:
|
|
target = link if link.startswith("http") else urljoin(base_url, link)
|
|
provider, slug = self.detect_ats_from_url_or_html(target)
|
|
if provider and slug:
|
|
return provider, slug, target
|
|
cr = self._safe_get(target, timeout=5, allow_redirects=True)
|
|
if cr is not None:
|
|
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
|
if provider and slug:
|
|
return provider, slug, cr.url
|
|
|
|
return None, None, None
|
|
|
|
def fetch_greenhouse_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
|
url = f"https://boards-api.greenhouse.io/v1/boards/{slug}/jobs?content=true"
|
|
collected = []
|
|
try:
|
|
r = self._safe_get(url, timeout=8)
|
|
if r is None or r.status_code != 200:
|
|
return collected
|
|
data = r.json()
|
|
jobs = data.get("jobs", [])
|
|
for j in jobs:
|
|
title = clean_html_text(j.get("title", ""))
|
|
job_url = j.get("absolute_url", "")
|
|
if not title or not job_url:
|
|
continue
|
|
|
|
location_obj = j.get("location", {})
|
|
location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj)
|
|
|
|
if not is_valid_us_location(location_name, title):
|
|
continue
|
|
|
|
is_remote = parse_is_us_remote(location_name, title)
|
|
departments = j.get("departments", [])
|
|
dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else ""
|
|
|
|
dept = determine_department(title, dept_text)
|
|
exp_level = determine_experience_level(title)
|
|
desc_clean = clean_html_text(j.get("content", "") or "")
|
|
|
|
collected.append({
|
|
"title": title,
|
|
"company": company_name,
|
|
"location": location_name if location_name != "Remote" else "Remote, USA",
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Greenhouse.",
|
|
"salary_min": None,
|
|
"salary_max": None,
|
|
"job_url": job_url,
|
|
"source": "greenhouse"
|
|
})
|
|
except Exception as e:
|
|
print(f"[Greenhouse Warning] {company_name} ({slug}) error: {e}")
|
|
return collected
|
|
|
|
def fetch_lever_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
|
url = f"https://api.lever.co/v0/postings/{slug}?mode=json"
|
|
collected = []
|
|
try:
|
|
r = self._safe_get(url, timeout=8)
|
|
if r is None or r.status_code != 200:
|
|
return collected
|
|
jobs = r.json()
|
|
for j in jobs:
|
|
title = clean_html_text(j.get("text", ""))
|
|
job_url = j.get("hostedUrl", "")
|
|
if not title or not job_url:
|
|
continue
|
|
|
|
categories = j.get("categories", {})
|
|
location_name = categories.get("location", "Remote, USA") if isinstance(categories, dict) else "Remote, USA"
|
|
workplace_type = categories.get("workplaceType", "") if isinstance(categories, dict) else ""
|
|
|
|
if not is_valid_us_location(location_name, title):
|
|
continue
|
|
|
|
is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title)
|
|
dept_text = categories.get("department", "") if isinstance(categories, dict) else ""
|
|
dept = determine_department(title, dept_text)
|
|
exp_level = determine_experience_level(title)
|
|
|
|
description_plain = j.get("descriptionPlain", "") or j.get("description", "")
|
|
desc_clean = clean_html_text(description_plain)
|
|
|
|
collected.append({
|
|
"title": title,
|
|
"company": company_name,
|
|
"location": location_name if location_name != "Remote" else "Remote, USA",
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Lever.",
|
|
"salary_min": None,
|
|
"salary_max": None,
|
|
"job_url": job_url,
|
|
"source": "lever"
|
|
})
|
|
except Exception as e:
|
|
print(f"[Lever Warning] {company_name} ({slug}) error: {e}")
|
|
return collected
|
|
|
|
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
|
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
|
collected = []
|
|
try:
|
|
r = self._safe_get(url, timeout=8)
|
|
if r is None or r.status_code != 200:
|
|
return collected
|
|
data = r.json()
|
|
jobs = data.get("jobs", [])
|
|
for j in jobs:
|
|
title = clean_html_text(j.get("title", ""))
|
|
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}"
|
|
if not title or not job_url:
|
|
continue
|
|
|
|
location_name = j.get("location", "Remote, USA") or "Remote, USA"
|
|
if not is_valid_us_location(location_name, title):
|
|
continue
|
|
|
|
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
|
|
dept_text = j.get("department", "")
|
|
dept = determine_department(title, dept_text)
|
|
exp_level = determine_experience_level(title)
|
|
|
|
description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
|
|
|
|
salary_min = None
|
|
salary_max = None
|
|
comp = j.get("compensation")
|
|
if isinstance(comp, dict):
|
|
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
|
|
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
|
|
if comp_min:
|
|
try:
|
|
salary_min = float(comp_min)
|
|
except (ValueError, TypeError):
|
|
pass
|
|
if comp_max:
|
|
try:
|
|
salary_max = float(comp_max)
|
|
except (ValueError, TypeError):
|
|
pass
|
|
|
|
collected.append({
|
|
"title": title,
|
|
"company": company_name,
|
|
"location": location_name,
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.",
|
|
"salary_min": salary_min,
|
|
"salary_max": salary_max,
|
|
"job_url": job_url,
|
|
"source": "ashby"
|
|
})
|
|
except Exception as e:
|
|
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
|
|
return collected
|
|
|
|
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
|
import json
|
|
from bs4 import BeautifulSoup
|
|
jobs = []
|
|
if not html_text:
|
|
return jobs
|
|
|
|
soup = BeautifulSoup(html_text, "html.parser")
|
|
scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I))
|
|
|
|
for s in scripts:
|
|
try:
|
|
data = json.loads(s.string or "{}")
|
|
except Exception:
|
|
continue
|
|
|
|
items = data if isinstance(data, list) else [data]
|
|
for item in items:
|
|
if not isinstance(item, dict):
|
|
continue
|
|
graph = item.get("@graph", [])
|
|
candidates = graph if isinstance(graph, list) and graph else [item]
|
|
|
|
for c in candidates:
|
|
if not isinstance(c, dict):
|
|
continue
|
|
item_type = str(c.get("@type", "")).lower()
|
|
if "jobposting" in item_type:
|
|
title = clean_html_text(c.get("title", ""))
|
|
if not title:
|
|
continue
|
|
|
|
hiring_org = c.get("hiringOrganization", {})
|
|
company = default_company
|
|
if isinstance(hiring_org, dict):
|
|
company = hiring_org.get("name") or default_company
|
|
elif isinstance(hiring_org, str):
|
|
company = hiring_org
|
|
|
|
job_loc = c.get("jobLocation", {})
|
|
location_name = "Remote, USA"
|
|
if isinstance(job_loc, dict):
|
|
addr = job_loc.get("address", {})
|
|
if isinstance(addr, dict):
|
|
city = addr.get("addressLocality", "")
|
|
region = addr.get("addressRegion", "")
|
|
country = addr.get("addressCountry", "")
|
|
parts = [p for p in [city, region, country] if p]
|
|
location_name = ", ".join(parts) or "USA"
|
|
elif isinstance(addr, str):
|
|
location_name = addr
|
|
|
|
work_location_type = str(c.get("jobLocationType", "")).upper()
|
|
is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower()
|
|
|
|
if not is_valid_us_location(location_name, title):
|
|
continue
|
|
|
|
desc_raw = c.get("description", "")
|
|
desc_clean = clean_html_text(desc_raw)
|
|
|
|
salary_min = None
|
|
salary_max = None
|
|
base_salary = c.get("baseSalary", {})
|
|
if isinstance(base_salary, dict):
|
|
val = base_salary.get("value", {})
|
|
if isinstance(val, dict):
|
|
s_min = val.get("minValue") or val.get("value")
|
|
s_max = val.get("maxValue") or val.get("value")
|
|
if s_min:
|
|
try: salary_min = float(s_min)
|
|
except: pass
|
|
if s_max:
|
|
try: salary_max = float(s_max)
|
|
except: pass
|
|
|
|
job_url = c.get("url") or page_url
|
|
if job_url.startswith("/"):
|
|
job_url = urljoin(page_url, job_url)
|
|
|
|
dept = determine_department(title, "")
|
|
exp_level = determine_experience_level(title)
|
|
|
|
jobs.append({
|
|
"title": title,
|
|
"company": company,
|
|
"location": location_name,
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.",
|
|
"salary_min": salary_min,
|
|
"salary_max": salary_max,
|
|
"job_url": job_url,
|
|
"source": "web_direct"
|
|
})
|
|
return jobs
|
|
|
|
def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
|
from bs4 import BeautifulSoup
|
|
jobs = []
|
|
if not html_text:
|
|
return jobs
|
|
|
|
soup = BeautifulSoup(html_text, "html.parser")
|
|
anchors = soup.find_all("a", href=True)
|
|
seen_links = set()
|
|
|
|
for a in anchors:
|
|
href = a.get("href", "").strip()
|
|
text = clean_html_text(a.get_text(strip=True))
|
|
if not href or not text or len(text) < 4 or len(text) > 90:
|
|
continue
|
|
|
|
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
|
|
if not is_job_href:
|
|
continue
|
|
|
|
full_url = href if href.startswith("http") else urljoin(page_url, href)
|
|
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
|
|
continue
|
|
seen_links.add(full_url)
|
|
|
|
lower_text = text.lower()
|
|
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
|
|
continue
|
|
|
|
dept = determine_department(text, "")
|
|
exp_level = determine_experience_level(text)
|
|
is_remote = "remote" in lower_text
|
|
|
|
jobs.append({
|
|
"title": text,
|
|
"company": default_company,
|
|
"location": "Remote, USA" if is_remote else "United States",
|
|
"is_remote": is_remote,
|
|
"department": dept,
|
|
"experience_level": exp_level,
|
|
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
|
|
"salary_min": None,
|
|
"salary_max": None,
|
|
"job_url": full_url,
|
|
"source": "web_direct"
|
|
})
|
|
if len(jobs) >= 50:
|
|
break
|
|
|
|
return jobs
|
|
|
|
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
|
try:
|
|
r = self._safe_get(url, timeout=8, allow_redirects=True)
|
|
if r is None or r.status_code != 200:
|
|
return []
|
|
html_text = r.text
|
|
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
|
|
jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company)
|
|
if jobs:
|
|
print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}")
|
|
return jobs
|
|
# 2. Try HTML job links
|
|
jobs = self.extract_html_job_links(html_text, r.url, default_company)
|
|
if jobs:
|
|
print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}")
|
|
return jobs
|
|
except Exception as e:
|
|
print(f"[Smart Crawler Warning] Scraping {url} failed: {e}")
|
|
return []
|