feat(scraper): universal web scraping with Schema.org JobPosting and HTML card extraction across domains
This commit is contained in:
parent
2d2aab2999
commit
f22b826193
2 changed files with 200 additions and 9 deletions
|
|
@ -10,8 +10,9 @@ from scrapers.jobspy_runner import run_jobspy_scrapes
|
||||||
from scrapers.smart_careers_crawler import SmartCareersCrawler
|
from scrapers.smart_careers_crawler import SmartCareersCrawler
|
||||||
from db import upsert_jobs
|
from db import upsert_jobs
|
||||||
|
|
||||||
# Targeted domains for smart careers auto-discovery across US
|
# Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial
|
||||||
DOMAINS_TO_PROBE = [
|
DOMAINS_TO_PROBE = [
|
||||||
|
# Top Tech & AI
|
||||||
("anthropic.com", "Anthropic"),
|
("anthropic.com", "Anthropic"),
|
||||||
("openai.com", "OpenAI"),
|
("openai.com", "OpenAI"),
|
||||||
("stripe.com", "Stripe"),
|
("stripe.com", "Stripe"),
|
||||||
|
|
@ -35,17 +36,34 @@ DOMAINS_TO_PROBE = [
|
||||||
("canva.com", "Canva"),
|
("canva.com", "Canva"),
|
||||||
("roblox.com", "Roblox"),
|
("roblox.com", "Roblox"),
|
||||||
("epicgames.com", "Epic Games"),
|
("epicgames.com", "Epic Games"),
|
||||||
("flexport.com", "Flexport")
|
("flexport.com", "Flexport"),
|
||||||
|
|
||||||
|
# Enterprise, Retail, Logistics & Healthcare
|
||||||
|
("target.com", "Target"),
|
||||||
|
("homedepot.com", "The Home Depot"),
|
||||||
|
("costco.com", "Costco Wholesale"),
|
||||||
|
("cvshealth.com", "CVS Health"),
|
||||||
|
("unitedhealthgroup.com", "UnitedHealth Group"),
|
||||||
|
("cigna.com", "The Cigna Group"),
|
||||||
|
("travelers.com", "Travelers Insurance"),
|
||||||
|
("hartfordhealthcare.org", "Hartford HealthCare"),
|
||||||
|
("yalehealth.org", "Yale Health"),
|
||||||
|
("lockheedmartin.com", "Lockheed Martin"),
|
||||||
|
("boeing.com", "Boeing"),
|
||||||
|
("caterpillar.com", "Caterpillar"),
|
||||||
|
("deere.com", "John Deere"),
|
||||||
|
("fedex.com", "FedEx"),
|
||||||
|
("ups.com", "UPS")
|
||||||
]
|
]
|
||||||
|
|
||||||
def run_smart_crawler_scrapes():
|
def run_smart_crawler_scrapes():
|
||||||
print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...")
|
print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...")
|
||||||
crawler = SmartCareersCrawler()
|
crawler = SmartCareersCrawler()
|
||||||
discovered_jobs = []
|
discovered_jobs = []
|
||||||
|
|
||||||
for domain, company_name in DOMAINS_TO_PROBE:
|
for domain, company_name in DOMAINS_TO_PROBE:
|
||||||
try:
|
try:
|
||||||
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
|
provider, slug, careers_url, html_text = crawler.probe_domain_for_careers(domain)
|
||||||
if provider and slug:
|
if provider and slug:
|
||||||
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
|
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
|
||||||
if provider == "greenhouse":
|
if provider == "greenhouse":
|
||||||
|
|
@ -57,10 +75,14 @@ def run_smart_crawler_scrapes():
|
||||||
elif provider == "ashby":
|
elif provider == "ashby":
|
||||||
jobs = crawler.fetch_ashby_board(slug, company_name)
|
jobs = crawler.fetch_ashby_board(slug, company_name)
|
||||||
discovered_jobs.extend(jobs)
|
discovered_jobs.extend(jobs)
|
||||||
|
elif careers_url:
|
||||||
|
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
|
||||||
|
jobs = crawler.scrape_native_career_page(careers_url, company_name)
|
||||||
|
discovered_jobs.extend(jobs)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
|
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
|
||||||
|
|
||||||
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.")
|
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
|
||||||
return discovered_jobs
|
return discovered_jobs
|
||||||
|
|
||||||
def execute_all_scrapes():
|
def execute_all_scrapes():
|
||||||
|
|
@ -70,7 +92,7 @@ def execute_all_scrapes():
|
||||||
|
|
||||||
all_jobs = []
|
all_jobs = []
|
||||||
|
|
||||||
# 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby)
|
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby)
|
||||||
try:
|
try:
|
||||||
crawler_jobs = run_smart_crawler_scrapes()
|
crawler_jobs = run_smart_crawler_scrapes()
|
||||||
all_jobs.extend(crawler_jobs)
|
all_jobs.extend(crawler_jobs)
|
||||||
|
|
@ -98,19 +120,19 @@ def execute_all_scrapes():
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"[Error] Expanded categories error: {e}")
|
print(f"[Error] Expanded categories error: {e}")
|
||||||
|
|
||||||
# 5. Major CT & Regional Enterprise Employers
|
# 5. Major Regional & Enterprise Employers
|
||||||
try:
|
try:
|
||||||
emp_jobs = run_major_ct_employers_scrape()
|
emp_jobs = run_major_ct_employers_scrape()
|
||||||
all_jobs.extend(emp_jobs)
|
all_jobs.extend(emp_jobs)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"[Error] Major Employers error: {e}")
|
print(f"[Error] Major Employers error: {e}")
|
||||||
|
|
||||||
# 6. CT State JobAps Government & Public Portal
|
# 6. Public Sector Portals
|
||||||
try:
|
try:
|
||||||
ct_jobs = run_ct_jobaps_scrape()
|
ct_jobs = run_ct_jobaps_scrape()
|
||||||
all_jobs.extend(ct_jobs)
|
all_jobs.extend(ct_jobs)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"[Error] CT JobAps execution error: {e}")
|
print(f"[Error] JobAps execution error: {e}")
|
||||||
|
|
||||||
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
|
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
|
||||||
try:
|
try:
|
||||||
|
|
|
||||||
|
|
@ -291,3 +291,172 @@ class SmartCareersCrawler:
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
|
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
|
||||||
return collected
|
return collected
|
||||||
|
|
||||||
|
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||||
|
import json
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
jobs = []
|
||||||
|
if not html_text:
|
||||||
|
return jobs
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html_text, "html.parser")
|
||||||
|
scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I))
|
||||||
|
|
||||||
|
for s in scripts:
|
||||||
|
try:
|
||||||
|
data = json.loads(s.string or "{}")
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
|
||||||
|
items = data if isinstance(data, list) else [data]
|
||||||
|
for item in items:
|
||||||
|
if not isinstance(item, dict):
|
||||||
|
continue
|
||||||
|
graph = item.get("@graph", [])
|
||||||
|
candidates = graph if isinstance(graph, list) and graph else [item]
|
||||||
|
|
||||||
|
for c in candidates:
|
||||||
|
if not isinstance(c, dict):
|
||||||
|
continue
|
||||||
|
item_type = str(c.get("@type", "")).lower()
|
||||||
|
if "jobposting" in item_type:
|
||||||
|
title = clean_html_text(c.get("title", ""))
|
||||||
|
if not title:
|
||||||
|
continue
|
||||||
|
|
||||||
|
hiring_org = c.get("hiringOrganization", {})
|
||||||
|
company = default_company
|
||||||
|
if isinstance(hiring_org, dict):
|
||||||
|
company = hiring_org.get("name") or default_company
|
||||||
|
elif isinstance(hiring_org, str):
|
||||||
|
company = hiring_org
|
||||||
|
|
||||||
|
job_loc = c.get("jobLocation", {})
|
||||||
|
location_name = "Remote, USA"
|
||||||
|
if isinstance(job_loc, dict):
|
||||||
|
addr = job_loc.get("address", {})
|
||||||
|
if isinstance(addr, dict):
|
||||||
|
city = addr.get("addressLocality", "")
|
||||||
|
region = addr.get("addressRegion", "")
|
||||||
|
country = addr.get("addressCountry", "")
|
||||||
|
parts = [p for p in [city, region, country] if p]
|
||||||
|
location_name = ", ".join(parts) or "USA"
|
||||||
|
elif isinstance(addr, str):
|
||||||
|
location_name = addr
|
||||||
|
|
||||||
|
work_location_type = str(c.get("jobLocationType", "")).upper()
|
||||||
|
is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower()
|
||||||
|
|
||||||
|
if not is_valid_us_location(location_name, title):
|
||||||
|
continue
|
||||||
|
|
||||||
|
desc_raw = c.get("description", "")
|
||||||
|
desc_clean = clean_html_text(desc_raw)
|
||||||
|
|
||||||
|
salary_min = None
|
||||||
|
salary_max = None
|
||||||
|
base_salary = c.get("baseSalary", {})
|
||||||
|
if isinstance(base_salary, dict):
|
||||||
|
val = base_salary.get("value", {})
|
||||||
|
if isinstance(val, dict):
|
||||||
|
s_min = val.get("minValue") or val.get("value")
|
||||||
|
s_max = val.get("maxValue") or val.get("value")
|
||||||
|
if s_min:
|
||||||
|
try: salary_min = float(s_min)
|
||||||
|
except: pass
|
||||||
|
if s_max:
|
||||||
|
try: salary_max = float(s_max)
|
||||||
|
except: pass
|
||||||
|
|
||||||
|
job_url = c.get("url") or page_url
|
||||||
|
if job_url.startswith("/"):
|
||||||
|
job_url = urljoin(page_url, job_url)
|
||||||
|
|
||||||
|
dept = determine_department(title, "")
|
||||||
|
exp_level = determine_experience_level(title)
|
||||||
|
|
||||||
|
jobs.append({
|
||||||
|
"title": title,
|
||||||
|
"company": company,
|
||||||
|
"location": location_name,
|
||||||
|
"is_remote": is_remote,
|
||||||
|
"department": dept,
|
||||||
|
"experience_level": exp_level,
|
||||||
|
"description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.",
|
||||||
|
"salary_min": salary_min,
|
||||||
|
"salary_max": salary_max,
|
||||||
|
"job_url": job_url,
|
||||||
|
"source": "web_direct"
|
||||||
|
})
|
||||||
|
return jobs
|
||||||
|
|
||||||
|
def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
jobs = []
|
||||||
|
if not html_text:
|
||||||
|
return jobs
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html_text, "html.parser")
|
||||||
|
anchors = soup.find_all("a", href=True)
|
||||||
|
seen_links = set()
|
||||||
|
|
||||||
|
for a in anchors:
|
||||||
|
href = a.get("href", "").strip()
|
||||||
|
text = clean_html_text(a.get_text(strip=True))
|
||||||
|
if not href or not text or len(text) < 4 or len(text) > 90:
|
||||||
|
continue
|
||||||
|
|
||||||
|
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
|
||||||
|
if not is_job_href:
|
||||||
|
continue
|
||||||
|
|
||||||
|
full_url = href if href.startswith("http") else urljoin(page_url, href)
|
||||||
|
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
|
||||||
|
continue
|
||||||
|
seen_links.add(full_url)
|
||||||
|
|
||||||
|
lower_text = text.lower()
|
||||||
|
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
|
||||||
|
continue
|
||||||
|
|
||||||
|
dept = determine_department(text, "")
|
||||||
|
exp_level = determine_experience_level(text)
|
||||||
|
is_remote = "remote" in lower_text
|
||||||
|
|
||||||
|
jobs.append({
|
||||||
|
"title": text,
|
||||||
|
"company": default_company,
|
||||||
|
"location": "Remote, USA" if is_remote else "United States",
|
||||||
|
"is_remote": is_remote,
|
||||||
|
"department": dept,
|
||||||
|
"experience_level": exp_level,
|
||||||
|
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
|
||||||
|
"salary_min": None,
|
||||||
|
"salary_max": None,
|
||||||
|
"job_url": full_url,
|
||||||
|
"source": "web_direct"
|
||||||
|
})
|
||||||
|
if len(jobs) >= 50:
|
||||||
|
break
|
||||||
|
|
||||||
|
return jobs
|
||||||
|
|
||||||
|
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||||
|
try:
|
||||||
|
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8, allow_redirects=True)
|
||||||
|
if r.status_code != 200:
|
||||||
|
return []
|
||||||
|
html_text = r.text
|
||||||
|
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
|
||||||
|
jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company)
|
||||||
|
if jobs:
|
||||||
|
print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}")
|
||||||
|
return jobs
|
||||||
|
# 2. Try HTML job links
|
||||||
|
jobs = self.extract_html_job_links(html_text, r.url, default_company)
|
||||||
|
if jobs:
|
||||||
|
print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}")
|
||||||
|
return jobs
|
||||||
|
except Exception as e:
|
||||||
|
print(f"[Smart Crawler Warning] Scraping {url} failed: {e}")
|
||||||
|
return []
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue