feat(scraper): universal web scraping with Schema.org JobPosting and HTML card extraction across domains

This commit is contained in:
JobsBoard Deployer 2026-09-05 12:38:22 -04:00
parent 2d2aab2999
commit f22b826193
2 changed files with 200 additions and 9 deletions

View file

@ -10,8 +10,9 @@ from scrapers.jobspy_runner import run_jobspy_scrapes
from scrapers.smart_careers_crawler import SmartCareersCrawler
from db import upsert_jobs
# Targeted domains for smart careers auto-discovery across US
# Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial
DOMAINS_TO_PROBE = [
# Top Tech & AI
("anthropic.com", "Anthropic"),
("openai.com", "OpenAI"),
("stripe.com", "Stripe"),
@ -35,17 +36,34 @@ DOMAINS_TO_PROBE = [
("canva.com", "Canva"),
("roblox.com", "Roblox"),
("epicgames.com", "Epic Games"),
("flexport.com", "Flexport")
("flexport.com", "Flexport"),
# Enterprise, Retail, Logistics & Healthcare
("target.com", "Target"),
("homedepot.com", "The Home Depot"),
("costco.com", "Costco Wholesale"),
("cvshealth.com", "CVS Health"),
("unitedhealthgroup.com", "UnitedHealth Group"),
("cigna.com", "The Cigna Group"),
("travelers.com", "Travelers Insurance"),
("hartfordhealthcare.org", "Hartford HealthCare"),
("yalehealth.org", "Yale Health"),
("lockheedmartin.com", "Lockheed Martin"),
("boeing.com", "Boeing"),
("caterpillar.com", "Caterpillar"),
("deere.com", "John Deere"),
("fedex.com", "FedEx"),
("ups.com", "UPS")
]
def run_smart_crawler_scrapes():
print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...")
print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...")
crawler = SmartCareersCrawler()
discovered_jobs = []
for domain, company_name in DOMAINS_TO_PROBE:
try:
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
provider, slug, careers_url, html_text = crawler.probe_domain_for_careers(domain)
if provider and slug:
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
if provider == "greenhouse":
@ -57,10 +75,14 @@ def run_smart_crawler_scrapes():
elif provider == "ashby":
jobs = crawler.fetch_ashby_board(slug, company_name)
discovered_jobs.extend(jobs)
elif careers_url:
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
jobs = crawler.scrape_native_career_page(careers_url, company_name)
discovered_jobs.extend(jobs)
except Exception as e:
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.")
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
return discovered_jobs
def execute_all_scrapes():
@ -70,7 +92,7 @@ def execute_all_scrapes():
all_jobs = []
# 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby)
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby)
try:
crawler_jobs = run_smart_crawler_scrapes()
all_jobs.extend(crawler_jobs)
@ -98,19 +120,19 @@ def execute_all_scrapes():
except Exception as e:
print(f"[Error] Expanded categories error: {e}")
# 5. Major CT & Regional Enterprise Employers
# 5. Major Regional & Enterprise Employers
try:
emp_jobs = run_major_ct_employers_scrape()
all_jobs.extend(emp_jobs)
except Exception as e:
print(f"[Error] Major Employers error: {e}")
# 6. CT State JobAps Government & Public Portal
# 6. Public Sector Portals
try:
ct_jobs = run_ct_jobaps_scrape()
all_jobs.extend(ct_jobs)
except Exception as e:
print(f"[Error] CT JobAps execution error: {e}")
print(f"[Error] JobAps execution error: {e}")
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
try:

View file

@ -291,3 +291,172 @@ class SmartCareersCrawler:
except Exception as e:
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
return collected
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
import json
from bs4 import BeautifulSoup
jobs = []
if not html_text:
return jobs
soup = BeautifulSoup(html_text, "html.parser")
scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I))
for s in scripts:
try:
data = json.loads(s.string or "{}")
except Exception:
continue
items = data if isinstance(data, list) else [data]
for item in items:
if not isinstance(item, dict):
continue
graph = item.get("@graph", [])
candidates = graph if isinstance(graph, list) and graph else [item]
for c in candidates:
if not isinstance(c, dict):
continue
item_type = str(c.get("@type", "")).lower()
if "jobposting" in item_type:
title = clean_html_text(c.get("title", ""))
if not title:
continue
hiring_org = c.get("hiringOrganization", {})
company = default_company
if isinstance(hiring_org, dict):
company = hiring_org.get("name") or default_company
elif isinstance(hiring_org, str):
company = hiring_org
job_loc = c.get("jobLocation", {})
location_name = "Remote, USA"
if isinstance(job_loc, dict):
addr = job_loc.get("address", {})
if isinstance(addr, dict):
city = addr.get("addressLocality", "")
region = addr.get("addressRegion", "")
country = addr.get("addressCountry", "")
parts = [p for p in [city, region, country] if p]
location_name = ", ".join(parts) or "USA"
elif isinstance(addr, str):
location_name = addr
work_location_type = str(c.get("jobLocationType", "")).upper()
is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower()
if not is_valid_us_location(location_name, title):
continue
desc_raw = c.get("description", "")
desc_clean = clean_html_text(desc_raw)
salary_min = None
salary_max = None
base_salary = c.get("baseSalary", {})
if isinstance(base_salary, dict):
val = base_salary.get("value", {})
if isinstance(val, dict):
s_min = val.get("minValue") or val.get("value")
s_max = val.get("maxValue") or val.get("value")
if s_min:
try: salary_min = float(s_min)
except: pass
if s_max:
try: salary_max = float(s_max)
except: pass
job_url = c.get("url") or page_url
if job_url.startswith("/"):
job_url = urljoin(page_url, job_url)
dept = determine_department(title, "")
exp_level = determine_experience_level(title)
jobs.append({
"title": title,
"company": company,
"location": location_name,
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "web_direct"
})
return jobs
def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
from bs4 import BeautifulSoup
jobs = []
if not html_text:
return jobs
soup = BeautifulSoup(html_text, "html.parser")
anchors = soup.find_all("a", href=True)
seen_links = set()
for a in anchors:
href = a.get("href", "").strip()
text = clean_html_text(a.get_text(strip=True))
if not href or not text or len(text) < 4 or len(text) > 90:
continue
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
if not is_job_href:
continue
full_url = href if href.startswith("http") else urljoin(page_url, href)
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
continue
seen_links.add(full_url)
lower_text = text.lower()
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
continue
dept = determine_department(text, "")
exp_level = determine_experience_level(text)
is_remote = "remote" in lower_text
jobs.append({
"title": text,
"company": default_company,
"location": "Remote, USA" if is_remote else "United States",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
"salary_min": None,
"salary_max": None,
"job_url": full_url,
"source": "web_direct"
})
if len(jobs) >= 50:
break
return jobs
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
try:
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8, allow_redirects=True)
if r.status_code != 200:
return []
html_text = r.text
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company)
if jobs:
print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}")
return jobs
# 2. Try HTML job links
jobs = self.extract_html_job_links(html_text, r.url, default_company)
if jobs:
print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}")
return jobs
except Exception as e:
print(f"[Smart Crawler Warning] Scraping {url} failed: {e}")
return []