feat(scraper): universal web scraping with Schema.org JobPosting and HTML card extraction across domains
This commit is contained in:
parent
2d2aab2999
commit
f22b826193
2 changed files with 200 additions and 9 deletions
|
|
@ -10,8 +10,9 @@ from scrapers.jobspy_runner import run_jobspy_scrapes
|
|||
from scrapers.smart_careers_crawler import SmartCareersCrawler
|
||||
from db import upsert_jobs
|
||||
|
||||
# Targeted domains for smart careers auto-discovery across US
|
||||
# Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial
|
||||
DOMAINS_TO_PROBE = [
|
||||
# Top Tech & AI
|
||||
("anthropic.com", "Anthropic"),
|
||||
("openai.com", "OpenAI"),
|
||||
("stripe.com", "Stripe"),
|
||||
|
|
@ -35,17 +36,34 @@ DOMAINS_TO_PROBE = [
|
|||
("canva.com", "Canva"),
|
||||
("roblox.com", "Roblox"),
|
||||
("epicgames.com", "Epic Games"),
|
||||
("flexport.com", "Flexport")
|
||||
("flexport.com", "Flexport"),
|
||||
|
||||
# Enterprise, Retail, Logistics & Healthcare
|
||||
("target.com", "Target"),
|
||||
("homedepot.com", "The Home Depot"),
|
||||
("costco.com", "Costco Wholesale"),
|
||||
("cvshealth.com", "CVS Health"),
|
||||
("unitedhealthgroup.com", "UnitedHealth Group"),
|
||||
("cigna.com", "The Cigna Group"),
|
||||
("travelers.com", "Travelers Insurance"),
|
||||
("hartfordhealthcare.org", "Hartford HealthCare"),
|
||||
("yalehealth.org", "Yale Health"),
|
||||
("lockheedmartin.com", "Lockheed Martin"),
|
||||
("boeing.com", "Boeing"),
|
||||
("caterpillar.com", "Caterpillar"),
|
||||
("deere.com", "John Deere"),
|
||||
("fedex.com", "FedEx"),
|
||||
("ups.com", "UPS")
|
||||
]
|
||||
|
||||
def run_smart_crawler_scrapes():
|
||||
print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...")
|
||||
print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...")
|
||||
crawler = SmartCareersCrawler()
|
||||
discovered_jobs = []
|
||||
|
||||
for domain, company_name in DOMAINS_TO_PROBE:
|
||||
try:
|
||||
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
|
||||
provider, slug, careers_url, html_text = crawler.probe_domain_for_careers(domain)
|
||||
if provider and slug:
|
||||
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
|
||||
if provider == "greenhouse":
|
||||
|
|
@ -57,10 +75,14 @@ def run_smart_crawler_scrapes():
|
|||
elif provider == "ashby":
|
||||
jobs = crawler.fetch_ashby_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
elif careers_url:
|
||||
print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}")
|
||||
jobs = crawler.scrape_native_career_page(careers_url, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
except Exception as e:
|
||||
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
|
||||
|
||||
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.")
|
||||
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.")
|
||||
return discovered_jobs
|
||||
|
||||
def execute_all_scrapes():
|
||||
|
|
@ -70,7 +92,7 @@ def execute_all_scrapes():
|
|||
|
||||
all_jobs = []
|
||||
|
||||
# 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby)
|
||||
# 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby)
|
||||
try:
|
||||
crawler_jobs = run_smart_crawler_scrapes()
|
||||
all_jobs.extend(crawler_jobs)
|
||||
|
|
@ -98,19 +120,19 @@ def execute_all_scrapes():
|
|||
except Exception as e:
|
||||
print(f"[Error] Expanded categories error: {e}")
|
||||
|
||||
# 5. Major CT & Regional Enterprise Employers
|
||||
# 5. Major Regional & Enterprise Employers
|
||||
try:
|
||||
emp_jobs = run_major_ct_employers_scrape()
|
||||
all_jobs.extend(emp_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] Major Employers error: {e}")
|
||||
|
||||
# 6. CT State JobAps Government & Public Portal
|
||||
# 6. Public Sector Portals
|
||||
try:
|
||||
ct_jobs = run_ct_jobaps_scrape()
|
||||
all_jobs.extend(ct_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] CT JobAps execution error: {e}")
|
||||
print(f"[Error] JobAps execution error: {e}")
|
||||
|
||||
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -291,3 +291,172 @@ class SmartCareersCrawler:
|
|||
except Exception as e:
|
||||
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
|
||||
return collected
|
||||
|
||||
def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||
import json
|
||||
from bs4 import BeautifulSoup
|
||||
jobs = []
|
||||
if not html_text:
|
||||
return jobs
|
||||
|
||||
soup = BeautifulSoup(html_text, "html.parser")
|
||||
scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I))
|
||||
|
||||
for s in scripts:
|
||||
try:
|
||||
data = json.loads(s.string or "{}")
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
items = data if isinstance(data, list) else [data]
|
||||
for item in items:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
graph = item.get("@graph", [])
|
||||
candidates = graph if isinstance(graph, list) and graph else [item]
|
||||
|
||||
for c in candidates:
|
||||
if not isinstance(c, dict):
|
||||
continue
|
||||
item_type = str(c.get("@type", "")).lower()
|
||||
if "jobposting" in item_type:
|
||||
title = clean_html_text(c.get("title", ""))
|
||||
if not title:
|
||||
continue
|
||||
|
||||
hiring_org = c.get("hiringOrganization", {})
|
||||
company = default_company
|
||||
if isinstance(hiring_org, dict):
|
||||
company = hiring_org.get("name") or default_company
|
||||
elif isinstance(hiring_org, str):
|
||||
company = hiring_org
|
||||
|
||||
job_loc = c.get("jobLocation", {})
|
||||
location_name = "Remote, USA"
|
||||
if isinstance(job_loc, dict):
|
||||
addr = job_loc.get("address", {})
|
||||
if isinstance(addr, dict):
|
||||
city = addr.get("addressLocality", "")
|
||||
region = addr.get("addressRegion", "")
|
||||
country = addr.get("addressCountry", "")
|
||||
parts = [p for p in [city, region, country] if p]
|
||||
location_name = ", ".join(parts) or "USA"
|
||||
elif isinstance(addr, str):
|
||||
location_name = addr
|
||||
|
||||
work_location_type = str(c.get("jobLocationType", "")).upper()
|
||||
is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower()
|
||||
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
desc_raw = c.get("description", "")
|
||||
desc_clean = clean_html_text(desc_raw)
|
||||
|
||||
salary_min = None
|
||||
salary_max = None
|
||||
base_salary = c.get("baseSalary", {})
|
||||
if isinstance(base_salary, dict):
|
||||
val = base_salary.get("value", {})
|
||||
if isinstance(val, dict):
|
||||
s_min = val.get("minValue") or val.get("value")
|
||||
s_max = val.get("maxValue") or val.get("value")
|
||||
if s_min:
|
||||
try: salary_min = float(s_min)
|
||||
except: pass
|
||||
if s_max:
|
||||
try: salary_max = float(s_max)
|
||||
except: pass
|
||||
|
||||
job_url = c.get("url") or page_url
|
||||
if job_url.startswith("/"):
|
||||
job_url = urljoin(page_url, job_url)
|
||||
|
||||
dept = determine_department(title, "")
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
jobs.append({
|
||||
"title": title,
|
||||
"company": company,
|
||||
"location": location_name,
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.",
|
||||
"salary_min": salary_min,
|
||||
"salary_max": salary_max,
|
||||
"job_url": job_url,
|
||||
"source": "web_direct"
|
||||
})
|
||||
return jobs
|
||||
|
||||
def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||
from bs4 import BeautifulSoup
|
||||
jobs = []
|
||||
if not html_text:
|
||||
return jobs
|
||||
|
||||
soup = BeautifulSoup(html_text, "html.parser")
|
||||
anchors = soup.find_all("a", href=True)
|
||||
seen_links = set()
|
||||
|
||||
for a in anchors:
|
||||
href = a.get("href", "").strip()
|
||||
text = clean_html_text(a.get_text(strip=True))
|
||||
if not href or not text or len(text) < 4 or len(text) > 90:
|
||||
continue
|
||||
|
||||
is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="])
|
||||
if not is_job_href:
|
||||
continue
|
||||
|
||||
full_url = href if href.startswith("http") else urljoin(page_url, href)
|
||||
if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"):
|
||||
continue
|
||||
seen_links.add(full_url)
|
||||
|
||||
lower_text = text.lower()
|
||||
if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]):
|
||||
continue
|
||||
|
||||
dept = determine_department(text, "")
|
||||
exp_level = determine_experience_level(text)
|
||||
is_remote = "remote" in lower_text
|
||||
|
||||
jobs.append({
|
||||
"title": text,
|
||||
"company": default_company,
|
||||
"location": "Remote, USA" if is_remote else "United States",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}",
|
||||
"salary_min": None,
|
||||
"salary_max": None,
|
||||
"job_url": full_url,
|
||||
"source": "web_direct"
|
||||
})
|
||||
if len(jobs) >= 50:
|
||||
break
|
||||
|
||||
return jobs
|
||||
|
||||
def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]:
|
||||
try:
|
||||
r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8, allow_redirects=True)
|
||||
if r.status_code != 200:
|
||||
return []
|
||||
html_text = r.text
|
||||
# 1. Try Schema.org JSON-LD first (highest quality structured jobs)
|
||||
jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company)
|
||||
if jobs:
|
||||
print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}")
|
||||
return jobs
|
||||
# 2. Try HTML job links
|
||||
jobs = self.extract_html_job_links(html_text, r.url, default_company)
|
||||
if jobs:
|
||||
print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}")
|
||||
return jobs
|
||||
except Exception as e:
|
||||
print(f"[Smart Crawler Warning] Scraping {url} failed: {e}")
|
||||
return []
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue