diff --git a/scraper/main.py b/scraper/main.py index 5ce0230..d4dd287 100644 --- a/scraper/main.py +++ b/scraper/main.py @@ -10,8 +10,9 @@ from scrapers.jobspy_runner import run_jobspy_scrapes from scrapers.smart_careers_crawler import SmartCareersCrawler from db import upsert_jobs -# Targeted domains for smart careers auto-discovery across US +# Targeted nationwide domains across Tech, Healthcare, Retail, Finance & Industrial DOMAINS_TO_PROBE = [ + # Top Tech & AI ("anthropic.com", "Anthropic"), ("openai.com", "OpenAI"), ("stripe.com", "Stripe"), @@ -35,17 +36,34 @@ DOMAINS_TO_PROBE = [ ("canva.com", "Canva"), ("roblox.com", "Roblox"), ("epicgames.com", "Epic Games"), - ("flexport.com", "Flexport") + ("flexport.com", "Flexport"), + + # Enterprise, Retail, Logistics & Healthcare + ("target.com", "Target"), + ("homedepot.com", "The Home Depot"), + ("costco.com", "Costco Wholesale"), + ("cvshealth.com", "CVS Health"), + ("unitedhealthgroup.com", "UnitedHealth Group"), + ("cigna.com", "The Cigna Group"), + ("travelers.com", "Travelers Insurance"), + ("hartfordhealthcare.org", "Hartford HealthCare"), + ("yalehealth.org", "Yale Health"), + ("lockheedmartin.com", "Lockheed Martin"), + ("boeing.com", "Boeing"), + ("caterpillar.com", "Caterpillar"), + ("deere.com", "John Deere"), + ("fedex.com", "FedEx"), + ("ups.com", "UPS") ] def run_smart_crawler_scrapes(): - print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...") + print("[Smart Crawler] Universal Web Crawler scanning domains for live /careers & /jobs...") crawler = SmartCareersCrawler() discovered_jobs = [] for domain, company_name in DOMAINS_TO_PROBE: try: - provider, slug, careers_url = crawler.probe_domain_for_careers(domain) + provider, slug, careers_url, html_text = crawler.probe_domain_for_careers(domain) if provider and slug: print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'") if provider == "greenhouse": @@ -57,10 +75,14 @@ def run_smart_crawler_scrapes(): elif provider == "ashby": jobs = crawler.fetch_ashby_board(slug, company_name) discovered_jobs.extend(jobs) + elif careers_url: + print(f"[Smart Crawler] Scraping native web careers page: {company_name} -> {careers_url}") + jobs = crawler.scrape_native_career_page(careers_url, company_name) + discovered_jobs.extend(jobs) except Exception as e: print(f"[Smart Crawler Warning] Probing {domain} failed: {e}") - print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.") + print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via universal web scraping.") return discovered_jobs def execute_all_scrapes(): @@ -70,7 +92,7 @@ def execute_all_scrapes(): all_jobs = [] - # 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby) + # 1. Universal Web Crawler (/careers, /jobs, Schema.org JobPosting, Greenhouse/Lever/Ashby) try: crawler_jobs = run_smart_crawler_scrapes() all_jobs.extend(crawler_jobs) @@ -98,19 +120,19 @@ def execute_all_scrapes(): except Exception as e: print(f"[Error] Expanded categories error: {e}") - # 5. Major CT & Regional Enterprise Employers + # 5. Major Regional & Enterprise Employers try: emp_jobs = run_major_ct_employers_scrape() all_jobs.extend(emp_jobs) except Exception as e: print(f"[Error] Major Employers error: {e}") - # 6. CT State JobAps Government & Public Portal + # 6. Public Sector Portals try: ct_jobs = run_ct_jobaps_scrape() all_jobs.extend(ct_jobs) except Exception as e: - print(f"[Error] CT JobAps execution error: {e}") + print(f"[Error] JobAps execution error: {e}") # 7. JobSpy Nationwide US Metros & Remote Broad Searches try: diff --git a/scraper/scrapers/smart_careers_crawler.py b/scraper/scrapers/smart_careers_crawler.py index 67f7e43..1848b71 100644 --- a/scraper/scrapers/smart_careers_crawler.py +++ b/scraper/scrapers/smart_careers_crawler.py @@ -291,3 +291,172 @@ class SmartCareersCrawler: except Exception as e: print(f"[Ashby Warning] {company_name} ({slug}) error: {e}") return collected + + def extract_schema_org_job_postings(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]: + import json + from bs4 import BeautifulSoup + jobs = [] + if not html_text: + return jobs + + soup = BeautifulSoup(html_text, "html.parser") + scripts = soup.find_all("script", type=re.compile(r"application/ld\+json", re.I)) + + for s in scripts: + try: + data = json.loads(s.string or "{}") + except Exception: + continue + + items = data if isinstance(data, list) else [data] + for item in items: + if not isinstance(item, dict): + continue + graph = item.get("@graph", []) + candidates = graph if isinstance(graph, list) and graph else [item] + + for c in candidates: + if not isinstance(c, dict): + continue + item_type = str(c.get("@type", "")).lower() + if "jobposting" in item_type: + title = clean_html_text(c.get("title", "")) + if not title: + continue + + hiring_org = c.get("hiringOrganization", {}) + company = default_company + if isinstance(hiring_org, dict): + company = hiring_org.get("name") or default_company + elif isinstance(hiring_org, str): + company = hiring_org + + job_loc = c.get("jobLocation", {}) + location_name = "Remote, USA" + if isinstance(job_loc, dict): + addr = job_loc.get("address", {}) + if isinstance(addr, dict): + city = addr.get("addressLocality", "") + region = addr.get("addressRegion", "") + country = addr.get("addressCountry", "") + parts = [p for p in [city, region, country] if p] + location_name = ", ".join(parts) or "USA" + elif isinstance(addr, str): + location_name = addr + + work_location_type = str(c.get("jobLocationType", "")).upper() + is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower() + + if not is_valid_us_location(location_name, title): + continue + + desc_raw = c.get("description", "") + desc_clean = clean_html_text(desc_raw) + + salary_min = None + salary_max = None + base_salary = c.get("baseSalary", {}) + if isinstance(base_salary, dict): + val = base_salary.get("value", {}) + if isinstance(val, dict): + s_min = val.get("minValue") or val.get("value") + s_max = val.get("maxValue") or val.get("value") + if s_min: + try: salary_min = float(s_min) + except: pass + if s_max: + try: salary_max = float(s_max) + except: pass + + job_url = c.get("url") or page_url + if job_url.startswith("/"): + job_url = urljoin(page_url, job_url) + + dept = determine_department(title, "") + exp_level = determine_experience_level(title) + + jobs.append({ + "title": title, + "company": company, + "location": location_name, + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": desc_clean[:2500] or f"Job position at {company}. Apply direct on careers portal.", + "salary_min": salary_min, + "salary_max": salary_max, + "job_url": job_url, + "source": "web_direct" + }) + return jobs + + def extract_html_job_links(self, html_text: str, page_url: str, default_company: str) -> List[Dict[str, Any]]: + from bs4 import BeautifulSoup + jobs = [] + if not html_text: + return jobs + + soup = BeautifulSoup(html_text, "html.parser") + anchors = soup.find_all("a", href=True) + seen_links = set() + + for a in anchors: + href = a.get("href", "").strip() + text = clean_html_text(a.get_text(strip=True)) + if not href or not text or len(text) < 4 or len(text) > 90: + continue + + is_job_href = any(k in href.lower() for k in ["/job/", "/jobs/", "/careers/", "/positions/", "/position/", "/opening/", "gh_jid="]) + if not is_job_href: + continue + + full_url = href if href.startswith("http") else urljoin(page_url, href) + if full_url in seen_links or full_url.rstrip("/") == page_url.rstrip("/"): + continue + seen_links.add(full_url) + + lower_text = text.lower() + if any(k in lower_text for k in ["view all", "see all", "back to", "apply now", "privacy", "terms", "learn more", "search", "cookies"]): + continue + + dept = determine_department(text, "") + exp_level = determine_experience_level(text) + is_remote = "remote" in lower_text + + jobs.append({ + "title": text, + "company": default_company, + "location": "Remote, USA" if is_remote else "United States", + "is_remote": is_remote, + "department": dept, + "experience_level": exp_level, + "description": f"Direct opening for {text} at {default_company}. Application details and qualifications available at {full_url}", + "salary_min": None, + "salary_max": None, + "job_url": full_url, + "source": "web_direct" + }) + if len(jobs) >= 50: + break + + return jobs + + def scrape_native_career_page(self, url: str, default_company: str) -> List[Dict[str, Any]]: + try: + r = requests.get(url, headers=self.headers, proxies=self.proxies, timeout=8, allow_redirects=True) + if r.status_code != 200: + return [] + html_text = r.text + # 1. Try Schema.org JSON-LD first (highest quality structured jobs) + jobs = self.extract_schema_org_job_postings(html_text, r.url, default_company) + if jobs: + print(f"[Smart Crawler] Extracted {len(jobs)} Schema.org JobPosting records from {url}") + return jobs + # 2. Try HTML job links + jobs = self.extract_html_job_links(html_text, r.url, default_company) + if jobs: + print(f"[Smart Crawler] Extracted {len(jobs)} HTML career links from {url}") + return jobs + except Exception as e: + print(f"[Smart Crawler Warning] Scraping {url} failed: {e}") + return []