feat(scraper,web): nationwide multi-source scaling, smart careers crawler, and resume location matching
This commit is contained in:
parent
83ee7378da
commit
5792a79377
8 changed files with 591 additions and 76 deletions
|
|
@ -7,51 +7,112 @@ from scrapers.expanded_categories_ingestion import run_expanded_categories_inges
|
|||
from scrapers.major_ct_employers import run_major_ct_employers_scrape
|
||||
from scrapers.jobaps_ct import run_ct_jobaps_scrape
|
||||
from scrapers.jobspy_runner import run_jobspy_scrapes
|
||||
from scrapers.smart_careers_crawler import SmartCareersCrawler
|
||||
from db import upsert_jobs
|
||||
|
||||
# Targeted domains for smart careers auto-discovery across US
|
||||
DOMAINS_TO_PROBE = [
|
||||
("anthropic.com", "Anthropic"),
|
||||
("openai.com", "OpenAI"),
|
||||
("stripe.com", "Stripe"),
|
||||
("ramp.com", "Ramp"),
|
||||
("brex.com", "Brex"),
|
||||
("plaid.com", "Plaid"),
|
||||
("figma.com", "Figma"),
|
||||
("linear.app", "Linear"),
|
||||
("notion.so", "Notion"),
|
||||
("cursor.com", "Cursor"),
|
||||
("superhuman.com", "Superhuman"),
|
||||
("vercel.com", "Vercel"),
|
||||
("supabase.com", "Supabase"),
|
||||
("datadoghq.com", "Datadog"),
|
||||
("cloudflare.com", "Cloudflare"),
|
||||
("posthog.com", "PostHog"),
|
||||
("sentry.io", "Sentry"),
|
||||
("resend.com", "Resend"),
|
||||
("retool.com", "Retool"),
|
||||
("duolingo.com", "Duolingo"),
|
||||
("canva.com", "Canva"),
|
||||
("roblox.com", "Roblox"),
|
||||
("epicgames.com", "Epic Games"),
|
||||
("flexport.com", "Flexport")
|
||||
]
|
||||
|
||||
def run_smart_crawler_scrapes():
|
||||
print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...")
|
||||
crawler = SmartCareersCrawler()
|
||||
discovered_jobs = []
|
||||
|
||||
for domain, company_name in DOMAINS_TO_PROBE:
|
||||
try:
|
||||
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
|
||||
if provider and slug:
|
||||
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
|
||||
if provider == "greenhouse":
|
||||
jobs = crawler.fetch_greenhouse_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
elif provider == "lever":
|
||||
jobs = crawler.fetch_lever_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
elif provider == "ashby":
|
||||
jobs = crawler.fetch_ashby_board(slug, company_name)
|
||||
discovered_jobs.extend(jobs)
|
||||
except Exception as e:
|
||||
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
|
||||
|
||||
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.")
|
||||
return discovered_jobs
|
||||
|
||||
def execute_all_scrapes():
|
||||
print("\n==============================================")
|
||||
print("Starting CareerHound-Class Multi-Source Ingestion...")
|
||||
print("Starting Nationwide Multi-Source Ingestion Pipeline...")
|
||||
print("==============================================")
|
||||
|
||||
all_jobs = []
|
||||
|
||||
# 1. Expanded Legal, Education, Trades, Logistics, HR, Biotech
|
||||
# 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby)
|
||||
try:
|
||||
exp_jobs = run_expanded_categories_ingestion()
|
||||
all_jobs.extend(exp_jobs)
|
||||
crawler_jobs = run_smart_crawler_scrapes()
|
||||
all_jobs.extend(crawler_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] Expanded categories error: {e}")
|
||||
print(f"[Error] Smart Careers Crawler error: {e}")
|
||||
|
||||
# 2. Dedicated Art, Creative, Gaming & Design Ingestion
|
||||
try:
|
||||
art_jobs = run_art_and_design_ingestion()
|
||||
all_jobs.extend(art_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] Art & Design Ingestion error: {e}")
|
||||
|
||||
# 3. Direct Public ATS Board Ingestion (Greenhouse, Lever)
|
||||
# 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped)
|
||||
try:
|
||||
ats_jobs = run_ats_direct_ingestion()
|
||||
all_jobs.extend(ats_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] ATS Ingestion error: {e}")
|
||||
|
||||
# 4. Major CT Enterprise & Healthcare Employers
|
||||
# 3. Dedicated Art, Creative, Gaming & Design Ingestion
|
||||
try:
|
||||
art_jobs = run_art_and_design_ingestion()
|
||||
all_jobs.extend(art_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] Art & Design Ingestion error: {e}")
|
||||
|
||||
# 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech
|
||||
try:
|
||||
exp_jobs = run_expanded_categories_ingestion()
|
||||
all_jobs.extend(exp_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] Expanded categories error: {e}")
|
||||
|
||||
# 5. Major CT & Regional Enterprise Employers
|
||||
try:
|
||||
emp_jobs = run_major_ct_employers_scrape()
|
||||
all_jobs.extend(emp_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] Major CT Employers error: {e}")
|
||||
print(f"[Error] Major Employers error: {e}")
|
||||
|
||||
# 5. CT State JobAps Government & Public Portal
|
||||
# 6. CT State JobAps Government & Public Portal
|
||||
try:
|
||||
ct_jobs = run_ct_jobaps_scrape()
|
||||
all_jobs.extend(ct_jobs)
|
||||
except Exception as e:
|
||||
print(f"[Error] CT JobAps execution error: {e}")
|
||||
|
||||
# 6. JobSpy Local & Remote Broad Searches
|
||||
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
|
||||
try:
|
||||
jobspy_jobs = run_jobspy_scrapes()
|
||||
all_jobs.extend(jobspy_jobs)
|
||||
|
|
|
|||
|
|
@ -104,6 +104,27 @@ LEVER_BOARDS = [
|
|||
("sentry", "Sentry")
|
||||
]
|
||||
|
||||
ASHBY_BOARDS = [
|
||||
("ramp", "Ramp"),
|
||||
("openai", "OpenAI"),
|
||||
("anthropic", "Anthropic"),
|
||||
("linear", "Linear"),
|
||||
("cursor", "Cursor (Anysphere)"),
|
||||
("replit", "Replit"),
|
||||
("dust", "Dust"),
|
||||
("ironclad", "Ironclad"),
|
||||
("deel", "Deel"),
|
||||
("superhuman", "Superhuman"),
|
||||
("notion", "Notion"),
|
||||
("retell", "Retell AI"),
|
||||
("postman", "Postman"),
|
||||
("vapi", "Vapi"),
|
||||
("browserbase", "Browserbase"),
|
||||
("tavus", "Tavus"),
|
||||
("pave", "Pave"),
|
||||
("cohere", "Cohere")
|
||||
]
|
||||
|
||||
NON_US_KEYWORDS = [
|
||||
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
|
||||
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
|
||||
|
|
@ -204,7 +225,7 @@ def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
|
|||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
|
||||
for j in jobs[:30]:
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("absolute_url", "")
|
||||
if not title or not job_url:
|
||||
|
|
@ -249,7 +270,7 @@ def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
|
|||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
jobs = r.json()
|
||||
for j in jobs[:30]:
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("text", ""))
|
||||
job_url = j.get("hostedUrl", "")
|
||||
if not title or not job_url:
|
||||
|
|
@ -287,5 +308,63 @@ def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
|
|||
except Exception as e:
|
||||
print(f"[Lever Warning] {company_name} failed: {e}")
|
||||
|
||||
print("[ATS Direct] Fetching Ashby public API boards...")
|
||||
for board_slug, company_name in ASHBY_BOARDS:
|
||||
try:
|
||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
|
||||
r = requests.get(url, headers=headers, timeout=6)
|
||||
if r.status_code == 200:
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
location_name = j.get("location", "Remote, USA") or "Remote, USA"
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
|
||||
dept_text = j.get("department", "")
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
|
||||
|
||||
salary_min = None
|
||||
salary_max = None
|
||||
comp = j.get("compensation")
|
||||
if isinstance(comp, dict):
|
||||
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
|
||||
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
|
||||
if comp_min:
|
||||
try:
|
||||
salary_min = float(comp_min)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if comp_max:
|
||||
try:
|
||||
salary_max = float(comp_max)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name if location_name != "Remote" else "Remote, USA",
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
|
||||
"salary_min": salary_min,
|
||||
"salary_max": salary_max,
|
||||
"job_url": job_url,
|
||||
"source": "ashby"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Ashby Warning] {company_name} failed: {e}")
|
||||
|
||||
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
|
||||
return collected
|
||||
|
|
|
|||
|
|
@ -4,8 +4,8 @@ from typing import List, Dict, Any
|
|||
|
||||
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||
"""
|
||||
Executes JobSpy searches for CT local jobs across all industries
|
||||
and nationwide remote roles across major job categories.
|
||||
Executes JobSpy searches across nationwide US hubs and remote roles
|
||||
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
|
||||
"""
|
||||
collected_jobs = []
|
||||
|
||||
|
|
@ -15,64 +15,77 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|||
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
|
||||
return []
|
||||
|
||||
# CT local queries (all industries / general)
|
||||
ct_queries = [
|
||||
"Customer Support", "Administrative", "Healthcare", "Education",
|
||||
"Operations", "Retail", "Finance", "Software", "Engineering", "General"
|
||||
# Target key employment hubs across US states
|
||||
us_regions = [
|
||||
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
||||
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
|
||||
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
|
||||
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
|
||||
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
|
||||
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
|
||||
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
|
||||
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
|
||||
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
|
||||
]
|
||||
|
||||
# Remote queries
|
||||
# Nationwide Remote queries
|
||||
remote_queries = [
|
||||
"Customer Support", "Administrative Assistant", "IT Support",
|
||||
"Data Analyst", "Project Manager", "Software Engineer", "Marketing"
|
||||
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
|
||||
"Customer Support", "Administrative Assistant", "IT Support Specialist",
|
||||
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
||||
]
|
||||
|
||||
print("[JobSpy] Starting CT Local Scrapes...")
|
||||
for query in ct_queries:
|
||||
try:
|
||||
print(f"[JobSpy] Searching CT Local: '{query}'")
|
||||
jobs_df = scrape_jobs(
|
||||
site_name=["indeed"],
|
||||
search_term=query,
|
||||
location="Connecticut",
|
||||
results_wanted=15,
|
||||
hours_old=72,
|
||||
country_indeed='USA',
|
||||
is_remote=False
|
||||
)
|
||||
print("[JobSpy] Starting Nationwide Regional Scrapes...")
|
||||
for loc, queries in us_regions:
|
||||
for query in queries:
|
||||
try:
|
||||
print(f"[JobSpy] Searching {loc}: '{query}'")
|
||||
jobs_df = scrape_jobs(
|
||||
site_name=["indeed", "zip_recruiter"],
|
||||
search_term=query,
|
||||
location=loc,
|
||||
results_wanted=35,
|
||||
hours_old=72,
|
||||
country_indeed="USA",
|
||||
is_remote=False
|
||||
)
|
||||
|
||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||
for idx, row in jobs_df.iterrows():
|
||||
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
||||
if not job_url or job_url == "nan":
|
||||
continue
|
||||
|
||||
collected_jobs.append({
|
||||
"title": str(row.get("title", "Untitled")),
|
||||
"company": str(row.get("company", "Unknown")),
|
||||
"location": str(row.get("location", "Connecticut, USA")),
|
||||
"is_remote": False,
|
||||
"description": str(row.get("description", "") or "No description provided."),
|
||||
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
||||
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
||||
"job_url": job_url,
|
||||
"source": str(row.get("site", "jobspy")),
|
||||
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
||||
})
|
||||
time.sleep(2) # Graceful delay between queries
|
||||
except Exception as e:
|
||||
print(f"[JobSpy Warning] CT search '{query}' failed: {e}")
|
||||
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
|
||||
for idx, row in jobs_df.iterrows():
|
||||
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
|
||||
if not job_url or job_url == "nan":
|
||||
continue
|
||||
|
||||
loc_val = str(row.get("location", loc) or loc)
|
||||
if loc_val == "nan":
|
||||
loc_val = loc
|
||||
|
||||
print("[JobSpy] Starting Remote Scrapes...")
|
||||
collected_jobs.append({
|
||||
"title": str(row.get("title", "Untitled")),
|
||||
"company": str(row.get("company", "Unknown")),
|
||||
"location": loc_val,
|
||||
"is_remote": False,
|
||||
"description": str(row.get("description", "") or "No description provided."),
|
||||
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
|
||||
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
|
||||
"job_url": job_url,
|
||||
"source": str(row.get("site", "jobspy")),
|
||||
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
||||
})
|
||||
time.sleep(1.5)
|
||||
except Exception as e:
|
||||
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
|
||||
|
||||
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
|
||||
for query in remote_queries:
|
||||
try:
|
||||
print(f"[JobSpy] Searching Remote: '{query}'")
|
||||
print(f"[JobSpy] Searching US Remote: '{query}'")
|
||||
jobs_df = scrape_jobs(
|
||||
site_name=["indeed"],
|
||||
site_name=["indeed", "zip_recruiter"],
|
||||
search_term=query,
|
||||
results_wanted=15,
|
||||
results_wanted=35,
|
||||
hours_old=72,
|
||||
country_indeed='USA',
|
||||
country_indeed="USA",
|
||||
is_remote=True
|
||||
)
|
||||
|
||||
|
|
@ -94,7 +107,7 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|||
"source": str(row.get("site", "jobspy")),
|
||||
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
|
||||
})
|
||||
time.sleep(2)
|
||||
time.sleep(1.5)
|
||||
except Exception as e:
|
||||
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")
|
||||
|
||||
|
|
|
|||
190
scraper/scrapers/smart_careers_crawler.py
Normal file
190
scraper/scrapers/smart_careers_crawler.py
Normal file
|
|
@ -0,0 +1,190 @@
|
|||
import requests
|
||||
import re
|
||||
from urllib.parse import urlparse, urljoin
|
||||
from typing import Optional, Tuple, Dict, Any, List
|
||||
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level
|
||||
|
||||
COMMON_CAREER_PATHS = [
|
||||
"/careers",
|
||||
"/career",
|
||||
"/jobs",
|
||||
"/job",
|
||||
"/work-with-us",
|
||||
"/join-us",
|
||||
"/join",
|
||||
"/open-positions",
|
||||
"/vacancies"
|
||||
]
|
||||
|
||||
NON_US_KEYWORDS = [
|
||||
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
|
||||
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
|
||||
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
|
||||
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
|
||||
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
|
||||
"ireland", "dublin", "switzerland", "zurich"
|
||||
]
|
||||
|
||||
def is_valid_us_location(location_name: str, title: str = "") -> bool:
|
||||
loc_lower = (location_name or "").lower()
|
||||
title_lower = (title or "").lower()
|
||||
for non_us in NON_US_KEYWORDS:
|
||||
if non_us in loc_lower or non_us in title_lower:
|
||||
return False
|
||||
return True
|
||||
|
||||
def parse_is_us_remote(location_name: str, title: str = "") -> bool:
|
||||
loc_lower = (location_name or "").lower()
|
||||
title_lower = (title or "").lower()
|
||||
if not is_valid_us_location(location_name, title):
|
||||
return False
|
||||
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
|
||||
return bool(is_remote_mention)
|
||||
|
||||
class SmartCareersCrawler:
|
||||
"""
|
||||
Crawls company domains, finds their /careers or /jobs pages,
|
||||
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
|
||||
"""
|
||||
def __init__(self, headers: Optional[Dict[str, str]] = None):
|
||||
self.headers = headers or {
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
|
||||
}
|
||||
|
||||
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
|
||||
parsed = urlparse(url)
|
||||
host = parsed.netloc.lower()
|
||||
path = parsed.path.strip("/")
|
||||
|
||||
if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
|
||||
parts = [p for p in path.split("/") if p and p != "embed"]
|
||||
if parts:
|
||||
return "greenhouse", parts[0]
|
||||
|
||||
if "jobs.lever.co" in host:
|
||||
parts = [p for p in path.split("/") if p]
|
||||
if parts:
|
||||
return "lever", parts[0]
|
||||
|
||||
if "jobs.ashbyhq.com" in host:
|
||||
parts = [p for p in path.split("/") if p]
|
||||
if parts:
|
||||
return "ashby", parts[0]
|
||||
|
||||
if not html_text:
|
||||
return None, None
|
||||
|
||||
gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
||||
if gh_match:
|
||||
return "greenhouse", gh_match.group(1)
|
||||
|
||||
lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
||||
if lever_match:
|
||||
return "lever", lever_match.group(1)
|
||||
|
||||
ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
|
||||
if ashby_match:
|
||||
return "ashby", ashby_match.group(1)
|
||||
|
||||
return None, None
|
||||
|
||||
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
||||
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
|
||||
base_url = f"https://{clean_domain}"
|
||||
|
||||
for path in COMMON_CAREER_PATHS:
|
||||
test_url = f"{base_url}{path}"
|
||||
try:
|
||||
r = requests.get(test_url, headers=self.headers, timeout=5, allow_redirects=True)
|
||||
final_url = r.url
|
||||
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
|
||||
if provider and slug:
|
||||
return provider, slug, final_url
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
try:
|
||||
r = requests.get(base_url, headers=self.headers, timeout=5, allow_redirects=True)
|
||||
if r.status_code == 200:
|
||||
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
|
||||
if provider and slug:
|
||||
return provider, slug, r.url
|
||||
|
||||
links = re.findall(r"href=['"]([^'"]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^'"]*)['"]", r.text, re.IGNORECASE)
|
||||
for link in links[:5]:
|
||||
target = link if link.startswith("http") else urljoin(base_url, link)
|
||||
provider, slug = self.detect_ats_from_url_or_html(target)
|
||||
if provider and slug:
|
||||
return provider, slug, target
|
||||
try:
|
||||
cr = requests.get(target, headers=self.headers, timeout=5, allow_redirects=True)
|
||||
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
|
||||
if provider and slug:
|
||||
return provider, slug, cr.url
|
||||
except Exception:
|
||||
continue
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return None, None, None
|
||||
|
||||
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
|
||||
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
|
||||
collected = []
|
||||
try:
|
||||
r = requests.get(url, headers=self.headers, timeout=8)
|
||||
if r.status_code != 200:
|
||||
return collected
|
||||
data = r.json()
|
||||
jobs = data.get("jobs", [])
|
||||
for j in jobs:
|
||||
title = clean_html_text(j.get("title", ""))
|
||||
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}"
|
||||
if not title or not job_url:
|
||||
continue
|
||||
|
||||
location_name = j.get("location", "Remote, USA") or "Remote, USA"
|
||||
if not is_valid_us_location(location_name, title):
|
||||
continue
|
||||
|
||||
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
|
||||
dept_text = j.get("department", "")
|
||||
dept = determine_department(title, dept_text)
|
||||
exp_level = determine_experience_level(title)
|
||||
|
||||
description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
|
||||
|
||||
salary_min = None
|
||||
salary_max = None
|
||||
comp = j.get("compensation")
|
||||
if isinstance(comp, dict):
|
||||
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
|
||||
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
|
||||
if comp_min:
|
||||
try:
|
||||
salary_min = float(comp_min)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
if comp_max:
|
||||
try:
|
||||
salary_max = float(comp_max)
|
||||
except (ValueError, TypeError):
|
||||
pass
|
||||
|
||||
collected.append({
|
||||
"title": title,
|
||||
"company": company_name,
|
||||
"location": location_name,
|
||||
"is_remote": is_remote,
|
||||
"department": dept,
|
||||
"experience_level": exp_level,
|
||||
"description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.",
|
||||
"salary_min": salary_min,
|
||||
"salary_max": salary_max,
|
||||
"job_url": job_url,
|
||||
"source": "ashby"
|
||||
})
|
||||
except Exception as e:
|
||||
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
|
||||
return collected
|
||||
|
|
@ -16,6 +16,8 @@ export async function GET(req: Request) {
|
|||
const search = searchParams.get("search") || "";
|
||||
const remoteOnly = searchParams.get("remoteOnly") === "true";
|
||||
const ctOnly = searchParams.get("ctOnly") === "true";
|
||||
const state = searchParams.get("state") || "";
|
||||
const location = searchParams.get("location") || "";
|
||||
const source = searchParams.get("source") || "";
|
||||
const department = searchParams.get("department") || "";
|
||||
const experienceLevel = searchParams.get("experienceLevel") || "";
|
||||
|
|
@ -39,7 +41,17 @@ export async function GET(req: Request) {
|
|||
];
|
||||
}
|
||||
|
||||
if (ctOnly) {
|
||||
if (state) {
|
||||
where.AND = [
|
||||
...(where.AND || []),
|
||||
{
|
||||
OR: [
|
||||
{ location: { contains: state } },
|
||||
{ isRemote: true }
|
||||
]
|
||||
}
|
||||
];
|
||||
} else if (ctOnly) {
|
||||
where.AND = [
|
||||
...(where.AND || []),
|
||||
{
|
||||
|
|
@ -55,6 +67,18 @@ export async function GET(req: Request) {
|
|||
];
|
||||
}
|
||||
|
||||
if (location) {
|
||||
where.AND = [
|
||||
...(where.AND || []),
|
||||
{
|
||||
OR: [
|
||||
{ location: { contains: location } },
|
||||
{ isRemote: true }
|
||||
]
|
||||
}
|
||||
];
|
||||
}
|
||||
|
||||
if (source) {
|
||||
where.source = source;
|
||||
}
|
||||
|
|
@ -219,7 +243,7 @@ export async function GET(req: Request) {
|
|||
const processedJobs = jobs.map((job) => {
|
||||
const interaction = userId && (job as any).interactions?.[0];
|
||||
const matchScore = hasActiveResume && activeResumeData
|
||||
? calculateMatchScore(activeResumeData, job.title, job.description)
|
||||
? calculateMatchScore(activeResumeData, job.title, job.description, job.location, job.isRemote)
|
||||
: null;
|
||||
|
||||
return {
|
||||
|
|
|
|||
|
|
@ -12,6 +12,7 @@ const initialFilters: FilterState = {
|
|||
tab: "all",
|
||||
remoteOnly: false,
|
||||
ctOnly: false,
|
||||
state: "",
|
||||
source: "",
|
||||
department: "",
|
||||
experienceLevel: "",
|
||||
|
|
@ -61,6 +62,7 @@ export function JobFeed() {
|
|||
const params = new URLSearchParams();
|
||||
if (filters.search) params.set("search", filters.search);
|
||||
if (filters.remoteOnly) params.set("remoteOnly", "true");
|
||||
if (filters.state) params.set("state", filters.state);
|
||||
if (filters.ctOnly) params.set("ctOnly", "true");
|
||||
if (filters.source) params.set("source", filters.source);
|
||||
if (filters.department) params.set("department", filters.department);
|
||||
|
|
@ -107,6 +109,7 @@ export function JobFeed() {
|
|||
const isFiltered = Boolean(
|
||||
filters.search ||
|
||||
filters.remoteOnly ||
|
||||
filters.state ||
|
||||
filters.ctOnly ||
|
||||
filters.source ||
|
||||
filters.department ||
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ export interface FilterState {
|
|||
tab: "all" | "saved" | "applied";
|
||||
remoteOnly: boolean;
|
||||
ctOnly: boolean;
|
||||
state: string;
|
||||
source: string;
|
||||
department: string;
|
||||
experienceLevel: string;
|
||||
|
|
@ -19,6 +20,60 @@ interface JobFiltersProps {
|
|||
isFiltered: boolean;
|
||||
}
|
||||
|
||||
const US_STATES_LIST = [
|
||||
{ code: "AL", name: "Alabama" },
|
||||
{ code: "AK", name: "Alaska" },
|
||||
{ code: "AZ", name: "Arizona" },
|
||||
{ code: "AR", name: "Arkansas" },
|
||||
{ code: "CA", name: "California" },
|
||||
{ code: "CO", name: "Colorado" },
|
||||
{ code: "CT", name: "Connecticut" },
|
||||
{ code: "DE", name: "Delaware" },
|
||||
{ code: "FL", name: "Florida" },
|
||||
{ code: "GA", name: "Georgia" },
|
||||
{ code: "HI", name: "Hawaii" },
|
||||
{ code: "ID", name: "Idaho" },
|
||||
{ code: "IL", name: "Illinois" },
|
||||
{ code: "IN", name: "Indiana" },
|
||||
{ code: "IA", name: "Iowa" },
|
||||
{ code: "KS", name: "Kansas" },
|
||||
{ code: "KY", name: "Kentucky" },
|
||||
{ code: "LA", name: "Louisiana" },
|
||||
{ code: "ME", name: "Maine" },
|
||||
{ code: "MD", name: "Maryland" },
|
||||
{ code: "MA", name: "Massachusetts" },
|
||||
{ code: "MI", name: "Michigan" },
|
||||
{ code: "MN", name: "Minnesota" },
|
||||
{ code: "MS", name: "Mississippi" },
|
||||
{ code: "MO", name: "Missouri" },
|
||||
{ code: "MT", name: "Montana" },
|
||||
{ code: "NE", name: "Nebraska" },
|
||||
{ code: "NV", name: "Nevada" },
|
||||
{ code: "NH", name: "New Hampshire" },
|
||||
{ code: "NJ", name: "New Jersey" },
|
||||
{ code: "NM", name: "New Mexico" },
|
||||
{ code: "NY", name: "New York" },
|
||||
{ code: "NC", name: "North Carolina" },
|
||||
{ code: "ND", name: "North Dakota" },
|
||||
{ code: "OH", name: "Ohio" },
|
||||
{ code: "OK", name: "Oklahoma" },
|
||||
{ code: "OR", name: "Oregon" },
|
||||
{ code: "PA", name: "Pennsylvania" },
|
||||
{ code: "RI", name: "Rhode Island" },
|
||||
{ code: "SC", name: "South Carolina" },
|
||||
{ code: "SD", name: "South Dakota" },
|
||||
{ code: "TN", name: "Tennessee" },
|
||||
{ code: "TX", name: "Texas" },
|
||||
{ code: "UT", name: "Utah" },
|
||||
{ code: "VT", name: "Vermont" },
|
||||
{ code: "VA", name: "Virginia" },
|
||||
{ code: "WA", name: "Washington" },
|
||||
{ code: "WV", name: "West Virginia" },
|
||||
{ code: "WI", name: "Wisconsin" },
|
||||
{ code: "WY", name: "Wyoming" },
|
||||
{ code: "DC", name: "Washington D.C." },
|
||||
];
|
||||
|
||||
export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFiltered }: JobFiltersProps) {
|
||||
return (
|
||||
<div className="bg-white border border-stone-300 p-4 mb-6 space-y-3.5 shadow-[2px_2px_0px_0px_rgba(231,229,228,1)]">
|
||||
|
|
@ -51,9 +106,9 @@ export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFilt
|
|||
? "bg-stone-900 text-stone-50 border-stone-900"
|
||||
: "bg-stone-50 text-stone-700 border-stone-300 hover:border-stone-900"
|
||||
}`}
|
||||
title={hasActiveResume ? "Sort postings by match score against your active resume" : "Create or upload a resume in Resume Builder to enable matching"}
|
||||
title={hasActiveResume ? "Sort postings by match score & resume location fit" : "Create or upload a resume in Resume Builder to enable smart matching"}
|
||||
>
|
||||
{filters.matchResume ? "Resume Scoring: ON" : "Resume Scoring: OFF"}
|
||||
{filters.matchResume ? "Resume & Location Fit: ON" : "Resume Scoring: OFF"}
|
||||
</button>
|
||||
</div>
|
||||
|
||||
|
|
@ -89,6 +144,21 @@ export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFilt
|
|||
|
||||
{/* Dropdown Selects */}
|
||||
<div className="flex items-center space-x-2 flex-wrap gap-y-1.5">
|
||||
{/* US State Selector */}
|
||||
<select
|
||||
value={filters.state}
|
||||
aria-label="Filter by US State"
|
||||
onChange={(e) => onChange({ state: e.target.value, ctOnly: false })}
|
||||
className="text-xs border border-stone-300 px-2.5 py-1 bg-stone-50 text-stone-800 font-sans focus:outline-none focus:border-stone-900"
|
||||
>
|
||||
<option value="">All US Locations</option>
|
||||
{US_STATES_LIST.map((s) => (
|
||||
<option key={s.code} value={s.code}>
|
||||
{s.name} ({s.code})
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
|
||||
<select
|
||||
value={filters.department}
|
||||
aria-label="Filter by Job Category"
|
||||
|
|
@ -129,7 +199,8 @@ export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFilt
|
|||
<option value="greenhouse">Greenhouse</option>
|
||||
<option value="lever">Lever</option>
|
||||
<option value="ashby">Ashby</option>
|
||||
<option value="jobaps_ct">JobAps CT</option>
|
||||
<option value="jobspy">Indeed / ZipRecruiter</option>
|
||||
<option value="jobaps_ct">State Portals</option>
|
||||
</select>
|
||||
|
||||
<label className="inline-flex items-center space-x-1.5 text-xs text-stone-800 cursor-pointer font-medium select-none pl-1">
|
||||
|
|
|
|||
|
|
@ -16,6 +16,52 @@ const STOP_WORDS = new Set([
|
|||
"you've", "your", "yours", "yourself", "yourselves", "work", "job", "experience", "role", "years"
|
||||
]);
|
||||
|
||||
const US_STATES: Record<string, string> = {
|
||||
"al": "alabama", "ak": "alaska", "az": "arizona", "ar": "arkansas", "ca": "california",
|
||||
"co": "colorado", "ct": "connecticut", "de": "delaware", "fl": "florida", "ga": "georgia",
|
||||
"hi": "hawaii", "id": "idaho", "il": "illinois", "in": "indiana", "ia": "iowa",
|
||||
"ks": "kansas", "ky": "kentucky", "la": "louisiana", "me": "maine", "md": "maryland",
|
||||
"ma": "massachusetts", "mi": "michigan", "mn": "minnesota", "ms": "mississippi", "mo": "missouri",
|
||||
"mt": "montana", "ne": "nebraska", "nv": "nevada", "nh": "new hampshire", "nj": "new jersey",
|
||||
"nm": "new mexico", "ny": "new york", "nc": "north carolina", "nd": "north dakota", "oh": "ohio",
|
||||
"ok": "oklahoma", "or": "oregon", "pa": "pennsylvania", "ri": "rhode island", "sc": "south carolina",
|
||||
"sd": "south dakota", "tn": "tennessee", "tx": "texas", "ut": "utah", "vt": "vermont",
|
||||
"va": "virginia", "wa": "washington", "wv": "west virginia", "wi": "wisconsin", "wy": "wyoming",
|
||||
"dc": "district of columbia"
|
||||
};
|
||||
|
||||
export function extractLocationEntities(locationStr: string): { stateCode?: string; stateName?: string; city?: string; isRemote: boolean } {
|
||||
if (!locationStr) return { isRemote: false };
|
||||
const lower = locationStr.toLowerCase();
|
||||
const isRemote = lower.includes("remote") || lower.includes("anywhere") || lower.includes("telecommute");
|
||||
|
||||
let stateCode: string | undefined;
|
||||
let stateName: string | undefined;
|
||||
|
||||
// Check state abbreviations (e.g. "Hartford, CT", "Austin, TX")
|
||||
const abbrMatch = locationStr.match(/\b([A-Z]{2})\b/);
|
||||
if (abbrMatch) {
|
||||
const code = abbrMatch[1].toLowerCase();
|
||||
if (US_STATES[code]) {
|
||||
stateCode = code.toUpperCase();
|
||||
stateName = US_STATES[code];
|
||||
}
|
||||
}
|
||||
|
||||
// Check full state names
|
||||
if (!stateCode) {
|
||||
for (const [code, name] of Object.entries(US_STATES)) {
|
||||
if (lower.includes(name)) {
|
||||
stateCode = code.toUpperCase();
|
||||
stateName = name;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return { stateCode, stateName, isRemote };
|
||||
}
|
||||
|
||||
function tokenize(text: string): Set<string> {
|
||||
if (!text) return new Set();
|
||||
const words = text
|
||||
|
|
@ -56,13 +102,19 @@ export interface ResumeData {
|
|||
}>;
|
||||
}
|
||||
|
||||
export function calculateMatchScore(resumeData: ResumeData, jobTitle: string, jobDescription: string): number {
|
||||
export function calculateMatchScore(
|
||||
resumeData: ResumeData,
|
||||
jobTitle: string,
|
||||
jobDescription: string,
|
||||
jobLocation?: string,
|
||||
jobIsRemote?: boolean
|
||||
): number {
|
||||
if (!resumeData || (!jobTitle && !jobDescription)) return 0;
|
||||
|
||||
const skillTokens = new Set<string>();
|
||||
if (Array.isArray(resumeData.skills)) {
|
||||
for (const skill of resumeData.skills) {
|
||||
if (typeof skill === 'string') {
|
||||
if (typeof skill === "string") {
|
||||
tokenize(skill).forEach(t => skillTokens.add(t));
|
||||
}
|
||||
}
|
||||
|
|
@ -118,7 +170,29 @@ export function calculateMatchScore(resumeData: ResumeData, jobTitle: string, jo
|
|||
if (maxPossibleScore === 0) return 0;
|
||||
|
||||
const rawScore = (skillMatches * 2.5) + (titleMatches * 3.0) + (descMatches * 1.0);
|
||||
const percentage = Math.min(Math.round((rawScore / maxPossibleScore) * 100), 99);
|
||||
let percentage = Math.min(Math.round((rawScore / maxPossibleScore) * 100), 99);
|
||||
|
||||
// Smart Location Boost & Adjustment
|
||||
const userLocStr = resumeData.profile?.location || "";
|
||||
if (userLocStr && (jobLocation || jobIsRemote !== undefined)) {
|
||||
const userLoc = extractLocationEntities(userLocStr);
|
||||
const jobLoc = extractLocationEntities(jobLocation || "");
|
||||
|
||||
const isJobRemote = jobIsRemote || jobLoc.isRemote;
|
||||
|
||||
if (isJobRemote) {
|
||||
// Remote jobs are high fit for all nationwide candidates
|
||||
percentage = Math.min(percentage + 10, 99);
|
||||
} else if (userLoc.stateCode && jobLoc.stateCode && userLoc.stateCode === jobLoc.stateCode) {
|
||||
// Same US state match boost
|
||||
percentage = Math.min(percentage + 15, 99);
|
||||
} else if (userLoc.stateName && jobLoc.stateName && userLoc.stateName === jobLoc.stateName) {
|
||||
percentage = Math.min(percentage + 15, 99);
|
||||
} else if (!userLoc.isRemote && userLoc.stateCode && jobLoc.stateCode && userLoc.stateCode !== jobLoc.stateCode) {
|
||||
// Different state and neither is remote: slight proximity reduction
|
||||
percentage = Math.max(percentage - 15, 5);
|
||||
}
|
||||
}
|
||||
|
||||
return Math.max(percentage, 5);
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue