feat(scraper,web): nationwide multi-source scaling, smart careers crawler, and resume location matching

This commit is contained in:
JobsBoard Deployer 2026-09-05 12:22:04 -04:00
parent 83ee7378da
commit 5792a79377
8 changed files with 591 additions and 76 deletions

View file

@ -7,51 +7,112 @@ from scrapers.expanded_categories_ingestion import run_expanded_categories_inges
from scrapers.major_ct_employers import run_major_ct_employers_scrape
from scrapers.jobaps_ct import run_ct_jobaps_scrape
from scrapers.jobspy_runner import run_jobspy_scrapes
from scrapers.smart_careers_crawler import SmartCareersCrawler
from db import upsert_jobs
# Targeted domains for smart careers auto-discovery across US
DOMAINS_TO_PROBE = [
("anthropic.com", "Anthropic"),
("openai.com", "OpenAI"),
("stripe.com", "Stripe"),
("ramp.com", "Ramp"),
("brex.com", "Brex"),
("plaid.com", "Plaid"),
("figma.com", "Figma"),
("linear.app", "Linear"),
("notion.so", "Notion"),
("cursor.com", "Cursor"),
("superhuman.com", "Superhuman"),
("vercel.com", "Vercel"),
("supabase.com", "Supabase"),
("datadoghq.com", "Datadog"),
("cloudflare.com", "Cloudflare"),
("posthog.com", "PostHog"),
("sentry.io", "Sentry"),
("resend.com", "Resend"),
("retool.com", "Retool"),
("duolingo.com", "Duolingo"),
("canva.com", "Canva"),
("roblox.com", "Roblox"),
("epicgames.com", "Epic Games"),
("flexport.com", "Flexport")
]
def run_smart_crawler_scrapes():
print("[Smart Crawler] Probing company domains for live /careers, /jobs & ATS endpoints...")
crawler = SmartCareersCrawler()
discovered_jobs = []
for domain, company_name in DOMAINS_TO_PROBE:
try:
provider, slug, careers_url = crawler.probe_domain_for_careers(domain)
if provider and slug:
print(f"[Smart Crawler] Discovered {company_name} ATS: {provider.upper()} -> '{slug}'")
if provider == "greenhouse":
jobs = crawler.fetch_greenhouse_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "lever":
jobs = crawler.fetch_lever_board(slug, company_name)
discovered_jobs.extend(jobs)
elif provider == "ashby":
jobs = crawler.fetch_ashby_board(slug, company_name)
discovered_jobs.extend(jobs)
except Exception as e:
print(f"[Smart Crawler Warning] Probing {domain} failed: {e}")
print(f"[Smart Crawler] Successfully gathered {len(discovered_jobs)} postings via smart domain probing.")
return discovered_jobs
def execute_all_scrapes():
print("\n==============================================")
print("Starting CareerHound-Class Multi-Source Ingestion...")
print("Starting Nationwide Multi-Source Ingestion Pipeline...")
print("==============================================")
all_jobs = []
# 1. Expanded Legal, Education, Trades, Logistics, HR, Biotech
# 1. Smart Domain Careers Crawler (/careers, /jobs, Greenhouse/Lever/Ashby)
try:
exp_jobs = run_expanded_categories_ingestion()
all_jobs.extend(exp_jobs)
crawler_jobs = run_smart_crawler_scrapes()
all_jobs.extend(crawler_jobs)
except Exception as e:
print(f"[Error] Expanded categories error: {e}")
print(f"[Error] Smart Careers Crawler error: {e}")
# 2. Dedicated Art, Creative, Gaming & Design Ingestion
try:
art_jobs = run_art_and_design_ingestion()
all_jobs.extend(art_jobs)
except Exception as e:
print(f"[Error] Art & Design Ingestion error: {e}")
# 3. Direct Public ATS Board Ingestion (Greenhouse, Lever)
# 2. Direct Public ATS Board Ingestion (Greenhouse, Lever, Ashby uncapped)
try:
ats_jobs = run_ats_direct_ingestion()
all_jobs.extend(ats_jobs)
except Exception as e:
print(f"[Error] ATS Ingestion error: {e}")
# 4. Major CT Enterprise & Healthcare Employers
# 3. Dedicated Art, Creative, Gaming & Design Ingestion
try:
art_jobs = run_art_and_design_ingestion()
all_jobs.extend(art_jobs)
except Exception as e:
print(f"[Error] Art & Design Ingestion error: {e}")
# 4. Expanded Legal, Education, Trades, Logistics, HR, Biotech
try:
exp_jobs = run_expanded_categories_ingestion()
all_jobs.extend(exp_jobs)
except Exception as e:
print(f"[Error] Expanded categories error: {e}")
# 5. Major CT & Regional Enterprise Employers
try:
emp_jobs = run_major_ct_employers_scrape()
all_jobs.extend(emp_jobs)
except Exception as e:
print(f"[Error] Major CT Employers error: {e}")
print(f"[Error] Major Employers error: {e}")
# 5. CT State JobAps Government & Public Portal
# 6. CT State JobAps Government & Public Portal
try:
ct_jobs = run_ct_jobaps_scrape()
all_jobs.extend(ct_jobs)
except Exception as e:
print(f"[Error] CT JobAps execution error: {e}")
# 6. JobSpy Local & Remote Broad Searches
# 7. JobSpy Nationwide US Metros & Remote Broad Searches
try:
jobspy_jobs = run_jobspy_scrapes()
all_jobs.extend(jobspy_jobs)

View file

@ -104,6 +104,27 @@ LEVER_BOARDS = [
("sentry", "Sentry")
]
ASHBY_BOARDS = [
("ramp", "Ramp"),
("openai", "OpenAI"),
("anthropic", "Anthropic"),
("linear", "Linear"),
("cursor", "Cursor (Anysphere)"),
("replit", "Replit"),
("dust", "Dust"),
("ironclad", "Ironclad"),
("deel", "Deel"),
("superhuman", "Superhuman"),
("notion", "Notion"),
("retell", "Retell AI"),
("postman", "Postman"),
("vapi", "Vapi"),
("browserbase", "Browserbase"),
("tavus", "Tavus"),
("pave", "Pave"),
("cohere", "Cohere")
]
NON_US_KEYWORDS = [
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
@ -204,7 +225,7 @@ def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs[:30]:
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("absolute_url", "")
if not title or not job_url:
@ -249,7 +270,7 @@ def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
jobs = r.json()
for j in jobs[:30]:
for j in jobs:
title = clean_html_text(j.get("text", ""))
job_url = j.get("hostedUrl", "")
if not title or not job_url:
@ -287,5 +308,63 @@ def run_ats_direct_ingestion() -> List[Dict[Any, Any]]:
except Exception as e:
print(f"[Lever Warning] {company_name} failed: {e}")
print("[ATS Direct] Fetching Ashby public API boards...")
for board_slug, company_name in ASHBY_BOARDS:
try:
url = f"https://api.ashbyhq.com/posting-api/job-board/{board_slug}"
r = requests.get(url, headers=headers, timeout=6)
if r.status_code == 200:
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{board_slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
dept_text = j.get("department", "")
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
collected.append({
"title": title,
"company": company_name,
"location": location_name if location_name != "Remote" else "Remote, USA",
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": desc_clean[:2500] or f"Direct posting at {company_name}. Apply on Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} failed: {e}")
print(f"[ATS Direct] Total US-filtered direct ATS postings ingested: {len(collected)}")
return collected

View file

@ -4,8 +4,8 @@ from typing import List, Dict, Any
def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"""
Executes JobSpy searches for CT local jobs across all industries
and nationwide remote roles across major job categories.
Executes JobSpy searches across nationwide US hubs and remote roles
covering major industries: Tech, Finance, Healthcare, Retail, Trades, etc.
"""
collected_jobs = []
@ -15,64 +15,77 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
print("[JobSpy] python-jobspy is not installed. Skipping JobSpy runner.")
return []
# CT local queries (all industries / general)
ct_queries = [
"Customer Support", "Administrative", "Healthcare", "Education",
"Operations", "Retail", "Finance", "Software", "Engineering", "General"
# Target key employment hubs across US states
us_regions = [
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
]
# Remote queries
# Nationwide Remote queries
remote_queries = [
"Customer Support", "Administrative Assistant", "IT Support",
"Data Analyst", "Project Manager", "Software Engineer", "Marketing"
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
"Customer Support", "Administrative Assistant", "IT Support Specialist",
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
]
print("[JobSpy] Starting CT Local Scrapes...")
for query in ct_queries:
try:
print(f"[JobSpy] Searching CT Local: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed"],
search_term=query,
location="Connecticut",
results_wanted=15,
hours_old=72,
country_indeed='USA',
is_remote=False
)
print("[JobSpy] Starting Nationwide Regional Scrapes...")
for loc, queries in us_regions:
for query in queries:
try:
print(f"[JobSpy] Searching {loc}: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed", "zip_recruiter"],
search_term=query,
location=loc,
results_wanted=35,
hours_old=72,
country_indeed="USA",
is_remote=False
)
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
collected_jobs.append({
"title": str(row.get("title", "Untitled")),
"company": str(row.get("company", "Unknown")),
"location": str(row.get("location", "Connecticut, USA")),
"is_remote": False,
"description": str(row.get("description", "") or "No description provided."),
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(2) # Graceful delay between queries
except Exception as e:
print(f"[JobSpy Warning] CT search '{query}' failed: {e}")
if isinstance(jobs_df, pd.DataFrame) and not jobs_df.empty:
for idx, row in jobs_df.iterrows():
job_url = str(row.get("job_url", "") or row.get("job_url_direct", ""))
if not job_url or job_url == "nan":
continue
loc_val = str(row.get("location", loc) or loc)
if loc_val == "nan":
loc_val = loc
print("[JobSpy] Starting Remote Scrapes...")
collected_jobs.append({
"title": str(row.get("title", "Untitled")),
"company": str(row.get("company", "Unknown")),
"location": loc_val,
"is_remote": False,
"description": str(row.get("description", "") or "No description provided."),
"salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None,
"salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None,
"job_url": job_url,
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(1.5)
except Exception as e:
print(f"[JobSpy Warning] Search '{query}' in '{loc}' failed: {e}")
print("[JobSpy] Starting US Nationwide Remote Scrapes...")
for query in remote_queries:
try:
print(f"[JobSpy] Searching Remote: '{query}'")
print(f"[JobSpy] Searching US Remote: '{query}'")
jobs_df = scrape_jobs(
site_name=["indeed"],
site_name=["indeed", "zip_recruiter"],
search_term=query,
results_wanted=15,
results_wanted=35,
hours_old=72,
country_indeed='USA',
country_indeed="USA",
is_remote=True
)
@ -94,7 +107,7 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
"source": str(row.get("site", "jobspy")),
"date_posted": str(row.get("date_posted")) if pd.notnull(row.get("date_posted")) else None
})
time.sleep(2)
time.sleep(1.5)
except Exception as e:
print(f"[JobSpy Warning] Remote search '{query}' failed: {e}")

View file

@ -0,0 +1,190 @@
import requests
import re
from urllib.parse import urlparse, urljoin
from typing import Optional, Tuple, Dict, Any, List
from scrapers.ats_ingestion import clean_html_text, determine_department, determine_experience_level
COMMON_CAREER_PATHS = [
"/careers",
"/career",
"/jobs",
"/job",
"/work-with-us",
"/join-us",
"/join",
"/open-positions",
"/vacancies"
]
NON_US_KEYWORDS = [
"london", "uk", "united kingdom", "england", "germany", "berlin", "munich",
"france", "paris", "canada", "toronto", "vancouver", "montreal", "india",
"bengaluru", "bangalore", "delhi", "singapore", "australia", "sydney",
"melbourne", "tokyo", "japan", "brazil", "sao paulo", "amsterdam", "netherlands",
"emea", "apac", "latam", "poland", "warsaw", "romania", "spain", "madrid", "barcelona",
"ireland", "dublin", "switzerland", "zurich"
]
def is_valid_us_location(location_name: str, title: str = "") -> bool:
loc_lower = (location_name or "").lower()
title_lower = (title or "").lower()
for non_us in NON_US_KEYWORDS:
if non_us in loc_lower or non_us in title_lower:
return False
return True
def parse_is_us_remote(location_name: str, title: str = "") -> bool:
loc_lower = (location_name or "").lower()
title_lower = (title or "").lower()
if not is_valid_us_location(location_name, title):
return False
is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower
return bool(is_remote_mention)
class SmartCareersCrawler:
"""
Crawls company domains, finds their /careers or /jobs pages,
and automatically detects and ingests from Greenhouse, Lever, or Ashby.
"""
def __init__(self, headers: Optional[Dict[str, str]] = None):
self.headers = headers or {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"
}
def detect_ats_from_url_or_html(self, url: str, html_text: str = "") -> Tuple[Optional[str], Optional[str]]:
parsed = urlparse(url)
host = parsed.netloc.lower()
path = parsed.path.strip("/")
if "boards.greenhouse.io" in host or "job-boards.greenhouse.io" in host:
parts = [p for p in path.split("/") if p and p != "embed"]
if parts:
return "greenhouse", parts[0]
if "jobs.lever.co" in host:
parts = [p for p in path.split("/") if p]
if parts:
return "lever", parts[0]
if "jobs.ashbyhq.com" in host:
parts = [p for p in path.split("/") if p]
if parts:
return "ashby", parts[0]
if not html_text:
return None, None
gh_match = re.search(r"boards\.greenhouse\.io\/(?:embed\/job_board\?for=|)([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if gh_match:
return "greenhouse", gh_match.group(1)
lever_match = re.search(r"jobs\.lever\.co\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if lever_match:
return "lever", lever_match.group(1)
ashby_match = re.search(r"jobs\.ashbyhq\.com\/([a-zA-Z0-9_\-]+)", html_text, re.IGNORECASE)
if ashby_match:
return "ashby", ashby_match.group(1)
return None, None
def probe_domain_for_careers(self, domain: str) -> Tuple[Optional[str], Optional[str], Optional[str]]:
clean_domain = domain.replace("https://", "").replace("http://", "").split("/")[0].strip().lower()
base_url = f"https://{clean_domain}"
for path in COMMON_CAREER_PATHS:
test_url = f"{base_url}{path}"
try:
r = requests.get(test_url, headers=self.headers, timeout=5, allow_redirects=True)
final_url = r.url
provider, slug = self.detect_ats_from_url_or_html(final_url, r.text if r.status_code == 200 else "")
if provider and slug:
return provider, slug, final_url
except Exception:
continue
try:
r = requests.get(base_url, headers=self.headers, timeout=5, allow_redirects=True)
if r.status_code == 200:
provider, slug = self.detect_ats_from_url_or_html(r.url, r.text)
if provider and slug:
return provider, slug, r.url
links = re.findall(r"href=['"]([^'"]*(?:career|jobs|work-at|join|greenhouse|lever|ashby)[^'"]*)['"]", r.text, re.IGNORECASE)
for link in links[:5]:
target = link if link.startswith("http") else urljoin(base_url, link)
provider, slug = self.detect_ats_from_url_or_html(target)
if provider and slug:
return provider, slug, target
try:
cr = requests.get(target, headers=self.headers, timeout=5, allow_redirects=True)
provider, slug = self.detect_ats_from_url_or_html(cr.url, cr.text)
if provider and slug:
return provider, slug, cr.url
except Exception:
continue
except Exception:
pass
return None, None, None
def fetch_ashby_board(self, slug: str, company_name: str) -> List[Dict[str, Any]]:
url = f"https://api.ashbyhq.com/posting-api/job-board/{slug}"
collected = []
try:
r = requests.get(url, headers=self.headers, timeout=8)
if r.status_code != 200:
return collected
data = r.json()
jobs = data.get("jobs", [])
for j in jobs:
title = clean_html_text(j.get("title", ""))
job_url = j.get("jobUrl") or j.get("applyUrl") or f"https://jobs.ashbyhq.com/{slug}/{j.get('id', '')}"
if not title or not job_url:
continue
location_name = j.get("location", "Remote, USA") or "Remote, USA"
if not is_valid_us_location(location_name, title):
continue
is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title)
dept_text = j.get("department", "")
dept = determine_department(title, dept_text)
exp_level = determine_experience_level(title)
description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "")
salary_min = None
salary_max = None
comp = j.get("compensation")
if isinstance(comp, dict):
comp_min = comp.get("compensationTierSummary", {}).get("min") or comp.get("min")
comp_max = comp.get("compensationTierSummary", {}).get("max") or comp.get("max")
if comp_min:
try:
salary_min = float(comp_min)
except (ValueError, TypeError):
pass
if comp_max:
try:
salary_max = float(comp_max)
except (ValueError, TypeError):
pass
collected.append({
"title": title,
"company": company_name,
"location": location_name,
"is_remote": is_remote,
"department": dept,
"experience_level": exp_level,
"description": description[:2500] or f"Direct posting at {company_name}. Apply via Ashby.",
"salary_min": salary_min,
"salary_max": salary_max,
"job_url": job_url,
"source": "ashby"
})
except Exception as e:
print(f"[Ashby Warning] {company_name} ({slug}) error: {e}")
return collected

View file

@ -16,6 +16,8 @@ export async function GET(req: Request) {
const search = searchParams.get("search") || "";
const remoteOnly = searchParams.get("remoteOnly") === "true";
const ctOnly = searchParams.get("ctOnly") === "true";
const state = searchParams.get("state") || "";
const location = searchParams.get("location") || "";
const source = searchParams.get("source") || "";
const department = searchParams.get("department") || "";
const experienceLevel = searchParams.get("experienceLevel") || "";
@ -39,7 +41,17 @@ export async function GET(req: Request) {
];
}
if (ctOnly) {
if (state) {
where.AND = [
...(where.AND || []),
{
OR: [
{ location: { contains: state } },
{ isRemote: true }
]
}
];
} else if (ctOnly) {
where.AND = [
...(where.AND || []),
{
@ -55,6 +67,18 @@ export async function GET(req: Request) {
];
}
if (location) {
where.AND = [
...(where.AND || []),
{
OR: [
{ location: { contains: location } },
{ isRemote: true }
]
}
];
}
if (source) {
where.source = source;
}
@ -219,7 +243,7 @@ export async function GET(req: Request) {
const processedJobs = jobs.map((job) => {
const interaction = userId && (job as any).interactions?.[0];
const matchScore = hasActiveResume && activeResumeData
? calculateMatchScore(activeResumeData, job.title, job.description)
? calculateMatchScore(activeResumeData, job.title, job.description, job.location, job.isRemote)
: null;
return {

View file

@ -12,6 +12,7 @@ const initialFilters: FilterState = {
tab: "all",
remoteOnly: false,
ctOnly: false,
state: "",
source: "",
department: "",
experienceLevel: "",
@ -61,6 +62,7 @@ export function JobFeed() {
const params = new URLSearchParams();
if (filters.search) params.set("search", filters.search);
if (filters.remoteOnly) params.set("remoteOnly", "true");
if (filters.state) params.set("state", filters.state);
if (filters.ctOnly) params.set("ctOnly", "true");
if (filters.source) params.set("source", filters.source);
if (filters.department) params.set("department", filters.department);
@ -107,6 +109,7 @@ export function JobFeed() {
const isFiltered = Boolean(
filters.search ||
filters.remoteOnly ||
filters.state ||
filters.ctOnly ||
filters.source ||
filters.department ||

View file

@ -5,6 +5,7 @@ export interface FilterState {
tab: "all" | "saved" | "applied";
remoteOnly: boolean;
ctOnly: boolean;
state: string;
source: string;
department: string;
experienceLevel: string;
@ -19,6 +20,60 @@ interface JobFiltersProps {
isFiltered: boolean;
}
const US_STATES_LIST = [
{ code: "AL", name: "Alabama" },
{ code: "AK", name: "Alaska" },
{ code: "AZ", name: "Arizona" },
{ code: "AR", name: "Arkansas" },
{ code: "CA", name: "California" },
{ code: "CO", name: "Colorado" },
{ code: "CT", name: "Connecticut" },
{ code: "DE", name: "Delaware" },
{ code: "FL", name: "Florida" },
{ code: "GA", name: "Georgia" },
{ code: "HI", name: "Hawaii" },
{ code: "ID", name: "Idaho" },
{ code: "IL", name: "Illinois" },
{ code: "IN", name: "Indiana" },
{ code: "IA", name: "Iowa" },
{ code: "KS", name: "Kansas" },
{ code: "KY", name: "Kentucky" },
{ code: "LA", name: "Louisiana" },
{ code: "ME", name: "Maine" },
{ code: "MD", name: "Maryland" },
{ code: "MA", name: "Massachusetts" },
{ code: "MI", name: "Michigan" },
{ code: "MN", name: "Minnesota" },
{ code: "MS", name: "Mississippi" },
{ code: "MO", name: "Missouri" },
{ code: "MT", name: "Montana" },
{ code: "NE", name: "Nebraska" },
{ code: "NV", name: "Nevada" },
{ code: "NH", name: "New Hampshire" },
{ code: "NJ", name: "New Jersey" },
{ code: "NM", name: "New Mexico" },
{ code: "NY", name: "New York" },
{ code: "NC", name: "North Carolina" },
{ code: "ND", name: "North Dakota" },
{ code: "OH", name: "Ohio" },
{ code: "OK", name: "Oklahoma" },
{ code: "OR", name: "Oregon" },
{ code: "PA", name: "Pennsylvania" },
{ code: "RI", name: "Rhode Island" },
{ code: "SC", name: "South Carolina" },
{ code: "SD", name: "South Dakota" },
{ code: "TN", name: "Tennessee" },
{ code: "TX", name: "Texas" },
{ code: "UT", name: "Utah" },
{ code: "VT", name: "Vermont" },
{ code: "VA", name: "Virginia" },
{ code: "WA", name: "Washington" },
{ code: "WV", name: "West Virginia" },
{ code: "WI", name: "Wisconsin" },
{ code: "WY", name: "Wyoming" },
{ code: "DC", name: "Washington D.C." },
];
export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFiltered }: JobFiltersProps) {
return (
<div className="bg-white border border-stone-300 p-4 mb-6 space-y-3.5 shadow-[2px_2px_0px_0px_rgba(231,229,228,1)]">
@ -51,9 +106,9 @@ export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFilt
? "bg-stone-900 text-stone-50 border-stone-900"
: "bg-stone-50 text-stone-700 border-stone-300 hover:border-stone-900"
}`}
title={hasActiveResume ? "Sort postings by match score against your active resume" : "Create or upload a resume in Resume Builder to enable matching"}
title={hasActiveResume ? "Sort postings by match score & resume location fit" : "Create or upload a resume in Resume Builder to enable smart matching"}
>
{filters.matchResume ? "Resume Scoring: ON" : "Resume Scoring: OFF"}
{filters.matchResume ? "Resume & Location Fit: ON" : "Resume Scoring: OFF"}
</button>
</div>
@ -89,6 +144,21 @@ export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFilt
{/* Dropdown Selects */}
<div className="flex items-center space-x-2 flex-wrap gap-y-1.5">
{/* US State Selector */}
<select
value={filters.state}
aria-label="Filter by US State"
onChange={(e) => onChange({ state: e.target.value, ctOnly: false })}
className="text-xs border border-stone-300 px-2.5 py-1 bg-stone-50 text-stone-800 font-sans focus:outline-none focus:border-stone-900"
>
<option value="">All US Locations</option>
{US_STATES_LIST.map((s) => (
<option key={s.code} value={s.code}>
{s.name} ({s.code})
</option>
))}
</select>
<select
value={filters.department}
aria-label="Filter by Job Category"
@ -129,7 +199,8 @@ export function JobFilters({ filters, onChange, onReset, hasActiveResume, isFilt
<option value="greenhouse">Greenhouse</option>
<option value="lever">Lever</option>
<option value="ashby">Ashby</option>
<option value="jobaps_ct">JobAps CT</option>
<option value="jobspy">Indeed / ZipRecruiter</option>
<option value="jobaps_ct">State Portals</option>
</select>
<label className="inline-flex items-center space-x-1.5 text-xs text-stone-800 cursor-pointer font-medium select-none pl-1">

View file

@ -16,6 +16,52 @@ const STOP_WORDS = new Set([
"you've", "your", "yours", "yourself", "yourselves", "work", "job", "experience", "role", "years"
]);
const US_STATES: Record<string, string> = {
"al": "alabama", "ak": "alaska", "az": "arizona", "ar": "arkansas", "ca": "california",
"co": "colorado", "ct": "connecticut", "de": "delaware", "fl": "florida", "ga": "georgia",
"hi": "hawaii", "id": "idaho", "il": "illinois", "in": "indiana", "ia": "iowa",
"ks": "kansas", "ky": "kentucky", "la": "louisiana", "me": "maine", "md": "maryland",
"ma": "massachusetts", "mi": "michigan", "mn": "minnesota", "ms": "mississippi", "mo": "missouri",
"mt": "montana", "ne": "nebraska", "nv": "nevada", "nh": "new hampshire", "nj": "new jersey",
"nm": "new mexico", "ny": "new york", "nc": "north carolina", "nd": "north dakota", "oh": "ohio",
"ok": "oklahoma", "or": "oregon", "pa": "pennsylvania", "ri": "rhode island", "sc": "south carolina",
"sd": "south dakota", "tn": "tennessee", "tx": "texas", "ut": "utah", "vt": "vermont",
"va": "virginia", "wa": "washington", "wv": "west virginia", "wi": "wisconsin", "wy": "wyoming",
"dc": "district of columbia"
};
export function extractLocationEntities(locationStr: string): { stateCode?: string; stateName?: string; city?: string; isRemote: boolean } {
if (!locationStr) return { isRemote: false };
const lower = locationStr.toLowerCase();
const isRemote = lower.includes("remote") || lower.includes("anywhere") || lower.includes("telecommute");
let stateCode: string | undefined;
let stateName: string | undefined;
// Check state abbreviations (e.g. "Hartford, CT", "Austin, TX")
const abbrMatch = locationStr.match(/\b([A-Z]{2})\b/);
if (abbrMatch) {
const code = abbrMatch[1].toLowerCase();
if (US_STATES[code]) {
stateCode = code.toUpperCase();
stateName = US_STATES[code];
}
}
// Check full state names
if (!stateCode) {
for (const [code, name] of Object.entries(US_STATES)) {
if (lower.includes(name)) {
stateCode = code.toUpperCase();
stateName = name;
break;
}
}
}
return { stateCode, stateName, isRemote };
}
function tokenize(text: string): Set<string> {
if (!text) return new Set();
const words = text
@ -56,13 +102,19 @@ export interface ResumeData {
}>;
}
export function calculateMatchScore(resumeData: ResumeData, jobTitle: string, jobDescription: string): number {
export function calculateMatchScore(
resumeData: ResumeData,
jobTitle: string,
jobDescription: string,
jobLocation?: string,
jobIsRemote?: boolean
): number {
if (!resumeData || (!jobTitle && !jobDescription)) return 0;
const skillTokens = new Set<string>();
if (Array.isArray(resumeData.skills)) {
for (const skill of resumeData.skills) {
if (typeof skill === 'string') {
if (typeof skill === "string") {
tokenize(skill).forEach(t => skillTokens.add(t));
}
}
@ -118,7 +170,29 @@ export function calculateMatchScore(resumeData: ResumeData, jobTitle: string, jo
if (maxPossibleScore === 0) return 0;
const rawScore = (skillMatches * 2.5) + (titleMatches * 3.0) + (descMatches * 1.0);
const percentage = Math.min(Math.round((rawScore / maxPossibleScore) * 100), 99);
let percentage = Math.min(Math.round((rawScore / maxPossibleScore) * 100), 99);
// Smart Location Boost & Adjustment
const userLocStr = resumeData.profile?.location || "";
if (userLocStr && (jobLocation || jobIsRemote !== undefined)) {
const userLoc = extractLocationEntities(userLocStr);
const jobLoc = extractLocationEntities(jobLocation || "");
const isJobRemote = jobIsRemote || jobLoc.isRemote;
if (isJobRemote) {
// Remote jobs are high fit for all nationwide candidates
percentage = Math.min(percentage + 10, 99);
} else if (userLoc.stateCode && jobLoc.stateCode && userLoc.stateCode === jobLoc.stateCode) {
// Same US state match boost
percentage = Math.min(percentage + 15, 99);
} else if (userLoc.stateName && jobLoc.stateName && userLoc.stateName === jobLoc.stateName) {
percentage = Math.min(percentage + 15, 99);
} else if (!userLoc.isRemote && userLoc.stateCode && jobLoc.stateCode && userLoc.stateCode !== jobLoc.stateCode) {
// Different state and neither is remote: slight proximity reduction
percentage = Math.max(percentage - 15, 5);
}
}
return Math.max(percentage, 5);
}