From 0aba1993f89adfaeeaca280937f99bcb9c117d3a Mon Sep 17 00:00:00 2001 From: JobsBoard Deployer Date: Sat, 5 Sep 2026 15:53:28 -0400 Subject: [PATCH] Enhance remote categorization and prevent false remote classifications across all scrapers --- scraper/db.py | 44 ++++++++++++ scraper/fetch_real_jobs.py | 2 +- scraper/scrapers/art_and_design_ingestion.py | 14 ++-- scraper/scrapers/ats_ingestion.py | 62 +++++++++++----- scraper/scrapers/jobspy_runner.py | 49 +++++++++++-- scraper/scrapers/major_ct_employers.py | 4 +- scraper/scrapers/smart_careers_crawler.py | 75 ++++++++++++++------ 7 files changed, 194 insertions(+), 56 deletions(-) diff --git a/scraper/db.py b/scraper/db.py index 8ab890e..bb81e7c 100644 --- a/scraper/db.py +++ b/scraper/db.py @@ -187,6 +187,28 @@ def _upsert_sqlite(jobs_list: list): try: cursor.executemany(query, records) + # Self-healing sanitize step: reset any existing jobs falsely tagged as remote if location or title contains negative indicators + cursor.execute(""" + UPDATE Job + SET isRemote = 0 + WHERE isRemote = 1 + AND ( + LOWER(location) LIKE '%hybrid%' + OR LOWER(location) LIKE '%onsite%' + OR LOWER(location) LIKE '%on-site%' + OR LOWER(location) LIKE '%not remote%' + OR LOWER(location) LIKE '%in-office%' + OR LOWER(location) LIKE '%in office%' + OR LOWER(title) LIKE '%hybrid%' + OR LOWER(title) LIKE '%onsite%' + OR LOWER(title) LIKE '%on-site%' + OR LOWER(title) LIKE '%not remote%' + ) + AND LOWER(location) NOT LIKE '%100% remote%' + AND LOWER(location) NOT LIKE '%fully remote%' + AND LOWER(title) NOT LIKE '%100% remote%' + AND LOWER(title) NOT LIKE '%fully remote%'; + """) conn.commit() count = len(records) cursor.close() @@ -344,6 +366,28 @@ def _upsert_postgres(jobs_list: list, db_url: str): try: execute_values(cursor, query, records) + # Self-healing sanitize step: reset any existing jobs falsely tagged as remote if location or title contains negative indicators + cursor.execute(""" + UPDATE "Job" + SET "isRemote" = FALSE + WHERE "isRemote" = TRUE + AND ( + LOWER("location") LIKE '%hybrid%' + OR LOWER("location") LIKE '%onsite%' + OR LOWER("location") LIKE '%on-site%' + OR LOWER("location") LIKE '%not remote%' + OR LOWER("location") LIKE '%in-office%' + OR LOWER("location") LIKE '%in office%' + OR LOWER("title") LIKE '%hybrid%' + OR LOWER("title") LIKE '%onsite%' + OR LOWER("title") LIKE '%on-site%' + OR LOWER("title") LIKE '%not remote%' + ) + AND LOWER("location") NOT LIKE '%100% remote%' + AND LOWER("location") NOT LIKE '%fully remote%' + AND LOWER("title") NOT LIKE '%100% remote%' + AND LOWER("title") NOT LIKE '%fully remote%'; + """) conn.commit() count = len(records) cursor.close() diff --git a/scraper/fetch_real_jobs.py b/scraper/fetch_real_jobs.py index a599949..817a9cf 100644 --- a/scraper/fetch_real_jobs.py +++ b/scraper/fetch_real_jobs.py @@ -54,7 +54,7 @@ def fetch_ct_jobaps_jobs(): "title": title, "company": f"State of CT — {agency}", "location": location, - "is_remote": "remote" in title.lower() or "hybrid" in title.lower(), + "is_remote": "remote" in title.lower() and "hybrid" not in title.lower(), "description": desc, "salary_min": 65000 if "director" in title.lower() or "chief" in title.lower() else (52000 if "trainee" in title.lower() else 72000), "salary_max": 135000 if "director" in title.lower() or "chief" in title.lower() else (70000 if "trainee" in title.lower() else 105000), diff --git a/scraper/scrapers/art_and_design_ingestion.py b/scraper/scrapers/art_and_design_ingestion.py index 98365d1..612a99d 100644 --- a/scraper/scrapers/art_and_design_ingestion.py +++ b/scraper/scrapers/art_and_design_ingestion.py @@ -3,7 +3,7 @@ import html import re import concurrent.futures from typing import List, Dict, Any -from scrapers.ats_ingestion import clean_html_text, determine_experience_level +from scrapers.ats_ingestion import clean_html_text, determine_experience_level, parse_is_us_remote # Specialized Art, Design, Creative, Game Studio & Media Greenhouse & Lever boards ART_DESIGN_GREENHOUSE_BOARDS = [ @@ -161,22 +161,22 @@ def fetch_art_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, Any departments = j.get("departments", []) dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" - if is_art_and_design_job(title, dept_text): location_obj = j.get("location", {}) - location_name = location_obj.get("name", "Remote") if isinstance(location_obj, dict) else str(location_obj) - is_remote = "remote" in location_name.lower() or "remote" in title.lower() + location_name = location_obj.get("name", "Remote, USA") if isinstance(location_obj, dict) else str(location_obj) - exp_level = determine_experience_level(title) content_raw = j.get("content", "") or "" desc_clean = clean_html_text(content_raw) + is_remote = parse_is_us_remote(location_name, title, desc_clean) + + exp_level = determine_experience_level(title) results.append({ "title": title, "company": company_name, - "location": location_name, + "location": location_name if location_name != "Remote" else "Remote, USA", "is_remote": is_remote, - "department": "Art & Design", + "department": "Art, Design & Creative", "experience_level": exp_level, "description": desc_clean[:2500] or f"Art & Design position at {company_name}.", "salary_min": None, diff --git a/scraper/scrapers/ats_ingestion.py b/scraper/scrapers/ats_ingestion.py index b892bcc..4f1359f 100644 --- a/scraper/scrapers/ats_ingestion.py +++ b/scraper/scrapers/ats_ingestion.py @@ -172,17 +172,43 @@ def is_valid_us_location(location_name: str, title: str = "") -> bool: combined = f"{location_name} {title}" return not bool(NON_US_REGEX.search(combined)) -def parse_is_us_remote(location_name: str, title: str) -> bool: - loc_lower = location_name.lower() - title_lower = title.lower() +NEGATIVE_REMOTE_REGEX = re.compile( + r'\b(?:not\s+remote|non-remote|no\s+remote|on-site|onsite|in-office|in\s+office|hybrid|office\s+only|relocation\s+required|must\s+report\s+to\s+office)\b', + re.IGNORECASE +) - if not is_valid_us_location(location_name, title): +POSITIVE_REMOTE_REGEX = re.compile( + r'\b(?:100%\s+remote|fully\s+remote|remote\s+only|strictly\s+remote|anywhere\s+in\s+(?:the\s+)?(?:us|usa|united states))\b', + re.IGNORECASE +) + +def parse_is_us_remote(location_name: str, title: str = "", description: str = "") -> bool: + loc = (location_name or "").strip() + tit = (title or "").strip() + desc = (description or "").strip() + + if not is_valid_us_location(loc, tit): return False - is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower + loc_lower = loc.lower() + tit_lower = tit.lower() + combined_header = f"{loc_lower} {tit_lower}" + + # Negative indicator overrides on location/title unless explicitly 100% / fully remote + if NEGATIVE_REMOTE_REGEX.search(combined_header): + if not POSITIVE_REMOTE_REGEX.search(combined_header): + return False + + # Check first 1500 chars of description for explicit on-site or hybrid mandate + if desc: + desc_start = desc[:1500].lower() + if re.search(r'\b(?:this\s+position\s+is\s+not\s+remote|not\s+a\s+remote\s+position|must\s+be\s+willing\s+to\s+work\s+on-site|requires\s+working\s+on-site|on-site\s+attendance\s+is\s+required|hybrid\s+work\s+schedule|in-person\s+attendance\s+required|must\s+commute\s+to\s+the\s+office)\b', desc_start): + return False + + is_remote_mention = bool(re.search(r'\b(?:remote|telecommute|work\s+from\s+home|virtual|anywhere)\b', combined_header)) has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"]) - return is_remote_mention and (has_us_indicator or "remote" in loc_lower) + return is_remote_mention and (has_us_indicator or "remote" in loc_lower or "anywhere" in loc_lower) def clean_html_text(raw: str) -> str: if not raw: @@ -277,16 +303,15 @@ def fetch_single_greenhouse_board(item: tuple, headers: dict) -> List[Dict[str, if not is_valid_us_location(location_name, title): continue - is_remote = parse_is_us_remote(location_name, title) + content_raw = j.get("content", "") or "" + desc_clean = clean_html_text(content_raw) + is_remote = parse_is_us_remote(location_name, title, desc_clean) departments = j.get("departments", []) dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) - content_raw = j.get("content", "") or "" - desc_clean = clean_html_text(content_raw) - results.append({ "title": title, "company": company_name, @@ -325,14 +350,13 @@ def fetch_single_lever_board(item: tuple, headers: dict) -> List[Dict[str, Any]] if not is_valid_us_location(location_name, title): continue - is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) + description_plain = j.get("descriptionPlain", "") or j.get("description", "") + desc_clean = clean_html_text(description_plain) + is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title, desc_clean) dept_text = categories.get("department", "") if isinstance(categories, dict) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) - description_plain = j.get("descriptionPlain", "") or j.get("description", "") - desc_clean = clean_html_text(description_plain) - results.append({ "title": title, "company": company_name, @@ -369,12 +393,12 @@ def fetch_single_ashby_board(item: tuple, headers: dict) -> List[Dict[str, Any]] if not is_valid_us_location(location_name, title): continue - is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) - dept_text = j.get("department", "") - dept = determine_department(title, dept_text) - exp_level = determine_experience_level(title) - desc_clean = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") + is_remote = parse_is_us_remote(location_name, title, desc_clean) if not bool(j.get("isRemote", False)) else parse_is_us_remote(location_name, title, desc_clean) or ("remote" in location_name.lower()) + if bool(j.get("isRemote", False)) and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()): + is_remote = True + else: + is_remote = parse_is_us_remote(location_name, title, desc_clean) salary_min = None salary_max = None diff --git a/scraper/scrapers/jobspy_runner.py b/scraper/scrapers/jobspy_runner.py index 19d5521..9b4a0d2 100644 --- a/scraper/scrapers/jobspy_runner.py +++ b/scraper/scrapers/jobspy_runner.py @@ -3,7 +3,7 @@ import os import time import pandas as pd from typing import List, Dict, Any -from scrapers.ats_ingestion import determine_experience_level, determine_department +from scrapers.ats_ingestion import determine_experience_level, determine_department, parse_is_us_remote def get_jobspy_proxies(): proxies = get_privado_proxy_list() @@ -138,14 +138,25 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: loc_val = loc title_str = str(row.get("title", "Untitled")) + desc_str = str(row.get("description", "") or "No description provided.") + + # Even in regional searches, verify if JobSpy returned a true remote role + raw_is_remote = row.get("is_remote") + is_remote_flag = bool(raw_is_remote) if pd.notnull(raw_is_remote) and raw_is_remote not in ["nan", "None", ""] else False + is_remote = is_remote_flag or parse_is_us_remote(loc_val, title_str, desc_str) + + # But strictly verify negative indicators + if is_remote and not parse_is_us_remote(loc_val, title_str, desc_str): + is_remote = False + collected_jobs.append({ "title": title_str, "company": str(row.get("company", "Unknown")), - "location": loc_val, - "is_remote": False, + "location": "Remote, USA" if is_remote and ("remote" in loc_val.lower() or loc_val == loc) else loc_val, + "is_remote": is_remote, "department": determine_department(title_str, ""), "experience_level": determine_experience_level(title_str), - "description": str(row.get("description", "") or "No description provided."), + "description": desc_str, "salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None, "salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None, "job_url": job_url, @@ -169,14 +180,38 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: continue title_str = str(row.get("title", "Untitled")) + desc_str = str(row.get("description", "") or "No description provided.") + loc_val = str(row.get("location", "") or "").strip() + if loc_val in ["nan", "None"]: + loc_val = "" + + # Verify remote status strictly + raw_is_remote = row.get("is_remote") + is_remote_flag = bool(raw_is_remote) if pd.notnull(raw_is_remote) and raw_is_remote not in ["nan", "None", ""] else True + + # Pass through our strict parse_is_us_remote validator + header_loc = loc_val or "Remote, USA" + is_truly_remote = parse_is_us_remote(header_loc, title_str, desc_str) + + # If the returned location is clearly a physical location (e.g. "Sacramento, CA") and not remote + if not is_truly_remote and not is_remote_flag: + resolved_loc = loc_val or "United States" + is_remote = False + elif not is_truly_remote and loc_val and ("remote" not in loc_val.lower()): + resolved_loc = loc_val + is_remote = False + else: + resolved_loc = loc_val if ("remote" in loc_val.lower()) else ("Remote, USA" if not loc_val else f"{loc_val} (Remote)") + is_remote = is_truly_remote + collected_jobs.append({ "title": title_str, "company": str(row.get("company", "Unknown")), - "location": "Remote, USA", - "is_remote": True, + "location": resolved_loc, + "is_remote": is_remote, "department": determine_department(title_str, ""), "experience_level": determine_experience_level(title_str), - "description": str(row.get("description", "") or "No description provided."), + "description": desc_str, "salary_min": float(row.get("min_amount")) if pd.notnull(row.get("min_amount")) else None, "salary_max": float(row.get("max_amount")) if pd.notnull(row.get("max_amount")) else None, "job_url": job_url, diff --git a/scraper/scrapers/major_ct_employers.py b/scraper/scrapers/major_ct_employers.py index 5cbfe08..f16bcff 100644 --- a/scraper/scrapers/major_ct_employers.py +++ b/scraper/scrapers/major_ct_employers.py @@ -10,7 +10,7 @@ def run_major_ct_employers_scrape() -> List[Dict[Any, Any]]: "title": "Senior Cloud Software Engineer (Full Stack)", "company": "Travelers Insurance", "location": "Hartford, CT (Hybrid)", - "is_remote": True, + "is_remote": False, "department": "Engineering", "experience_level": "Senior", "description": "Design and construct core cloud platform services using TypeScript, Next.js, Node.js, microservices, and PostgreSQL for Travelers digital insurance systems.", @@ -49,7 +49,7 @@ def run_major_ct_employers_scrape() -> List[Dict[Any, Any]]: "title": "Cybersecurity Operations & Threat Analyst", "company": "Cigna Group", "location": "Hartford, CT (Hybrid)", - "is_remote": True, + "is_remote": False, "department": "Engineering", "experience_level": "Senior", "description": "Monitor security event telemetry, execute incident response playbooks, and secure healthcare infrastructure.", diff --git a/scraper/scrapers/smart_careers_crawler.py b/scraper/scrapers/smart_careers_crawler.py index bc34cc3..7482c4a 100644 --- a/scraper/scrapers/smart_careers_crawler.py +++ b/scraper/scrapers/smart_careers_crawler.py @@ -23,13 +23,43 @@ def is_valid_us_location(location_name: str, title: str = "") -> bool: combined = f"{location_name} {title}" return not bool(NON_US_REGEX.search(combined)) -def parse_is_us_remote(location_name: str, title: str = "") -> bool: - loc_lower = (location_name or "").lower() - title_lower = (title or "").lower() - if not is_valid_us_location(location_name, title): +NEGATIVE_REMOTE_REGEX = re.compile( + r'\b(?:not\s+remote|non-remote|no\s+remote|on-site|onsite|in-office|in\s+office|hybrid|office\s+only|relocation\s+required|must\s+report\s+to\s+office)\b', + re.IGNORECASE +) + +POSITIVE_REMOTE_REGEX = re.compile( + r'\b(?:100%\s+remote|fully\s+remote|remote\s+only|strictly\s+remote|anywhere\s+in\s+(?:the\s+)?(?:us|usa|united states))\b', + re.IGNORECASE +) + +def parse_is_us_remote(location_name: str, title: str = "", description: str = "") -> bool: + loc = (location_name or "").strip() + tit = (title or "").strip() + desc = (description or "").strip() + + if not is_valid_us_location(loc, tit): return False - is_remote_mention = "remote" in loc_lower or "remote" in title_lower or "anywhere" in loc_lower - return bool(is_remote_mention) + + loc_lower = loc.lower() + tit_lower = tit.lower() + combined_header = f"{loc_lower} {tit_lower}" + + # Negative indicator overrides on location/title unless explicitly 100% / fully remote + if NEGATIVE_REMOTE_REGEX.search(combined_header): + if not POSITIVE_REMOTE_REGEX.search(combined_header): + return False + + # Check first 1500 chars of description for explicit on-site or hybrid mandate + if desc: + desc_start = desc[:1500].lower() + if re.search(r'\b(?:this\s+position\s+is\s+not\s+remote|not\s+a\s+remote\s+position|must\s+be\s+willing\s+to\s+work\s+on-site|requires\s+working\s+on-site|on-site\s+attendance\s+is\s+required|hybrid\s+work\s+schedule|in-person\s+attendance\s+required|must\s+commute\s+to\s+the\s+office)\b', desc_start): + return False + + is_remote_mention = bool(re.search(r'\b(?:remote|telecommute|work\s+from\s+home|virtual|anywhere)\b', combined_header)) + has_us_indicator = any(u in loc_lower for u in ["us", "usa", "united states", "americas", "ct", "connecticut", "ny", "new york", "ca", "texas", "tx", "fl", "florida", "various", "nationwide"]) + + return is_remote_mention and (has_us_indicator or "remote" in loc_lower or "anywhere" in loc_lower) class SmartCareersCrawler: """ @@ -173,13 +203,13 @@ class SmartCareersCrawler: if not is_valid_us_location(location_name, title): continue - is_remote = parse_is_us_remote(location_name, title) + desc_clean = clean_html_text(j.get("content", "") or "") + is_remote = parse_is_us_remote(location_name, title, desc_clean) departments = j.get("departments", []) dept_text = departments[0].get("name", "") if departments and isinstance(departments, list) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) - desc_clean = clean_html_text(j.get("content", "") or "") collected.append({ "title": title, @@ -219,14 +249,13 @@ class SmartCareersCrawler: if not is_valid_us_location(location_name, title): continue - is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title) + description_plain = j.get("descriptionPlain", "") or j.get("description", "") + desc_clean = clean_html_text(description_plain) + is_remote = parse_is_us_remote(f"{location_name} {workplace_type}", title, desc_clean) dept_text = categories.get("department", "") if isinstance(categories, dict) else "" dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) - description_plain = j.get("descriptionPlain", "") or j.get("description", "") - desc_clean = clean_html_text(description_plain) - collected.append({ "title": title, "company": company_name, @@ -263,12 +292,15 @@ class SmartCareersCrawler: if not is_valid_us_location(location_name, title): continue - is_remote = bool(j.get("isRemote", False)) or parse_is_us_remote(location_name, title) + description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") + if bool(j.get("isRemote", False)) and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()): + is_remote = True + else: + is_remote = parse_is_us_remote(location_name, title, description) + dept_text = j.get("department", "") dept = determine_department(title, dept_text) exp_level = determine_experience_level(title) - - description = clean_html_text(j.get("descriptionHtml", "") or j.get("descriptionPlain", "") or "") salary_min = None salary_max = None @@ -413,15 +445,18 @@ class SmartCareersCrawler: elif isinstance(addr, str): location_name = addr + desc_raw = c.get("description", "") + desc_clean = clean_html_text(desc_raw) + work_location_type = str(c.get("jobLocationType", "")).upper() - is_remote = "TELECOMMUTE" in work_location_type or "remote" in location_name.lower() or "remote" in title.lower() + if "TELECOMMUTE" in work_location_type and not NEGATIVE_REMOTE_REGEX.search(f"{location_name} {title}".lower()): + is_remote = True + else: + is_remote = parse_is_us_remote(location_name, title, desc_clean) if not is_valid_us_location(location_name, title): continue - desc_raw = c.get("description", "") - desc_clean = clean_html_text(desc_raw) - salary_min = None salary_max = None base_salary = c.get("baseSalary", {}) @@ -497,7 +532,7 @@ class SmartCareersCrawler: dept = determine_department(text, "") exp_level = determine_experience_level(text) - is_remote = "remote" in low_text + is_remote = parse_is_us_remote("United States", text) jobs.append({ "title": text,