import requests from bs4 import BeautifulSoup from typing import List, Dict, Any from scrapers.ats_ingestion import determine_experience_level def run_ct_jobaps_scrape() -> List[Dict[Any, Any]]: """ Scrapes public employment announcements from the State of Connecticut JobAps portal. """ print("[JobAps CT] Scraper starting...") url = "https://www.jobapscloud.com/CT/" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36" } jobs = [] try: resp = requests.get(url, headers=headers, timeout=15) if resp.status_code != 200: print(f"[JobAps CT Warning] HTTP status {resp.status_code}") return jobs soup = BeautifulSoup(resp.text, 'html.parser') rows = soup.find_all('tr') for row in rows: link_tag = row.find('a', href=True) if not link_tag: continue href_val = link_tag.get('href', '') if 'target=' in href_val or 'JobListing' in href_val or 'specs' in href_val.lower(): title = link_tag.get_text(strip=True) full_url = href_val if href_val.startswith('http') else f"https://www.jobapscloud.com/CT/{href_val.lstrip('/')}" cells = row.find_all('td') agency = "State of Connecticut" if len(cells) > 1: agency_text = cells[1].get_text(strip=True) if agency_text: agency = f"State of CT - {agency_text}" jobs.append({ "title": title or "State of CT Job Position", "company": agency, "location": "Hartford & CT Statewide", "is_remote": False, "department": "Government & Public Services", "experience_level": determine_experience_level(title), "description": f"Official State of Connecticut position. Application details available at {full_url}", "salary_min": None, "salary_max": None, "job_url": full_url, "source": "jobaps_ct", "date_posted": None }) print(f"[JobAps CT] Scraped {len(jobs)} state postings.") except Exception as e: print(f"[JobAps CT Warning] Failed to scrape: {e}") return jobs