JB/scraper/scrapers/jobaps_ct.py

61 lines
2.4 KiB
Python

import requests
from bs4 import BeautifulSoup
from typing import List, Dict, Any
def run_ct_jobaps_scrape() -> List[Dict[Any, Any]]:
"""
Scrapes public employment announcements from the State of Connecticut JobAps portal.
"""
print("[JobAps CT] Scraper starting...")
url = "https://www.jobapscloud.com/CT/"
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}
jobs = []
try:
resp = requests.get(url, headers=headers, timeout=15)
if resp.status_code != 200:
print(f"[JobAps CT Warning] HTTP status {resp.status_code}")
return jobs
soup = BeautifulSoup(resp.text, 'html.parser')
rows = soup.find_all('tr')
for row in rows:
link_tag = row.find('a', href=True)
if not link_tag:
continue
href_val = link_tag.get('href', '')
if 'target=' in href_val or 'JobListing' in href_val or 'specs' in href_val.lower():
title = link_tag.get_text(strip=True)
full_url = href_val if href_val.startswith('http') else f"https://www.jobapscloud.com/CT/{href_val.lstrip('/')}"
cells = row.find_all('td')
agency = "State of Connecticut"
if len(cells) > 1:
agency_text = cells[1].get_text(strip=True)
if agency_text:
agency = f"State of CT - {agency_text}"
jobs.append({
"title": title or "State of CT Job Position",
"company": agency,
"location": "Hartford & CT Statewide",
"is_remote": False,
"department": "Government & Public Services",
"experience_level": "Mid-Level",
"description": f"Official State of Connecticut position. Application details available at {full_url}",
"salary_min": None,
"salary_max": None,
"job_url": full_url,
"source": "jobaps_ct",
"date_posted": None
})
print(f"[JobAps CT] Scraped {len(jobs)} state postings.")
except Exception as e:
print(f"[JobAps CT Warning] Failed to scrape: {e}")
return jobs