61 lines
2.4 KiB
Python
61 lines
2.4 KiB
Python
import requests
|
|
from bs4 import BeautifulSoup
|
|
from typing import List, Dict, Any
|
|
|
|
def run_ct_jobaps_scrape() -> List[Dict[Any, Any]]:
|
|
"""
|
|
Scrapes public employment announcements from the State of Connecticut JobAps portal.
|
|
"""
|
|
print("[JobAps CT] Scraper starting...")
|
|
url = "https://www.jobapscloud.com/CT/"
|
|
headers = {
|
|
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
|
}
|
|
|
|
jobs = []
|
|
try:
|
|
resp = requests.get(url, headers=headers, timeout=15)
|
|
if resp.status_code != 200:
|
|
print(f"[JobAps CT Warning] HTTP status {resp.status_code}")
|
|
return jobs
|
|
|
|
soup = BeautifulSoup(resp.text, 'html.parser')
|
|
|
|
rows = soup.find_all('tr')
|
|
for row in rows:
|
|
link_tag = row.find('a', href=True)
|
|
if not link_tag:
|
|
continue
|
|
|
|
href_val = link_tag.get('href', '')
|
|
if 'target=' in href_val or 'JobListing' in href_val or 'specs' in href_val.lower():
|
|
title = link_tag.get_text(strip=True)
|
|
full_url = href_val if href_val.startswith('http') else f"https://www.jobapscloud.com/CT/{href_val.lstrip('/')}"
|
|
|
|
cells = row.find_all('td')
|
|
agency = "State of Connecticut"
|
|
if len(cells) > 1:
|
|
agency_text = cells[1].get_text(strip=True)
|
|
if agency_text:
|
|
agency = f"State of CT - {agency_text}"
|
|
|
|
jobs.append({
|
|
"title": title or "State of CT Job Position",
|
|
"company": agency,
|
|
"location": "Hartford & CT Statewide",
|
|
"is_remote": False,
|
|
"department": "Government & Public Services",
|
|
"experience_level": "Mid-Level",
|
|
"description": f"Official State of Connecticut position. Application details available at {full_url}",
|
|
"salary_min": None,
|
|
"salary_max": None,
|
|
"job_url": full_url,
|
|
"source": "jobaps_ct",
|
|
"date_posted": None
|
|
})
|
|
|
|
print(f"[JobAps CT] Scraped {len(jobs)} state postings.")
|
|
except Exception as e:
|
|
print(f"[JobAps CT Warning] Failed to scrape: {e}")
|
|
|
|
return jobs
|