diff --git a/scraper/db.py b/scraper/db.py index e64bf86..8ab890e 100644 --- a/scraper/db.py +++ b/scraper/db.py @@ -111,6 +111,27 @@ def _upsert_sqlite(jobs_list: list): seen_hashes = set() records = [] + # 1. First ensure all distinct companies exist in Company table + distinct_companies = {} + for j in jobs_list: + c_name = (j.get("company") or "").strip() + c_loc = (j.get("location") or "").strip() + if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]: + if c_name not in distinct_companies: + distinct_companies[c_name] = c_loc + + for c_name, c_loc in distinct_companies.items(): + comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20] + cursor.execute(""" + INSERT OR IGNORE INTO Company ( + id, name, location, verificationStatus, trustStatus, createdAt, updatedAt + ) VALUES (?, ?, ?, 'UNCLAIMED', 'UNVERIFIED', ?, ?) + """, (comp_id, c_name, c_loc or "USA", now_iso, now_iso)) + + # Fetch mapping of company name to id + cursor.execute("SELECT id, name FROM Company") + comp_map = {row[1]: row[0] for row in cursor.fetchall()} + for j in jobs_list: job_url = j.get("job_url", "") if not job_url: @@ -120,12 +141,14 @@ def _upsert_sqlite(jobs_list: list): continue seen_hashes.add(job_hash) job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20] + c_name = j.get("company", "Unknown Company")[:255] + c_id = comp_map.get(c_name) records.append(( job_id, job_hash, j.get("title", "Untitled Position")[:255], - j.get("company", "Unknown Company")[:255], + c_name, j.get("location", "Not Specified")[:255], 1 if j.get("is_remote") else 0, j.get("department") or "Other", @@ -136,10 +159,32 @@ def _upsert_sqlite(jobs_list: list): job_url, j.get("source", "jobspy"), format_iso(j.get("date_posted")), + c_id, now_iso, now_iso )) + query = """ + INSERT INTO Job ( + id, jobUrlHash, title, company, location, isRemote, + department, experienceLevel, description, salaryMin, salaryMax, jobUrl, source, datePosted, + companyId, createdAt, updatedAt + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT (jobUrlHash) DO UPDATE SET + title = excluded.title, + company = excluded.company, + location = excluded.location, + isRemote = excluded.isRemote, + department = COALESCE(excluded.department, Job.department), + experienceLevel = COALESCE(excluded.experienceLevel, Job.experienceLevel), + description = excluded.description, + salaryMin = COALESCE(excluded.salaryMin, Job.salaryMin), + salaryMax = COALESCE(excluded.salaryMax, Job.salaryMax), + datePosted = COALESCE(excluded.datePosted, Job.datePosted), + companyId = COALESCE(excluded.companyId, Job.companyId), + updatedAt = excluded.updatedAt; + """ + try: cursor.executemany(query, records) conn.commit() @@ -190,11 +235,58 @@ def _upsert_postgres(jobs_list: list, db_url: str): else: cursor = conn.cursor() + now = datetime.datetime.now(datetime.timezone.utc) + + # 1. First ensure all distinct companies exist in "Company" table + distinct_companies = {} + for j in jobs_list: + c_name = (j.get("company") or "").strip() + c_loc = (j.get("location") or "").strip() + if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]: + if c_name not in distinct_companies: + distinct_companies[c_name] = c_loc + + if distinct_companies: + company_records = [] + for c_name, c_loc in distinct_companies.items(): + comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20] + company_records.append(( + comp_id, + c_name, + c_loc or "USA", + "UNCLAIMED", + "UNVERIFIED", + now, + now + )) + comp_upsert_query = """ + INSERT INTO "Company" ("id", "name", "location", "verificationStatus", "trustStatus", "createdAt", "updatedAt") + VALUES %s + ON CONFLICT ("name") DO UPDATE SET + "location" = COALESCE("Company"."location", EXCLUDED."location"), + "updatedAt" = EXCLUDED."updatedAt"; + """ + try: + execute_values(cursor, comp_upsert_query, company_records) + conn.commit() + except Exception as comp_err: + conn.rollback() + print(f"[Postgres Warning] Company pre-population error: {comp_err}") + + # Fetch mapping of company name to id + comp_map = {} + try: + cursor.execute('SELECT "id", "name" FROM "Company"') + for row in cursor.fetchall(): + comp_map[row[1]] = row[0] + except Exception as map_err: + print(f"[Postgres Warning] Failed to fetch company map: {map_err}") + query = """ INSERT INTO "Job" ( "id", "jobUrlHash", "title", "company", "location", "isRemote", "department", "experienceLevel", "description", "salaryMin", "salaryMax", "jobUrl", "source", "datePosted", - "lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt" + "companyId", "lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt" ) VALUES %s ON CONFLICT ("jobUrlHash") DO UPDATE SET "title" = EXCLUDED."title", @@ -207,6 +299,7 @@ def _upsert_postgres(jobs_list: list, db_url: str): "salaryMin" = COALESCE(EXCLUDED."salaryMin", "Job"."salaryMin"), "salaryMax" = COALESCE(EXCLUDED."salaryMax", "Job"."salaryMax"), "datePosted" = COALESCE(EXCLUDED."datePosted", "Job"."datePosted"), + "companyId" = COALESCE(EXCLUDED."companyId", "Job"."companyId"), "lifecycleStatus" = 'ACTIVE', "lastSeenAt" = NOW(), "updatedAt" = NOW(); @@ -214,7 +307,6 @@ def _upsert_postgres(jobs_list: list, db_url: str): records = [] seen_hashes = set() - now = datetime.datetime.now(datetime.timezone.utc) for j in jobs_list: job_url = j.get("job_url", "") @@ -225,12 +317,14 @@ def _upsert_postgres(jobs_list: list, db_url: str): continue seen_hashes.add(job_hash) job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20] + c_name = j.get("company", "Unknown Company")[:255] + c_id = comp_map.get(c_name) records.append(( job_id, job_hash, j.get("title", "Untitled Position")[:255], - j.get("company", "Unknown Company")[:255], + c_name, j.get("location", "Not Specified")[:255], bool(j.get("is_remote", False)), j.get("department") or "Other", @@ -241,6 +335,7 @@ def _upsert_postgres(jobs_list: list, db_url: str): job_url, j.get("source", "jobspy"), now, + c_id, "ACTIVE", now, now, diff --git a/scraper/main.py b/scraper/main.py index 5cbb34c..9610280 100644 --- a/scraper/main.py +++ b/scraper/main.py @@ -49,12 +49,18 @@ DOMAINS_TO_PROBE = [ ("travelers.com", "Travelers Insurance"), ("hartfordhealthcare.org", "Hartford HealthCare"), ("yalehealth.org", "Yale Health"), + ("labcorp.com", "Labcorp"), + ("questdiagnostics.com", "Quest Diagnostics"), ("lockheedmartin.com", "Lockheed Martin"), ("boeing.com", "Boeing"), ("caterpillar.com", "Caterpillar"), ("deere.com", "John Deere"), ("fedex.com", "FedEx"), - ("ups.com", "UPS") + ("ups.com", "UPS"), + ("dhl.com", "DHL Express"), + ("marriott.com", "Marriott International"), + ("hilton.com", "Hilton"), + ("starbucks.com", "Starbucks") ] def run_smart_crawler_scrapes(): diff --git a/scraper/scrapers/ats_ingestion.py b/scraper/scrapers/ats_ingestion.py index a3a86f2..32f6020 100644 --- a/scraper/scrapers/ats_ingestion.py +++ b/scraper/scrapers/ats_ingestion.py @@ -88,7 +88,44 @@ GREENHOUSE_BOARDS = [ ("tempus", "Tempus Labs"), ("guardanthealth", "Guardant Health"), ("flatiron", "Flatiron Health"), - ("moderna", "Moderna") + ("moderna", "Moderna"), + ("cityblock", "Cityblock Health"), + ("hims", "Hims & Hers Health"), + ("springhealth", "Spring Health"), + ("headway", "Headway"), + ("talkspace", "Talkspace"), + ("carbonhealth", "Carbon Health"), + ("omadahealth", "Omada Health"), + ("mavenclinic", "Maven Clinic"), + ("goodrx", "GoodRx"), + ("color", "Color Health"), + ("invitae", "Invitae"), + + # Logistics, Supply Chain & Industrial + ("deliverr", "Deliverr (Logistics)"), + ("convoy", "Convoy (Freight & Logistics)"), + ("fulfill", "Fulfill.com"), + ("shipbob", "ShipBob"), + ("flockfreight", "Flock Freight"), + + # Real Estate, Property & Construction + ("compass", "Compass Real Estate"), + ("opendoor", "Opendoor"), + ("redfin", "Redfin"), + ("cbre", "CBRE"), + + # Hospitality, Food & Travel + ("sweetgreen", "Sweetgreen"), + ("instacart", "Instacart"), + ("grubhub", "Grubhub"), + ("goldbelly", "Goldbelly"), + + # Professional Services, Accounting & Legal + ("pilot", "Pilot (Bookkeeping & Tax)"), + ("bench", "Bench Accounting"), + ("brex", "Brex"), + ("gusto", "Gusto (Payroll & HR)"), + ("rippling", "Rippling (HR & Workforce)") ] LEVER_BOARDS = [ diff --git a/scraper/scrapers/jobspy_runner.py b/scraper/scrapers/jobspy_runner.py index 5f3a739..42717e7 100644 --- a/scraper/scrapers/jobspy_runner.py +++ b/scraper/scrapers/jobspy_runner.py @@ -35,24 +35,35 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]: if proxies: sites.extend(["zip_recruiter", "glassdoor"]) - # Target key employment hubs across US states + # Target key employment hubs across US states covering all industries us_regions = [ - ("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]), - ("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]), - ("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]), - ("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]), - ("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]), - ("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]), - ("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]), - ("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]), - ("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"]) + ("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]), + ("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]), + ("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]), + ("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]), + ("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]), + ("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]), + ("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]), + ("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]), + ("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"]) ] - # Nationwide Remote queries + # Nationwide Remote queries across ALL professional disciplines remote_queries = [ - "Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager", - "Customer Support", "Administrative Assistant", "IT Support Specialist", - "DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer" + # Healthcare & Medical + "Medical Biller", "Telehealth Nurse", "Clinical Research Coordinator", "Healthcare Recruiter", + # Finance, Accounting & Legal + "Staff Accountant", "Financial Analyst", "Bookkeeper", "Paralegal", "Compliance Specialist", "Underwriter", + # Sales, Marketing & Customer Support + "Customer Support Representative", "Customer Success Manager", "Account Executive", "Digital Marketing Specialist", "Content Writer", + # Human Resources & Operations + "HR Generalist", "Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Project Coordinator", + # Art, Design & Creative + "Graphic Designer", "UX Designer", "Video Editor", "Instructional Designer", + # Logistics, Supply Chain & Purchasing + "Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist", + # Tech & Engineering + "Software Engineer", "IT Support Specialist", "Data Analyst", "Systems Administrator", "DevOps Engineer" ] def _execute_scrape(site_list, term, loc, is_rem): diff --git a/web/src/app/api/companies/[id]/route.ts b/web/src/app/api/companies/[id]/route.ts index 76f7923..a3d58f0 100644 --- a/web/src/app/api/companies/[id]/route.ts +++ b/web/src/app/api/companies/[id]/route.ts @@ -13,10 +13,47 @@ export async function GET( const companyId = params.id; // 1. Fetch Company - const company = await prisma.company.findUnique({ + let company = await prisma.company.findUnique({ where: { id: companyId }, }); + if (!company) { + // Check if companyId matches a company name directly or decoded + const decodedName = decodeURIComponent(companyId); + company = await prisma.company.findFirst({ + where: { + OR: [ + { name: { equals: decodedName } }, + { name: { equals: companyId } }, + ], + }, + }); + + // If still not found, check if jobs exist with this company name and auto-create the company record + if (!company) { + const sampleJob = await prisma.job.findFirst({ + where: { + OR: [ + { company: { equals: decodedName } }, + { company: { equals: companyId } }, + ], + }, + select: { company: true, location: true }, + }); + + if (sampleJob && sampleJob.company) { + company = await prisma.company.create({ + data: { + name: sampleJob.company.trim(), + location: sampleJob.location || "USA", + verificationStatus: "UNCLAIMED", + trustStatus: "UNVERIFIED", + }, + }); + } + } + } + if (!company) { return NextResponse.json({ error: "Company not found" }, { status: 404 }); } diff --git a/web/src/app/api/companies/route.ts b/web/src/app/api/companies/route.ts index f4d06e9..385b615 100644 --- a/web/src/app/api/companies/route.ts +++ b/web/src/app/api/companies/route.ts @@ -7,8 +7,32 @@ export async function GET(req: Request) { const { searchParams } = new URL(req.url); const search = searchParams.get("search")?.trim().toLowerCase() || ""; - // 1. Query companies with approved reviews - const companies = await prisma.company.findMany({ + // 1. Fetch all distinct companies referenced in Job table to ensure full directory coverage + const jobsWithCompany = await prisma.job.findMany({ + where: { + company: { not: "" }, + }, + select: { company: true, location: true }, + }); + + const companyJobCountMap = new Map(); + const companyLocationMap = new Map(); + + for (const j of jobsWithCompany) { + if (j.company) { + const cRaw = j.company.trim(); + const cLower = cRaw.toLowerCase(); + if (cLower !== "unknown" && cLower !== "n/a" && cRaw.length >= 2) { + companyJobCountMap.set(cLower, (companyJobCountMap.get(cLower) || 0) + 1); + if (j.location && !companyLocationMap.has(cLower)) { + companyLocationMap.set(cLower, j.location); + } + } + } + } + + // 2. Query existing Company table records with approved reviews + const existingCompanies = await prisma.company.findMany({ include: { reviews: { where: { isApproved: true }, @@ -17,19 +41,55 @@ export async function GET(req: Request) { orderBy: { name: "asc" }, }); - // 2. Count open jobs per company from Job table - const jobs = await prisma.job.findMany({ - select: { company: true }, - }); + const existingNamesSet = new Set(existingCompanies.map((c) => c.name.trim().toLowerCase())); - const companyJobCountMap = new Map(); - for (const j of jobs) { - if (j.company) { - const cName = j.company.trim().toLowerCase(); - companyJobCountMap.set(cName, (companyJobCountMap.get(cName) || 0) + 1); + // 3. Find any distinct job employers not yet in Company table and auto-insert them + const missingCompanies: { name: string; location: string }[] = []; + for (const [cLower] of companyJobCountMap.entries()) { + if (!existingNamesSet.has(cLower)) { + // Find original casing + const orig = jobsWithCompany.find((j) => j.company.trim().toLowerCase() === cLower); + if (orig) { + missingCompanies.push({ + name: orig.company.trim(), + location: orig.location || "USA", + }); + } } } + if (missingCompanies.length > 0) { + try { + // Auto-create missing companies + for (const m of missingCompanies.slice(0, 100)) { + try { + await prisma.company.create({ + data: { + name: m.name, + location: m.location, + verificationStatus: "UNCLAIMED", + trustStatus: "UNVERIFIED", + }, + }); + } catch { + // Safe to ignore duplicate or race condition + } + } + } catch { + // Non-blocking + } + } + + // 4. Reload full companies list + const companies = await prisma.company.findMany({ + include: { + reviews: { + where: { isApproved: true }, + }, + }, + orderBy: { name: "asc" }, + }); + const companyList = companies .map((c) => { const approvedReviews = c.reviews; diff --git a/web/src/lib/sources/pipeline.ts b/web/src/lib/sources/pipeline.ts index c34ed7a..f5f87cf 100644 --- a/web/src/lib/sources/pipeline.ts +++ b/web/src/lib/sources/pipeline.ts @@ -89,6 +89,27 @@ export async function executeSourceIngestion( totalDuplicates++; } + // Auto-ensure Company record exists + let companyId: string | undefined = undefined; + try { + const trimmedCompany = (job.company || "").trim(); + if (trimmedCompany && trimmedCompany.length >= 2 && trimmedCompany.toLowerCase() !== "unknown") { + const comp = await prisma.company.upsert({ + where: { name: trimmedCompany }, + create: { + name: trimmedCompany, + location: job.location || "USA", + verificationStatus: "UNCLAIMED", + trustStatus: "UNVERIFIED", + }, + update: {}, + }); + companyId = comp.id; + } + } catch { + // Non-blocking if Company constraint or race occurs + } + // Create new job await prisma.job.create({ data: { @@ -96,6 +117,7 @@ export async function executeSourceIngestion( fingerprintHash: job.fingerprintHash, title: job.title, company: job.company, + companyId: companyId, location: job.location, isRemote: job.isRemote, department: job.department,