Auto-populate companies from job scrapers and broaden multi-industry coverage across nationwide hubs
This commit is contained in:
parent
16efc58a5c
commit
e0ae08e023
7 changed files with 300 additions and 32 deletions
103
scraper/db.py
103
scraper/db.py
|
|
@ -111,6 +111,27 @@ def _upsert_sqlite(jobs_list: list):
|
||||||
seen_hashes = set()
|
seen_hashes = set()
|
||||||
records = []
|
records = []
|
||||||
|
|
||||||
|
# 1. First ensure all distinct companies exist in Company table
|
||||||
|
distinct_companies = {}
|
||||||
|
for j in jobs_list:
|
||||||
|
c_name = (j.get("company") or "").strip()
|
||||||
|
c_loc = (j.get("location") or "").strip()
|
||||||
|
if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]:
|
||||||
|
if c_name not in distinct_companies:
|
||||||
|
distinct_companies[c_name] = c_loc
|
||||||
|
|
||||||
|
for c_name, c_loc in distinct_companies.items():
|
||||||
|
comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20]
|
||||||
|
cursor.execute("""
|
||||||
|
INSERT OR IGNORE INTO Company (
|
||||||
|
id, name, location, verificationStatus, trustStatus, createdAt, updatedAt
|
||||||
|
) VALUES (?, ?, ?, 'UNCLAIMED', 'UNVERIFIED', ?, ?)
|
||||||
|
""", (comp_id, c_name, c_loc or "USA", now_iso, now_iso))
|
||||||
|
|
||||||
|
# Fetch mapping of company name to id
|
||||||
|
cursor.execute("SELECT id, name FROM Company")
|
||||||
|
comp_map = {row[1]: row[0] for row in cursor.fetchall()}
|
||||||
|
|
||||||
for j in jobs_list:
|
for j in jobs_list:
|
||||||
job_url = j.get("job_url", "")
|
job_url = j.get("job_url", "")
|
||||||
if not job_url:
|
if not job_url:
|
||||||
|
|
@ -120,12 +141,14 @@ def _upsert_sqlite(jobs_list: list):
|
||||||
continue
|
continue
|
||||||
seen_hashes.add(job_hash)
|
seen_hashes.add(job_hash)
|
||||||
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
|
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
|
||||||
|
c_name = j.get("company", "Unknown Company")[:255]
|
||||||
|
c_id = comp_map.get(c_name)
|
||||||
|
|
||||||
records.append((
|
records.append((
|
||||||
job_id,
|
job_id,
|
||||||
job_hash,
|
job_hash,
|
||||||
j.get("title", "Untitled Position")[:255],
|
j.get("title", "Untitled Position")[:255],
|
||||||
j.get("company", "Unknown Company")[:255],
|
c_name,
|
||||||
j.get("location", "Not Specified")[:255],
|
j.get("location", "Not Specified")[:255],
|
||||||
1 if j.get("is_remote") else 0,
|
1 if j.get("is_remote") else 0,
|
||||||
j.get("department") or "Other",
|
j.get("department") or "Other",
|
||||||
|
|
@ -136,10 +159,32 @@ def _upsert_sqlite(jobs_list: list):
|
||||||
job_url,
|
job_url,
|
||||||
j.get("source", "jobspy"),
|
j.get("source", "jobspy"),
|
||||||
format_iso(j.get("date_posted")),
|
format_iso(j.get("date_posted")),
|
||||||
|
c_id,
|
||||||
now_iso,
|
now_iso,
|
||||||
now_iso
|
now_iso
|
||||||
))
|
))
|
||||||
|
|
||||||
|
query = """
|
||||||
|
INSERT INTO Job (
|
||||||
|
id, jobUrlHash, title, company, location, isRemote,
|
||||||
|
department, experienceLevel, description, salaryMin, salaryMax, jobUrl, source, datePosted,
|
||||||
|
companyId, createdAt, updatedAt
|
||||||
|
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||||
|
ON CONFLICT (jobUrlHash) DO UPDATE SET
|
||||||
|
title = excluded.title,
|
||||||
|
company = excluded.company,
|
||||||
|
location = excluded.location,
|
||||||
|
isRemote = excluded.isRemote,
|
||||||
|
department = COALESCE(excluded.department, Job.department),
|
||||||
|
experienceLevel = COALESCE(excluded.experienceLevel, Job.experienceLevel),
|
||||||
|
description = excluded.description,
|
||||||
|
salaryMin = COALESCE(excluded.salaryMin, Job.salaryMin),
|
||||||
|
salaryMax = COALESCE(excluded.salaryMax, Job.salaryMax),
|
||||||
|
datePosted = COALESCE(excluded.datePosted, Job.datePosted),
|
||||||
|
companyId = COALESCE(excluded.companyId, Job.companyId),
|
||||||
|
updatedAt = excluded.updatedAt;
|
||||||
|
"""
|
||||||
|
|
||||||
try:
|
try:
|
||||||
cursor.executemany(query, records)
|
cursor.executemany(query, records)
|
||||||
conn.commit()
|
conn.commit()
|
||||||
|
|
@ -190,11 +235,58 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
||||||
else:
|
else:
|
||||||
cursor = conn.cursor()
|
cursor = conn.cursor()
|
||||||
|
|
||||||
|
now = datetime.datetime.now(datetime.timezone.utc)
|
||||||
|
|
||||||
|
# 1. First ensure all distinct companies exist in "Company" table
|
||||||
|
distinct_companies = {}
|
||||||
|
for j in jobs_list:
|
||||||
|
c_name = (j.get("company") or "").strip()
|
||||||
|
c_loc = (j.get("location") or "").strip()
|
||||||
|
if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]:
|
||||||
|
if c_name not in distinct_companies:
|
||||||
|
distinct_companies[c_name] = c_loc
|
||||||
|
|
||||||
|
if distinct_companies:
|
||||||
|
company_records = []
|
||||||
|
for c_name, c_loc in distinct_companies.items():
|
||||||
|
comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20]
|
||||||
|
company_records.append((
|
||||||
|
comp_id,
|
||||||
|
c_name,
|
||||||
|
c_loc or "USA",
|
||||||
|
"UNCLAIMED",
|
||||||
|
"UNVERIFIED",
|
||||||
|
now,
|
||||||
|
now
|
||||||
|
))
|
||||||
|
comp_upsert_query = """
|
||||||
|
INSERT INTO "Company" ("id", "name", "location", "verificationStatus", "trustStatus", "createdAt", "updatedAt")
|
||||||
|
VALUES %s
|
||||||
|
ON CONFLICT ("name") DO UPDATE SET
|
||||||
|
"location" = COALESCE("Company"."location", EXCLUDED."location"),
|
||||||
|
"updatedAt" = EXCLUDED."updatedAt";
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
execute_values(cursor, comp_upsert_query, company_records)
|
||||||
|
conn.commit()
|
||||||
|
except Exception as comp_err:
|
||||||
|
conn.rollback()
|
||||||
|
print(f"[Postgres Warning] Company pre-population error: {comp_err}")
|
||||||
|
|
||||||
|
# Fetch mapping of company name to id
|
||||||
|
comp_map = {}
|
||||||
|
try:
|
||||||
|
cursor.execute('SELECT "id", "name" FROM "Company"')
|
||||||
|
for row in cursor.fetchall():
|
||||||
|
comp_map[row[1]] = row[0]
|
||||||
|
except Exception as map_err:
|
||||||
|
print(f"[Postgres Warning] Failed to fetch company map: {map_err}")
|
||||||
|
|
||||||
query = """
|
query = """
|
||||||
INSERT INTO "Job" (
|
INSERT INTO "Job" (
|
||||||
"id", "jobUrlHash", "title", "company", "location", "isRemote",
|
"id", "jobUrlHash", "title", "company", "location", "isRemote",
|
||||||
"department", "experienceLevel", "description", "salaryMin", "salaryMax", "jobUrl", "source", "datePosted",
|
"department", "experienceLevel", "description", "salaryMin", "salaryMax", "jobUrl", "source", "datePosted",
|
||||||
"lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt"
|
"companyId", "lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt"
|
||||||
) VALUES %s
|
) VALUES %s
|
||||||
ON CONFLICT ("jobUrlHash") DO UPDATE SET
|
ON CONFLICT ("jobUrlHash") DO UPDATE SET
|
||||||
"title" = EXCLUDED."title",
|
"title" = EXCLUDED."title",
|
||||||
|
|
@ -207,6 +299,7 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
||||||
"salaryMin" = COALESCE(EXCLUDED."salaryMin", "Job"."salaryMin"),
|
"salaryMin" = COALESCE(EXCLUDED."salaryMin", "Job"."salaryMin"),
|
||||||
"salaryMax" = COALESCE(EXCLUDED."salaryMax", "Job"."salaryMax"),
|
"salaryMax" = COALESCE(EXCLUDED."salaryMax", "Job"."salaryMax"),
|
||||||
"datePosted" = COALESCE(EXCLUDED."datePosted", "Job"."datePosted"),
|
"datePosted" = COALESCE(EXCLUDED."datePosted", "Job"."datePosted"),
|
||||||
|
"companyId" = COALESCE(EXCLUDED."companyId", "Job"."companyId"),
|
||||||
"lifecycleStatus" = 'ACTIVE',
|
"lifecycleStatus" = 'ACTIVE',
|
||||||
"lastSeenAt" = NOW(),
|
"lastSeenAt" = NOW(),
|
||||||
"updatedAt" = NOW();
|
"updatedAt" = NOW();
|
||||||
|
|
@ -214,7 +307,6 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
||||||
|
|
||||||
records = []
|
records = []
|
||||||
seen_hashes = set()
|
seen_hashes = set()
|
||||||
now = datetime.datetime.now(datetime.timezone.utc)
|
|
||||||
|
|
||||||
for j in jobs_list:
|
for j in jobs_list:
|
||||||
job_url = j.get("job_url", "")
|
job_url = j.get("job_url", "")
|
||||||
|
|
@ -225,12 +317,14 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
||||||
continue
|
continue
|
||||||
seen_hashes.add(job_hash)
|
seen_hashes.add(job_hash)
|
||||||
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
|
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
|
||||||
|
c_name = j.get("company", "Unknown Company")[:255]
|
||||||
|
c_id = comp_map.get(c_name)
|
||||||
|
|
||||||
records.append((
|
records.append((
|
||||||
job_id,
|
job_id,
|
||||||
job_hash,
|
job_hash,
|
||||||
j.get("title", "Untitled Position")[:255],
|
j.get("title", "Untitled Position")[:255],
|
||||||
j.get("company", "Unknown Company")[:255],
|
c_name,
|
||||||
j.get("location", "Not Specified")[:255],
|
j.get("location", "Not Specified")[:255],
|
||||||
bool(j.get("is_remote", False)),
|
bool(j.get("is_remote", False)),
|
||||||
j.get("department") or "Other",
|
j.get("department") or "Other",
|
||||||
|
|
@ -241,6 +335,7 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
||||||
job_url,
|
job_url,
|
||||||
j.get("source", "jobspy"),
|
j.get("source", "jobspy"),
|
||||||
now,
|
now,
|
||||||
|
c_id,
|
||||||
"ACTIVE",
|
"ACTIVE",
|
||||||
now,
|
now,
|
||||||
now,
|
now,
|
||||||
|
|
|
||||||
|
|
@ -49,12 +49,18 @@ DOMAINS_TO_PROBE = [
|
||||||
("travelers.com", "Travelers Insurance"),
|
("travelers.com", "Travelers Insurance"),
|
||||||
("hartfordhealthcare.org", "Hartford HealthCare"),
|
("hartfordhealthcare.org", "Hartford HealthCare"),
|
||||||
("yalehealth.org", "Yale Health"),
|
("yalehealth.org", "Yale Health"),
|
||||||
|
("labcorp.com", "Labcorp"),
|
||||||
|
("questdiagnostics.com", "Quest Diagnostics"),
|
||||||
("lockheedmartin.com", "Lockheed Martin"),
|
("lockheedmartin.com", "Lockheed Martin"),
|
||||||
("boeing.com", "Boeing"),
|
("boeing.com", "Boeing"),
|
||||||
("caterpillar.com", "Caterpillar"),
|
("caterpillar.com", "Caterpillar"),
|
||||||
("deere.com", "John Deere"),
|
("deere.com", "John Deere"),
|
||||||
("fedex.com", "FedEx"),
|
("fedex.com", "FedEx"),
|
||||||
("ups.com", "UPS")
|
("ups.com", "UPS"),
|
||||||
|
("dhl.com", "DHL Express"),
|
||||||
|
("marriott.com", "Marriott International"),
|
||||||
|
("hilton.com", "Hilton"),
|
||||||
|
("starbucks.com", "Starbucks")
|
||||||
]
|
]
|
||||||
|
|
||||||
def run_smart_crawler_scrapes():
|
def run_smart_crawler_scrapes():
|
||||||
|
|
|
||||||
|
|
@ -88,7 +88,44 @@ GREENHOUSE_BOARDS = [
|
||||||
("tempus", "Tempus Labs"),
|
("tempus", "Tempus Labs"),
|
||||||
("guardanthealth", "Guardant Health"),
|
("guardanthealth", "Guardant Health"),
|
||||||
("flatiron", "Flatiron Health"),
|
("flatiron", "Flatiron Health"),
|
||||||
("moderna", "Moderna")
|
("moderna", "Moderna"),
|
||||||
|
("cityblock", "Cityblock Health"),
|
||||||
|
("hims", "Hims & Hers Health"),
|
||||||
|
("springhealth", "Spring Health"),
|
||||||
|
("headway", "Headway"),
|
||||||
|
("talkspace", "Talkspace"),
|
||||||
|
("carbonhealth", "Carbon Health"),
|
||||||
|
("omadahealth", "Omada Health"),
|
||||||
|
("mavenclinic", "Maven Clinic"),
|
||||||
|
("goodrx", "GoodRx"),
|
||||||
|
("color", "Color Health"),
|
||||||
|
("invitae", "Invitae"),
|
||||||
|
|
||||||
|
# Logistics, Supply Chain & Industrial
|
||||||
|
("deliverr", "Deliverr (Logistics)"),
|
||||||
|
("convoy", "Convoy (Freight & Logistics)"),
|
||||||
|
("fulfill", "Fulfill.com"),
|
||||||
|
("shipbob", "ShipBob"),
|
||||||
|
("flockfreight", "Flock Freight"),
|
||||||
|
|
||||||
|
# Real Estate, Property & Construction
|
||||||
|
("compass", "Compass Real Estate"),
|
||||||
|
("opendoor", "Opendoor"),
|
||||||
|
("redfin", "Redfin"),
|
||||||
|
("cbre", "CBRE"),
|
||||||
|
|
||||||
|
# Hospitality, Food & Travel
|
||||||
|
("sweetgreen", "Sweetgreen"),
|
||||||
|
("instacart", "Instacart"),
|
||||||
|
("grubhub", "Grubhub"),
|
||||||
|
("goldbelly", "Goldbelly"),
|
||||||
|
|
||||||
|
# Professional Services, Accounting & Legal
|
||||||
|
("pilot", "Pilot (Bookkeeping & Tax)"),
|
||||||
|
("bench", "Bench Accounting"),
|
||||||
|
("brex", "Brex"),
|
||||||
|
("gusto", "Gusto (Payroll & HR)"),
|
||||||
|
("rippling", "Rippling (HR & Workforce)")
|
||||||
]
|
]
|
||||||
|
|
||||||
LEVER_BOARDS = [
|
LEVER_BOARDS = [
|
||||||
|
|
|
||||||
|
|
@ -35,24 +35,35 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
||||||
if proxies:
|
if proxies:
|
||||||
sites.extend(["zip_recruiter", "glassdoor"])
|
sites.extend(["zip_recruiter", "glassdoor"])
|
||||||
|
|
||||||
# Target key employment hubs across US states
|
# Target key employment hubs across US states covering all industries
|
||||||
us_regions = [
|
us_regions = [
|
||||||
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]),
|
||||||
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
|
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]),
|
||||||
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
|
("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]),
|
||||||
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
|
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]),
|
||||||
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
|
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]),
|
||||||
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
|
("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]),
|
||||||
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
|
("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]),
|
||||||
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
|
("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]),
|
||||||
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
|
("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"])
|
||||||
]
|
]
|
||||||
|
|
||||||
# Nationwide Remote queries
|
# Nationwide Remote queries across ALL professional disciplines
|
||||||
remote_queries = [
|
remote_queries = [
|
||||||
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
|
# Healthcare & Medical
|
||||||
"Customer Support", "Administrative Assistant", "IT Support Specialist",
|
"Medical Biller", "Telehealth Nurse", "Clinical Research Coordinator", "Healthcare Recruiter",
|
||||||
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
# Finance, Accounting & Legal
|
||||||
|
"Staff Accountant", "Financial Analyst", "Bookkeeper", "Paralegal", "Compliance Specialist", "Underwriter",
|
||||||
|
# Sales, Marketing & Customer Support
|
||||||
|
"Customer Support Representative", "Customer Success Manager", "Account Executive", "Digital Marketing Specialist", "Content Writer",
|
||||||
|
# Human Resources & Operations
|
||||||
|
"HR Generalist", "Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Project Coordinator",
|
||||||
|
# Art, Design & Creative
|
||||||
|
"Graphic Designer", "UX Designer", "Video Editor", "Instructional Designer",
|
||||||
|
# Logistics, Supply Chain & Purchasing
|
||||||
|
"Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist",
|
||||||
|
# Tech & Engineering
|
||||||
|
"Software Engineer", "IT Support Specialist", "Data Analyst", "Systems Administrator", "DevOps Engineer"
|
||||||
]
|
]
|
||||||
|
|
||||||
def _execute_scrape(site_list, term, loc, is_rem):
|
def _execute_scrape(site_list, term, loc, is_rem):
|
||||||
|
|
|
||||||
|
|
@ -13,10 +13,47 @@ export async function GET(
|
||||||
const companyId = params.id;
|
const companyId = params.id;
|
||||||
|
|
||||||
// 1. Fetch Company
|
// 1. Fetch Company
|
||||||
const company = await prisma.company.findUnique({
|
let company = await prisma.company.findUnique({
|
||||||
where: { id: companyId },
|
where: { id: companyId },
|
||||||
});
|
});
|
||||||
|
|
||||||
|
if (!company) {
|
||||||
|
// Check if companyId matches a company name directly or decoded
|
||||||
|
const decodedName = decodeURIComponent(companyId);
|
||||||
|
company = await prisma.company.findFirst({
|
||||||
|
where: {
|
||||||
|
OR: [
|
||||||
|
{ name: { equals: decodedName } },
|
||||||
|
{ name: { equals: companyId } },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
// If still not found, check if jobs exist with this company name and auto-create the company record
|
||||||
|
if (!company) {
|
||||||
|
const sampleJob = await prisma.job.findFirst({
|
||||||
|
where: {
|
||||||
|
OR: [
|
||||||
|
{ company: { equals: decodedName } },
|
||||||
|
{ company: { equals: companyId } },
|
||||||
|
],
|
||||||
|
},
|
||||||
|
select: { company: true, location: true },
|
||||||
|
});
|
||||||
|
|
||||||
|
if (sampleJob && sampleJob.company) {
|
||||||
|
company = await prisma.company.create({
|
||||||
|
data: {
|
||||||
|
name: sampleJob.company.trim(),
|
||||||
|
location: sampleJob.location || "USA",
|
||||||
|
verificationStatus: "UNCLAIMED",
|
||||||
|
trustStatus: "UNVERIFIED",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if (!company) {
|
if (!company) {
|
||||||
return NextResponse.json({ error: "Company not found" }, { status: 404 });
|
return NextResponse.json({ error: "Company not found" }, { status: 404 });
|
||||||
}
|
}
|
||||||
|
|
|
||||||
|
|
@ -7,8 +7,32 @@ export async function GET(req: Request) {
|
||||||
const { searchParams } = new URL(req.url);
|
const { searchParams } = new URL(req.url);
|
||||||
const search = searchParams.get("search")?.trim().toLowerCase() || "";
|
const search = searchParams.get("search")?.trim().toLowerCase() || "";
|
||||||
|
|
||||||
// 1. Query companies with approved reviews
|
// 1. Fetch all distinct companies referenced in Job table to ensure full directory coverage
|
||||||
const companies = await prisma.company.findMany({
|
const jobsWithCompany = await prisma.job.findMany({
|
||||||
|
where: {
|
||||||
|
company: { not: "" },
|
||||||
|
},
|
||||||
|
select: { company: true, location: true },
|
||||||
|
});
|
||||||
|
|
||||||
|
const companyJobCountMap = new Map<string, number>();
|
||||||
|
const companyLocationMap = new Map<string, string>();
|
||||||
|
|
||||||
|
for (const j of jobsWithCompany) {
|
||||||
|
if (j.company) {
|
||||||
|
const cRaw = j.company.trim();
|
||||||
|
const cLower = cRaw.toLowerCase();
|
||||||
|
if (cLower !== "unknown" && cLower !== "n/a" && cRaw.length >= 2) {
|
||||||
|
companyJobCountMap.set(cLower, (companyJobCountMap.get(cLower) || 0) + 1);
|
||||||
|
if (j.location && !companyLocationMap.has(cLower)) {
|
||||||
|
companyLocationMap.set(cLower, j.location);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 2. Query existing Company table records with approved reviews
|
||||||
|
const existingCompanies = await prisma.company.findMany({
|
||||||
include: {
|
include: {
|
||||||
reviews: {
|
reviews: {
|
||||||
where: { isApproved: true },
|
where: { isApproved: true },
|
||||||
|
|
@ -17,19 +41,55 @@ export async function GET(req: Request) {
|
||||||
orderBy: { name: "asc" },
|
orderBy: { name: "asc" },
|
||||||
});
|
});
|
||||||
|
|
||||||
// 2. Count open jobs per company from Job table
|
const existingNamesSet = new Set(existingCompanies.map((c) => c.name.trim().toLowerCase()));
|
||||||
const jobs = await prisma.job.findMany({
|
|
||||||
select: { company: true },
|
|
||||||
});
|
|
||||||
|
|
||||||
const companyJobCountMap = new Map<string, number>();
|
// 3. Find any distinct job employers not yet in Company table and auto-insert them
|
||||||
for (const j of jobs) {
|
const missingCompanies: { name: string; location: string }[] = [];
|
||||||
if (j.company) {
|
for (const [cLower] of companyJobCountMap.entries()) {
|
||||||
const cName = j.company.trim().toLowerCase();
|
if (!existingNamesSet.has(cLower)) {
|
||||||
companyJobCountMap.set(cName, (companyJobCountMap.get(cName) || 0) + 1);
|
// Find original casing
|
||||||
|
const orig = jobsWithCompany.find((j) => j.company.trim().toLowerCase() === cLower);
|
||||||
|
if (orig) {
|
||||||
|
missingCompanies.push({
|
||||||
|
name: orig.company.trim(),
|
||||||
|
location: orig.location || "USA",
|
||||||
|
});
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (missingCompanies.length > 0) {
|
||||||
|
try {
|
||||||
|
// Auto-create missing companies
|
||||||
|
for (const m of missingCompanies.slice(0, 100)) {
|
||||||
|
try {
|
||||||
|
await prisma.company.create({
|
||||||
|
data: {
|
||||||
|
name: m.name,
|
||||||
|
location: m.location,
|
||||||
|
verificationStatus: "UNCLAIMED",
|
||||||
|
trustStatus: "UNVERIFIED",
|
||||||
|
},
|
||||||
|
});
|
||||||
|
} catch {
|
||||||
|
// Safe to ignore duplicate or race condition
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
// Non-blocking
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// 4. Reload full companies list
|
||||||
|
const companies = await prisma.company.findMany({
|
||||||
|
include: {
|
||||||
|
reviews: {
|
||||||
|
where: { isApproved: true },
|
||||||
|
},
|
||||||
|
},
|
||||||
|
orderBy: { name: "asc" },
|
||||||
|
});
|
||||||
|
|
||||||
const companyList = companies
|
const companyList = companies
|
||||||
.map((c) => {
|
.map((c) => {
|
||||||
const approvedReviews = c.reviews;
|
const approvedReviews = c.reviews;
|
||||||
|
|
|
||||||
|
|
@ -89,6 +89,27 @@ export async function executeSourceIngestion(
|
||||||
totalDuplicates++;
|
totalDuplicates++;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Auto-ensure Company record exists
|
||||||
|
let companyId: string | undefined = undefined;
|
||||||
|
try {
|
||||||
|
const trimmedCompany = (job.company || "").trim();
|
||||||
|
if (trimmedCompany && trimmedCompany.length >= 2 && trimmedCompany.toLowerCase() !== "unknown") {
|
||||||
|
const comp = await prisma.company.upsert({
|
||||||
|
where: { name: trimmedCompany },
|
||||||
|
create: {
|
||||||
|
name: trimmedCompany,
|
||||||
|
location: job.location || "USA",
|
||||||
|
verificationStatus: "UNCLAIMED",
|
||||||
|
trustStatus: "UNVERIFIED",
|
||||||
|
},
|
||||||
|
update: {},
|
||||||
|
});
|
||||||
|
companyId = comp.id;
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
// Non-blocking if Company constraint or race occurs
|
||||||
|
}
|
||||||
|
|
||||||
// Create new job
|
// Create new job
|
||||||
await prisma.job.create({
|
await prisma.job.create({
|
||||||
data: {
|
data: {
|
||||||
|
|
@ -96,6 +117,7 @@ export async function executeSourceIngestion(
|
||||||
fingerprintHash: job.fingerprintHash,
|
fingerprintHash: job.fingerprintHash,
|
||||||
title: job.title,
|
title: job.title,
|
||||||
company: job.company,
|
company: job.company,
|
||||||
|
companyId: companyId,
|
||||||
location: job.location,
|
location: job.location,
|
||||||
isRemote: job.isRemote,
|
isRemote: job.isRemote,
|
||||||
department: job.department,
|
department: job.department,
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue