Auto-populate companies from job scrapers and broaden multi-industry coverage across nationwide hubs

This commit is contained in:
JobsBoard Deployer 2026-09-05 14:43:44 -04:00
parent 16efc58a5c
commit e0ae08e023
7 changed files with 300 additions and 32 deletions

View file

@ -111,6 +111,27 @@ def _upsert_sqlite(jobs_list: list):
seen_hashes = set()
records = []
# 1. First ensure all distinct companies exist in Company table
distinct_companies = {}
for j in jobs_list:
c_name = (j.get("company") or "").strip()
c_loc = (j.get("location") or "").strip()
if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]:
if c_name not in distinct_companies:
distinct_companies[c_name] = c_loc
for c_name, c_loc in distinct_companies.items():
comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20]
cursor.execute("""
INSERT OR IGNORE INTO Company (
id, name, location, verificationStatus, trustStatus, createdAt, updatedAt
) VALUES (?, ?, ?, 'UNCLAIMED', 'UNVERIFIED', ?, ?)
""", (comp_id, c_name, c_loc or "USA", now_iso, now_iso))
# Fetch mapping of company name to id
cursor.execute("SELECT id, name FROM Company")
comp_map = {row[1]: row[0] for row in cursor.fetchall()}
for j in jobs_list:
job_url = j.get("job_url", "")
if not job_url:
@ -120,12 +141,14 @@ def _upsert_sqlite(jobs_list: list):
continue
seen_hashes.add(job_hash)
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
c_name = j.get("company", "Unknown Company")[:255]
c_id = comp_map.get(c_name)
records.append((
job_id,
job_hash,
j.get("title", "Untitled Position")[:255],
j.get("company", "Unknown Company")[:255],
c_name,
j.get("location", "Not Specified")[:255],
1 if j.get("is_remote") else 0,
j.get("department") or "Other",
@ -136,10 +159,32 @@ def _upsert_sqlite(jobs_list: list):
job_url,
j.get("source", "jobspy"),
format_iso(j.get("date_posted")),
c_id,
now_iso,
now_iso
))
query = """
INSERT INTO Job (
id, jobUrlHash, title, company, location, isRemote,
department, experienceLevel, description, salaryMin, salaryMax, jobUrl, source, datePosted,
companyId, createdAt, updatedAt
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
ON CONFLICT (jobUrlHash) DO UPDATE SET
title = excluded.title,
company = excluded.company,
location = excluded.location,
isRemote = excluded.isRemote,
department = COALESCE(excluded.department, Job.department),
experienceLevel = COALESCE(excluded.experienceLevel, Job.experienceLevel),
description = excluded.description,
salaryMin = COALESCE(excluded.salaryMin, Job.salaryMin),
salaryMax = COALESCE(excluded.salaryMax, Job.salaryMax),
datePosted = COALESCE(excluded.datePosted, Job.datePosted),
companyId = COALESCE(excluded.companyId, Job.companyId),
updatedAt = excluded.updatedAt;
"""
try:
cursor.executemany(query, records)
conn.commit()
@ -190,11 +235,58 @@ def _upsert_postgres(jobs_list: list, db_url: str):
else:
cursor = conn.cursor()
now = datetime.datetime.now(datetime.timezone.utc)
# 1. First ensure all distinct companies exist in "Company" table
distinct_companies = {}
for j in jobs_list:
c_name = (j.get("company") or "").strip()
c_loc = (j.get("location") or "").strip()
if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]:
if c_name not in distinct_companies:
distinct_companies[c_name] = c_loc
if distinct_companies:
company_records = []
for c_name, c_loc in distinct_companies.items():
comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20]
company_records.append((
comp_id,
c_name,
c_loc or "USA",
"UNCLAIMED",
"UNVERIFIED",
now,
now
))
comp_upsert_query = """
INSERT INTO "Company" ("id", "name", "location", "verificationStatus", "trustStatus", "createdAt", "updatedAt")
VALUES %s
ON CONFLICT ("name") DO UPDATE SET
"location" = COALESCE("Company"."location", EXCLUDED."location"),
"updatedAt" = EXCLUDED."updatedAt";
"""
try:
execute_values(cursor, comp_upsert_query, company_records)
conn.commit()
except Exception as comp_err:
conn.rollback()
print(f"[Postgres Warning] Company pre-population error: {comp_err}")
# Fetch mapping of company name to id
comp_map = {}
try:
cursor.execute('SELECT "id", "name" FROM "Company"')
for row in cursor.fetchall():
comp_map[row[1]] = row[0]
except Exception as map_err:
print(f"[Postgres Warning] Failed to fetch company map: {map_err}")
query = """
INSERT INTO "Job" (
"id", "jobUrlHash", "title", "company", "location", "isRemote",
"department", "experienceLevel", "description", "salaryMin", "salaryMax", "jobUrl", "source", "datePosted",
"lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt"
"companyId", "lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt"
) VALUES %s
ON CONFLICT ("jobUrlHash") DO UPDATE SET
"title" = EXCLUDED."title",
@ -207,6 +299,7 @@ def _upsert_postgres(jobs_list: list, db_url: str):
"salaryMin" = COALESCE(EXCLUDED."salaryMin", "Job"."salaryMin"),
"salaryMax" = COALESCE(EXCLUDED."salaryMax", "Job"."salaryMax"),
"datePosted" = COALESCE(EXCLUDED."datePosted", "Job"."datePosted"),
"companyId" = COALESCE(EXCLUDED."companyId", "Job"."companyId"),
"lifecycleStatus" = 'ACTIVE',
"lastSeenAt" = NOW(),
"updatedAt" = NOW();
@ -214,7 +307,6 @@ def _upsert_postgres(jobs_list: list, db_url: str):
records = []
seen_hashes = set()
now = datetime.datetime.now(datetime.timezone.utc)
for j in jobs_list:
job_url = j.get("job_url", "")
@ -225,12 +317,14 @@ def _upsert_postgres(jobs_list: list, db_url: str):
continue
seen_hashes.add(job_hash)
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
c_name = j.get("company", "Unknown Company")[:255]
c_id = comp_map.get(c_name)
records.append((
job_id,
job_hash,
j.get("title", "Untitled Position")[:255],
j.get("company", "Unknown Company")[:255],
c_name,
j.get("location", "Not Specified")[:255],
bool(j.get("is_remote", False)),
j.get("department") or "Other",
@ -241,6 +335,7 @@ def _upsert_postgres(jobs_list: list, db_url: str):
job_url,
j.get("source", "jobspy"),
now,
c_id,
"ACTIVE",
now,
now,

View file

@ -49,12 +49,18 @@ DOMAINS_TO_PROBE = [
("travelers.com", "Travelers Insurance"),
("hartfordhealthcare.org", "Hartford HealthCare"),
("yalehealth.org", "Yale Health"),
("labcorp.com", "Labcorp"),
("questdiagnostics.com", "Quest Diagnostics"),
("lockheedmartin.com", "Lockheed Martin"),
("boeing.com", "Boeing"),
("caterpillar.com", "Caterpillar"),
("deere.com", "John Deere"),
("fedex.com", "FedEx"),
("ups.com", "UPS")
("ups.com", "UPS"),
("dhl.com", "DHL Express"),
("marriott.com", "Marriott International"),
("hilton.com", "Hilton"),
("starbucks.com", "Starbucks")
]
def run_smart_crawler_scrapes():

View file

@ -88,7 +88,44 @@ GREENHOUSE_BOARDS = [
("tempus", "Tempus Labs"),
("guardanthealth", "Guardant Health"),
("flatiron", "Flatiron Health"),
("moderna", "Moderna")
("moderna", "Moderna"),
("cityblock", "Cityblock Health"),
("hims", "Hims & Hers Health"),
("springhealth", "Spring Health"),
("headway", "Headway"),
("talkspace", "Talkspace"),
("carbonhealth", "Carbon Health"),
("omadahealth", "Omada Health"),
("mavenclinic", "Maven Clinic"),
("goodrx", "GoodRx"),
("color", "Color Health"),
("invitae", "Invitae"),
# Logistics, Supply Chain & Industrial
("deliverr", "Deliverr (Logistics)"),
("convoy", "Convoy (Freight & Logistics)"),
("fulfill", "Fulfill.com"),
("shipbob", "ShipBob"),
("flockfreight", "Flock Freight"),
# Real Estate, Property & Construction
("compass", "Compass Real Estate"),
("opendoor", "Opendoor"),
("redfin", "Redfin"),
("cbre", "CBRE"),
# Hospitality, Food & Travel
("sweetgreen", "Sweetgreen"),
("instacart", "Instacart"),
("grubhub", "Grubhub"),
("goldbelly", "Goldbelly"),
# Professional Services, Accounting & Legal
("pilot", "Pilot (Bookkeeping & Tax)"),
("bench", "Bench Accounting"),
("brex", "Brex"),
("gusto", "Gusto (Payroll & HR)"),
("rippling", "Rippling (HR & Workforce)")
]
LEVER_BOARDS = [

View file

@ -35,24 +35,35 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
if proxies:
sites.extend(["zip_recruiter", "glassdoor"])
# Target key employment hubs across US states
# Target key employment hubs across US states covering all industries
us_regions = [
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]),
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]),
("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]),
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]),
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]),
("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]),
("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]),
("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]),
("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"])
]
# Nationwide Remote queries
# Nationwide Remote queries across ALL professional disciplines
remote_queries = [
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
"Customer Support", "Administrative Assistant", "IT Support Specialist",
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
# Healthcare & Medical
"Medical Biller", "Telehealth Nurse", "Clinical Research Coordinator", "Healthcare Recruiter",
# Finance, Accounting & Legal
"Staff Accountant", "Financial Analyst", "Bookkeeper", "Paralegal", "Compliance Specialist", "Underwriter",
# Sales, Marketing & Customer Support
"Customer Support Representative", "Customer Success Manager", "Account Executive", "Digital Marketing Specialist", "Content Writer",
# Human Resources & Operations
"HR Generalist", "Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Project Coordinator",
# Art, Design & Creative
"Graphic Designer", "UX Designer", "Video Editor", "Instructional Designer",
# Logistics, Supply Chain & Purchasing
"Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist",
# Tech & Engineering
"Software Engineer", "IT Support Specialist", "Data Analyst", "Systems Administrator", "DevOps Engineer"
]
def _execute_scrape(site_list, term, loc, is_rem):

View file

@ -13,10 +13,47 @@ export async function GET(
const companyId = params.id;
// 1. Fetch Company
const company = await prisma.company.findUnique({
let company = await prisma.company.findUnique({
where: { id: companyId },
});
if (!company) {
// Check if companyId matches a company name directly or decoded
const decodedName = decodeURIComponent(companyId);
company = await prisma.company.findFirst({
where: {
OR: [
{ name: { equals: decodedName } },
{ name: { equals: companyId } },
],
},
});
// If still not found, check if jobs exist with this company name and auto-create the company record
if (!company) {
const sampleJob = await prisma.job.findFirst({
where: {
OR: [
{ company: { equals: decodedName } },
{ company: { equals: companyId } },
],
},
select: { company: true, location: true },
});
if (sampleJob && sampleJob.company) {
company = await prisma.company.create({
data: {
name: sampleJob.company.trim(),
location: sampleJob.location || "USA",
verificationStatus: "UNCLAIMED",
trustStatus: "UNVERIFIED",
},
});
}
}
}
if (!company) {
return NextResponse.json({ error: "Company not found" }, { status: 404 });
}

View file

@ -7,8 +7,32 @@ export async function GET(req: Request) {
const { searchParams } = new URL(req.url);
const search = searchParams.get("search")?.trim().toLowerCase() || "";
// 1. Query companies with approved reviews
const companies = await prisma.company.findMany({
// 1. Fetch all distinct companies referenced in Job table to ensure full directory coverage
const jobsWithCompany = await prisma.job.findMany({
where: {
company: { not: "" },
},
select: { company: true, location: true },
});
const companyJobCountMap = new Map<string, number>();
const companyLocationMap = new Map<string, string>();
for (const j of jobsWithCompany) {
if (j.company) {
const cRaw = j.company.trim();
const cLower = cRaw.toLowerCase();
if (cLower !== "unknown" && cLower !== "n/a" && cRaw.length >= 2) {
companyJobCountMap.set(cLower, (companyJobCountMap.get(cLower) || 0) + 1);
if (j.location && !companyLocationMap.has(cLower)) {
companyLocationMap.set(cLower, j.location);
}
}
}
}
// 2. Query existing Company table records with approved reviews
const existingCompanies = await prisma.company.findMany({
include: {
reviews: {
where: { isApproved: true },
@ -17,19 +41,55 @@ export async function GET(req: Request) {
orderBy: { name: "asc" },
});
// 2. Count open jobs per company from Job table
const jobs = await prisma.job.findMany({
select: { company: true },
});
const existingNamesSet = new Set(existingCompanies.map((c) => c.name.trim().toLowerCase()));
const companyJobCountMap = new Map<string, number>();
for (const j of jobs) {
if (j.company) {
const cName = j.company.trim().toLowerCase();
companyJobCountMap.set(cName, (companyJobCountMap.get(cName) || 0) + 1);
// 3. Find any distinct job employers not yet in Company table and auto-insert them
const missingCompanies: { name: string; location: string }[] = [];
for (const [cLower] of companyJobCountMap.entries()) {
if (!existingNamesSet.has(cLower)) {
// Find original casing
const orig = jobsWithCompany.find((j) => j.company.trim().toLowerCase() === cLower);
if (orig) {
missingCompanies.push({
name: orig.company.trim(),
location: orig.location || "USA",
});
}
}
}
if (missingCompanies.length > 0) {
try {
// Auto-create missing companies
for (const m of missingCompanies.slice(0, 100)) {
try {
await prisma.company.create({
data: {
name: m.name,
location: m.location,
verificationStatus: "UNCLAIMED",
trustStatus: "UNVERIFIED",
},
});
} catch {
// Safe to ignore duplicate or race condition
}
}
} catch {
// Non-blocking
}
}
// 4. Reload full companies list
const companies = await prisma.company.findMany({
include: {
reviews: {
where: { isApproved: true },
},
},
orderBy: { name: "asc" },
});
const companyList = companies
.map((c) => {
const approvedReviews = c.reviews;

View file

@ -89,6 +89,27 @@ export async function executeSourceIngestion(
totalDuplicates++;
}
// Auto-ensure Company record exists
let companyId: string | undefined = undefined;
try {
const trimmedCompany = (job.company || "").trim();
if (trimmedCompany && trimmedCompany.length >= 2 && trimmedCompany.toLowerCase() !== "unknown") {
const comp = await prisma.company.upsert({
where: { name: trimmedCompany },
create: {
name: trimmedCompany,
location: job.location || "USA",
verificationStatus: "UNCLAIMED",
trustStatus: "UNVERIFIED",
},
update: {},
});
companyId = comp.id;
}
} catch {
// Non-blocking if Company constraint or race occurs
}
// Create new job
await prisma.job.create({
data: {
@ -96,6 +117,7 @@ export async function executeSourceIngestion(
fingerprintHash: job.fingerprintHash,
title: job.title,
company: job.company,
companyId: companyId,
location: job.location,
isRemote: job.isRemote,
department: job.department,