Auto-populate companies from job scrapers and broaden multi-industry coverage across nationwide hubs
This commit is contained in:
parent
16efc58a5c
commit
e0ae08e023
7 changed files with 300 additions and 32 deletions
103
scraper/db.py
103
scraper/db.py
|
|
@ -111,6 +111,27 @@ def _upsert_sqlite(jobs_list: list):
|
|||
seen_hashes = set()
|
||||
records = []
|
||||
|
||||
# 1. First ensure all distinct companies exist in Company table
|
||||
distinct_companies = {}
|
||||
for j in jobs_list:
|
||||
c_name = (j.get("company") or "").strip()
|
||||
c_loc = (j.get("location") or "").strip()
|
||||
if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]:
|
||||
if c_name not in distinct_companies:
|
||||
distinct_companies[c_name] = c_loc
|
||||
|
||||
for c_name, c_loc in distinct_companies.items():
|
||||
comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20]
|
||||
cursor.execute("""
|
||||
INSERT OR IGNORE INTO Company (
|
||||
id, name, location, verificationStatus, trustStatus, createdAt, updatedAt
|
||||
) VALUES (?, ?, ?, 'UNCLAIMED', 'UNVERIFIED', ?, ?)
|
||||
""", (comp_id, c_name, c_loc or "USA", now_iso, now_iso))
|
||||
|
||||
# Fetch mapping of company name to id
|
||||
cursor.execute("SELECT id, name FROM Company")
|
||||
comp_map = {row[1]: row[0] for row in cursor.fetchall()}
|
||||
|
||||
for j in jobs_list:
|
||||
job_url = j.get("job_url", "")
|
||||
if not job_url:
|
||||
|
|
@ -120,12 +141,14 @@ def _upsert_sqlite(jobs_list: list):
|
|||
continue
|
||||
seen_hashes.add(job_hash)
|
||||
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
|
||||
c_name = j.get("company", "Unknown Company")[:255]
|
||||
c_id = comp_map.get(c_name)
|
||||
|
||||
records.append((
|
||||
job_id,
|
||||
job_hash,
|
||||
j.get("title", "Untitled Position")[:255],
|
||||
j.get("company", "Unknown Company")[:255],
|
||||
c_name,
|
||||
j.get("location", "Not Specified")[:255],
|
||||
1 if j.get("is_remote") else 0,
|
||||
j.get("department") or "Other",
|
||||
|
|
@ -136,10 +159,32 @@ def _upsert_sqlite(jobs_list: list):
|
|||
job_url,
|
||||
j.get("source", "jobspy"),
|
||||
format_iso(j.get("date_posted")),
|
||||
c_id,
|
||||
now_iso,
|
||||
now_iso
|
||||
))
|
||||
|
||||
query = """
|
||||
INSERT INTO Job (
|
||||
id, jobUrlHash, title, company, location, isRemote,
|
||||
department, experienceLevel, description, salaryMin, salaryMax, jobUrl, source, datePosted,
|
||||
companyId, createdAt, updatedAt
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
ON CONFLICT (jobUrlHash) DO UPDATE SET
|
||||
title = excluded.title,
|
||||
company = excluded.company,
|
||||
location = excluded.location,
|
||||
isRemote = excluded.isRemote,
|
||||
department = COALESCE(excluded.department, Job.department),
|
||||
experienceLevel = COALESCE(excluded.experienceLevel, Job.experienceLevel),
|
||||
description = excluded.description,
|
||||
salaryMin = COALESCE(excluded.salaryMin, Job.salaryMin),
|
||||
salaryMax = COALESCE(excluded.salaryMax, Job.salaryMax),
|
||||
datePosted = COALESCE(excluded.datePosted, Job.datePosted),
|
||||
companyId = COALESCE(excluded.companyId, Job.companyId),
|
||||
updatedAt = excluded.updatedAt;
|
||||
"""
|
||||
|
||||
try:
|
||||
cursor.executemany(query, records)
|
||||
conn.commit()
|
||||
|
|
@ -190,11 +235,58 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
|||
else:
|
||||
cursor = conn.cursor()
|
||||
|
||||
now = datetime.datetime.now(datetime.timezone.utc)
|
||||
|
||||
# 1. First ensure all distinct companies exist in "Company" table
|
||||
distinct_companies = {}
|
||||
for j in jobs_list:
|
||||
c_name = (j.get("company") or "").strip()
|
||||
c_loc = (j.get("location") or "").strip()
|
||||
if c_name and len(c_name) >= 2 and c_name.lower() not in ["unknown", "confidential", "confidential company", "n/a"]:
|
||||
if c_name not in distinct_companies:
|
||||
distinct_companies[c_name] = c_loc
|
||||
|
||||
if distinct_companies:
|
||||
company_records = []
|
||||
for c_name, c_loc in distinct_companies.items():
|
||||
comp_id = "comp_" + hashlib.sha256(c_name.lower().encode("utf-8")).hexdigest()[:20]
|
||||
company_records.append((
|
||||
comp_id,
|
||||
c_name,
|
||||
c_loc or "USA",
|
||||
"UNCLAIMED",
|
||||
"UNVERIFIED",
|
||||
now,
|
||||
now
|
||||
))
|
||||
comp_upsert_query = """
|
||||
INSERT INTO "Company" ("id", "name", "location", "verificationStatus", "trustStatus", "createdAt", "updatedAt")
|
||||
VALUES %s
|
||||
ON CONFLICT ("name") DO UPDATE SET
|
||||
"location" = COALESCE("Company"."location", EXCLUDED."location"),
|
||||
"updatedAt" = EXCLUDED."updatedAt";
|
||||
"""
|
||||
try:
|
||||
execute_values(cursor, comp_upsert_query, company_records)
|
||||
conn.commit()
|
||||
except Exception as comp_err:
|
||||
conn.rollback()
|
||||
print(f"[Postgres Warning] Company pre-population error: {comp_err}")
|
||||
|
||||
# Fetch mapping of company name to id
|
||||
comp_map = {}
|
||||
try:
|
||||
cursor.execute('SELECT "id", "name" FROM "Company"')
|
||||
for row in cursor.fetchall():
|
||||
comp_map[row[1]] = row[0]
|
||||
except Exception as map_err:
|
||||
print(f"[Postgres Warning] Failed to fetch company map: {map_err}")
|
||||
|
||||
query = """
|
||||
INSERT INTO "Job" (
|
||||
"id", "jobUrlHash", "title", "company", "location", "isRemote",
|
||||
"department", "experienceLevel", "description", "salaryMin", "salaryMax", "jobUrl", "source", "datePosted",
|
||||
"lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt"
|
||||
"companyId", "lifecycleStatus", "lastSeenAt", "createdAt", "updatedAt"
|
||||
) VALUES %s
|
||||
ON CONFLICT ("jobUrlHash") DO UPDATE SET
|
||||
"title" = EXCLUDED."title",
|
||||
|
|
@ -207,6 +299,7 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
|||
"salaryMin" = COALESCE(EXCLUDED."salaryMin", "Job"."salaryMin"),
|
||||
"salaryMax" = COALESCE(EXCLUDED."salaryMax", "Job"."salaryMax"),
|
||||
"datePosted" = COALESCE(EXCLUDED."datePosted", "Job"."datePosted"),
|
||||
"companyId" = COALESCE(EXCLUDED."companyId", "Job"."companyId"),
|
||||
"lifecycleStatus" = 'ACTIVE',
|
||||
"lastSeenAt" = NOW(),
|
||||
"updatedAt" = NOW();
|
||||
|
|
@ -214,7 +307,6 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
|||
|
||||
records = []
|
||||
seen_hashes = set()
|
||||
now = datetime.datetime.now(datetime.timezone.utc)
|
||||
|
||||
for j in jobs_list:
|
||||
job_url = j.get("job_url", "")
|
||||
|
|
@ -225,12 +317,14 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
|||
continue
|
||||
seen_hashes.add(job_hash)
|
||||
job_id = "job_" + str(uuid.uuid4()).replace("-", "")[:20]
|
||||
c_name = j.get("company", "Unknown Company")[:255]
|
||||
c_id = comp_map.get(c_name)
|
||||
|
||||
records.append((
|
||||
job_id,
|
||||
job_hash,
|
||||
j.get("title", "Untitled Position")[:255],
|
||||
j.get("company", "Unknown Company")[:255],
|
||||
c_name,
|
||||
j.get("location", "Not Specified")[:255],
|
||||
bool(j.get("is_remote", False)),
|
||||
j.get("department") or "Other",
|
||||
|
|
@ -241,6 +335,7 @@ def _upsert_postgres(jobs_list: list, db_url: str):
|
|||
job_url,
|
||||
j.get("source", "jobspy"),
|
||||
now,
|
||||
c_id,
|
||||
"ACTIVE",
|
||||
now,
|
||||
now,
|
||||
|
|
|
|||
|
|
@ -49,12 +49,18 @@ DOMAINS_TO_PROBE = [
|
|||
("travelers.com", "Travelers Insurance"),
|
||||
("hartfordhealthcare.org", "Hartford HealthCare"),
|
||||
("yalehealth.org", "Yale Health"),
|
||||
("labcorp.com", "Labcorp"),
|
||||
("questdiagnostics.com", "Quest Diagnostics"),
|
||||
("lockheedmartin.com", "Lockheed Martin"),
|
||||
("boeing.com", "Boeing"),
|
||||
("caterpillar.com", "Caterpillar"),
|
||||
("deere.com", "John Deere"),
|
||||
("fedex.com", "FedEx"),
|
||||
("ups.com", "UPS")
|
||||
("ups.com", "UPS"),
|
||||
("dhl.com", "DHL Express"),
|
||||
("marriott.com", "Marriott International"),
|
||||
("hilton.com", "Hilton"),
|
||||
("starbucks.com", "Starbucks")
|
||||
]
|
||||
|
||||
def run_smart_crawler_scrapes():
|
||||
|
|
|
|||
|
|
@ -88,7 +88,44 @@ GREENHOUSE_BOARDS = [
|
|||
("tempus", "Tempus Labs"),
|
||||
("guardanthealth", "Guardant Health"),
|
||||
("flatiron", "Flatiron Health"),
|
||||
("moderna", "Moderna")
|
||||
("moderna", "Moderna"),
|
||||
("cityblock", "Cityblock Health"),
|
||||
("hims", "Hims & Hers Health"),
|
||||
("springhealth", "Spring Health"),
|
||||
("headway", "Headway"),
|
||||
("talkspace", "Talkspace"),
|
||||
("carbonhealth", "Carbon Health"),
|
||||
("omadahealth", "Omada Health"),
|
||||
("mavenclinic", "Maven Clinic"),
|
||||
("goodrx", "GoodRx"),
|
||||
("color", "Color Health"),
|
||||
("invitae", "Invitae"),
|
||||
|
||||
# Logistics, Supply Chain & Industrial
|
||||
("deliverr", "Deliverr (Logistics)"),
|
||||
("convoy", "Convoy (Freight & Logistics)"),
|
||||
("fulfill", "Fulfill.com"),
|
||||
("shipbob", "ShipBob"),
|
||||
("flockfreight", "Flock Freight"),
|
||||
|
||||
# Real Estate, Property & Construction
|
||||
("compass", "Compass Real Estate"),
|
||||
("opendoor", "Opendoor"),
|
||||
("redfin", "Redfin"),
|
||||
("cbre", "CBRE"),
|
||||
|
||||
# Hospitality, Food & Travel
|
||||
("sweetgreen", "Sweetgreen"),
|
||||
("instacart", "Instacart"),
|
||||
("grubhub", "Grubhub"),
|
||||
("goldbelly", "Goldbelly"),
|
||||
|
||||
# Professional Services, Accounting & Legal
|
||||
("pilot", "Pilot (Bookkeeping & Tax)"),
|
||||
("bench", "Bench Accounting"),
|
||||
("brex", "Brex"),
|
||||
("gusto", "Gusto (Payroll & HR)"),
|
||||
("rippling", "Rippling (HR & Workforce)")
|
||||
]
|
||||
|
||||
LEVER_BOARDS = [
|
||||
|
|
|
|||
|
|
@ -35,24 +35,35 @@ def run_jobspy_scrapes() -> List[Dict[Any, Any]]:
|
|||
if proxies:
|
||||
sites.extend(["zip_recruiter", "glassdoor"])
|
||||
|
||||
# Target key employment hubs across US states
|
||||
# Target key employment hubs across US states covering all industries
|
||||
us_regions = [
|
||||
("New York, NY", ["Finance", "Software Engineer", "Marketing", "Data Analyst"]),
|
||||
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support"]),
|
||||
("San Francisco, CA", ["AI Engineer", "Product Manager", "DevOps"]),
|
||||
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant"]),
|
||||
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative"]),
|
||||
("Seattle, WA", ["Cloud Architect", "Software Developer", "Data Scientist"]),
|
||||
("Boston, MA", ["Biotech", "Software Engineer", "Healthcare"]),
|
||||
("Denver, CO", ["Customer Success", "Cybersecurity", "Engineering"]),
|
||||
("Connecticut", ["Healthcare", "Finance", "Insurance", "Software", "Engineering"])
|
||||
("New York, NY", ["Finance", "Accountant", "Marketing", "Data Analyst", "Nurse", "Paralegal", "Sales"]),
|
||||
("Austin, TX", ["Software Engineer", "Project Manager", "Customer Support", "Operations", "Electrician"]),
|
||||
("San Francisco, CA", ["AI Engineer", "Product Manager", "Graphic Designer", "Recruiter"]),
|
||||
("Chicago, IL", ["Operations", "Healthcare", "Logistics", "Accountant", "Warehouse", "HR Specialist"]),
|
||||
("Atlanta, GA", ["IT Support", "Sales", "Supply Chain", "Administrative", "Medical Assistant", "Customer Service"]),
|
||||
("Seattle, WA", ["Software Developer", "Data Scientist", "Procurement", "Compliance Officer"]),
|
||||
("Boston, MA", ["Biotech", "Clinical Research", "Healthcare", "Financial Analyst", "Teacher"]),
|
||||
("Denver, CO", ["Customer Success", "Cybersecurity", "Construction Manager", "Account Executive"]),
|
||||
("Connecticut", ["Healthcare", "Nurse", "Finance", "Insurance Underwriter", "Manufacturing", "Electrician", "Administrative"])
|
||||
]
|
||||
|
||||
# Nationwide Remote queries
|
||||
# Nationwide Remote queries across ALL professional disciplines
|
||||
remote_queries = [
|
||||
"Software Engineer", "Full Stack Developer", "Data Analyst", "Product Manager",
|
||||
"Customer Support", "Administrative Assistant", "IT Support Specialist",
|
||||
"DevOps Engineer", "Account Executive", "Marketing Manager", "UX Designer"
|
||||
# Healthcare & Medical
|
||||
"Medical Biller", "Telehealth Nurse", "Clinical Research Coordinator", "Healthcare Recruiter",
|
||||
# Finance, Accounting & Legal
|
||||
"Staff Accountant", "Financial Analyst", "Bookkeeper", "Paralegal", "Compliance Specialist", "Underwriter",
|
||||
# Sales, Marketing & Customer Support
|
||||
"Customer Support Representative", "Customer Success Manager", "Account Executive", "Digital Marketing Specialist", "Content Writer",
|
||||
# Human Resources & Operations
|
||||
"HR Generalist", "Technical Recruiter", "Executive Assistant", "Operations Coordinator", "Project Coordinator",
|
||||
# Art, Design & Creative
|
||||
"Graphic Designer", "UX Designer", "Video Editor", "Instructional Designer",
|
||||
# Logistics, Supply Chain & Purchasing
|
||||
"Logistics Coordinator", "Supply Chain Analyst", "Procurement Specialist",
|
||||
# Tech & Engineering
|
||||
"Software Engineer", "IT Support Specialist", "Data Analyst", "Systems Administrator", "DevOps Engineer"
|
||||
]
|
||||
|
||||
def _execute_scrape(site_list, term, loc, is_rem):
|
||||
|
|
|
|||
|
|
@ -13,10 +13,47 @@ export async function GET(
|
|||
const companyId = params.id;
|
||||
|
||||
// 1. Fetch Company
|
||||
const company = await prisma.company.findUnique({
|
||||
let company = await prisma.company.findUnique({
|
||||
where: { id: companyId },
|
||||
});
|
||||
|
||||
if (!company) {
|
||||
// Check if companyId matches a company name directly or decoded
|
||||
const decodedName = decodeURIComponent(companyId);
|
||||
company = await prisma.company.findFirst({
|
||||
where: {
|
||||
OR: [
|
||||
{ name: { equals: decodedName } },
|
||||
{ name: { equals: companyId } },
|
||||
],
|
||||
},
|
||||
});
|
||||
|
||||
// If still not found, check if jobs exist with this company name and auto-create the company record
|
||||
if (!company) {
|
||||
const sampleJob = await prisma.job.findFirst({
|
||||
where: {
|
||||
OR: [
|
||||
{ company: { equals: decodedName } },
|
||||
{ company: { equals: companyId } },
|
||||
],
|
||||
},
|
||||
select: { company: true, location: true },
|
||||
});
|
||||
|
||||
if (sampleJob && sampleJob.company) {
|
||||
company = await prisma.company.create({
|
||||
data: {
|
||||
name: sampleJob.company.trim(),
|
||||
location: sampleJob.location || "USA",
|
||||
verificationStatus: "UNCLAIMED",
|
||||
trustStatus: "UNVERIFIED",
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!company) {
|
||||
return NextResponse.json({ error: "Company not found" }, { status: 404 });
|
||||
}
|
||||
|
|
|
|||
|
|
@ -7,8 +7,32 @@ export async function GET(req: Request) {
|
|||
const { searchParams } = new URL(req.url);
|
||||
const search = searchParams.get("search")?.trim().toLowerCase() || "";
|
||||
|
||||
// 1. Query companies with approved reviews
|
||||
const companies = await prisma.company.findMany({
|
||||
// 1. Fetch all distinct companies referenced in Job table to ensure full directory coverage
|
||||
const jobsWithCompany = await prisma.job.findMany({
|
||||
where: {
|
||||
company: { not: "" },
|
||||
},
|
||||
select: { company: true, location: true },
|
||||
});
|
||||
|
||||
const companyJobCountMap = new Map<string, number>();
|
||||
const companyLocationMap = new Map<string, string>();
|
||||
|
||||
for (const j of jobsWithCompany) {
|
||||
if (j.company) {
|
||||
const cRaw = j.company.trim();
|
||||
const cLower = cRaw.toLowerCase();
|
||||
if (cLower !== "unknown" && cLower !== "n/a" && cRaw.length >= 2) {
|
||||
companyJobCountMap.set(cLower, (companyJobCountMap.get(cLower) || 0) + 1);
|
||||
if (j.location && !companyLocationMap.has(cLower)) {
|
||||
companyLocationMap.set(cLower, j.location);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Query existing Company table records with approved reviews
|
||||
const existingCompanies = await prisma.company.findMany({
|
||||
include: {
|
||||
reviews: {
|
||||
where: { isApproved: true },
|
||||
|
|
@ -17,19 +41,55 @@ export async function GET(req: Request) {
|
|||
orderBy: { name: "asc" },
|
||||
});
|
||||
|
||||
// 2. Count open jobs per company from Job table
|
||||
const jobs = await prisma.job.findMany({
|
||||
select: { company: true },
|
||||
});
|
||||
const existingNamesSet = new Set(existingCompanies.map((c) => c.name.trim().toLowerCase()));
|
||||
|
||||
const companyJobCountMap = new Map<string, number>();
|
||||
for (const j of jobs) {
|
||||
if (j.company) {
|
||||
const cName = j.company.trim().toLowerCase();
|
||||
companyJobCountMap.set(cName, (companyJobCountMap.get(cName) || 0) + 1);
|
||||
// 3. Find any distinct job employers not yet in Company table and auto-insert them
|
||||
const missingCompanies: { name: string; location: string }[] = [];
|
||||
for (const [cLower] of companyJobCountMap.entries()) {
|
||||
if (!existingNamesSet.has(cLower)) {
|
||||
// Find original casing
|
||||
const orig = jobsWithCompany.find((j) => j.company.trim().toLowerCase() === cLower);
|
||||
if (orig) {
|
||||
missingCompanies.push({
|
||||
name: orig.company.trim(),
|
||||
location: orig.location || "USA",
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (missingCompanies.length > 0) {
|
||||
try {
|
||||
// Auto-create missing companies
|
||||
for (const m of missingCompanies.slice(0, 100)) {
|
||||
try {
|
||||
await prisma.company.create({
|
||||
data: {
|
||||
name: m.name,
|
||||
location: m.location,
|
||||
verificationStatus: "UNCLAIMED",
|
||||
trustStatus: "UNVERIFIED",
|
||||
},
|
||||
});
|
||||
} catch {
|
||||
// Safe to ignore duplicate or race condition
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Non-blocking
|
||||
}
|
||||
}
|
||||
|
||||
// 4. Reload full companies list
|
||||
const companies = await prisma.company.findMany({
|
||||
include: {
|
||||
reviews: {
|
||||
where: { isApproved: true },
|
||||
},
|
||||
},
|
||||
orderBy: { name: "asc" },
|
||||
});
|
||||
|
||||
const companyList = companies
|
||||
.map((c) => {
|
||||
const approvedReviews = c.reviews;
|
||||
|
|
|
|||
|
|
@ -89,6 +89,27 @@ export async function executeSourceIngestion(
|
|||
totalDuplicates++;
|
||||
}
|
||||
|
||||
// Auto-ensure Company record exists
|
||||
let companyId: string | undefined = undefined;
|
||||
try {
|
||||
const trimmedCompany = (job.company || "").trim();
|
||||
if (trimmedCompany && trimmedCompany.length >= 2 && trimmedCompany.toLowerCase() !== "unknown") {
|
||||
const comp = await prisma.company.upsert({
|
||||
where: { name: trimmedCompany },
|
||||
create: {
|
||||
name: trimmedCompany,
|
||||
location: job.location || "USA",
|
||||
verificationStatus: "UNCLAIMED",
|
||||
trustStatus: "UNVERIFIED",
|
||||
},
|
||||
update: {},
|
||||
});
|
||||
companyId = comp.id;
|
||||
}
|
||||
} catch {
|
||||
// Non-blocking if Company constraint or race occurs
|
||||
}
|
||||
|
||||
// Create new job
|
||||
await prisma.job.create({
|
||||
data: {
|
||||
|
|
@ -96,6 +117,7 @@ export async function executeSourceIngestion(
|
|||
fingerprintHash: job.fingerprintHash,
|
||||
title: job.title,
|
||||
company: job.company,
|
||||
companyId: companyId,
|
||||
location: job.location,
|
||||
isRemote: job.isRemote,
|
||||
department: job.department,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue