From e84871926bd4a9f33851975e1a520115b93ccc3d Mon Sep 17 00:00:00 2001 From: JobsBoard Deployer Date: Sat, 5 Sep 2026 14:02:58 -0400 Subject: [PATCH] Enable high-precision OpenRouter LLM extraction for resume parsing to prevent scrambled titles and organization names --- web/src/app/api/resume/parse/route.ts | 86 +++++++++++++++++++++++---- 1 file changed, 73 insertions(+), 13 deletions(-) diff --git a/web/src/app/api/resume/parse/route.ts b/web/src/app/api/resume/parse/route.ts index 7658787..9562306 100644 --- a/web/src/app/api/resume/parse/route.ts +++ b/web/src/app/api/resume/parse/route.ts @@ -340,7 +340,10 @@ function parseResumeFull(rawText: string) { async function parseResumeWithLLM(rawText: string) { const apiKey = process.env.OPENROUTER_API_KEY || process.env.OPENAI_API_KEY; - if (!apiKey) return null; + if (!apiKey) { + console.log("[Resume Parser] No OPENROUTER_API_KEY or OPENAI_API_KEY configured; using heuristic parser."); + return null; + } const isRouter = !!process.env.OPENROUTER_API_KEY; const endpoint = isRouter @@ -348,38 +351,94 @@ async function parseResumeWithLLM(rawText: string) { : `${process.env.OPENAI_BASE_URL || "https://api.openai.com/v1"}/chat/completions`; const model = process.env.OPENROUTER_MODEL || process.env.OPENAI_MODEL || (isRouter ? "google/gemini-2.5-flash" : "gpt-4o-mini"); + console.log(`[Resume Parser] Invoking LLM parser via ${endpoint} using model: ${model}`); + + const systemPrompt = `You are a high-precision ATS resume parser and information extraction engine. +CRITICAL EXTRACTION GUIDELINES: +1. Parse raw, messy, or multi-column resume text into clean, structured JSON. +2. Accurately untangle company names and position titles: + - Identify the real organization/employer (e.g. "Pokemoto", "Christmas Wish CT", "Personal Homelab"). + - NEVER output generic placeholders like "Organization" or "Company" if the actual organization name is stated anywhere nearby. + - Do NOT place dates or location text into the company name field. +3. Identify the candidate's actual name, email, phone number, and location accurately. +4. Extract every distinct job experience entry with its authentic position, company, and bullet points (highlights). +5. Extract explicit skills into the skills array. +6. Extract schools/universities and degrees into the education array. +7. Return strictly a JSON object with this schema: +{ + "profile": { + "name": string, + "headline": string, + "email": string, + "phone": string, + "location": string, + "summary": string + }, + "work": [ + { + "position": string, + "company": string, + "summary": string, + "highlights": string[] + } + ], + "skills": string[], + "education": [ + { + "degree": string, + "institution": string + } + ] +}`; + try { + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), 15000); + const res = await fetch(endpoint, { method: "POST", headers: { Authorization: `Bearer ${apiKey}`, "Content-Type": "application/json", + ...(isRouter ? { "HTTP-Referer": "https://directwire.io", "X-Title": "DirectWire Resume Parser" } : {}), }, + signal: controller.signal, body: JSON.stringify({ model, + temperature: 0.1, messages: [ - { - role: "system", - content: - "You are a professional ATS resume parser. Extract structured JSON strictly matching: { profile: { name: string, headline: string, email: string, phone: string, location: string, summary: string }, work: [{ position: string, company: string, summary: string, highlights: string[] }], skills: string[], education: [{ degree: string, institution: string }] }", - }, + { role: "system", content: systemPrompt }, { role: "user", - content: `Extract structured JSON resume data:\n\n${rawText.slice(0, 10000)}`, + content: `Extract structured JSON resume from this text:\n\n${rawText.slice(0, 12000)}`, }, ], response_format: { type: "json_object" }, }), }); - if (!res.ok) return null; + clearTimeout(timeout); + + if (!res.ok) { + console.warn("[Resume Parser] LLM API responded with error:", res.status, await res.text()); + return null; + } + const json = await res.json(); - const content = json.choices?.[0]?.message?.content; + let content = json.choices?.[0]?.message?.content; if (!content) return null; - return JSON.parse(content); - } catch (e) { - console.error("LLM parse fallback triggered:", e); + // Clean markdown code fence wrappers if present + content = content.replace(/^```(?:json)?\s*/i, "").replace(/\s*```$/i, "").trim(); + + const parsed = JSON.parse(content); + if (parsed && (parsed.profile || Array.isArray(parsed.work))) { + console.log("[Resume Parser] Successfully extracted structured resume with LLM."); + return parsed; + } + return null; + } catch (e: any) { + console.error("[Resume Parser] LLM parse call failed or timed out:", e.message); return null; } } @@ -469,7 +528,8 @@ export async function POST(req: Request) { // Try LLM parsing first if API key configured, otherwise use high-precision local parsing let parsedData = await parseResumeWithLLM(rawText); - if (!parsedData || !parsedData.profile || !parsedData.profile.name) { + if (!parsedData || !parsedData.profile || (!parsedData.profile.name && (!Array.isArray(parsedData.work) || parsedData.work.length === 0))) { + console.log("[Resume Parser] LLM returned empty or incomplete data; applying local heuristic extraction."); parsedData = parseResumeFull(rawText); }