Enable high-precision OpenRouter LLM extraction for resume parsing to prevent scrambled titles and organization names
This commit is contained in:
parent
9205b88458
commit
e84871926b
1 changed files with 73 additions and 13 deletions
|
|
@ -340,7 +340,10 @@ function parseResumeFull(rawText: string) {
|
|||
|
||||
async function parseResumeWithLLM(rawText: string) {
|
||||
const apiKey = process.env.OPENROUTER_API_KEY || process.env.OPENAI_API_KEY;
|
||||
if (!apiKey) return null;
|
||||
if (!apiKey) {
|
||||
console.log("[Resume Parser] No OPENROUTER_API_KEY or OPENAI_API_KEY configured; using heuristic parser.");
|
||||
return null;
|
||||
}
|
||||
|
||||
const isRouter = !!process.env.OPENROUTER_API_KEY;
|
||||
const endpoint = isRouter
|
||||
|
|
@ -348,38 +351,94 @@ async function parseResumeWithLLM(rawText: string) {
|
|||
: `${process.env.OPENAI_BASE_URL || "https://api.openai.com/v1"}/chat/completions`;
|
||||
const model = process.env.OPENROUTER_MODEL || process.env.OPENAI_MODEL || (isRouter ? "google/gemini-2.5-flash" : "gpt-4o-mini");
|
||||
|
||||
console.log(`[Resume Parser] Invoking LLM parser via ${endpoint} using model: ${model}`);
|
||||
|
||||
const systemPrompt = `You are a high-precision ATS resume parser and information extraction engine.
|
||||
CRITICAL EXTRACTION GUIDELINES:
|
||||
1. Parse raw, messy, or multi-column resume text into clean, structured JSON.
|
||||
2. Accurately untangle company names and position titles:
|
||||
- Identify the real organization/employer (e.g. "Pokemoto", "Christmas Wish CT", "Personal Homelab").
|
||||
- NEVER output generic placeholders like "Organization" or "Company" if the actual organization name is stated anywhere nearby.
|
||||
- Do NOT place dates or location text into the company name field.
|
||||
3. Identify the candidate's actual name, email, phone number, and location accurately.
|
||||
4. Extract every distinct job experience entry with its authentic position, company, and bullet points (highlights).
|
||||
5. Extract explicit skills into the skills array.
|
||||
6. Extract schools/universities and degrees into the education array.
|
||||
7. Return strictly a JSON object with this schema:
|
||||
{
|
||||
"profile": {
|
||||
"name": string,
|
||||
"headline": string,
|
||||
"email": string,
|
||||
"phone": string,
|
||||
"location": string,
|
||||
"summary": string
|
||||
},
|
||||
"work": [
|
||||
{
|
||||
"position": string,
|
||||
"company": string,
|
||||
"summary": string,
|
||||
"highlights": string[]
|
||||
}
|
||||
],
|
||||
"skills": string[],
|
||||
"education": [
|
||||
{
|
||||
"degree": string,
|
||||
"institution": string
|
||||
}
|
||||
]
|
||||
}`;
|
||||
|
||||
try {
|
||||
const controller = new AbortController();
|
||||
const timeout = setTimeout(() => controller.abort(), 15000);
|
||||
|
||||
const res = await fetch(endpoint, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
Authorization: `Bearer ${apiKey}`,
|
||||
"Content-Type": "application/json",
|
||||
...(isRouter ? { "HTTP-Referer": "https://directwire.io", "X-Title": "DirectWire Resume Parser" } : {}),
|
||||
},
|
||||
signal: controller.signal,
|
||||
body: JSON.stringify({
|
||||
model,
|
||||
temperature: 0.1,
|
||||
messages: [
|
||||
{
|
||||
role: "system",
|
||||
content:
|
||||
"You are a professional ATS resume parser. Extract structured JSON strictly matching: { profile: { name: string, headline: string, email: string, phone: string, location: string, summary: string }, work: [{ position: string, company: string, summary: string, highlights: string[] }], skills: string[], education: [{ degree: string, institution: string }] }",
|
||||
},
|
||||
{ role: "system", content: systemPrompt },
|
||||
{
|
||||
role: "user",
|
||||
content: `Extract structured JSON resume data:\n\n${rawText.slice(0, 10000)}`,
|
||||
content: `Extract structured JSON resume from this text:\n\n${rawText.slice(0, 12000)}`,
|
||||
},
|
||||
],
|
||||
response_format: { type: "json_object" },
|
||||
}),
|
||||
});
|
||||
|
||||
if (!res.ok) return null;
|
||||
clearTimeout(timeout);
|
||||
|
||||
if (!res.ok) {
|
||||
console.warn("[Resume Parser] LLM API responded with error:", res.status, await res.text());
|
||||
return null;
|
||||
}
|
||||
|
||||
const json = await res.json();
|
||||
const content = json.choices?.[0]?.message?.content;
|
||||
let content = json.choices?.[0]?.message?.content;
|
||||
if (!content) return null;
|
||||
|
||||
return JSON.parse(content);
|
||||
} catch (e) {
|
||||
console.error("LLM parse fallback triggered:", e);
|
||||
// Clean markdown code fence wrappers if present
|
||||
content = content.replace(/^```(?:json)?\s*/i, "").replace(/\s*```$/i, "").trim();
|
||||
|
||||
const parsed = JSON.parse(content);
|
||||
if (parsed && (parsed.profile || Array.isArray(parsed.work))) {
|
||||
console.log("[Resume Parser] Successfully extracted structured resume with LLM.");
|
||||
return parsed;
|
||||
}
|
||||
return null;
|
||||
} catch (e: any) {
|
||||
console.error("[Resume Parser] LLM parse call failed or timed out:", e.message);
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
|
@ -469,7 +528,8 @@ export async function POST(req: Request) {
|
|||
|
||||
// Try LLM parsing first if API key configured, otherwise use high-precision local parsing
|
||||
let parsedData = await parseResumeWithLLM(rawText);
|
||||
if (!parsedData || !parsedData.profile || !parsedData.profile.name) {
|
||||
if (!parsedData || !parsedData.profile || (!parsedData.profile.name && (!Array.isArray(parsedData.work) || parsedData.work.length === 0))) {
|
||||
console.log("[Resume Parser] LLM returned empty or incomplete data; applying local heuristic extraction.");
|
||||
parsedData = parseResumeFull(rawText);
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue