Enable high-precision OpenRouter LLM extraction for resume parsing to prevent scrambled titles and organization names
This commit is contained in:
parent
9205b88458
commit
e84871926b
1 changed files with 73 additions and 13 deletions
|
|
@ -340,7 +340,10 @@ function parseResumeFull(rawText: string) {
|
||||||
|
|
||||||
async function parseResumeWithLLM(rawText: string) {
|
async function parseResumeWithLLM(rawText: string) {
|
||||||
const apiKey = process.env.OPENROUTER_API_KEY || process.env.OPENAI_API_KEY;
|
const apiKey = process.env.OPENROUTER_API_KEY || process.env.OPENAI_API_KEY;
|
||||||
if (!apiKey) return null;
|
if (!apiKey) {
|
||||||
|
console.log("[Resume Parser] No OPENROUTER_API_KEY or OPENAI_API_KEY configured; using heuristic parser.");
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
const isRouter = !!process.env.OPENROUTER_API_KEY;
|
const isRouter = !!process.env.OPENROUTER_API_KEY;
|
||||||
const endpoint = isRouter
|
const endpoint = isRouter
|
||||||
|
|
@ -348,38 +351,94 @@ async function parseResumeWithLLM(rawText: string) {
|
||||||
: `${process.env.OPENAI_BASE_URL || "https://api.openai.com/v1"}/chat/completions`;
|
: `${process.env.OPENAI_BASE_URL || "https://api.openai.com/v1"}/chat/completions`;
|
||||||
const model = process.env.OPENROUTER_MODEL || process.env.OPENAI_MODEL || (isRouter ? "google/gemini-2.5-flash" : "gpt-4o-mini");
|
const model = process.env.OPENROUTER_MODEL || process.env.OPENAI_MODEL || (isRouter ? "google/gemini-2.5-flash" : "gpt-4o-mini");
|
||||||
|
|
||||||
|
console.log(`[Resume Parser] Invoking LLM parser via ${endpoint} using model: ${model}`);
|
||||||
|
|
||||||
|
const systemPrompt = `You are a high-precision ATS resume parser and information extraction engine.
|
||||||
|
CRITICAL EXTRACTION GUIDELINES:
|
||||||
|
1. Parse raw, messy, or multi-column resume text into clean, structured JSON.
|
||||||
|
2. Accurately untangle company names and position titles:
|
||||||
|
- Identify the real organization/employer (e.g. "Pokemoto", "Christmas Wish CT", "Personal Homelab").
|
||||||
|
- NEVER output generic placeholders like "Organization" or "Company" if the actual organization name is stated anywhere nearby.
|
||||||
|
- Do NOT place dates or location text into the company name field.
|
||||||
|
3. Identify the candidate's actual name, email, phone number, and location accurately.
|
||||||
|
4. Extract every distinct job experience entry with its authentic position, company, and bullet points (highlights).
|
||||||
|
5. Extract explicit skills into the skills array.
|
||||||
|
6. Extract schools/universities and degrees into the education array.
|
||||||
|
7. Return strictly a JSON object with this schema:
|
||||||
|
{
|
||||||
|
"profile": {
|
||||||
|
"name": string,
|
||||||
|
"headline": string,
|
||||||
|
"email": string,
|
||||||
|
"phone": string,
|
||||||
|
"location": string,
|
||||||
|
"summary": string
|
||||||
|
},
|
||||||
|
"work": [
|
||||||
|
{
|
||||||
|
"position": string,
|
||||||
|
"company": string,
|
||||||
|
"summary": string,
|
||||||
|
"highlights": string[]
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"skills": string[],
|
||||||
|
"education": [
|
||||||
|
{
|
||||||
|
"degree": string,
|
||||||
|
"institution": string
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}`;
|
||||||
|
|
||||||
try {
|
try {
|
||||||
|
const controller = new AbortController();
|
||||||
|
const timeout = setTimeout(() => controller.abort(), 15000);
|
||||||
|
|
||||||
const res = await fetch(endpoint, {
|
const res = await fetch(endpoint, {
|
||||||
method: "POST",
|
method: "POST",
|
||||||
headers: {
|
headers: {
|
||||||
Authorization: `Bearer ${apiKey}`,
|
Authorization: `Bearer ${apiKey}`,
|
||||||
"Content-Type": "application/json",
|
"Content-Type": "application/json",
|
||||||
|
...(isRouter ? { "HTTP-Referer": "https://directwire.io", "X-Title": "DirectWire Resume Parser" } : {}),
|
||||||
},
|
},
|
||||||
|
signal: controller.signal,
|
||||||
body: JSON.stringify({
|
body: JSON.stringify({
|
||||||
model,
|
model,
|
||||||
|
temperature: 0.1,
|
||||||
messages: [
|
messages: [
|
||||||
{
|
{ role: "system", content: systemPrompt },
|
||||||
role: "system",
|
|
||||||
content:
|
|
||||||
"You are a professional ATS resume parser. Extract structured JSON strictly matching: { profile: { name: string, headline: string, email: string, phone: string, location: string, summary: string }, work: [{ position: string, company: string, summary: string, highlights: string[] }], skills: string[], education: [{ degree: string, institution: string }] }",
|
|
||||||
},
|
|
||||||
{
|
{
|
||||||
role: "user",
|
role: "user",
|
||||||
content: `Extract structured JSON resume data:\n\n${rawText.slice(0, 10000)}`,
|
content: `Extract structured JSON resume from this text:\n\n${rawText.slice(0, 12000)}`,
|
||||||
},
|
},
|
||||||
],
|
],
|
||||||
response_format: { type: "json_object" },
|
response_format: { type: "json_object" },
|
||||||
}),
|
}),
|
||||||
});
|
});
|
||||||
|
|
||||||
if (!res.ok) return null;
|
clearTimeout(timeout);
|
||||||
|
|
||||||
|
if (!res.ok) {
|
||||||
|
console.warn("[Resume Parser] LLM API responded with error:", res.status, await res.text());
|
||||||
|
return null;
|
||||||
|
}
|
||||||
|
|
||||||
const json = await res.json();
|
const json = await res.json();
|
||||||
const content = json.choices?.[0]?.message?.content;
|
let content = json.choices?.[0]?.message?.content;
|
||||||
if (!content) return null;
|
if (!content) return null;
|
||||||
|
|
||||||
return JSON.parse(content);
|
// Clean markdown code fence wrappers if present
|
||||||
} catch (e) {
|
content = content.replace(/^```(?:json)?\s*/i, "").replace(/\s*```$/i, "").trim();
|
||||||
console.error("LLM parse fallback triggered:", e);
|
|
||||||
|
const parsed = JSON.parse(content);
|
||||||
|
if (parsed && (parsed.profile || Array.isArray(parsed.work))) {
|
||||||
|
console.log("[Resume Parser] Successfully extracted structured resume with LLM.");
|
||||||
|
return parsed;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
} catch (e: any) {
|
||||||
|
console.error("[Resume Parser] LLM parse call failed or timed out:", e.message);
|
||||||
return null;
|
return null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -469,7 +528,8 @@ export async function POST(req: Request) {
|
||||||
|
|
||||||
// Try LLM parsing first if API key configured, otherwise use high-precision local parsing
|
// Try LLM parsing first if API key configured, otherwise use high-precision local parsing
|
||||||
let parsedData = await parseResumeWithLLM(rawText);
|
let parsedData = await parseResumeWithLLM(rawText);
|
||||||
if (!parsedData || !parsedData.profile || !parsedData.profile.name) {
|
if (!parsedData || !parsedData.profile || (!parsedData.profile.name && (!Array.isArray(parsedData.work) || parsedData.work.length === 0))) {
|
||||||
|
console.log("[Resume Parser] LLM returned empty or incomplete data; applying local heuristic extraction.");
|
||||||
parsedData = parseResumeFull(rawText);
|
parsedData = parseResumeFull(rawText);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue