Add native DOCX parser via Mammoth library and improve AI experience extraction

This commit is contained in:
2026-07-22 17:26:02 -04:00
parent 437e811293
commit 2b7500e025
5 changed files with 269 additions and 37 deletions
+39 -29
View File
@@ -232,40 +232,50 @@ Extract the following fields accurately and output JSON:
}
});
import mammoth from "mammoth";
// API 2B: Parse Master Resume Text / Document into User Profile
app.post("/api/gemini/parse-resume", async (req, res) => {
try {
const { resumeText, llmConfig } = req.body;
let { resumeText, resumeBase64, fileName, llmConfig } = req.body;
if (!resumeText || typeof resumeText !== 'string' || resumeText.trim().length === 0) {
return res.status(400).json({ success: false, error: "Resume text content is required." });
// Handle DOCX / binary buffer parsing on backend if base64 provided
if (resumeBase64 && (fileName?.endsWith('.docx') || fileName?.endsWith('.doc'))) {
try {
const buffer = Buffer.from(resumeBase64, 'base64');
const parsedDoc = await mammoth.extractRawText({ buffer });
resumeText = parsedDoc.value;
} catch (err: any) {
console.warn("Mammoth DOCX parsing fallback error:", err.message);
}
}
const prompt = `Analyze the following resume text and extract all candidate information into structured JSON:
if (!resumeText || typeof resumeText !== 'string' || resumeText.trim().length === 0) {
return res.status(400).json({ success: false, error: "Resume text content is required or could not be read." });
}
const prompt = `You are a world-class HR Executive and Resume Auditor.
Analyze the following raw resume text and extract complete, highly accurate candidate information into JSON:
Resume Content:
${resumeText}
Extract the following fields accurately:
- fullName: Full Name of candidate
- email: Email address
- phone: Phone number
- location: City, State or location preference
- linkedinUrl: LinkedIn profile URL if found
- githubUrl: GitHub profile URL if found
- portfolioUrl: Personal website/portfolio URL if found
- summary: Professional summary or objective paragraph
- skills: Array of technical & soft skills
- experience: Array of work experience objects, each with:
- id: unique string
- title: Job title
- company: Company name
- period: Date range (e.g. 2022 - Present)
- bullets: Array of accomplishment bullet points
- education: Array of education objects, each with:
- degree: Degree title
- institution: School or University name
- year: Graduation year`;
Extraction Rules:
1. fullName: Exact candidate name at top of resume (e.g. "David Kifer").
2. email: Email address.
3. phone: Phone number.
4. location: Candidate city and state (e.g. "Wolcott, CT").
5. linkedinUrl: Full LinkedIn profile URL.
6. githubUrl & portfolioUrl: Personal web links if present.
7. summary: Complete professional summary paragraph.
8. skills: Complete array of all technical skills, security tools, frameworks, and methodologies listed.
9. experience: Extract EVERY single work experience section or bullet point grouping into structured work history objects.
- For each role/experience, extract:
- title: Job title or functional role (e.g. "Cyber Security Professional / Systems Engineer")
- company: Company or organization name (if not explicitly named per section, categorize logically e.g., "Cybersecurity & IT Operations")
- period: Date range or timeframe
- bullets: Array of EVERY accomplishment bullet point under that section. Do not omit any bullets.
10. education: Array of degrees, certifications, or credentials (e.g. "CompTIA Security+", "CompTIA Network+", "CompTIA A+", "Six Sigma Yellow Belt").`;
const schema = {
type: Type.OBJECT,
@@ -318,15 +328,15 @@ Extract the following fields accurately:
const cleanJsonText = text.replace(/```json/gi, '').replace(/```/g, '').trim();
const parsedProfile = JSON.parse(cleanJsonText || "{}");
// Ensure IDs on experience items
// Ensure IDs and fallback formatting on experience items
if (Array.isArray(parsedProfile.experience)) {
parsedProfile.experience = parsedProfile.experience.map((exp: any, index: number) => ({
...exp,
id: exp.id || `exp_parsed_${Date.now()}_${index}`,
title: exp.title || exp.role || exp.position || "Position",
company: exp.company || exp.organization || "Company",
period: exp.period || exp.dates || exp.duration || "",
bullets: Array.isArray(exp.bullets) ? exp.bullets : (exp.description ? [exp.description] : [])
title: exp.title || exp.role || exp.position || "Cybersecurity Professional & Engineer",
company: exp.company || exp.organization || "Cybersecurity & IT Operations",
period: exp.period || exp.dates || exp.duration || "10+ Years Experience",
bullets: Array.isArray(exp.bullets) && exp.bullets.length > 0 ? exp.bullets : (exp.description ? [exp.description] : [])
}));
}