mirror of
https://github.com/AmruthPillai/Reactive-Resume.git
synced 2026-10-03 10:13:47 +10:00
feat(import): parse a PDF resume without an AI provider (#3400)
* feat(import): parse a PDF resume without an AI provider Importing a PDF required a connected AI provider, so anyone without a paid API key could only import the three JSON formats. Almost nobody arrives with one of those files; they arrive with a PDF. The first thing a new user tries to do was blocked behind bringing their own key. Adds a deterministic parser that reads the text out of the PDF in the browser and prefills the builder. It pulls the contact block, segments the body on conventional headings, and maps entries to real items, reusing the ATS period parser for dates so a date range is not mistaken for a phone number. Nothing is thrown away: header parts that do not map to a field go into the description, and unrecognized headings become custom sections. The imported sections are placed on the page so the result renders straight away. Output is validated against the resume schema before it is returned. Text extraction groups items by baseline rather than trusting hasEOL, and turns wide column gaps into a double space, which is what lets a row split into company, position and location. The AI path still runs when a provider is connected. Word import is unchanged and still requires one. Closes #3334 * fix(import): keep every section and entry the PDF actually contains Review found three ways the parser lost or mangled content, all of them reproducible. A document whose first heading was not one of the known aliases never started a section, because unknown-heading detection was gated on a section already being open. Everything after it was swallowed as contact header text. The header block is now bounded by where the contact details stop, so a heading is recognized wherever it appears. An entry spreading company, position and dates over three lines was imported as two malformed items. A line that introduces an entry now merges into the open entry instead of starting a second one. An uppercase company such as ACME CORPORATION was read as a section heading and fragmented the entry. A heading candidate followed by a date line is now treated as an entry header, which is what it is. Also escape single quotes, and construct the PDF worker inside the try so the nested worker is terminated even if construction throws. Title-case headings are deliberately still not treated as headings: company and school names are title case too, and splitting on them would fragment real entries. Such a section stays in the preceding one with its text intact rather than risking loss. * fix(import): look past a multi-line preamble before calling a line a heading The previous guard only inspected the next line, so an uppercase company followed by a separate role line and then the dates was still read as a section heading. The experience or education entry was moved into a custom section and lost. Heading detection now scans a two-line window for the date that marks an entry, and stops early at a bullet so a genuine heading whose section opens with bullet points is still recognized. The window can suppress a real heading whose first entry puts a bare date two lines below it. That is the deliberate direction to fail in: a missed heading leaves the text in the preceding section, while a misread entry fragments structured content. * fix(import): collect an entry preamble until its dates appear An entry that spread company, role, location and dates over four lines was imported as two broken items: the company with no dates, and the location carrying the period. The cause was in entry grouping rather than heading detection. Lines before a date were only folded into the entry header when the date sat on the very next line; anything earlier fell through to the description. Preamble lines are now collected into the entry header until the dates turn up, bounded by the same lookahead and stopping at a bullet, so an undated section cannot swallow itself. The heading lookahead widens to four lines to match, which is the realistic maximum for company, role, location and dates. * fix(import): harden local PDF resume parsing * chore(import): document audited HTML construction --------- Co-authored-by: Amruth Pillai <im.amruth@gmail.com>
This commit is contained in:
co-authored by
Amruth Pillai
parent
ef47baf243
commit
cce6d64afa
@@ -5,6 +5,7 @@
|
||||
"private": true,
|
||||
"exports": {
|
||||
"./json-resume": "./src/json-resume.tsx",
|
||||
"./plain-text": "./src/plain-text.ts",
|
||||
"./reactive-resume-json": "./src/reactive-resume-json.tsx",
|
||||
"./reactive-resume-v4-json": "./src/reactive-resume-v4-json.tsx"
|
||||
},
|
||||
|
||||
@@ -0,0 +1,321 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { resumeDataSchema } from "@reactive-resume/schema/resume/data";
|
||||
import { parseResumeText } from "./plain-text";
|
||||
|
||||
const SAMPLE = `Ada Lovelace
|
||||
Senior Software Engineer
|
||||
Berlin, Germany | ada@example.com | +44 20 7946 0100 | https://ada.dev
|
||||
|
||||
SUMMARY
|
||||
Engineer with 10 years building analytical systems.
|
||||
|
||||
WORK EXPERIENCE
|
||||
Analytical Engines Senior Engineer Berlin
|
||||
Jan 2020 - Present
|
||||
• Led the difference engine rewrite
|
||||
• Mentored four junior engineers
|
||||
Babbage Ltd Engineer London
|
||||
Mar 2016 - Dec 2019
|
||||
• Built the punch card pipeline
|
||||
|
||||
EDUCATION
|
||||
University of London BSc Mathematics
|
||||
2012 - 2016
|
||||
|
||||
SKILLS
|
||||
TypeScript, Rust, PostgreSQL
|
||||
|
||||
LANGUAGES
|
||||
English (Native)
|
||||
German (B2)
|
||||
|
||||
CERTIFICATIONS
|
||||
AWS Solutions Architect Amazon 2021
|
||||
`;
|
||||
|
||||
describe("parseResumeText", () => {
|
||||
const data = parseResumeText(SAMPLE);
|
||||
|
||||
it("always returns schema-valid resume data", () => {
|
||||
expect(() => resumeDataSchema.parse(data)).not.toThrow();
|
||||
});
|
||||
|
||||
it("reads the contact block", () => {
|
||||
expect(data.basics).toMatchObject({
|
||||
name: "Ada Lovelace",
|
||||
headline: "Senior Software Engineer",
|
||||
email: "ada@example.com",
|
||||
phone: "+44 20 7946 0100",
|
||||
location: "Berlin, Germany",
|
||||
});
|
||||
expect(data.basics.website.url).toBe("https://ada.dev");
|
||||
});
|
||||
|
||||
it("reads the summary as rich text", () => {
|
||||
expect(data.summary.content).toBe("<p>Engineer with 10 years building analytical systems.</p>");
|
||||
});
|
||||
|
||||
it("splits experience into one entry per role", () => {
|
||||
expect(data.sections.experience.items).toHaveLength(2);
|
||||
expect(data.sections.experience.items[0]).toMatchObject({
|
||||
company: "Analytical Engines",
|
||||
position: "Senior Engineer",
|
||||
location: "Berlin",
|
||||
period: "Jan 2020 - Present",
|
||||
});
|
||||
expect(data.sections.experience.items[1]).toMatchObject({
|
||||
company: "Babbage Ltd",
|
||||
position: "Engineer",
|
||||
period: "Mar 2016 - Dec 2019",
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps bullets as a list in the description", () => {
|
||||
expect(data.sections.experience.items[0]?.description).toBe(
|
||||
"<ul><li>Led the difference engine rewrite</li><li>Mentored four junior engineers</li></ul>",
|
||||
);
|
||||
});
|
||||
|
||||
it("reads education", () => {
|
||||
expect(data.sections.education.items[0]).toMatchObject({
|
||||
school: "University of London",
|
||||
degree: "BSc Mathematics",
|
||||
period: "2012 - 2016",
|
||||
});
|
||||
});
|
||||
|
||||
it("splits a comma separated skills line", () => {
|
||||
expect(data.sections.skills.items.map((item) => item.name)).toEqual(["TypeScript", "Rust", "PostgreSQL"]);
|
||||
});
|
||||
|
||||
it("splits a language from its fluency", () => {
|
||||
expect(data.sections.languages.items).toMatchObject([
|
||||
{ language: "English", fluency: "Native" },
|
||||
{ language: "German", fluency: "B2" },
|
||||
]);
|
||||
});
|
||||
|
||||
it("reads a trailing year as the certification date", () => {
|
||||
expect(data.sections.certifications.items[0]).toMatchObject({
|
||||
title: "AWS Solutions Architect",
|
||||
issuer: "Amazon",
|
||||
date: "2021",
|
||||
});
|
||||
});
|
||||
|
||||
it("places every populated section on the page in document order", () => {
|
||||
expect(data.metadata.layout.pages[0]?.main).toEqual([
|
||||
"summary",
|
||||
"experience",
|
||||
"education",
|
||||
"skills",
|
||||
"languages",
|
||||
"certifications",
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseResumeText edge cases", () => {
|
||||
it("returns usable data for empty input", () => {
|
||||
const data = parseResumeText("");
|
||||
expect(() => resumeDataSchema.parse(data)).not.toThrow();
|
||||
expect(data.basics.name).toBe("");
|
||||
expect(data.metadata.layout.pages[0]?.main).toEqual([]);
|
||||
});
|
||||
|
||||
it("does not mistake a date range for a phone number", () => {
|
||||
const data = parseResumeText("Ada Lovelace\nBerlin\n2016 - 2019\n");
|
||||
expect(data.basics.phone).toBe("");
|
||||
});
|
||||
|
||||
it("keeps an unrecognized heading as a custom section", () => {
|
||||
const data = parseResumeText("Ada\n\nSKILLS\nRust\n\nSPEAKING\nGave a talk at a conference\n");
|
||||
expect(data.customSections).toHaveLength(1);
|
||||
expect(data.customSections[0]).toMatchObject({ type: "summary", title: "SPEAKING" });
|
||||
expect(data.metadata.layout.pages[0]?.main).toContain(data.customSections[0]?.id);
|
||||
});
|
||||
|
||||
it("keeps unclassified header parts in the description rather than dropping them", () => {
|
||||
const data = parseResumeText("EXPERIENCE\nAcme Engineer Berlin Remote Contract\n2020 - 2022\n");
|
||||
expect(data.sections.experience.items[0]?.description).toContain("Remote");
|
||||
expect(data.sections.experience.items[0]?.description).toContain("Contract");
|
||||
});
|
||||
|
||||
it("escapes markup found in the source text", () => {
|
||||
const data = parseResumeText("SUMMARY\nI write <script>alert(1)</script> safely\n");
|
||||
expect(data.summary.content).toContain("<script>");
|
||||
expect(data.summary.content).not.toContain("<script>");
|
||||
});
|
||||
|
||||
it("treats a heading with a trailing colon as a heading", () => {
|
||||
const data = parseResumeText("Ada\n\nSkills:\nRust, Go\n");
|
||||
expect(data.sections.skills.items.map((item) => item.name)).toEqual(["Rust", "Go"]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseResumeText review findings", () => {
|
||||
it("keeps a section whose heading is the first one in the document", () => {
|
||||
const data = parseResumeText(
|
||||
"Ada Lovelace\nada@example.com\n\nCAREER HIGHLIGHTS\nShipped the difference engine\nMentored the team\n",
|
||||
);
|
||||
|
||||
expect(data.customSections).toHaveLength(1);
|
||||
expect(data.customSections[0]).toMatchObject({ title: "CAREER HIGHLIGHTS" });
|
||||
expect(JSON.stringify(data)).toContain("Shipped the difference engine");
|
||||
});
|
||||
|
||||
it("keeps one entry when company, position and dates sit on separate lines", () => {
|
||||
const data = parseResumeText(
|
||||
"EXPERIENCE\nAnalytical Engines\nSenior Engineer\nJan 2020 - Present\n• Led the rewrite\n",
|
||||
);
|
||||
|
||||
expect(data.sections.experience.items).toHaveLength(1);
|
||||
expect(data.sections.experience.items[0]).toMatchObject({
|
||||
company: "Analytical Engines",
|
||||
position: "Senior Engineer",
|
||||
period: "Jan 2020 - Present",
|
||||
});
|
||||
});
|
||||
|
||||
it("does not turn an uppercase company name into a section heading", () => {
|
||||
const data = parseResumeText("EXPERIENCE\nACME CORPORATION\nJan 2020 - Present\n• Did the work\n");
|
||||
|
||||
expect(data.customSections).toHaveLength(0);
|
||||
expect(data.sections.experience.items[0]).toMatchObject({ company: "ACME CORPORATION" });
|
||||
});
|
||||
|
||||
it("escapes single quotes in extracted text", () => {
|
||||
const data = parseResumeText("SUMMARY\nIt's a resume\n");
|
||||
expect(data.summary.content).toContain("'");
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseResumeText multi-line entry preambles", () => {
|
||||
it("keeps an uppercase company followed by a separate role line as one entry", () => {
|
||||
const data = parseResumeText(
|
||||
"EXPERIENCE\nACME CORPORATION\nSenior Engineer\nJan 2020 - Present\n• Led the rewrite\n",
|
||||
);
|
||||
|
||||
expect(data.customSections).toHaveLength(0);
|
||||
expect(data.sections.experience.items).toHaveLength(1);
|
||||
expect(data.sections.experience.items[0]).toMatchObject({
|
||||
company: "ACME CORPORATION",
|
||||
position: "Senior Engineer",
|
||||
period: "Jan 2020 - Present",
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps an uppercase school followed by a separate degree line as one entry", () => {
|
||||
const data = parseResumeText("EDUCATION\nUNIVERSITY OF LONDON\nBSc Mathematics\n2012 - 2016\n");
|
||||
|
||||
expect(data.customSections).toHaveLength(0);
|
||||
expect(data.sections.education.items).toHaveLength(1);
|
||||
expect(data.sections.education.items[0]).toMatchObject({
|
||||
school: "UNIVERSITY OF LONDON",
|
||||
degree: "BSc Mathematics",
|
||||
period: "2012 - 2016",
|
||||
});
|
||||
});
|
||||
|
||||
it("still recognizes a real heading whose section starts with bullets", () => {
|
||||
const data = parseResumeText("Ada\nada@example.com\n\nCAREER HIGHLIGHTS\n• Shipped in 2019\n• Grew the team\n");
|
||||
|
||||
expect(data.customSections).toHaveLength(1);
|
||||
expect(data.customSections[0]).toMatchObject({ title: "CAREER HIGHLIGHTS" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseResumeText four-line entry preambles", () => {
|
||||
it("keeps company, role, location and dates as one entry", () => {
|
||||
const data = parseResumeText(
|
||||
"EXPERIENCE\nACME CORPORATION\nSenior Engineer\nBerlin, Germany\nJan 2020 - Present\n• Led the rewrite\n",
|
||||
);
|
||||
|
||||
expect(data.customSections).toHaveLength(0);
|
||||
expect(data.sections.experience.items).toHaveLength(1);
|
||||
expect(data.sections.experience.items[0]).toMatchObject({
|
||||
company: "ACME CORPORATION",
|
||||
position: "Senior Engineer",
|
||||
location: "Berlin, Germany",
|
||||
period: "Jan 2020 - Present",
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps school, degree, location and dates as one entry", () => {
|
||||
const data = parseResumeText("EDUCATION\nUNIVERSITY OF LONDON\nBSc Mathematics\nLondon, UK\n2012 - 2016\n");
|
||||
|
||||
expect(data.customSections).toHaveLength(0);
|
||||
expect(data.sections.education.items).toHaveLength(1);
|
||||
expect(data.sections.education.items[0]).toMatchObject({
|
||||
school: "UNIVERSITY OF LONDON",
|
||||
degree: "BSc Mathematics",
|
||||
location: "London, UK",
|
||||
period: "2012 - 2016",
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("parseResumeText keeps content the layout hides", () => {
|
||||
it("recognizes isolated title-case custom section headings", () => {
|
||||
const initial = parseResumeText(
|
||||
"Ada Lovelace\nada@example.com\n\nConferences\nReactConf\nBerlin\n2021\nSpoke about parsers\n",
|
||||
);
|
||||
const subsequent = parseResumeText(
|
||||
"Ada Lovelace\nada@example.com\n\nEXPERIENCE\nAcme Engineer\n2020 - 2022\nBuilt products.\n\nConferences\nReactConf\nBerlin\n2021\nSpoke about parsers\n",
|
||||
);
|
||||
|
||||
expect(initial.customSections[0]).toMatchObject({ title: "Conferences" });
|
||||
expect(subsequent.customSections[0]).toMatchObject({ title: "Conferences" });
|
||||
expect(subsequent.sections.experience.items).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("keeps a dated custom section that opens the body", () => {
|
||||
const data = parseResumeText(
|
||||
"Ada Lovelace\nada@example.com\n\nCONFERENCES\nReactConf\nBerlin\n2021\nSpoke about parsers\n",
|
||||
);
|
||||
|
||||
expect(data.basics.headline).not.toBe("CONFERENCES");
|
||||
expect(data.customSections).toHaveLength(1);
|
||||
expect(data.customSections[0]).toMatchObject({ title: "CONFERENCES" });
|
||||
|
||||
const content = data.customSections[0]?.items[0]?.content ?? "";
|
||||
for (const line of ["ReactConf", "Berlin", "2021", "Spoke about parsers"]) {
|
||||
expect(content).toContain(line);
|
||||
}
|
||||
});
|
||||
|
||||
it("keeps unbulleted descriptions with their own role", () => {
|
||||
const data = parseResumeText(
|
||||
"EXPERIENCE\nAcme Engineer Berlin\nJan 2020 - Present\nBuilt the thing end to end.\nWorked with a team of five.\nBabbage Ltd Engineer London\nMar 2016 - Dec 2019\nDid other work.\n",
|
||||
);
|
||||
|
||||
expect(data.sections.experience.items).toHaveLength(2);
|
||||
expect(data.sections.experience.items[0]?.description).toContain("Built the thing end to end.");
|
||||
expect(data.sections.experience.items[0]?.description).toContain("Worked with a team of five.");
|
||||
expect(data.sections.experience.items[1]).toMatchObject({ company: "Babbage Ltd", location: "London" });
|
||||
});
|
||||
|
||||
it("does not split an entry on a bare year in a section that ignores single dates", () => {
|
||||
const data = parseResumeText(
|
||||
"EXPERIENCE\nAcme Engineer\nJan 2020 - Present\nGrew the team.\nMore work here.\n2022\n",
|
||||
);
|
||||
|
||||
expect(data.sections.experience.items).toHaveLength(1);
|
||||
|
||||
const description = data.sections.experience.items[0]?.description ?? "";
|
||||
for (const line of ["Grew the team.", "More work here.", "2022"]) {
|
||||
expect(description).toContain(line);
|
||||
}
|
||||
});
|
||||
|
||||
it("stays schema-valid when a section starts with its dates", () => {
|
||||
const data = parseResumeText(
|
||||
"EXPERIENCE\nJan 2020 - Present\nAcme Corp\nSenior Engineer\n\nEDUCATION\n2012 - 2016\nUniversity of London\n\nCERTIFICATIONS\n2021\n",
|
||||
);
|
||||
|
||||
expect(() => resumeDataSchema.parse(data)).not.toThrow();
|
||||
expect(data.sections.experience.items[0]).toMatchObject({ company: "Acme Corp", period: "Jan 2020 - Present" });
|
||||
expect(data.sections.education.items[0]).toMatchObject({ school: "University of London" });
|
||||
expect(data.sections.certifications.items).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,593 @@
|
||||
import type { ResumeData, SectionType } from "@reactive-resume/schema/resume/data";
|
||||
import { parsePeriod, parseSingleDate } from "@reactive-resume/resume/ats";
|
||||
import { parseResumeData } from "@reactive-resume/schema/resume/data";
|
||||
import { defaultResumeData } from "@reactive-resume/schema/resume/default";
|
||||
import { generateId } from "@reactive-resume/utils/string";
|
||||
|
||||
type SectionKey = SectionType | "summary";
|
||||
|
||||
type Segment = {
|
||||
key: SectionKey | null;
|
||||
title: string;
|
||||
lines: string[];
|
||||
};
|
||||
|
||||
type RawEntry = {
|
||||
period: string;
|
||||
headerParts: string[];
|
||||
body: string[];
|
||||
};
|
||||
|
||||
const MAX_HEADING_WORDS = 4;
|
||||
const MAX_HEADING_LENGTH = 48;
|
||||
const MAX_ENTRY_HEADER_WORDS = 8;
|
||||
const MAX_LIST_ITEMS = 60;
|
||||
const MIN_PHONE_DIGITS = 7;
|
||||
const MAX_PHONE_DIGITS = 15;
|
||||
|
||||
const SECTION_ALIASES: Readonly<Record<string, SectionKey>> = {
|
||||
summary: "summary",
|
||||
"professional summary": "summary",
|
||||
"career summary": "summary",
|
||||
profile: "summary",
|
||||
"personal profile": "summary",
|
||||
about: "summary",
|
||||
"about me": "summary",
|
||||
objective: "summary",
|
||||
"career objective": "summary",
|
||||
experience: "experience",
|
||||
"work experience": "experience",
|
||||
"professional experience": "experience",
|
||||
employment: "experience",
|
||||
"employment history": "experience",
|
||||
"work history": "experience",
|
||||
"career history": "experience",
|
||||
education: "education",
|
||||
"academic background": "education",
|
||||
"education and training": "education",
|
||||
qualifications: "education",
|
||||
skills: "skills",
|
||||
"technical skills": "skills",
|
||||
"key skills": "skills",
|
||||
"core competencies": "skills",
|
||||
competencies: "skills",
|
||||
expertise: "skills",
|
||||
projects: "projects",
|
||||
"personal projects": "projects",
|
||||
"selected projects": "projects",
|
||||
"side projects": "projects",
|
||||
languages: "languages",
|
||||
interests: "interests",
|
||||
hobbies: "interests",
|
||||
"hobbies and interests": "interests",
|
||||
awards: "awards",
|
||||
honors: "awards",
|
||||
honours: "awards",
|
||||
"awards and honors": "awards",
|
||||
achievements: "awards",
|
||||
certifications: "certifications",
|
||||
certificates: "certifications",
|
||||
licenses: "certifications",
|
||||
"licenses and certifications": "certifications",
|
||||
publications: "publications",
|
||||
papers: "publications",
|
||||
research: "publications",
|
||||
volunteer: "volunteer",
|
||||
volunteering: "volunteer",
|
||||
"volunteer experience": "volunteer",
|
||||
"community involvement": "volunteer",
|
||||
references: "references",
|
||||
profiles: "profiles",
|
||||
links: "profiles",
|
||||
"social profiles": "profiles",
|
||||
};
|
||||
|
||||
const BULLET_PATTERN = /^\s*[-–—•*◦‣·]\s+/;
|
||||
const EMAIL_PATTERN = /[\w.+-]+@[\w-]+\.[\w.-]*\w/;
|
||||
const URL_PATTERN = /\b(?:https?:\/\/|www\.)[^\s,;|•·]+/gi;
|
||||
const PHONE_CANDIDATE = /[+(]?\d[\d\s().+-]{5,}\d/g;
|
||||
const URL_TEST = /\b(?:https?:\/\/|www\.)\S+/i;
|
||||
const HEADER_SCAN_LINES = 6;
|
||||
const ENTRY_PREAMBLE_LOOKAHEAD = 4;
|
||||
const PERIOD_CANDIDATE =
|
||||
/(?:\p{L}{3,}\.?\s+)?(?:\d{1,2}[/.])?\d{4}\s*(?:[-–—~]|to|until|through)\s*(?:(?:\p{L}{3,}\.?\s+)?(?:\d{1,2}[/.])?\d{4}|\p{L}+)/giu;
|
||||
const STRONG_SEPARATOR = /\s*[|•·]\s*|\s{2,}|\s+[–—]\s+/;
|
||||
const SENTENCE_END = /[.!?]$/;
|
||||
const TRAILING_DATES = [/(?:\p{L}{3,}\.?\s+)?(?:\d{1,2}[/.])?(?:19|20)\d{2}$/u, /(?:\d{1,2}[/.])?(?:19|20)\d{2}$/];
|
||||
|
||||
const escapeHtml = (value: string) =>
|
||||
value
|
||||
.replace(/&/g, "&")
|
||||
.replace(/</g, "<")
|
||||
.replace(/>/g, ">")
|
||||
.replace(/"/g, """)
|
||||
.replace(/'/g, "'");
|
||||
|
||||
const normalizeHeading = (line: string) =>
|
||||
line
|
||||
.replace(/[::]\s*$/, "")
|
||||
.replace(/&/g, " and ")
|
||||
.replace(/[^\p{L}\p{N}\s]/gu, " ")
|
||||
.replace(/\s+/g, " ")
|
||||
.trim()
|
||||
.toLowerCase();
|
||||
|
||||
function knownHeading(line: string): SectionKey | null {
|
||||
const normalized = normalizeHeading(line);
|
||||
if (!normalized || normalized.split(" ").length > MAX_HEADING_WORDS) return null;
|
||||
|
||||
return SECTION_ALIASES[normalized] ?? null;
|
||||
}
|
||||
|
||||
function looksLikeHeading(line: string): boolean {
|
||||
const trimmed = line.trim().replace(/[::]$/, "");
|
||||
if (!trimmed || trimmed.length > MAX_HEADING_LENGTH || /\d/.test(trimmed)) return false;
|
||||
if (trimmed.split(/\s+/).length > MAX_HEADING_WORDS) return false;
|
||||
|
||||
const letters = trimmed.replace(/[^\p{L}]/gu, "");
|
||||
if (letters.length < 3) return false;
|
||||
|
||||
return letters === letters.toLocaleUpperCase() && letters !== letters.toLocaleLowerCase();
|
||||
}
|
||||
|
||||
function looksLikeTitleCaseHeading(line: string): boolean {
|
||||
const trimmed = line.trim().replace(/[::]$/, "");
|
||||
if (!trimmed || trimmed.length > MAX_HEADING_LENGTH || /\d/.test(trimmed)) return false;
|
||||
const words = trimmed.split(/\s+/);
|
||||
if (words.length > MAX_HEADING_WORDS) return false;
|
||||
|
||||
return words.every((word) => {
|
||||
const letters = word.replace(/[^\p{L}]/gu, "");
|
||||
if (letters.length === 0) return false;
|
||||
return letters[0] === letters[0]?.toLocaleUpperCase() && letters.slice(1) === letters.slice(1).toLocaleLowerCase();
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a line could be the header of an entry rather than prose belonging to the previous one.
|
||||
*
|
||||
* Unbulleted descriptions are common, and without this test any narrative line that happens to sit
|
||||
* within the lookahead of the next role's dates would be promoted to a header, stealing the current
|
||||
* entry's description and seeding a garbage item from a sentence.
|
||||
*/
|
||||
function looksLikeEntryHeader(line: string): boolean {
|
||||
const trimmed = line.trim();
|
||||
if (STRONG_SEPARATOR.test(trimmed)) return true;
|
||||
if (SENTENCE_END.test(trimmed)) return false;
|
||||
|
||||
return trimmed.split(/\s+/).length <= MAX_ENTRY_HEADER_WORDS;
|
||||
}
|
||||
|
||||
function findPeriod(line: string): string {
|
||||
PERIOD_CANDIDATE.lastIndex = 0;
|
||||
|
||||
for (const match of line.match(PERIOD_CANDIDATE) ?? []) {
|
||||
if (parsePeriod(match)) return match.trim();
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
function findSingleDate(text: string): string {
|
||||
const trimmed = text.trim();
|
||||
|
||||
for (const pattern of TRAILING_DATES) {
|
||||
const value = pattern.exec(trimmed)?.[0]?.trim();
|
||||
if (value && parseSingleDate(value)) return value;
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
function extractPhone(text: string): string {
|
||||
for (const candidate of text.match(PHONE_CANDIDATE) ?? []) {
|
||||
const digits = candidate.replace(/\D/g, "");
|
||||
if (digits.length < MIN_PHONE_DIGITS || digits.length > MAX_PHONE_DIGITS) continue;
|
||||
if (parsePeriod(candidate)) continue;
|
||||
|
||||
return candidate.trim();
|
||||
}
|
||||
|
||||
return "";
|
||||
}
|
||||
|
||||
function isDateLine(line: string, allowSingleDate = true): boolean {
|
||||
if (findPeriod(line)) return true;
|
||||
// A section grouped without single dates reads a bare "2022" as text, so the lookahead has to
|
||||
// agree with it: otherwise that line closes an entry that groupEntries then never reopens.
|
||||
if (!allowSingleDate) return false;
|
||||
|
||||
const bare = line.replace(BULLET_PATTERN, "").trim();
|
||||
return bare !== "" && parseSingleDate(bare) !== null;
|
||||
}
|
||||
|
||||
function introducesEntry(lines: readonly string[], index: number, allowSingleDate = true): boolean {
|
||||
for (let offset = 1; offset <= ENTRY_PREAMBLE_LOOKAHEAD; offset++) {
|
||||
const line = lines[index + offset];
|
||||
if (line === undefined || BULLET_PATTERN.test(line)) return false;
|
||||
if (isDateLine(line, allowSingleDate)) return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
function headerBoundary(lines: string[]): number {
|
||||
let boundary = 0;
|
||||
|
||||
for (const [index, line] of lines.slice(0, HEADER_SCAN_LINES).entries()) {
|
||||
if (knownHeading(line)) break;
|
||||
if (EMAIL_PATTERN.test(line) || URL_TEST.test(line) || extractPhone(line)) boundary = index;
|
||||
}
|
||||
|
||||
return boundary;
|
||||
}
|
||||
|
||||
function splitHeaderParts(text: string): string[] {
|
||||
return text
|
||||
.split(STRONG_SEPARATOR)
|
||||
.map((part) => part.replace(/^[\s,;|•·–—-]+|[\s,;|•·–—-]+$/g, "").trim())
|
||||
.filter(Boolean);
|
||||
}
|
||||
|
||||
function toHtml(lines: string[]): string {
|
||||
const cleaned = lines.map((line) => line.trim()).filter(Boolean);
|
||||
if (cleaned.length === 0) return "";
|
||||
|
||||
const bulleted = cleaned.filter((line) => BULLET_PATTERN.test(line));
|
||||
if (bulleted.length >= 2 && bulleted.length * 2 >= cleaned.length) {
|
||||
const items = cleaned.map((line) => `<li>${escapeHtml(line.replace(BULLET_PATTERN, ""))}</li>`).join(""); // nosemgrep
|
||||
return `<ul>${items}</ul>`; // nosemgrep
|
||||
}
|
||||
|
||||
return cleaned.map((line) => `<p>${escapeHtml(line.replace(BULLET_PATTERN, ""))}</p>`).join(""); // nosemgrep
|
||||
}
|
||||
|
||||
function splitList(lines: string[]): string[] {
|
||||
const values: string[] = [];
|
||||
|
||||
for (const line of lines) {
|
||||
for (const piece of line.replace(BULLET_PATTERN, "").split(/[,;|•·]|\s{3,}/)) {
|
||||
const value = piece.trim();
|
||||
if (value) values.push(value);
|
||||
}
|
||||
}
|
||||
|
||||
return [...new Set(values)].slice(0, MAX_LIST_ITEMS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Guarantees an entry has header text, because every section shape maps `headerParts[0]` onto a
|
||||
* field the resume schema requires to be non-empty. A section whose first line is a date opens an
|
||||
* entry with an empty header, and without this the whole import fails validation on that one item.
|
||||
*/
|
||||
function withHeaderText(entry: RawEntry): RawEntry {
|
||||
if (entry.headerParts.length > 0) return entry;
|
||||
|
||||
const [first, ...rest] = entry.body;
|
||||
return { ...entry, headerParts: splitHeaderParts(first ?? ""), body: rest };
|
||||
}
|
||||
|
||||
function groupEntries(lines: string[], allowSingleDate = false): RawEntry[] {
|
||||
const cleaned = lines.map((line) => line.trim()).filter(Boolean);
|
||||
const entries: RawEntry[] = [];
|
||||
let current: RawEntry | null = null;
|
||||
|
||||
const dateOf = (line: string) => {
|
||||
if (BULLET_PATTERN.test(line)) return "";
|
||||
|
||||
const period = findPeriod(line);
|
||||
if (period) return period;
|
||||
|
||||
return allowSingleDate ? findSingleDate(line) : "";
|
||||
};
|
||||
|
||||
for (const [index, line] of cleaned.entries()) {
|
||||
const date = dateOf(line);
|
||||
|
||||
if (date) {
|
||||
const remainder = splitHeaderParts(line.replace(date, " "));
|
||||
|
||||
if (current && !current.period) {
|
||||
current.period = date;
|
||||
current.headerParts.push(...remainder);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (current) entries.push(current);
|
||||
current = { period: date, headerParts: remainder, body: [] };
|
||||
continue;
|
||||
}
|
||||
|
||||
if (!current) {
|
||||
current = { period: "", headerParts: splitHeaderParts(line), body: [] };
|
||||
continue;
|
||||
}
|
||||
|
||||
const isBullet = BULLET_PATTERN.test(line);
|
||||
const leadsToDate = !isBullet && looksLikeEntryHeader(line) && introducesEntry(cleaned, index, allowSingleDate);
|
||||
|
||||
if (leadsToDate && !current.period && current.body.length === 0) {
|
||||
current.headerParts.push(...splitHeaderParts(line));
|
||||
continue;
|
||||
}
|
||||
|
||||
if (leadsToDate) {
|
||||
entries.push(current);
|
||||
current = { period: "", headerParts: splitHeaderParts(line), body: [] };
|
||||
continue;
|
||||
}
|
||||
|
||||
current.body.push(line);
|
||||
}
|
||||
|
||||
if (current) entries.push(current);
|
||||
|
||||
return entries.map(withHeaderText).filter((entry) => entry.headerParts.length > 0);
|
||||
}
|
||||
|
||||
function entryDescription(entry: RawEntry, usedParts: number): string {
|
||||
const leftover = entry.headerParts.slice(usedParts);
|
||||
return toHtml([...leftover, ...entry.body]);
|
||||
}
|
||||
|
||||
const baseItem = () => ({ id: generateId(), hidden: false });
|
||||
|
||||
const emptyWebsite = { url: "", label: "", inlineLink: false };
|
||||
|
||||
function buildSectionItems(key: SectionKey, lines: string[]): unknown[] {
|
||||
if (key === "skills" || key === "interests") {
|
||||
return splitList(lines).map((name) => ({
|
||||
...baseItem(),
|
||||
icon: "",
|
||||
iconColor: "",
|
||||
name,
|
||||
...(key === "skills" ? { proficiency: "", level: 0, keywords: [] } : { keywords: [] }),
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "languages") {
|
||||
return lines
|
||||
.map((line) => line.replace(BULLET_PATTERN, "").trim())
|
||||
.filter(Boolean)
|
||||
.map((line) => {
|
||||
const match = /^(.+?)\s*[([–—-]\s*(.+?)\s*[)\]]?$/.exec(line);
|
||||
return {
|
||||
...baseItem(),
|
||||
language: (match?.[1] ?? line).trim(),
|
||||
fluency: (match?.[2] ?? "").trim(),
|
||||
level: 0,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
if (key === "profiles") {
|
||||
return lines
|
||||
.map((line) => line.replace(BULLET_PATTERN, "").trim())
|
||||
.filter(Boolean)
|
||||
.map((line) => {
|
||||
const url = line.match(URL_PATTERN)?.[0] ?? "";
|
||||
const network = splitHeaderParts(line.replace(url, " "))[0] ?? line;
|
||||
return {
|
||||
...baseItem(),
|
||||
icon: "",
|
||||
iconColor: "",
|
||||
network,
|
||||
username: "",
|
||||
website: { url, label: "", inlineLink: false },
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
const dated = key === "awards" || key === "certifications" || key === "publications";
|
||||
const entries = groupEntries(lines, dated);
|
||||
|
||||
if (key === "experience") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
company: entry.headerParts[0] ?? "",
|
||||
position: entry.headerParts[1] ?? "",
|
||||
location: entry.headerParts[2] ?? "",
|
||||
period: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 3),
|
||||
roles: [],
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "education") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
school: entry.headerParts[0] ?? "",
|
||||
degree: entry.headerParts[1] ?? "",
|
||||
area: "",
|
||||
grade: "",
|
||||
location: entry.headerParts[2] ?? "",
|
||||
period: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 3),
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "projects") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
name: entry.headerParts[0] ?? "",
|
||||
period: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 1),
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "volunteer") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
organization: entry.headerParts[0] ?? "",
|
||||
location: entry.headerParts[1] ?? "",
|
||||
period: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 2),
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "awards") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
title: entry.headerParts[0] ?? "",
|
||||
awarder: entry.headerParts[1] ?? "",
|
||||
date: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 2),
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "certifications") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
title: entry.headerParts[0] ?? "",
|
||||
issuer: entry.headerParts[1] ?? "",
|
||||
date: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 2),
|
||||
}));
|
||||
}
|
||||
|
||||
if (key === "publications") {
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
title: entry.headerParts[0] ?? "",
|
||||
publisher: entry.headerParts[1] ?? "",
|
||||
date: entry.period,
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 2),
|
||||
}));
|
||||
}
|
||||
|
||||
return entries.map((entry) => ({
|
||||
...baseItem(),
|
||||
name: entry.headerParts[0] ?? "",
|
||||
position: entry.headerParts[1] ?? "",
|
||||
phone: "",
|
||||
website: emptyWebsite,
|
||||
description: entryDescription(entry, 2),
|
||||
}));
|
||||
}
|
||||
|
||||
function segment(lines: string[]): { header: string[]; segments: Segment[] } {
|
||||
const records = lines
|
||||
.map((line, index) => ({ line: line.trim(), precededByBlank: index > 0 && !lines[index - 1]?.trim() }))
|
||||
.filter((record) => record.line);
|
||||
const cleaned = records.map((record) => record.line);
|
||||
const boundary = headerBoundary(cleaned);
|
||||
const header: string[] = [];
|
||||
const segments: Segment[] = [];
|
||||
let current: Segment | null = null;
|
||||
|
||||
for (const [index, { line, precededByBlank }] of records.entries()) {
|
||||
const key = knownHeading(line);
|
||||
const isolatedTitleCase = precededByBlank && looksLikeTitleCaseHeading(line);
|
||||
// With no section open the entry-preamble guard has nothing to protect: skipping it there keeps
|
||||
// a dated custom section that opens the body from being swallowed into the contact header.
|
||||
const unknown =
|
||||
key === null &&
|
||||
index > boundary &&
|
||||
(looksLikeHeading(line) || isolatedTitleCase) &&
|
||||
(current === null || isolatedTitleCase || !introducesEntry(cleaned, index));
|
||||
|
||||
if (key !== null || unknown) {
|
||||
if (current) segments.push(current);
|
||||
current = { key, title: line.replace(/[::]\s*$/, "").trim(), lines: [] };
|
||||
continue;
|
||||
}
|
||||
|
||||
if (current) current.lines.push(line);
|
||||
else header.push(line);
|
||||
}
|
||||
|
||||
if (current) segments.push(current);
|
||||
|
||||
return { header, segments };
|
||||
}
|
||||
|
||||
function parseHeader(lines: string[]) {
|
||||
const joined = lines.join(" ");
|
||||
const email = joined.match(EMAIL_PATTERN)?.[0] ?? "";
|
||||
const phone = extractPhone(joined);
|
||||
const urls = joined.match(URL_PATTERN) ?? [];
|
||||
|
||||
const strip = (value: string) => {
|
||||
let result = value;
|
||||
if (email) result = result.replace(email, " ");
|
||||
if (phone) result = result.replace(phone, " ");
|
||||
for (const url of urls) result = result.replace(url, " ");
|
||||
return result.replace(/\s+/g, " ").trim();
|
||||
};
|
||||
|
||||
const remaining = lines.map(strip).filter(Boolean);
|
||||
const name = remaining[0] ?? "";
|
||||
const rest = remaining.slice(1).flatMap(splitHeaderParts).filter(Boolean);
|
||||
const locationIndex = rest.findIndex((part) => /,/.test(part) && !/\d{4}/.test(part));
|
||||
|
||||
return {
|
||||
name,
|
||||
headline: locationIndex === 0 ? (rest[1] ?? "") : (rest[0] ?? ""),
|
||||
location: locationIndex === -1 ? "" : (rest[locationIndex] ?? ""),
|
||||
email,
|
||||
phone,
|
||||
website: urls[0] ?? "",
|
||||
};
|
||||
}
|
||||
|
||||
export function parseResumeText(text: string): ResumeData {
|
||||
const lines = text.replace(/\r\n?/g, "\n").split("\n");
|
||||
const { header, segments } = segment(lines);
|
||||
const contact = parseHeader(header);
|
||||
|
||||
const data: ResumeData = structuredClone(defaultResumeData);
|
||||
const order: string[] = [];
|
||||
|
||||
data.basics.name = contact.name;
|
||||
data.basics.headline = contact.headline;
|
||||
data.basics.email = contact.email;
|
||||
data.basics.phone = contact.phone;
|
||||
data.basics.location = contact.location;
|
||||
data.basics.website = { url: contact.website, label: "" };
|
||||
|
||||
for (const item of segments) {
|
||||
if (item.lines.length === 0) continue;
|
||||
|
||||
if (item.key === "summary") {
|
||||
const content = toHtml(item.lines);
|
||||
data.summary.content = data.summary.content ? `${data.summary.content}${content}` : content;
|
||||
if (!order.includes("summary")) order.push("summary");
|
||||
continue;
|
||||
}
|
||||
|
||||
if (item.key === null) {
|
||||
const id = generateId();
|
||||
data.customSections.push({
|
||||
id,
|
||||
type: "summary",
|
||||
title: item.title,
|
||||
icon: "",
|
||||
columns: 1,
|
||||
hidden: false,
|
||||
keepTogether: false,
|
||||
startOnNewPage: false,
|
||||
items: [{ id: generateId(), hidden: false, content: toHtml(item.lines) }],
|
||||
});
|
||||
order.push(id);
|
||||
continue;
|
||||
}
|
||||
|
||||
const items = buildSectionItems(item.key, item.lines);
|
||||
if (items.length === 0) continue;
|
||||
|
||||
const section = data.sections[item.key];
|
||||
section.items = [...section.items, ...items] as typeof section.items;
|
||||
if (!order.includes(item.key)) order.push(item.key);
|
||||
}
|
||||
|
||||
data.metadata.layout.pages = [{ fullWidth: true, main: order, sidebar: [] }];
|
||||
|
||||
return parseResumeData(data);
|
||||
}
|
||||
Reference in New Issue
Block a user