From f89873f083653f310548a06ee33af902d4b07fb7 Mon Sep 17 00:00:00 2001 From: Amruth Pillai Date: Sun, 6 Sep 2026 06:23:16 +0200 Subject: [PATCH] test: evaluate ATS PDF and DOCX extraction --- pnpm-lock.yaml | 9 + tooling/ats-export-evaluation/README.md | 17 ++ .../evaluation.integration.test.ts | 171 ++++++++++++ tooling/ats-export-evaluation/extract.ts | 154 +++++++++++ tooling/ats-export-evaluation/fixture.ts | 250 ++++++++++++++++++ tooling/ats-export-evaluation/metrics.test.ts | 43 +++ tooling/ats-export-evaluation/metrics.ts | 118 +++++++++ tooling/package.json | 3 + tooling/vitest.config.ts | 14 + 9 files changed, 779 insertions(+) create mode 100644 tooling/ats-export-evaluation/README.md create mode 100644 tooling/ats-export-evaluation/evaluation.integration.test.ts create mode 100644 tooling/ats-export-evaluation/extract.ts create mode 100644 tooling/ats-export-evaluation/fixture.ts create mode 100644 tooling/ats-export-evaluation/metrics.test.ts create mode 100644 tooling/ats-export-evaluation/metrics.ts create mode 100644 tooling/vitest.config.ts diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index f1d350c16..017cc32f1 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -1349,9 +1349,15 @@ importers: '@reactive-resume/config': specifier: workspace:* version: link:../packages/config + '@reactive-resume/docx': + specifier: workspace:* + version: link:../packages/docx '@reactive-resume/env': specifier: workspace:* version: link:../packages/env + '@reactive-resume/pdf': + specifier: workspace:* + version: link:../packages/pdf '@reactive-resume/resume': specifier: workspace:* version: link:../packages/resume @@ -1364,6 +1370,9 @@ importers: drizzle-orm: specifier: 1.0.0-rc.4 version: 1.0.0-rc.4(@types/pg@8.23.1)(pg@8.23.0)(zod@4.5.4) + pdfjs-dist: + specifier: 6.3.289 + version: 6.3.289 pg: specifier: ^8.23.0 version: 8.23.0 diff --git a/tooling/ats-export-evaluation/README.md b/tooling/ats-export-evaluation/README.md new file mode 100644 index 000000000..70292a9d1 --- /dev/null +++ b/tooling/ats-export-evaluation/README.md @@ -0,0 +1,17 @@ +# ATS export evaluation + +`evaluation.integration.test.ts` renders deterministic synthetic resume data through current +`ResumeDocument` and `buildDocx`, then measures extraction with installed PDF.js and DOCX XML. +`metrics.test.ts` locks raw distinct-token recall, order, duplicate, and semantic-grouping behavior, +including deliberate drop/duplicate regressions. + +Run from repository root: + +```sh +pnpm --filter @reactive-resume/tooling test +``` + +The integration test writes PDF/DOCX fixtures and raw-count reports to +`tooling/ats-export-evaluation/test-results/` (ignored test output). Results explicitly distinguish +local extraction measurements from vendor parser accuracy. Steps 1–2 add no ATS preset; a later +product decision can use measured deficiencies from `ats-export-report.md`. diff --git a/tooling/ats-export-evaluation/evaluation.integration.test.ts b/tooling/ats-export-evaluation/evaluation.integration.test.ts new file mode 100644 index 000000000..280dd6694 --- /dev/null +++ b/tooling/ats-export-evaluation/evaluation.integration.test.ts @@ -0,0 +1,171 @@ +// @vitest-environment happy-dom + +import type { SectionTitleResolver } from "@reactive-resume/pdf/section-title"; +import type { ExportVariant } from "./fixture"; +import type { ExportMetrics } from "./metrics"; +import { mkdir, writeFile } from "node:fs/promises"; +import { join, resolve } from "node:path"; +import { describe, expect, it } from "vitest"; +import { buildDocx } from "@reactive-resume/docx"; +import { createResumePdfFile } from "@reactive-resume/pdf/server"; +import { extractDocx, extractPdf } from "./extract"; +import { createSyntheticCorpus } from "./fixture"; +import { evaluateExport, tokenize } from "./metrics"; + +const outputDirectory = resolve(process.cwd(), "ats-export-evaluation/test-results"); + +const pdfTitleResolver: SectionTitleResolver = ({ defaultEnglishTitle, sectionId }) => defaultEnglishTitle ?? sectionId; +const docxTitleResolver = (sectionId: string) => sectionId; + +type FormatResult = { + format: "pdf" | "docx"; + metrics: ExportMetrics; + links: readonly string[]; + paragraphs: number; + pageCount?: number; + fontCount?: number; + numberingDefinitions?: number; + numberedParagraphs?: number; + hiddenLeaks: readonly string[]; +}; + +type VariantResult = { + variant: ExportVariant; + formats: FormatResult[]; +}; + +function hiddenLeaks(paragraphs: readonly string[], hiddenTokens: readonly string[]): string[] { + const observed = paragraphs.flatMap(tokenize); + return hiddenTokens.filter((value) => { + const expected = tokenize(value); + return observed.some((_, index) => expected.every((token, offset) => observed[index + offset] === token)); + }); +} + +async function measureVariant(variant: ExportVariant): Promise { + const corpus = createSyntheticCorpus(variant); + const before = JSON.stringify(corpus.data); + const pdfFile = await createResumePdfFile({ + data: corpus.data, + filename: `${corpus.name}.pdf`, + template: corpus.data.metadata.template, + resolveSectionTitle: pdfTitleResolver, + }); + const pdfBytes = new Uint8Array(await pdfFile.arrayBuffer()); + const pdf = await extractPdf(pdfBytes); + await writeFile(join(outputDirectory, `${corpus.name}.pdf`), pdfBytes); + + const docxBlob = await buildDocx(corpus.data, docxTitleResolver); + const docxBytes = new Uint8Array(await docxBlob.arrayBuffer()); + const docx = await extractDocx(docxBytes); + await writeFile(join(outputDirectory, `${corpus.name}.docx`), docxBytes); + + expect(JSON.stringify(corpus.data)).toBe(before); + + return { + variant, + formats: [ + { + format: "pdf", + metrics: evaluateExport(corpus, { paragraphs: pdf.paragraphs, links: pdf.links }), + links: pdf.links, + paragraphs: pdf.paragraphs.length, + pageCount: pdf.raw.pageCount, + fontCount: pdf.raw.fonts.length, + hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens), + }, + { + format: "docx", + metrics: evaluateExport(corpus, { + paragraphs: docx.paragraphs.map((paragraph) => paragraph.text), + links: docx.links, + }), + links: docx.links, + paragraphs: docx.paragraphs.length, + numberingDefinitions: docx.numberingDefinitions, + numberedParagraphs: docx.numberedParagraphs, + hiddenLeaks: hiddenLeaks( + docx.paragraphs.map((paragraph) => paragraph.text), + corpus.hiddenTokens, + ), + }, + ], + }; +} + +function percentage(value: number): string { + return `${(value * 100).toFixed(1)}%`; +} + +function reportMarkdown(results: readonly VariantResult[]): string { + const lines = [ + "# ATS export evaluation", + "", + "Synthetic extraction measurements only. These are not vendor parsing accuracy claims.", + "", + "Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses pairs expected in the same authored field group and recovered in the same extracted paragraph/line.", + "", + "Corpus: one deterministic resume fixture per layout variant, covering header/contact, two roles, free-text dates, education, skills, project, custom section, long lines, links, hidden item, and CJK text. Hidden item tokens are intentionally excluded from expected recall and checked for leakage.", + "", + "| Variant | Format | Recall raw | Order raw | Duplicate raw | Grouping raw | Pages/paragraphs | Links | Numbering defs/paragraphs | Hidden leaks |", + "| --- | --- | --- | --- | --- | --- | ---: | ---: | ---: | --- |", + ]; + for (const result of results) { + for (const format of result.formats) { + const m = format.metrics; + lines.push( + `| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | ${format.links.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`, + ); + } + } + lines.push( + "", + "PDF font objects and raw link targets are recorded in adjacent JSON output.", + "", + "## Concrete loss/order evidence", + "", + ); + for (const result of results) { + for (const format of result.formats) { + const m = format.metrics; + lines.push( + `- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}.`, + ); + } + } + lines.push(""); + return lines.join("\n"); +} + +describe("current unchanged PDF and DOCX exports", () => { + it("measures two-column and full-width synthetic corpus without mutating input", { timeout: 120_000 }, async () => { + await mkdir(outputDirectory, { recursive: true }); + const results = [await measureVariant("two-column"), await measureVariant("full-width")]; + const report = { + claimBoundary: "Local extraction metrics only; no vendor parsing accuracy claim.", + tokenRules: "NFC, en-US case-folding, Unicode letter/number runs; punctuation separates tokens.", + corpus: { + variants: results.map((result) => result.variant), + expectedDistinctTokens: createSyntheticCorpus("full-width") + .tokens.flatMap((entry) => tokenize(entry.value)) + .filter((value, index, values) => values.indexOf(value) === index).length, + }, + results, + }; + await writeFile(join(outputDirectory, "ats-export-report.json"), `${JSON.stringify(report, null, 2)}\n`); + await writeFile(join(outputDirectory, "ats-export-report.md"), reportMarkdown(results)); + + for (const result of results) { + const pdf = result.formats.find((format) => format.format === "pdf"); + const docx = result.formats.find((format) => format.format === "docx"); + if (!pdf || !docx) throw new Error(`Missing measured format for ${result.variant}`); + expect(pdf.metrics.recall.denominator).toBeGreaterThan(20); + expect(docx.metrics.recall.denominator).toBe(pdf.metrics.recall.denominator); + expect(pdf.hiddenLeaks).toEqual([]); + expect(docx.hiddenLeaks).toEqual([]); + expect(pdf.fontCount).toBeGreaterThan(0); + expect(docx.numberingDefinitions).toBeGreaterThan(0); + expect(docx.numberedParagraphs).toBe(0); + } + }); +}); diff --git a/tooling/ats-export-evaluation/extract.ts b/tooling/ats-export-evaluation/extract.ts new file mode 100644 index 000000000..97db2ab99 --- /dev/null +++ b/tooling/ats-export-evaluation/extract.ts @@ -0,0 +1,154 @@ +import type { ExtractedDocument, PdfDocumentLike, RawExtraction } from "@reactive-resume/resume/ats-pdf"; +import { execFile } from "node:child_process"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { promisify } from "node:util"; +import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs"; +import { buildExtractedDocument, harvestPdfDocument } from "@reactive-resume/resume/ats-pdf"; + +const execFileAsync = promisify(execFile); + +export type PdfExtraction = { + raw: RawExtraction; + document: ExtractedDocument; + paragraphs: readonly string[]; + links: readonly string[]; +}; + +export type DocxParagraph = { + text: string; + numbering: { numId: string; level: string; format: string; marker: string } | null; +}; + +export type DocxExtraction = { + paragraphs: readonly DocxParagraph[]; + links: readonly string[]; + numberingDefinitions: number; + numberedParagraphs: number; +}; + +const unescapeXml = (value: string): string => + value + .replaceAll("<", "<") + .replaceAll(">", ">") + .replaceAll(""", '"') + .replaceAll("'", "'") + .replaceAll("&", "&"); + +const attribute = (attributes: string, name: string): string | null => { + const match = attributes.match(new RegExp(`(?:^|\\s)(?:[A-Za-z][\\w-]*:)?${name}="([^"]*)"`)); + return match ? unescapeXml(match[1] ?? "") : null; +}; + +const zipEntry = async (archivePath: string, entry: string): Promise => { + const { stdout } = await execFileAsync("unzip", ["-p", archivePath, entry]); + return stdout; +}; + +export async function extractPdf(bytes: Uint8Array): Promise { + const loadingTask = getDocument({ data: new Uint8Array(bytes), fontExtraProperties: true }); + try { + const document = (await loadingTask.promise) as unknown as PdfDocumentLike; + const raw = await harvestPdfDocument(document, { + file: { name: "synthetic-resume.pdf", sizeBytes: bytes.byteLength, magicBytesOk: true }, + }); + const extracted = buildExtractedDocument(raw); + return { + raw, + document: extracted, + paragraphs: extracted.lines.map((line) => line.text), + links: raw.links.flatMap((link) => (link.url ? [link.url] : [])), + }; + } finally { + await loadingTask.destroy(); + } +} + +function parseNumbering(numberingXml: string): Map { + const formats = new Map(); + const abstractDefinitions = new Map(); + for (const abstract of numberingXml.matchAll(/]*)>([\s\S]*?)<\/w:abstractNum>/g)) { + const abstractId = attribute(abstract[1] ?? "", "abstractNumId"); + if (!abstractId) continue; + for (const level of (abstract[2] ?? "").matchAll(/]*)>([\s\S]*?)<\/w:lvl>/g)) { + const levelId = attribute(level[1] ?? "", "ilvl") ?? "0"; + const format = attribute(level[2] ?? "", "val") ?? "unknown"; + const marker = attribute((level[2] ?? "").match(/]*)\/>/)?.[1] ?? "", "val") ?? ""; + abstractDefinitions.set(`${abstractId}:${levelId}`, { format, marker }); + } + } + for (const numbering of numberingXml.matchAll(/]*)>([\s\S]*?)<\/w:num>/g)) { + const numId = attribute(numbering[1] ?? "", "numId"); + const abstractId = attribute((numbering[2] ?? "").match(/]*)\/>/)?.[1] ?? "", "val"); + if (!numId || !abstractId) continue; + for (const level of ["0", "1", "2", "3", "4", "5", "6", "7", "8"]) { + const definition = abstractDefinitions.get(`${abstractId}:${level}`); + if (definition) formats.set(`${numId}:${level}`, definition); + } + } + return formats; +} + +function parseDocxParagraphs(documentXml: string, numbering: Map) { + const paragraphs: DocxParagraph[] = []; + for (const paragraph of documentXml.matchAll(/]*>([\s\S]*?)<\/w:p>/g)) { + const body = paragraph[1] ?? ""; + const text = [...body.matchAll(/]*>([\s\S]*?)<\/w:t>/g)] + .map((match) => unescapeXml(match[1] ?? "")) + .join(""); + const numPr = body.match(/]*>([\s\S]*?)<\/w:numPr>/)?.[1]; + const numId = numPr ? numPr.match(/]*)\/>/) : null; + const level = numPr ? numPr.match(/]*)\/>/) : null; + const numIdValue = numId ? attribute(numId[1] ?? "", "val") : null; + const levelValue = level ? (attribute(level[1] ?? "", "val") ?? "0") : null; + const definition = numIdValue && levelValue ? numbering.get(`${numIdValue}:${levelValue}`) : undefined; + paragraphs.push({ + text, + numbering: + numIdValue && levelValue && definition + ? { numId: numIdValue, level: levelValue, format: definition.format, marker: definition.marker } + : null, + }); + } + return paragraphs; +} + +function parseDocxLinks(documentXml: string, relationshipsXml: string): string[] { + const relationships = new Map(); + for (const relationship of relationshipsXml.matchAll(/]*)\/>/g)) { + const id = attribute(relationship[1] ?? "", "Id"); + const target = attribute(relationship[1] ?? "", "Target"); + if (id && target) relationships.set(id, target); + } + const links: string[] = []; + for (const hyperlink of documentXml.matchAll(/]*)>/g)) { + const id = attribute(hyperlink[1] ?? "", "id"); + const target = id ? relationships.get(id) : undefined; + if (target) links.push(target); + } + return links; +} + +export async function extractDocx(bytes: Uint8Array): Promise { + const tempDirectory = await mkdtemp(join(tmpdir(), "reactive-resume-ats-")); + const archivePath = join(tempDirectory, "resume.docx"); + try { + await writeFile(archivePath, bytes); + const [documentXml, numberingXml, relationshipsXml] = await Promise.all([ + zipEntry(archivePath, "word/document.xml"), + zipEntry(archivePath, "word/numbering.xml"), + zipEntry(archivePath, "word/_rels/document.xml.rels"), + ]); + const numbering = parseNumbering(numberingXml); + const paragraphs = parseDocxParagraphs(documentXml, numbering); + return { + paragraphs, + links: parseDocxLinks(documentXml, relationshipsXml), + numberingDefinitions: numbering.size, + numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length, + }; + } finally { + await rm(tempDirectory, { recursive: true, force: true }); + } +} diff --git a/tooling/ats-export-evaluation/fixture.ts b/tooling/ats-export-evaluation/fixture.ts new file mode 100644 index 000000000..5d5529d18 --- /dev/null +++ b/tooling/ats-export-evaluation/fixture.ts @@ -0,0 +1,250 @@ +import type { ResumeData } from "@reactive-resume/schema/resume/data"; +import type { EvaluationCorpus, ExpectedToken } from "./metrics"; +import { sampleResumeData } from "@reactive-resume/schema/resume/sample"; + +export type ExportVariant = "two-column" | "full-width"; + +export type SyntheticCorpus = EvaluationCorpus & { + data: ResumeData; + hiddenTokens: readonly string[]; + links: readonly string[]; +}; + +const token = (value: string, group: string): ExpectedToken => ({ value, group }); + +const text = (html: string) => html.replace(/<[^>]+>/g, " "); + +const SUMMARY_HTML = + "

summary-signal-alpha leads longline-calibration with multilingual 東京大学 context and durable systems.

"; +const EXPERIENCE_ROLE_ONE_HTML = + "
  • role-bullet-alpha shipped resilient pipeline-observability under longline-pressure.

  • role-bullet-beta measured queue-latency and improved release-safety.

"; +const EXPERIENCE_ROLE_TWO_HTML = + "

role-bullet-gamma guided distributed-runtime migration for platform-reliability.

"; +const EDUCATION_HTML = "

education-signal-delta researched multilingual retrieval and evaluation.

"; +const PROJECT_HTML = "

project-signal-epsilon demonstrates export-fixture determinism.

"; +const CUSTOM_HTML = "

custom-signal-zeta preserves authored custom content and free-text dates.

"; + +const expectedTokens: readonly ExpectedToken[] = [ + ...[ + "Mira Kova", + "Principal Systems Architect", + "mira.kova@example.com", + "+49 30 555 0142", + "Berlin 東京", + "mirakova.dev", + "orbit-field-omega", + ].map((value) => token(value, "header")), + ...[text(SUMMARY_HTML)].map((value) => token(value, "summary")), + ...[ + "Northstar Robotics", + "2018-02 — Present", + "Berlin", + "Staff Platform Engineer", + "2018-02 to 2020-12", + text(EXPERIENCE_ROLE_ONE_HTML), + "Principal Reliability Engineer", + "2021 / Present", + text(EXPERIENCE_ROLE_TWO_HTML), + ].map((value) => token(value, "experience")), + ...[ + "東京大学", + "Master of Computer Science", + "Distributed Systems", + "3.98 GPA", + "2014 — 2018 (long academic period)", + "東京", + text(EDUCATION_HTML), + ].map((value) => token(value, "education")), + ...["TypeScript", "skill-keyword-alpha", "Kubernetes", "skill-keyword-beta"].map((value) => token(value, "skills")), + ...["Export Observatory", "2022 to Winter 2024", text(PROJECT_HTML), "project.example/observatory"].map((value) => + token(value, "projects"), + ), + ...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")), +]; + +export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta"] as const; + +const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false }); + +export function createSyntheticCorpus(variant: ExportVariant): SyntheticCorpus { + const data = structuredClone(sampleResumeData); + data.picture.hidden = true; + data.basics = { + name: "Mira Kova", + headline: "Principal Systems Architect", + email: "mira.kova@example.com", + phone: "+49 30 555 0142", + location: "Berlin 東京", + website: resolveWebsite("https://mirakova.dev", "mirakova.dev"), + customFields: [ + { + id: "synthetic-field-omega", + icon: "", + text: "orbit-field-omega", + link: "https://orbit.example/omega", + }, + ], + }; + data.summary = { + ...data.summary, + title: "Summary", + content: SUMMARY_HTML, + }; + data.sections.profiles = { + ...data.sections.profiles, + title: "Profiles", + items: [ + { + id: "synthetic-profile-omega", + hidden: false, + icon: "", + iconColor: "", + network: "OrbitNet", + username: "orbit-profile-omega", + website: resolveWebsite("https://orbit.example/profile", "orbit.example/profile"), + }, + ], + }; + data.sections.experience = { + ...data.sections.experience, + title: "Experience", + items: [ + { + id: "synthetic-experience-northstar", + hidden: false, + company: "Northstar Robotics", + position: "", + location: "Berlin", + period: "2018-02 — Present", + website: resolveWebsite("https://northstar.example/jobs", "northstar.example/jobs"), + roles: [ + { + id: "synthetic-role-staff", + position: "Staff Platform Engineer", + period: "2018-02 to 2020-12", + description: EXPERIENCE_ROLE_ONE_HTML, + }, + { + id: "synthetic-role-principal", + position: "Principal Reliability Engineer", + period: "2021 / Present", + description: EXPERIENCE_ROLE_TWO_HTML, + }, + ], + description: "", + }, + { + id: "synthetic-hidden-experience", + hidden: true, + company: HIDDEN_TOKENS[0], + position: "", + location: "", + period: "", + website: resolveWebsite("https://hidden.example", "hidden.example"), + roles: [], + description: `

${HIDDEN_TOKENS[1]}

`, + }, + ], + }; + data.sections.education = { + ...data.sections.education, + title: "Education", + items: [ + { + id: "synthetic-education-tokyo", + hidden: false, + school: "東京大学", + degree: "Master of Computer Science", + area: "Distributed Systems", + grade: "3.98 GPA", + location: "東京", + period: "2014 — 2018 (long academic period)", + website: resolveWebsite("https://u-tokyo.example/program", "u-tokyo.example/program"), + description: EDUCATION_HTML, + }, + ], + }; + data.sections.skills = { + ...data.sections.skills, + title: "Skills", + items: [ + { + id: "synthetic-skill-typescript", + hidden: false, + icon: "", + iconColor: "", + name: "TypeScript", + proficiency: "Advanced", + level: 4, + keywords: ["skill-keyword-alpha"], + }, + { + id: "synthetic-skill-kubernetes", + hidden: false, + icon: "", + iconColor: "", + name: "Kubernetes", + proficiency: "Expert", + level: 5, + keywords: ["skill-keyword-beta"], + }, + ], + }; + data.sections.projects = { + ...data.sections.projects, + title: "Projects", + items: [ + { + id: "synthetic-project-observatory", + hidden: false, + name: "Export Observatory", + period: "2022 to Winter 2024", + website: resolveWebsite("https://project.example/observatory", "project.example/observatory"), + description: PROJECT_HTML, + }, + ], + }; + data.customSections = [ + { + id: "custom-ats-evidence", + type: "summary", + title: "Custom Evidence", + icon: "", + columns: 1, + hidden: false, + showHeading: true, + keepTogether: false, + startOnNewPage: false, + items: [{ id: "synthetic-custom-zeta", hidden: false, content: CUSTOM_HTML }], + }, + ]; + + const mainSections = ["profiles", "summary", "experience", "education", "projects"]; + const sidebarSections = ["skills", "custom-ats-evidence"]; + const page = { + fullWidth: variant === "full-width", + main: variant === "full-width" ? [...mainSections, ...sidebarSections] : mainSections, + sidebar: variant === "full-width" ? [] : sidebarSections, + }; + data.metadata = { + ...data.metadata, + template: variant === "full-width" ? "onyx" : "gengar", + layout: { ...data.metadata.layout, pages: [page] }, + page: { ...data.metadata.page, locale: "en-US", hideIcons: true, hideSectionIcons: true }, + }; + + return { + name: `ats-${variant}`, + tokens: expectedTokens, + data, + hiddenTokens: HIDDEN_TOKENS, + links: [ + "https://mirakova.dev", + "https://orbit.example/omega", + "https://orbit.example/profile", + "https://northstar.example/jobs", + "https://u-tokyo.example/program", + "https://project.example/observatory", + ], + }; +} diff --git a/tooling/ats-export-evaluation/metrics.test.ts b/tooling/ats-export-evaluation/metrics.test.ts new file mode 100644 index 000000000..4fe6b9d13 --- /dev/null +++ b/tooling/ats-export-evaluation/metrics.test.ts @@ -0,0 +1,43 @@ +import { describe, expect, it } from "vitest"; +import { evaluateExport } from "./metrics"; + +const corpus = { + name: "synthetic", + tokens: [ + { value: "Alpha", group: "experience" }, + { value: "Bravo", group: "experience" }, + { value: "Charlie", group: "education" }, + ] as const, +}; + +describe("evaluateExport", () => { + it("counts distinct recall, order, duplicate, and grouping losses from raw tokens", () => { + const result = evaluateExport(corpus, { + paragraphs: ["Alpha Bravo Bravo", "Charlie"], + links: [], + }); + + expect(result.recall).toEqual({ numerator: 3, denominator: 3, value: 1 }); + expect(result.order).toEqual({ numerator: 2, denominator: 2, value: 1 }); + expect(result.duplicates).toEqual({ expected: 3, observed: 4, extra: 1 }); + expect(result.missingTokens).toEqual([]); + expect(result.outOfOrderPairs).toEqual([]); + expect(result.grouping).toEqual({ numerator: 1, denominator: 2, value: 0.5 }); + }); + + it("detects a dropped token and an inverted pair", () => { + const result = evaluateExport(corpus, { + paragraphs: ["Bravo", "Alpha"], + links: [], + }); + + expect(result.recall).toEqual({ numerator: 2, denominator: 3, value: 2 / 3 }); + expect(result.order).toEqual({ numerator: 0, denominator: 2, value: 0 }); + expect(result.duplicates).toEqual({ expected: 3, observed: 2, extra: 0 }); + expect(result.missingTokens).toEqual(["charlie"]); + expect(result.outOfOrderPairs).toEqual([ + ["alpha", "bravo"], + ["bravo", "charlie"], + ]); + }); +}); diff --git a/tooling/ats-export-evaluation/metrics.ts b/tooling/ats-export-evaluation/metrics.ts new file mode 100644 index 000000000..8033a2713 --- /dev/null +++ b/tooling/ats-export-evaluation/metrics.ts @@ -0,0 +1,118 @@ +export type ExpectedToken = { + value: string; + group: string; +}; + +export type EvaluationCorpus = { + name: string; + tokens: readonly ExpectedToken[]; +}; + +export type ExtractedExport = { + /** Paragraphs/lines in the extractor's returned order. */ + paragraphs: readonly string[]; + links: readonly string[]; +}; + +export type ExportMetrics = { + recall: { numerator: number; denominator: number; value: number }; + order: { numerator: number; denominator: number; value: number }; + duplicates: { expected: number; observed: number; extra: number }; + grouping: { numerator: number; denominator: number; value: number }; + missingTokens: readonly string[]; + outOfOrderPairs: readonly (readonly [string, string])[]; + observedTokens: number; +}; + +/** + * Tokenization used by this evaluation only. It is deliberately transparent and locale-neutral: + * Unicode letters/numbers stay intact, punctuation is a separator, and matching is case-folded. + * This is a corpus metric, not a claim about any vendor parser. + */ +export function tokenize(value: string): string[] { + return ( + value + .normalize("NFC") + .match(/[\p{L}\p{N}]+/gu) + ?.map((token) => token.toLocaleLowerCase("en-US")) ?? [] + ); +} + +const metric = (numerator: number, denominator: number) => ({ + numerator, + denominator, + value: denominator === 0 ? 1 : numerator / denominator, +}); + +/** + * Computes raw extraction measurements from corpus tokens and extractor paragraphs. + * + * Recall uses distinct expected tokens. Duplicate accounting separately reports all matching + * occurrences, so dropping a token cannot be hidden by duplicate output. Order and grouping use + * the first occurrence of each distinct expected token, keeping those measures interpretable when + * an export repeats a heading or bullet. + */ +export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExport): ExportMetrics { + const expected = corpus.tokens.flatMap((entry) => + tokenize(entry.value).map((value) => ({ value, group: entry.group })), + ); + const expectedByValue = new Map(); + for (const [index, token] of expected.entries()) { + if (!expectedByValue.has(token.value)) expectedByValue.set(token.value, { group: token.group, index }); + } + + const observedByParagraph = extracted.paragraphs.map(tokenize); + const observed = observedByParagraph.flat(); + const expectedValues = [...expectedByValue.keys()]; + const observedPositions = new Map(); + for (const [index, token] of observed.entries()) { + if (!observedPositions.has(token)) observedPositions.set(token, index); + } + + const recoveredDistinct = expectedValues.filter((value) => observedPositions.has(value)).length; + const expectedTokenPairs = expectedValues.flatMap((value, index) => { + const next = expectedValues[index + 1]; + return next ? ([[value, next]] as const) : []; + }); + const outOfOrderPairs = expectedTokenPairs.filter(([left, right]) => { + const leftPosition = observedPositions.get(left); + const rightPosition = observedPositions.get(right); + return leftPosition === undefined || rightPosition === undefined || leftPosition >= rightPosition; + }); + + const observedParagraphPositions = new Map(); + for (const [paragraphIndex, paragraphTokens] of observedByParagraph.entries()) { + for (const token of paragraphTokens) { + if (!observedParagraphPositions.has(token)) observedParagraphPositions.set(token, paragraphIndex); + } + } + const groupedPairs = expectedTokenPairs.filter(([left, right]) => { + const leftEntry = expectedByValue.get(left); + const rightEntry = expectedByValue.get(right); + const leftParagraph = observedParagraphPositions.get(left); + const rightParagraph = observedParagraphPositions.get(right); + return ( + leftEntry?.group === rightEntry?.group && + leftParagraph !== undefined && + rightParagraph !== undefined && + leftParagraph === rightParagraph + ); + }).length; + + const expectedValuesSet = new Set(expectedValues); + const observedExpectedOccurrences = observed.filter((token) => expectedValuesSet.has(token)).length; + + return { + recall: metric(recoveredDistinct, expectedValues.length), + order: metric(expectedTokenPairs.length - outOfOrderPairs.length, expectedTokenPairs.length), + duplicates: { + expected: expected.length, + observed: observedExpectedOccurrences, + extra: Math.max(0, observedExpectedOccurrences - expected.length), + }, + grouping: metric(groupedPairs, expectedTokenPairs.length), + missingTokens: expectedValues.filter((value) => !observedPositions.has(value)), + outOfOrderPairs, + observedTokens: observed.length, + }; +} diff --git a/tooling/package.json b/tooling/package.json index 464279f4d..0e73b059e 100644 --- a/tooling/package.json +++ b/tooling/package.json @@ -18,9 +18,12 @@ "devDependencies": { "@lingui/format-po": "^6.6.0", "@reactive-resume/config": "workspace:*", + "@reactive-resume/docx": "workspace:*", "@reactive-resume/env": "workspace:*", + "@reactive-resume/pdf": "workspace:*", "@reactive-resume/resume": "workspace:*", "@types/pg": "^8.23.1", + "pdfjs-dist": "6.3.289", "@typescript/native-preview": "7.0.0-dev.20260707.2", "drizzle-orm": "1.0.0-rc.4", "pg": "^8.23.0", diff --git a/tooling/vitest.config.ts b/tooling/vitest.config.ts new file mode 100644 index 000000000..dce75bff4 --- /dev/null +++ b/tooling/vitest.config.ts @@ -0,0 +1,14 @@ +import { fileURLToPath } from "node:url"; +// @boundaries-ignore root shared Vitest config +import { createVitestProjectConfig } from "../vitest.shared.mts"; + +const config = createVitestProjectConfig({ + name: "@reactive-resume/tooling", + dirname: fileURLToPath(new URL(".", import.meta.url)), +}); + +export default { + ...config, + test: { ...config.test, include: ["**/*.{test,spec}.?(c|m)[jt]s?(x)"] }, + oxc: { jsx: { runtime: "automatic" as const } }, +};