From f89873f083653f310548a06ee33af902d4b07fb7 Mon Sep 17 00:00:00 2001 From: Amruth Pillai Date: Sun, 6 Sep 2026 06:23:16 +0200 Subject: [PATCH 1/4] test: evaluate ATS PDF and DOCX extraction --- pnpm-lock.yaml | 9 + tooling/ats-export-evaluation/README.md | 17 ++ .../evaluation.integration.test.ts | 171 ++++++++++++ tooling/ats-export-evaluation/extract.ts | 154 +++++++++++ tooling/ats-export-evaluation/fixture.ts | 250 ++++++++++++++++++ tooling/ats-export-evaluation/metrics.test.ts | 43 +++ tooling/ats-export-evaluation/metrics.ts | 118 +++++++++ tooling/package.json | 3 + tooling/vitest.config.ts | 14 + 9 files changed, 779 insertions(+) create mode 100644 tooling/ats-export-evaluation/README.md create mode 100644 tooling/ats-export-evaluation/evaluation.integration.test.ts create mode 100644 tooling/ats-export-evaluation/extract.ts create mode 100644 tooling/ats-export-evaluation/fixture.ts create mode 100644 tooling/ats-export-evaluation/metrics.test.ts create mode 100644 tooling/ats-export-evaluation/metrics.ts create mode 100644 tooling/vitest.config.ts diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index f1d350c16..017cc32f1 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -1349,9 +1349,15 @@ importers: '@reactive-resume/config': specifier: workspace:* version: link:../packages/config + '@reactive-resume/docx': + specifier: workspace:* + version: link:../packages/docx '@reactive-resume/env': specifier: workspace:* version: link:../packages/env + '@reactive-resume/pdf': + specifier: workspace:* + version: link:../packages/pdf '@reactive-resume/resume': specifier: workspace:* version: link:../packages/resume @@ -1364,6 +1370,9 @@ importers: drizzle-orm: specifier: 1.0.0-rc.4 version: 1.0.0-rc.4(@types/pg@8.23.1)(pg@8.23.0)(zod@4.5.4) + pdfjs-dist: + specifier: 6.3.289 + version: 6.3.289 pg: specifier: ^8.23.0 version: 8.23.0 diff --git a/tooling/ats-export-evaluation/README.md b/tooling/ats-export-evaluation/README.md new file mode 100644 index 000000000..70292a9d1 --- /dev/null +++ b/tooling/ats-export-evaluation/README.md @@ -0,0 +1,17 @@ +# ATS export evaluation + +`evaluation.integration.test.ts` renders deterministic synthetic resume data through current +`ResumeDocument` and `buildDocx`, then measures extraction with installed PDF.js and DOCX XML. +`metrics.test.ts` locks raw distinct-token recall, order, duplicate, and semantic-grouping behavior, +including deliberate drop/duplicate regressions. + +Run from repository root: + +```sh +pnpm --filter @reactive-resume/tooling test +``` + +The integration test writes PDF/DOCX fixtures and raw-count reports to +`tooling/ats-export-evaluation/test-results/` (ignored test output). Results explicitly distinguish +local extraction measurements from vendor parser accuracy. Steps 1–2 add no ATS preset; a later +product decision can use measured deficiencies from `ats-export-report.md`. diff --git a/tooling/ats-export-evaluation/evaluation.integration.test.ts b/tooling/ats-export-evaluation/evaluation.integration.test.ts new file mode 100644 index 000000000..280dd6694 --- /dev/null +++ b/tooling/ats-export-evaluation/evaluation.integration.test.ts @@ -0,0 +1,171 @@ +// @vitest-environment happy-dom + +import type { SectionTitleResolver } from "@reactive-resume/pdf/section-title"; +import type { ExportVariant } from "./fixture"; +import type { ExportMetrics } from "./metrics"; +import { mkdir, writeFile } from "node:fs/promises"; +import { join, resolve } from "node:path"; +import { describe, expect, it } from "vitest"; +import { buildDocx } from "@reactive-resume/docx"; +import { createResumePdfFile } from "@reactive-resume/pdf/server"; +import { extractDocx, extractPdf } from "./extract"; +import { createSyntheticCorpus } from "./fixture"; +import { evaluateExport, tokenize } from "./metrics"; + +const outputDirectory = resolve(process.cwd(), "ats-export-evaluation/test-results"); + +const pdfTitleResolver: SectionTitleResolver = ({ defaultEnglishTitle, sectionId }) => defaultEnglishTitle ?? sectionId; +const docxTitleResolver = (sectionId: string) => sectionId; + +type FormatResult = { + format: "pdf" | "docx"; + metrics: ExportMetrics; + links: readonly string[]; + paragraphs: number; + pageCount?: number; + fontCount?: number; + numberingDefinitions?: number; + numberedParagraphs?: number; + hiddenLeaks: readonly string[]; +}; + +type VariantResult = { + variant: ExportVariant; + formats: FormatResult[]; +}; + +function hiddenLeaks(paragraphs: readonly string[], hiddenTokens: readonly string[]): string[] { + const observed = paragraphs.flatMap(tokenize); + return hiddenTokens.filter((value) => { + const expected = tokenize(value); + return observed.some((_, index) => expected.every((token, offset) => observed[index + offset] === token)); + }); +} + +async function measureVariant(variant: ExportVariant): Promise { + const corpus = createSyntheticCorpus(variant); + const before = JSON.stringify(corpus.data); + const pdfFile = await createResumePdfFile({ + data: corpus.data, + filename: `${corpus.name}.pdf`, + template: corpus.data.metadata.template, + resolveSectionTitle: pdfTitleResolver, + }); + const pdfBytes = new Uint8Array(await pdfFile.arrayBuffer()); + const pdf = await extractPdf(pdfBytes); + await writeFile(join(outputDirectory, `${corpus.name}.pdf`), pdfBytes); + + const docxBlob = await buildDocx(corpus.data, docxTitleResolver); + const docxBytes = new Uint8Array(await docxBlob.arrayBuffer()); + const docx = await extractDocx(docxBytes); + await writeFile(join(outputDirectory, `${corpus.name}.docx`), docxBytes); + + expect(JSON.stringify(corpus.data)).toBe(before); + + return { + variant, + formats: [ + { + format: "pdf", + metrics: evaluateExport(corpus, { paragraphs: pdf.paragraphs, links: pdf.links }), + links: pdf.links, + paragraphs: pdf.paragraphs.length, + pageCount: pdf.raw.pageCount, + fontCount: pdf.raw.fonts.length, + hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens), + }, + { + format: "docx", + metrics: evaluateExport(corpus, { + paragraphs: docx.paragraphs.map((paragraph) => paragraph.text), + links: docx.links, + }), + links: docx.links, + paragraphs: docx.paragraphs.length, + numberingDefinitions: docx.numberingDefinitions, + numberedParagraphs: docx.numberedParagraphs, + hiddenLeaks: hiddenLeaks( + docx.paragraphs.map((paragraph) => paragraph.text), + corpus.hiddenTokens, + ), + }, + ], + }; +} + +function percentage(value: number): string { + return `${(value * 100).toFixed(1)}%`; +} + +function reportMarkdown(results: readonly VariantResult[]): string { + const lines = [ + "# ATS export evaluation", + "", + "Synthetic extraction measurements only. These are not vendor parsing accuracy claims.", + "", + "Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses pairs expected in the same authored field group and recovered in the same extracted paragraph/line.", + "", + "Corpus: one deterministic resume fixture per layout variant, covering header/contact, two roles, free-text dates, education, skills, project, custom section, long lines, links, hidden item, and CJK text. Hidden item tokens are intentionally excluded from expected recall and checked for leakage.", + "", + "| Variant | Format | Recall raw | Order raw | Duplicate raw | Grouping raw | Pages/paragraphs | Links | Numbering defs/paragraphs | Hidden leaks |", + "| --- | --- | --- | --- | --- | --- | ---: | ---: | ---: | --- |", + ]; + for (const result of results) { + for (const format of result.formats) { + const m = format.metrics; + lines.push( + `| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | ${format.links.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`, + ); + } + } + lines.push( + "", + "PDF font objects and raw link targets are recorded in adjacent JSON output.", + "", + "## Concrete loss/order evidence", + "", + ); + for (const result of results) { + for (const format of result.formats) { + const m = format.metrics; + lines.push( + `- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}.`, + ); + } + } + lines.push(""); + return lines.join("\n"); +} + +describe("current unchanged PDF and DOCX exports", () => { + it("measures two-column and full-width synthetic corpus without mutating input", { timeout: 120_000 }, async () => { + await mkdir(outputDirectory, { recursive: true }); + const results = [await measureVariant("two-column"), await measureVariant("full-width")]; + const report = { + claimBoundary: "Local extraction metrics only; no vendor parsing accuracy claim.", + tokenRules: "NFC, en-US case-folding, Unicode letter/number runs; punctuation separates tokens.", + corpus: { + variants: results.map((result) => result.variant), + expectedDistinctTokens: createSyntheticCorpus("full-width") + .tokens.flatMap((entry) => tokenize(entry.value)) + .filter((value, index, values) => values.indexOf(value) === index).length, + }, + results, + }; + await writeFile(join(outputDirectory, "ats-export-report.json"), `${JSON.stringify(report, null, 2)}\n`); + await writeFile(join(outputDirectory, "ats-export-report.md"), reportMarkdown(results)); + + for (const result of results) { + const pdf = result.formats.find((format) => format.format === "pdf"); + const docx = result.formats.find((format) => format.format === "docx"); + if (!pdf || !docx) throw new Error(`Missing measured format for ${result.variant}`); + expect(pdf.metrics.recall.denominator).toBeGreaterThan(20); + expect(docx.metrics.recall.denominator).toBe(pdf.metrics.recall.denominator); + expect(pdf.hiddenLeaks).toEqual([]); + expect(docx.hiddenLeaks).toEqual([]); + expect(pdf.fontCount).toBeGreaterThan(0); + expect(docx.numberingDefinitions).toBeGreaterThan(0); + expect(docx.numberedParagraphs).toBe(0); + } + }); +}); diff --git a/tooling/ats-export-evaluation/extract.ts b/tooling/ats-export-evaluation/extract.ts new file mode 100644 index 000000000..97db2ab99 --- /dev/null +++ b/tooling/ats-export-evaluation/extract.ts @@ -0,0 +1,154 @@ +import type { ExtractedDocument, PdfDocumentLike, RawExtraction } from "@reactive-resume/resume/ats-pdf"; +import { execFile } from "node:child_process"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { promisify } from "node:util"; +import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs"; +import { buildExtractedDocument, harvestPdfDocument } from "@reactive-resume/resume/ats-pdf"; + +const execFileAsync = promisify(execFile); + +export type PdfExtraction = { + raw: RawExtraction; + document: ExtractedDocument; + paragraphs: readonly string[]; + links: readonly string[]; +}; + +export type DocxParagraph = { + text: string; + numbering: { numId: string; level: string; format: string; marker: string } | null; +}; + +export type DocxExtraction = { + paragraphs: readonly DocxParagraph[]; + links: readonly string[]; + numberingDefinitions: number; + numberedParagraphs: number; +}; + +const unescapeXml = (value: string): string => + value + .replaceAll("<", "<") + .replaceAll(">", ">") + .replaceAll(""", '"') + .replaceAll("'", "'") + .replaceAll("&", "&"); + +const attribute = (attributes: string, name: string): string | null => { + const match = attributes.match(new RegExp(`(?:^|\\s)(?:[A-Za-z][\\w-]*:)?${name}="([^"]*)"`)); + return match ? unescapeXml(match[1] ?? "") : null; +}; + +const zipEntry = async (archivePath: string, entry: string): Promise => { + const { stdout } = await execFileAsync("unzip", ["-p", archivePath, entry]); + return stdout; +}; + +export async function extractPdf(bytes: Uint8Array): Promise { + const loadingTask = getDocument({ data: new Uint8Array(bytes), fontExtraProperties: true }); + try { + const document = (await loadingTask.promise) as unknown as PdfDocumentLike; + const raw = await harvestPdfDocument(document, { + file: { name: "synthetic-resume.pdf", sizeBytes: bytes.byteLength, magicBytesOk: true }, + }); + const extracted = buildExtractedDocument(raw); + return { + raw, + document: extracted, + paragraphs: extracted.lines.map((line) => line.text), + links: raw.links.flatMap((link) => (link.url ? [link.url] : [])), + }; + } finally { + await loadingTask.destroy(); + } +} + +function parseNumbering(numberingXml: string): Map { + const formats = new Map(); + const abstractDefinitions = new Map(); + for (const abstract of numberingXml.matchAll(/]*)>([\s\S]*?)<\/w:abstractNum>/g)) { + const abstractId = attribute(abstract[1] ?? "", "abstractNumId"); + if (!abstractId) continue; + for (const level of (abstract[2] ?? "").matchAll(/]*)>([\s\S]*?)<\/w:lvl>/g)) { + const levelId = attribute(level[1] ?? "", "ilvl") ?? "0"; + const format = attribute(level[2] ?? "", "val") ?? "unknown"; + const marker = attribute((level[2] ?? "").match(/]*)\/>/)?.[1] ?? "", "val") ?? ""; + abstractDefinitions.set(`${abstractId}:${levelId}`, { format, marker }); + } + } + for (const numbering of numberingXml.matchAll(/]*)>([\s\S]*?)<\/w:num>/g)) { + const numId = attribute(numbering[1] ?? "", "numId"); + const abstractId = attribute((numbering[2] ?? "").match(/]*)\/>/)?.[1] ?? "", "val"); + if (!numId || !abstractId) continue; + for (const level of ["0", "1", "2", "3", "4", "5", "6", "7", "8"]) { + const definition = abstractDefinitions.get(`${abstractId}:${level}`); + if (definition) formats.set(`${numId}:${level}`, definition); + } + } + return formats; +} + +function parseDocxParagraphs(documentXml: string, numbering: Map) { + const paragraphs: DocxParagraph[] = []; + for (const paragraph of documentXml.matchAll(/]*>([\s\S]*?)<\/w:p>/g)) { + const body = paragraph[1] ?? ""; + const text = [...body.matchAll(/]*>([\s\S]*?)<\/w:t>/g)] + .map((match) => unescapeXml(match[1] ?? "")) + .join(""); + const numPr = body.match(/]*>([\s\S]*?)<\/w:numPr>/)?.[1]; + const numId = numPr ? numPr.match(/]*)\/>/) : null; + const level = numPr ? numPr.match(/]*)\/>/) : null; + const numIdValue = numId ? attribute(numId[1] ?? "", "val") : null; + const levelValue = level ? (attribute(level[1] ?? "", "val") ?? "0") : null; + const definition = numIdValue && levelValue ? numbering.get(`${numIdValue}:${levelValue}`) : undefined; + paragraphs.push({ + text, + numbering: + numIdValue && levelValue && definition + ? { numId: numIdValue, level: levelValue, format: definition.format, marker: definition.marker } + : null, + }); + } + return paragraphs; +} + +function parseDocxLinks(documentXml: string, relationshipsXml: string): string[] { + const relationships = new Map(); + for (const relationship of relationshipsXml.matchAll(/]*)\/>/g)) { + const id = attribute(relationship[1] ?? "", "Id"); + const target = attribute(relationship[1] ?? "", "Target"); + if (id && target) relationships.set(id, target); + } + const links: string[] = []; + for (const hyperlink of documentXml.matchAll(/]*)>/g)) { + const id = attribute(hyperlink[1] ?? "", "id"); + const target = id ? relationships.get(id) : undefined; + if (target) links.push(target); + } + return links; +} + +export async function extractDocx(bytes: Uint8Array): Promise { + const tempDirectory = await mkdtemp(join(tmpdir(), "reactive-resume-ats-")); + const archivePath = join(tempDirectory, "resume.docx"); + try { + await writeFile(archivePath, bytes); + const [documentXml, numberingXml, relationshipsXml] = await Promise.all([ + zipEntry(archivePath, "word/document.xml"), + zipEntry(archivePath, "word/numbering.xml"), + zipEntry(archivePath, "word/_rels/document.xml.rels"), + ]); + const numbering = parseNumbering(numberingXml); + const paragraphs = parseDocxParagraphs(documentXml, numbering); + return { + paragraphs, + links: parseDocxLinks(documentXml, relationshipsXml), + numberingDefinitions: numbering.size, + numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length, + }; + } finally { + await rm(tempDirectory, { recursive: true, force: true }); + } +} diff --git a/tooling/ats-export-evaluation/fixture.ts b/tooling/ats-export-evaluation/fixture.ts new file mode 100644 index 000000000..5d5529d18 --- /dev/null +++ b/tooling/ats-export-evaluation/fixture.ts @@ -0,0 +1,250 @@ +import type { ResumeData } from "@reactive-resume/schema/resume/data"; +import type { EvaluationCorpus, ExpectedToken } from "./metrics"; +import { sampleResumeData } from "@reactive-resume/schema/resume/sample"; + +export type ExportVariant = "two-column" | "full-width"; + +export type SyntheticCorpus = EvaluationCorpus & { + data: ResumeData; + hiddenTokens: readonly string[]; + links: readonly string[]; +}; + +const token = (value: string, group: string): ExpectedToken => ({ value, group }); + +const text = (html: string) => html.replace(/<[^>]+>/g, " "); + +const SUMMARY_HTML = + "

summary-signal-alpha leads longline-calibration with multilingual 東京大学 context and durable systems.

"; +const EXPERIENCE_ROLE_ONE_HTML = + "
  • role-bullet-alpha shipped resilient pipeline-observability under longline-pressure.

  • role-bullet-beta measured queue-latency and improved release-safety.

"; +const EXPERIENCE_ROLE_TWO_HTML = + "

role-bullet-gamma guided distributed-runtime migration for platform-reliability.

"; +const EDUCATION_HTML = "

education-signal-delta researched multilingual retrieval and evaluation.

"; +const PROJECT_HTML = "

project-signal-epsilon demonstrates export-fixture determinism.

"; +const CUSTOM_HTML = "

custom-signal-zeta preserves authored custom content and free-text dates.

"; + +const expectedTokens: readonly ExpectedToken[] = [ + ...[ + "Mira Kova", + "Principal Systems Architect", + "mira.kova@example.com", + "+49 30 555 0142", + "Berlin 東京", + "mirakova.dev", + "orbit-field-omega", + ].map((value) => token(value, "header")), + ...[text(SUMMARY_HTML)].map((value) => token(value, "summary")), + ...[ + "Northstar Robotics", + "2018-02 — Present", + "Berlin", + "Staff Platform Engineer", + "2018-02 to 2020-12", + text(EXPERIENCE_ROLE_ONE_HTML), + "Principal Reliability Engineer", + "2021 / Present", + text(EXPERIENCE_ROLE_TWO_HTML), + ].map((value) => token(value, "experience")), + ...[ + "東京大学", + "Master of Computer Science", + "Distributed Systems", + "3.98 GPA", + "2014 — 2018 (long academic period)", + "東京", + text(EDUCATION_HTML), + ].map((value) => token(value, "education")), + ...["TypeScript", "skill-keyword-alpha", "Kubernetes", "skill-keyword-beta"].map((value) => token(value, "skills")), + ...["Export Observatory", "2022 to Winter 2024", text(PROJECT_HTML), "project.example/observatory"].map((value) => + token(value, "projects"), + ), + ...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")), +]; + +export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta"] as const; + +const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false }); + +export function createSyntheticCorpus(variant: ExportVariant): SyntheticCorpus { + const data = structuredClone(sampleResumeData); + data.picture.hidden = true; + data.basics = { + name: "Mira Kova", + headline: "Principal Systems Architect", + email: "mira.kova@example.com", + phone: "+49 30 555 0142", + location: "Berlin 東京", + website: resolveWebsite("https://mirakova.dev", "mirakova.dev"), + customFields: [ + { + id: "synthetic-field-omega", + icon: "", + text: "orbit-field-omega", + link: "https://orbit.example/omega", + }, + ], + }; + data.summary = { + ...data.summary, + title: "Summary", + content: SUMMARY_HTML, + }; + data.sections.profiles = { + ...data.sections.profiles, + title: "Profiles", + items: [ + { + id: "synthetic-profile-omega", + hidden: false, + icon: "", + iconColor: "", + network: "OrbitNet", + username: "orbit-profile-omega", + website: resolveWebsite("https://orbit.example/profile", "orbit.example/profile"), + }, + ], + }; + data.sections.experience = { + ...data.sections.experience, + title: "Experience", + items: [ + { + id: "synthetic-experience-northstar", + hidden: false, + company: "Northstar Robotics", + position: "", + location: "Berlin", + period: "2018-02 — Present", + website: resolveWebsite("https://northstar.example/jobs", "northstar.example/jobs"), + roles: [ + { + id: "synthetic-role-staff", + position: "Staff Platform Engineer", + period: "2018-02 to 2020-12", + description: EXPERIENCE_ROLE_ONE_HTML, + }, + { + id: "synthetic-role-principal", + position: "Principal Reliability Engineer", + period: "2021 / Present", + description: EXPERIENCE_ROLE_TWO_HTML, + }, + ], + description: "", + }, + { + id: "synthetic-hidden-experience", + hidden: true, + company: HIDDEN_TOKENS[0], + position: "", + location: "", + period: "", + website: resolveWebsite("https://hidden.example", "hidden.example"), + roles: [], + description: `

${HIDDEN_TOKENS[1]}

`, + }, + ], + }; + data.sections.education = { + ...data.sections.education, + title: "Education", + items: [ + { + id: "synthetic-education-tokyo", + hidden: false, + school: "東京大学", + degree: "Master of Computer Science", + area: "Distributed Systems", + grade: "3.98 GPA", + location: "東京", + period: "2014 — 2018 (long academic period)", + website: resolveWebsite("https://u-tokyo.example/program", "u-tokyo.example/program"), + description: EDUCATION_HTML, + }, + ], + }; + data.sections.skills = { + ...data.sections.skills, + title: "Skills", + items: [ + { + id: "synthetic-skill-typescript", + hidden: false, + icon: "", + iconColor: "", + name: "TypeScript", + proficiency: "Advanced", + level: 4, + keywords: ["skill-keyword-alpha"], + }, + { + id: "synthetic-skill-kubernetes", + hidden: false, + icon: "", + iconColor: "", + name: "Kubernetes", + proficiency: "Expert", + level: 5, + keywords: ["skill-keyword-beta"], + }, + ], + }; + data.sections.projects = { + ...data.sections.projects, + title: "Projects", + items: [ + { + id: "synthetic-project-observatory", + hidden: false, + name: "Export Observatory", + period: "2022 to Winter 2024", + website: resolveWebsite("https://project.example/observatory", "project.example/observatory"), + description: PROJECT_HTML, + }, + ], + }; + data.customSections = [ + { + id: "custom-ats-evidence", + type: "summary", + title: "Custom Evidence", + icon: "", + columns: 1, + hidden: false, + showHeading: true, + keepTogether: false, + startOnNewPage: false, + items: [{ id: "synthetic-custom-zeta", hidden: false, content: CUSTOM_HTML }], + }, + ]; + + const mainSections = ["profiles", "summary", "experience", "education", "projects"]; + const sidebarSections = ["skills", "custom-ats-evidence"]; + const page = { + fullWidth: variant === "full-width", + main: variant === "full-width" ? [...mainSections, ...sidebarSections] : mainSections, + sidebar: variant === "full-width" ? [] : sidebarSections, + }; + data.metadata = { + ...data.metadata, + template: variant === "full-width" ? "onyx" : "gengar", + layout: { ...data.metadata.layout, pages: [page] }, + page: { ...data.metadata.page, locale: "en-US", hideIcons: true, hideSectionIcons: true }, + }; + + return { + name: `ats-${variant}`, + tokens: expectedTokens, + data, + hiddenTokens: HIDDEN_TOKENS, + links: [ + "https://mirakova.dev", + "https://orbit.example/omega", + "https://orbit.example/profile", + "https://northstar.example/jobs", + "https://u-tokyo.example/program", + "https://project.example/observatory", + ], + }; +} diff --git a/tooling/ats-export-evaluation/metrics.test.ts b/tooling/ats-export-evaluation/metrics.test.ts new file mode 100644 index 000000000..4fe6b9d13 --- /dev/null +++ b/tooling/ats-export-evaluation/metrics.test.ts @@ -0,0 +1,43 @@ +import { describe, expect, it } from "vitest"; +import { evaluateExport } from "./metrics"; + +const corpus = { + name: "synthetic", + tokens: [ + { value: "Alpha", group: "experience" }, + { value: "Bravo", group: "experience" }, + { value: "Charlie", group: "education" }, + ] as const, +}; + +describe("evaluateExport", () => { + it("counts distinct recall, order, duplicate, and grouping losses from raw tokens", () => { + const result = evaluateExport(corpus, { + paragraphs: ["Alpha Bravo Bravo", "Charlie"], + links: [], + }); + + expect(result.recall).toEqual({ numerator: 3, denominator: 3, value: 1 }); + expect(result.order).toEqual({ numerator: 2, denominator: 2, value: 1 }); + expect(result.duplicates).toEqual({ expected: 3, observed: 4, extra: 1 }); + expect(result.missingTokens).toEqual([]); + expect(result.outOfOrderPairs).toEqual([]); + expect(result.grouping).toEqual({ numerator: 1, denominator: 2, value: 0.5 }); + }); + + it("detects a dropped token and an inverted pair", () => { + const result = evaluateExport(corpus, { + paragraphs: ["Bravo", "Alpha"], + links: [], + }); + + expect(result.recall).toEqual({ numerator: 2, denominator: 3, value: 2 / 3 }); + expect(result.order).toEqual({ numerator: 0, denominator: 2, value: 0 }); + expect(result.duplicates).toEqual({ expected: 3, observed: 2, extra: 0 }); + expect(result.missingTokens).toEqual(["charlie"]); + expect(result.outOfOrderPairs).toEqual([ + ["alpha", "bravo"], + ["bravo", "charlie"], + ]); + }); +}); diff --git a/tooling/ats-export-evaluation/metrics.ts b/tooling/ats-export-evaluation/metrics.ts new file mode 100644 index 000000000..8033a2713 --- /dev/null +++ b/tooling/ats-export-evaluation/metrics.ts @@ -0,0 +1,118 @@ +export type ExpectedToken = { + value: string; + group: string; +}; + +export type EvaluationCorpus = { + name: string; + tokens: readonly ExpectedToken[]; +}; + +export type ExtractedExport = { + /** Paragraphs/lines in the extractor's returned order. */ + paragraphs: readonly string[]; + links: readonly string[]; +}; + +export type ExportMetrics = { + recall: { numerator: number; denominator: number; value: number }; + order: { numerator: number; denominator: number; value: number }; + duplicates: { expected: number; observed: number; extra: number }; + grouping: { numerator: number; denominator: number; value: number }; + missingTokens: readonly string[]; + outOfOrderPairs: readonly (readonly [string, string])[]; + observedTokens: number; +}; + +/** + * Tokenization used by this evaluation only. It is deliberately transparent and locale-neutral: + * Unicode letters/numbers stay intact, punctuation is a separator, and matching is case-folded. + * This is a corpus metric, not a claim about any vendor parser. + */ +export function tokenize(value: string): string[] { + return ( + value + .normalize("NFC") + .match(/[\p{L}\p{N}]+/gu) + ?.map((token) => token.toLocaleLowerCase("en-US")) ?? [] + ); +} + +const metric = (numerator: number, denominator: number) => ({ + numerator, + denominator, + value: denominator === 0 ? 1 : numerator / denominator, +}); + +/** + * Computes raw extraction measurements from corpus tokens and extractor paragraphs. + * + * Recall uses distinct expected tokens. Duplicate accounting separately reports all matching + * occurrences, so dropping a token cannot be hidden by duplicate output. Order and grouping use + * the first occurrence of each distinct expected token, keeping those measures interpretable when + * an export repeats a heading or bullet. + */ +export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExport): ExportMetrics { + const expected = corpus.tokens.flatMap((entry) => + tokenize(entry.value).map((value) => ({ value, group: entry.group })), + ); + const expectedByValue = new Map(); + for (const [index, token] of expected.entries()) { + if (!expectedByValue.has(token.value)) expectedByValue.set(token.value, { group: token.group, index }); + } + + const observedByParagraph = extracted.paragraphs.map(tokenize); + const observed = observedByParagraph.flat(); + const expectedValues = [...expectedByValue.keys()]; + const observedPositions = new Map(); + for (const [index, token] of observed.entries()) { + if (!observedPositions.has(token)) observedPositions.set(token, index); + } + + const recoveredDistinct = expectedValues.filter((value) => observedPositions.has(value)).length; + const expectedTokenPairs = expectedValues.flatMap((value, index) => { + const next = expectedValues[index + 1]; + return next ? ([[value, next]] as const) : []; + }); + const outOfOrderPairs = expectedTokenPairs.filter(([left, right]) => { + const leftPosition = observedPositions.get(left); + const rightPosition = observedPositions.get(right); + return leftPosition === undefined || rightPosition === undefined || leftPosition >= rightPosition; + }); + + const observedParagraphPositions = new Map(); + for (const [paragraphIndex, paragraphTokens] of observedByParagraph.entries()) { + for (const token of paragraphTokens) { + if (!observedParagraphPositions.has(token)) observedParagraphPositions.set(token, paragraphIndex); + } + } + const groupedPairs = expectedTokenPairs.filter(([left, right]) => { + const leftEntry = expectedByValue.get(left); + const rightEntry = expectedByValue.get(right); + const leftParagraph = observedParagraphPositions.get(left); + const rightParagraph = observedParagraphPositions.get(right); + return ( + leftEntry?.group === rightEntry?.group && + leftParagraph !== undefined && + rightParagraph !== undefined && + leftParagraph === rightParagraph + ); + }).length; + + const expectedValuesSet = new Set(expectedValues); + const observedExpectedOccurrences = observed.filter((token) => expectedValuesSet.has(token)).length; + + return { + recall: metric(recoveredDistinct, expectedValues.length), + order: metric(expectedTokenPairs.length - outOfOrderPairs.length, expectedTokenPairs.length), + duplicates: { + expected: expected.length, + observed: observedExpectedOccurrences, + extra: Math.max(0, observedExpectedOccurrences - expected.length), + }, + grouping: metric(groupedPairs, expectedTokenPairs.length), + missingTokens: expectedValues.filter((value) => !observedPositions.has(value)), + outOfOrderPairs, + observedTokens: observed.length, + }; +} diff --git a/tooling/package.json b/tooling/package.json index 464279f4d..0e73b059e 100644 --- a/tooling/package.json +++ b/tooling/package.json @@ -18,9 +18,12 @@ "devDependencies": { "@lingui/format-po": "^6.6.0", "@reactive-resume/config": "workspace:*", + "@reactive-resume/docx": "workspace:*", "@reactive-resume/env": "workspace:*", + "@reactive-resume/pdf": "workspace:*", "@reactive-resume/resume": "workspace:*", "@types/pg": "^8.23.1", + "pdfjs-dist": "6.3.289", "@typescript/native-preview": "7.0.0-dev.20260707.2", "drizzle-orm": "1.0.0-rc.4", "pg": "^8.23.0", diff --git a/tooling/vitest.config.ts b/tooling/vitest.config.ts new file mode 100644 index 000000000..dce75bff4 --- /dev/null +++ b/tooling/vitest.config.ts @@ -0,0 +1,14 @@ +import { fileURLToPath } from "node:url"; +// @boundaries-ignore root shared Vitest config +import { createVitestProjectConfig } from "../vitest.shared.mts"; + +const config = createVitestProjectConfig({ + name: "@reactive-resume/tooling", + dirname: fileURLToPath(new URL(".", import.meta.url)), +}); + +export default { + ...config, + test: { ...config.test, include: ["**/*.{test,spec}.?(c|m)[jt]s?(x)"] }, + oxc: { jsx: { runtime: "automatic" as const } }, +}; From 0e5994f243270e57533fa6e6d618d533847106bb Mon Sep 17 00:00:00 2001 From: Amruth Pillai Date: Sun, 6 Sep 2026 06:44:48 +0200 Subject: [PATCH 2/4] test(tooling): harden ATS export evaluation --- pnpm-lock.yaml | 3 + tooling/ats-export-evaluation/README.md | 11 ++- .../evaluation.integration.test.ts | 28 ++++-- tooling/ats-export-evaluation/extract.test.ts | 58 +++++++++++ tooling/ats-export-evaluation/extract.ts | 49 ++++------ tooling/ats-export-evaluation/fixture.ts | 12 ++- tooling/ats-export-evaluation/metrics.test.ts | 37 ++++++- tooling/ats-export-evaluation/metrics.ts | 97 +++++++++++++++---- tooling/package.json | 1 + 9 files changed, 235 insertions(+), 61 deletions(-) create mode 100644 tooling/ats-export-evaluation/extract.test.ts diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 017cc32f1..0daaefa72 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -1370,6 +1370,9 @@ importers: drizzle-orm: specifier: 1.0.0-rc.4 version: 1.0.0-rc.4(@types/pg@8.23.1)(pg@8.23.0)(zod@4.5.4) + jszip: + specifier: 3.10.1 + version: 3.10.1 pdfjs-dist: specifier: 6.3.289 version: 6.3.289 diff --git a/tooling/ats-export-evaluation/README.md b/tooling/ats-export-evaluation/README.md index 70292a9d1..12aa8a85d 100644 --- a/tooling/ats-export-evaluation/README.md +++ b/tooling/ats-export-evaluation/README.md @@ -3,7 +3,16 @@ `evaluation.integration.test.ts` renders deterministic synthetic resume data through current `ResumeDocument` and `buildDocx`, then measures extraction with installed PDF.js and DOCX XML. `metrics.test.ts` locks raw distinct-token recall, order, duplicate, and semantic-grouping behavior, -including deliberate drop/duplicate regressions. +including deliberate drop/duplicate regressions. Grouping compares eligible same-group expected +occurrences, preserving repeated values' authored field identity. Link metrics normalize expected +and extracted targets and report dropped or changed targets; hidden-leak checks scan paragraph text +and link-target channels, including hidden URLs. + +Current DOCX output intentionally records its known missing `tel:` target as a measured loss; any +additional dropped or changed target fails integration assertions. + +`extract.test.ts` supplies a positive XML numbering fixture covering paragraph order, `numId`, level, +format, marker, and link-target extraction, plus valid archives with optional entries omitted. Run from repository root: diff --git a/tooling/ats-export-evaluation/evaluation.integration.test.ts b/tooling/ats-export-evaluation/evaluation.integration.test.ts index 280dd6694..be52c80d6 100644 --- a/tooling/ats-export-evaluation/evaluation.integration.test.ts +++ b/tooling/ats-export-evaluation/evaluation.integration.test.ts @@ -34,8 +34,12 @@ type VariantResult = { formats: FormatResult[]; }; -function hiddenLeaks(paragraphs: readonly string[], hiddenTokens: readonly string[]): string[] { - const observed = paragraphs.flatMap(tokenize); +function hiddenLeaks( + paragraphs: readonly string[], + hiddenTokens: readonly string[], + links: readonly string[] = [], +): string[] { + const observed = [...paragraphs, ...links].flatMap(tokenize); return hiddenTokens.filter((value) => { const expected = tokenize(value); return observed.some((_, index) => expected.every((token, offset) => observed[index + offset] === token)); @@ -72,7 +76,7 @@ async function measureVariant(variant: ExportVariant): Promise { paragraphs: pdf.paragraphs.length, pageCount: pdf.raw.pageCount, fontCount: pdf.raw.fonts.length, - hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens), + hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens, pdf.links), }, { format: "docx", @@ -87,6 +91,7 @@ async function measureVariant(variant: ExportVariant): Promise { hiddenLeaks: hiddenLeaks( docx.paragraphs.map((paragraph) => paragraph.text), corpus.hiddenTokens, + docx.links, ), }, ], @@ -103,7 +108,7 @@ function reportMarkdown(results: readonly VariantResult[]): string { "", "Synthetic extraction measurements only. These are not vendor parsing accuracy claims.", "", - "Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses pairs expected in the same authored field group and recovered in the same extracted paragraph/line.", + "Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses same-group expected occurrence pairs and recovered in the same extracted paragraph/line. Link targets are normalized before expected/observed set comparison.", "", "Corpus: one deterministic resume fixture per layout variant, covering header/contact, two roles, free-text dates, education, skills, project, custom section, long lines, links, hidden item, and CJK text. Hidden item tokens are intentionally excluded from expected recall and checked for leakage.", "", @@ -114,7 +119,7 @@ function reportMarkdown(results: readonly VariantResult[]): string { for (const format of result.formats) { const m = format.metrics; lines.push( - `| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | ${format.links.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`, + `| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | expected ${m.links.expected.length}, observed ${m.links.observed.length}, dropped ${m.links.missing.length}, changed/extra ${m.links.unexpected.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`, ); } } @@ -129,7 +134,7 @@ function reportMarkdown(results: readonly VariantResult[]): string { for (const format of result.formats) { const m = format.metrics; lines.push( - `- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}.`, + `- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}; dropped links = ${m.links.missing.length === 0 ? "none" : m.links.missing.join(", ")}; changed/extra links = ${m.links.unexpected.length === 0 ? "none" : m.links.unexpected.join(", ")}.`, ); } } @@ -138,6 +143,12 @@ function reportMarkdown(results: readonly VariantResult[]): string { } describe("current unchanged PDF and DOCX exports", () => { + it("detects hidden content emitted through link targets", () => { + expect(hiddenLeaks(["Visible"], ["https://hidden.example"], ["https://hidden.example"])).toEqual([ + "https://hidden.example", + ]); + }); + it("measures two-column and full-width synthetic corpus without mutating input", { timeout: 120_000 }, async () => { await mkdir(outputDirectory, { recursive: true }); const results = [await measureVariant("two-column"), await measureVariant("full-width")]; @@ -161,6 +172,11 @@ describe("current unchanged PDF and DOCX exports", () => { if (!pdf || !docx) throw new Error(`Missing measured format for ${result.variant}`); expect(pdf.metrics.recall.denominator).toBeGreaterThan(20); expect(docx.metrics.recall.denominator).toBe(pdf.metrics.recall.denominator); + expect(pdf.metrics.links.missing).toEqual([]); + expect(pdf.metrics.links.unexpected).toEqual([]); + // DOCX currently emits email but not telephone hyperlinks; retain this known loss in the measurement. + expect(docx.metrics.links.missing).toEqual(["tel:+49 30 555 0142"]); + expect(docx.metrics.links.unexpected).toEqual([]); expect(pdf.hiddenLeaks).toEqual([]); expect(docx.hiddenLeaks).toEqual([]); expect(pdf.fontCount).toBeGreaterThan(0); diff --git a/tooling/ats-export-evaluation/extract.test.ts b/tooling/ats-export-evaluation/extract.test.ts new file mode 100644 index 000000000..7bff11f94 --- /dev/null +++ b/tooling/ats-export-evaluation/extract.test.ts @@ -0,0 +1,58 @@ +import { describe, expect, it } from "vitest"; +import JSZip from "jszip"; +import { extractDocx } from "./extract"; + +const numberedDocument = ` + + + First itemlink + Second item + +`; + +const numberedDefinitions = ` + + + + + + +`; + +const relationships = ``; + +function createDocxZip(entries: Record): Promise { + const zip = new JSZip(); + for (const [name, content] of Object.entries(entries)) zip.file(name, content); + return zip.generateAsync({ type: "uint8array" }); +} + +describe("DOCX extraction", () => { + it("extracts numbering identity and link targets in paragraph order", async () => { + const bytes = await createDocxZip({ + "word/document.xml": numberedDocument, + "word/numbering.xml": numberedDefinitions, + "word/_rels/document.xml.rels": relationships, + }); + + const result = await extractDocx(bytes); + + expect(result.paragraphs).toEqual([ + { text: "First itemlink", numbering: { numId: "42", level: "1", format: "lowerLetter", marker: "%2)" } }, + { text: "Second item", numbering: { numId: "42", level: "0", format: "decimal", marker: "%1." } }, + ]); + expect(result.links).toEqual(["https://example.com/item"]); + expect(result.numberedParagraphs).toBe(2); + }); + + it("handles valid DOCX archives without optional XML entries", async () => { + const bytes = await createDocxZip({ "word/document.xml": numberedDocument }); + + const result = await extractDocx(bytes); + + expect(result.paragraphs.map((paragraph) => paragraph.numbering)).toEqual([null, null]); + expect(result.links).toEqual([]); + expect(result.numberingDefinitions).toBe(0); + expect(result.numberedParagraphs).toBe(0); + }); +}); diff --git a/tooling/ats-export-evaluation/extract.ts b/tooling/ats-export-evaluation/extract.ts index 97db2ab99..abbc33533 100644 --- a/tooling/ats-export-evaluation/extract.ts +++ b/tooling/ats-export-evaluation/extract.ts @@ -1,14 +1,8 @@ import type { ExtractedDocument, PdfDocumentLike, RawExtraction } from "@reactive-resume/resume/ats-pdf"; -import { execFile } from "node:child_process"; -import { mkdtemp, rm, writeFile } from "node:fs/promises"; -import { tmpdir } from "node:os"; -import { join } from "node:path"; -import { promisify } from "node:util"; +import JSZip from "jszip"; import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs"; import { buildExtractedDocument, harvestPdfDocument } from "@reactive-resume/resume/ats-pdf"; -const execFileAsync = promisify(execFile); - export type PdfExtraction = { raw: RawExtraction; document: ExtractedDocument; @@ -41,9 +35,9 @@ const attribute = (attributes: string, name: string): string | null => { return match ? unescapeXml(match[1] ?? "") : null; }; -const zipEntry = async (archivePath: string, entry: string): Promise => { - const { stdout } = await execFileAsync("unzip", ["-p", archivePath, entry]); - return stdout; +const zipEntry = (archive: JSZip, entry: string): Promise => { + const file = archive.file(entry); + return Promise.resolve(file ? file.async("string") : null); }; export async function extractPdf(bytes: Uint8Array): Promise { @@ -131,24 +125,19 @@ function parseDocxLinks(documentXml: string, relationshipsXml: string): string[] } export async function extractDocx(bytes: Uint8Array): Promise { - const tempDirectory = await mkdtemp(join(tmpdir(), "reactive-resume-ats-")); - const archivePath = join(tempDirectory, "resume.docx"); - try { - await writeFile(archivePath, bytes); - const [documentXml, numberingXml, relationshipsXml] = await Promise.all([ - zipEntry(archivePath, "word/document.xml"), - zipEntry(archivePath, "word/numbering.xml"), - zipEntry(archivePath, "word/_rels/document.xml.rels"), - ]); - const numbering = parseNumbering(numberingXml); - const paragraphs = parseDocxParagraphs(documentXml, numbering); - return { - paragraphs, - links: parseDocxLinks(documentXml, relationshipsXml), - numberingDefinitions: numbering.size, - numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length, - }; - } finally { - await rm(tempDirectory, { recursive: true, force: true }); - } + const archive = await JSZip.loadAsync(bytes); + const [documentXml, numberingXml, relationshipsXml] = await Promise.all([ + zipEntry(archive, "word/document.xml"), + zipEntry(archive, "word/numbering.xml"), + zipEntry(archive, "word/_rels/document.xml.rels"), + ]); + if (!documentXml) throw new Error("DOCX archive is missing word/document.xml"); + const numbering = parseNumbering(numberingXml ?? ""); + const paragraphs = parseDocxParagraphs(documentXml, numbering); + return { + paragraphs, + links: parseDocxLinks(documentXml, relationshipsXml ?? ""), + numberingDefinitions: numbering.size, + numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length, + }; } diff --git a/tooling/ats-export-evaluation/fixture.ts b/tooling/ats-export-evaluation/fixture.ts index 5d5529d18..641945229 100644 --- a/tooling/ats-export-evaluation/fixture.ts +++ b/tooling/ats-export-evaluation/fixture.ts @@ -34,7 +34,10 @@ const expectedTokens: readonly ExpectedToken[] = [ "mirakova.dev", "orbit-field-omega", ].map((value) => token(value, "header")), + ...["Profiles", "OrbitNet", "orbit-profile-omega"].map((value) => token(value, "profiles")), + ...["Summary"].map((value) => token(value, "summary")), ...[text(SUMMARY_HTML)].map((value) => token(value, "summary")), + ...["Experience"].map((value) => token(value, "experience")), ...[ "Northstar Robotics", "2018-02 — Present", @@ -46,6 +49,7 @@ const expectedTokens: readonly ExpectedToken[] = [ "2021 / Present", text(EXPERIENCE_ROLE_TWO_HTML), ].map((value) => token(value, "experience")), + ...["Education"].map((value) => token(value, "education")), ...[ "東京大学", "Master of Computer Science", @@ -55,14 +59,16 @@ const expectedTokens: readonly ExpectedToken[] = [ "東京", text(EDUCATION_HTML), ].map((value) => token(value, "education")), - ...["TypeScript", "skill-keyword-alpha", "Kubernetes", "skill-keyword-beta"].map((value) => token(value, "skills")), + ...["Skills", "TypeScript", "Advanced", "skill-keyword-alpha", "Kubernetes", "Expert", "skill-keyword-beta"].map( + (value) => token(value, "skills"), + ), ...["Export Observatory", "2022 to Winter 2024", text(PROJECT_HTML), "project.example/observatory"].map((value) => token(value, "projects"), ), ...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")), ]; -export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta"] as const; +export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta", "https://hidden.example"] as const; const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false }); @@ -240,6 +246,8 @@ export function createSyntheticCorpus(variant: ExportVariant): SyntheticCorpus { hiddenTokens: HIDDEN_TOKENS, links: [ "https://mirakova.dev", + "mailto:mira.kova@example.com", + "tel:+49 30 555 0142", "https://orbit.example/omega", "https://orbit.example/profile", "https://northstar.example/jobs", diff --git a/tooling/ats-export-evaluation/metrics.test.ts b/tooling/ats-export-evaluation/metrics.test.ts index 4fe6b9d13..cafcdb98b 100644 --- a/tooling/ats-export-evaluation/metrics.test.ts +++ b/tooling/ats-export-evaluation/metrics.test.ts @@ -22,7 +22,42 @@ describe("evaluateExport", () => { expect(result.duplicates).toEqual({ expected: 3, observed: 4, extra: 1 }); expect(result.missingTokens).toEqual([]); expect(result.outOfOrderPairs).toEqual([]); - expect(result.grouping).toEqual({ numerator: 1, denominator: 2, value: 0.5 }); + expect(result.grouping).toEqual({ numerator: 1, denominator: 1, value: 1 }); + }); + + it("keeps repeated expected values tied to their authored groups", () => { + const repeatedCorpus = { + name: "repeated", + tokens: [ + { value: "Alpha Bravo", group: "first" }, + { value: "Alpha Charlie", group: "second" }, + ], + } as const; + + const result = evaluateExport(repeatedCorpus, { + paragraphs: ["Alpha Bravo", "Alpha Charlie"], + links: [], + }); + + expect(result.grouping).toEqual({ numerator: 2, denominator: 2, value: 1 }); + }); + + it("reports dropped and changed link targets", () => { + const result = evaluateExport( + { + name: "links", + tokens: [], + links: ["HTTPS://Example.com/profile/", "https://example.com/jobs"], + }, + { paragraphs: [], links: ["https://example.com/profile", "https://changed.example/jobs"] }, + ); + + expect(result.links).toEqual({ + expected: ["https://example.com/profile", "https://example.com/jobs"], + observed: ["https://example.com/profile", "https://changed.example/jobs"], + missing: ["https://example.com/jobs"], + unexpected: ["https://changed.example/jobs"], + }); }); it("detects a dropped token and an inverted pair", () => { diff --git a/tooling/ats-export-evaluation/metrics.ts b/tooling/ats-export-evaluation/metrics.ts index 8033a2713..5461173ea 100644 --- a/tooling/ats-export-evaluation/metrics.ts +++ b/tooling/ats-export-evaluation/metrics.ts @@ -6,6 +6,7 @@ export type ExpectedToken = { export type EvaluationCorpus = { name: string; tokens: readonly ExpectedToken[]; + links?: readonly string[]; }; export type ExtractedExport = { @@ -19,6 +20,12 @@ export type ExportMetrics = { order: { numerator: number; denominator: number; value: number }; duplicates: { expected: number; observed: number; extra: number }; grouping: { numerator: number; denominator: number; value: number }; + links: { + expected: readonly string[]; + observed: readonly string[]; + missing: readonly string[]; + unexpected: readonly string[]; + }; missingTokens: readonly string[]; outOfOrderPairs: readonly (readonly [string, string])[]; observedTokens: number; @@ -44,29 +51,59 @@ const metric = (numerator: number, denominator: number) => ({ value: denominator === 0 ? 1 : numerator / denominator, }); +function normalizeLinkTarget(target: string): string { + const trimmed = target.trim(); + try { + const url = new URL(trimmed); + url.protocol = url.protocol.toLowerCase(); + url.hostname = url.hostname.toLowerCase(); + if (url.pathname.length > 1) url.pathname = url.pathname.replace(/\/+$/, ""); + return url.toString(); + } catch { + return trimmed; + } +} + +function normalizeLinks(links: readonly string[]): string[] { + return [...new Set(links.map(normalizeLinkTarget))]; +} + /** * Computes raw extraction measurements from corpus tokens and extractor paragraphs. * * Recall uses distinct expected tokens. Duplicate accounting separately reports all matching * occurrences, so dropping a token cannot be hidden by duplicate output. Order and grouping use - * the first occurrence of each distinct expected token, keeping those measures interpretable when - * an export repeats a heading or bullet. + * the first occurrence of each distinct expected token, keeping recall/order measures interpretable + * when an export repeats a heading or bullet. Grouping retains every expected token occurrence so + * repeated values keep their authored field group and occurrence identity. */ export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExport): ExportMetrics { const expected = corpus.tokens.flatMap((entry) => tokenize(entry.value).map((value) => ({ value, group: entry.group })), ); - const expectedByValue = new Map(); - for (const [index, token] of expected.entries()) { - if (!expectedByValue.has(token.value)) expectedByValue.set(token.value, { group: token.group, index }); + const expectedOccurrenceOrdinals: number[] = []; + const expectedValues: string[] = []; + const expectedValuesSet = new Set(); + const expectedCounts = new Map(); + for (const token of expected) { + const ordinal = expectedCounts.get(token.value) ?? 0; + expectedOccurrenceOrdinals.push(ordinal); + expectedCounts.set(token.value, ordinal + 1); + if (!expectedValuesSet.has(token.value)) { + expectedValuesSet.add(token.value); + expectedValues.push(token.value); + } } const observedByParagraph = extracted.paragraphs.map(tokenize); const observed = observedByParagraph.flat(); - const expectedValues = [...expectedByValue.keys()]; const observedPositions = new Map(); + const observedPositionsByValue = new Map(); for (const [index, token] of observed.entries()) { if (!observedPositions.has(token)) observedPositions.set(token, index); + const positions = observedPositionsByValue.get(token) ?? []; + positions.push(index); + observedPositionsByValue.set(token, positions); } const recoveredDistinct = expectedValues.filter((value) => observedPositions.has(value)).length; @@ -80,27 +117,39 @@ export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExp return leftPosition === undefined || rightPosition === undefined || leftPosition >= rightPosition; }); - const observedParagraphPositions = new Map(); + const observedParagraphPositionsByValue = new Map(); for (const [paragraphIndex, paragraphTokens] of observedByParagraph.entries()) { for (const token of paragraphTokens) { - if (!observedParagraphPositions.has(token)) observedParagraphPositions.set(token, paragraphIndex); + const positions = observedParagraphPositionsByValue.get(token) ?? []; + positions.push(paragraphIndex); + observedParagraphPositionsByValue.set(token, positions); } } - const groupedPairs = expectedTokenPairs.filter(([left, right]) => { - const leftEntry = expectedByValue.get(left); - const rightEntry = expectedByValue.get(right); - const leftParagraph = observedParagraphPositions.get(left); - const rightParagraph = observedParagraphPositions.get(right); - return ( - leftEntry?.group === rightEntry?.group && - leftParagraph !== undefined && - rightParagraph !== undefined && - leftParagraph === rightParagraph - ); + const eligibleExpectedPairs = expected.flatMap((token, index) => { + const next = expected[index + 1]; + return next && token.group === next.group ? [[index, index + 1] as const] : []; + }); + const groupedPairs = eligibleExpectedPairs.filter(([leftIndex, rightIndex]) => { + const left = expected[leftIndex]; + const right = expected[rightIndex]; + if (!left || !right) return false; + const leftPosition = observedPositionsByValue.get(left.value)?.[expectedOccurrenceOrdinals[leftIndex] ?? 0]; + const rightPosition = observedPositionsByValue.get(right.value)?.[expectedOccurrenceOrdinals[rightIndex] ?? 0]; + const leftParagraph = observedParagraphPositionsByValue.get(left.value)?.[ + expectedOccurrenceOrdinals[leftIndex] ?? 0 + ]; + const rightParagraph = observedParagraphPositionsByValue.get(right.value)?.[ + expectedOccurrenceOrdinals[rightIndex] ?? 0 + ]; + if (leftPosition === undefined || rightPosition === undefined) return false; + return leftParagraph !== undefined && rightParagraph !== undefined && leftParagraph === rightParagraph; }).length; - const expectedValuesSet = new Set(expectedValues); const observedExpectedOccurrences = observed.filter((token) => expectedValuesSet.has(token)).length; + const expectedLinks = normalizeLinks(corpus.links ?? []); + const observedLinks = normalizeLinks(extracted.links); + const observedLinksSet = new Set(observedLinks); + const expectedLinksSet = new Set(expectedLinks); return { recall: metric(recoveredDistinct, expectedValues.length), @@ -110,7 +159,13 @@ export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExp observed: observedExpectedOccurrences, extra: Math.max(0, observedExpectedOccurrences - expected.length), }, - grouping: metric(groupedPairs, expectedTokenPairs.length), + grouping: metric(groupedPairs, eligibleExpectedPairs.length), + links: { + expected: expectedLinks, + observed: observedLinks, + missing: expectedLinks.filter((link) => !observedLinksSet.has(link)), + unexpected: observedLinks.filter((link) => !expectedLinksSet.has(link)), + }, missingTokens: expectedValues.filter((value) => !observedPositions.has(value)), outOfOrderPairs, observedTokens: observed.length, diff --git a/tooling/package.json b/tooling/package.json index 0e73b059e..83a2a0ce2 100644 --- a/tooling/package.json +++ b/tooling/package.json @@ -24,6 +24,7 @@ "@reactive-resume/resume": "workspace:*", "@types/pg": "^8.23.1", "pdfjs-dist": "6.3.289", + "jszip": "3.10.1", "@typescript/native-preview": "7.0.0-dev.20260707.2", "drizzle-orm": "1.0.0-rc.4", "pg": "^8.23.0", From d17e188b0339a6e3289b0729423f7fe19d5373cd Mon Sep 17 00:00:00 2001 From: Amruth Pillai Date: Sun, 6 Sep 2026 06:57:08 +0200 Subject: [PATCH 3/4] test(tooling): cover visible website labels --- tooling/ats-export-evaluation/fixture.ts | 4 +++- tooling/ats-export-evaluation/metrics.test.ts | 21 ++++++++++++++++++- 2 files changed, 23 insertions(+), 2 deletions(-) diff --git a/tooling/ats-export-evaluation/fixture.ts b/tooling/ats-export-evaluation/fixture.ts index 641945229..14ee15845 100644 --- a/tooling/ats-export-evaluation/fixture.ts +++ b/tooling/ats-export-evaluation/fixture.ts @@ -34,7 +34,7 @@ const expectedTokens: readonly ExpectedToken[] = [ "mirakova.dev", "orbit-field-omega", ].map((value) => token(value, "header")), - ...["Profiles", "OrbitNet", "orbit-profile-omega"].map((value) => token(value, "profiles")), + ...["Profiles", "OrbitNet", "orbit-profile-omega", "orbit.example/profile"].map((value) => token(value, "profiles")), ...["Summary"].map((value) => token(value, "summary")), ...[text(SUMMARY_HTML)].map((value) => token(value, "summary")), ...["Experience"].map((value) => token(value, "experience")), @@ -48,6 +48,7 @@ const expectedTokens: readonly ExpectedToken[] = [ "Principal Reliability Engineer", "2021 / Present", text(EXPERIENCE_ROLE_TWO_HTML), + "northstar.example/jobs", ].map((value) => token(value, "experience")), ...["Education"].map((value) => token(value, "education")), ...[ @@ -58,6 +59,7 @@ const expectedTokens: readonly ExpectedToken[] = [ "2014 — 2018 (long academic period)", "東京", text(EDUCATION_HTML), + "u-tokyo.example/program", ].map((value) => token(value, "education")), ...["Skills", "TypeScript", "Advanced", "skill-keyword-alpha", "Kubernetes", "Expert", "skill-keyword-beta"].map( (value) => token(value, "skills"), diff --git a/tooling/ats-export-evaluation/metrics.test.ts b/tooling/ats-export-evaluation/metrics.test.ts index cafcdb98b..30d2ed83c 100644 --- a/tooling/ats-export-evaluation/metrics.test.ts +++ b/tooling/ats-export-evaluation/metrics.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from "vitest"; -import { evaluateExport } from "./metrics"; +import { createSyntheticCorpus } from "./fixture"; +import { evaluateExport, tokenize } from "./metrics"; const corpus = { name: "synthetic", @@ -60,6 +61,24 @@ describe("evaluateExport", () => { }); }); + it.each(["orbit.example/profile", "northstar.example/jobs", "u-tokyo.example/program"] as const)( + "fails when visible website label %s drops while link targets remain correct", + (label) => { + const corpus = createSyntheticCorpus("full-width"); + const paragraphs = corpus.tokens.map((entry) => entry.value); + const baseline = evaluateExport(corpus, { paragraphs, links: corpus.links }); + const dropped = evaluateExport(corpus, { + paragraphs: paragraphs.map((paragraph) => (paragraph === label ? "" : paragraph)), + links: corpus.links, + }); + + expect(baseline.links.missing).toEqual([]); + expect(baseline.links.unexpected).toEqual([]); + expect(dropped.links).toEqual(baseline.links); + expect(dropped.duplicates.observed).toBe(baseline.duplicates.observed - tokenize(label).length); + }, + ); + it("detects a dropped token and an inverted pair", () => { const result = evaluateExport(corpus, { paragraphs: ["Bravo", "Alpha"], From f447f429a9cc6879a16a9b8ea6e0a2af40a38932 Mon Sep 17 00:00:00 2001 From: "autofix-ci[bot]" <114827586+autofix-ci[bot]@users.noreply.github.com> Date: Sun, 6 Sep 2026 04:59:17 +0000 Subject: [PATCH 4/4] [autofix.ci] apply automated fixes --- tooling/ats-export-evaluation/extract.ts | 2 +- tooling/ats-export-evaluation/fixture.ts | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tooling/ats-export-evaluation/extract.ts b/tooling/ats-export-evaluation/extract.ts index abbc33533..67a8662a1 100644 --- a/tooling/ats-export-evaluation/extract.ts +++ b/tooling/ats-export-evaluation/extract.ts @@ -10,7 +10,7 @@ export type PdfExtraction = { links: readonly string[]; }; -export type DocxParagraph = { +type DocxParagraph = { text: string; numbering: { numId: string; level: string; format: string; marker: string } | null; }; diff --git a/tooling/ats-export-evaluation/fixture.ts b/tooling/ats-export-evaluation/fixture.ts index 14ee15845..820c6930f 100644 --- a/tooling/ats-export-evaluation/fixture.ts +++ b/tooling/ats-export-evaluation/fixture.ts @@ -70,7 +70,7 @@ const expectedTokens: readonly ExpectedToken[] = [ ...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")), ]; -export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta", "https://hidden.example"] as const; +const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta", "https://hidden.example"] as const; const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false });