test: evaluate ATS PDF and DOCX extraction

This commit is contained in:
Amruth Pillai
2026-09-06 06:23:16 +02:00
parent e71b5e6e91
commit f89873f083
9 changed files with 779 additions and 0 deletions
+9
View File
@@ -1349,9 +1349,15 @@ importers:
'@reactive-resume/config':
specifier: workspace:*
version: link:../packages/config
'@reactive-resume/docx':
specifier: workspace:*
version: link:../packages/docx
'@reactive-resume/env':
specifier: workspace:*
version: link:../packages/env
'@reactive-resume/pdf':
specifier: workspace:*
version: link:../packages/pdf
'@reactive-resume/resume':
specifier: workspace:*
version: link:../packages/resume
@@ -1364,6 +1370,9 @@ importers:
drizzle-orm:
specifier: 1.0.0-rc.4
version: 1.0.0-rc.4(@types/pg@8.23.1)(pg@8.23.0)(zod@4.5.4)
pdfjs-dist:
specifier: 6.3.289
version: 6.3.289
pg:
specifier: ^8.23.0
version: 8.23.0
+17
View File
@@ -0,0 +1,17 @@
# ATS export evaluation
`evaluation.integration.test.ts` renders deterministic synthetic resume data through current
`ResumeDocument` and `buildDocx`, then measures extraction with installed PDF.js and DOCX XML.
`metrics.test.ts` locks raw distinct-token recall, order, duplicate, and semantic-grouping behavior,
including deliberate drop/duplicate regressions.
Run from repository root:
```sh
pnpm --filter @reactive-resume/tooling test
```
The integration test writes PDF/DOCX fixtures and raw-count reports to
`tooling/ats-export-evaluation/test-results/` (ignored test output). Results explicitly distinguish
local extraction measurements from vendor parser accuracy. Steps 1–2 add no ATS preset; a later
product decision can use measured deficiencies from `ats-export-report.md`.
@@ -0,0 +1,171 @@
// @vitest-environment happy-dom
import type { SectionTitleResolver } from "@reactive-resume/pdf/section-title";
import type { ExportVariant } from "./fixture";
import type { ExportMetrics } from "./metrics";
import { mkdir, writeFile } from "node:fs/promises";
import { join, resolve } from "node:path";
import { describe, expect, it } from "vitest";
import { buildDocx } from "@reactive-resume/docx";
import { createResumePdfFile } from "@reactive-resume/pdf/server";
import { extractDocx, extractPdf } from "./extract";
import { createSyntheticCorpus } from "./fixture";
import { evaluateExport, tokenize } from "./metrics";
const outputDirectory = resolve(process.cwd(), "ats-export-evaluation/test-results");
const pdfTitleResolver: SectionTitleResolver = ({ defaultEnglishTitle, sectionId }) => defaultEnglishTitle ?? sectionId;
const docxTitleResolver = (sectionId: string) => sectionId;
type FormatResult = {
format: "pdf" | "docx";
metrics: ExportMetrics;
links: readonly string[];
paragraphs: number;
pageCount?: number;
fontCount?: number;
numberingDefinitions?: number;
numberedParagraphs?: number;
hiddenLeaks: readonly string[];
};
type VariantResult = {
variant: ExportVariant;
formats: FormatResult[];
};
function hiddenLeaks(paragraphs: readonly string[], hiddenTokens: readonly string[]): string[] {
const observed = paragraphs.flatMap(tokenize);
return hiddenTokens.filter((value) => {
const expected = tokenize(value);
return observed.some((_, index) => expected.every((token, offset) => observed[index + offset] === token));
});
}
async function measureVariant(variant: ExportVariant): Promise<VariantResult> {
const corpus = createSyntheticCorpus(variant);
const before = JSON.stringify(corpus.data);
const pdfFile = await createResumePdfFile({
data: corpus.data,
filename: `${corpus.name}.pdf`,
template: corpus.data.metadata.template,
resolveSectionTitle: pdfTitleResolver,
});
const pdfBytes = new Uint8Array(await pdfFile.arrayBuffer());
const pdf = await extractPdf(pdfBytes);
await writeFile(join(outputDirectory, `${corpus.name}.pdf`), pdfBytes);
const docxBlob = await buildDocx(corpus.data, docxTitleResolver);
const docxBytes = new Uint8Array(await docxBlob.arrayBuffer());
const docx = await extractDocx(docxBytes);
await writeFile(join(outputDirectory, `${corpus.name}.docx`), docxBytes);
expect(JSON.stringify(corpus.data)).toBe(before);
return {
variant,
formats: [
{
format: "pdf",
metrics: evaluateExport(corpus, { paragraphs: pdf.paragraphs, links: pdf.links }),
links: pdf.links,
paragraphs: pdf.paragraphs.length,
pageCount: pdf.raw.pageCount,
fontCount: pdf.raw.fonts.length,
hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens),
},
{
format: "docx",
metrics: evaluateExport(corpus, {
paragraphs: docx.paragraphs.map((paragraph) => paragraph.text),
links: docx.links,
}),
links: docx.links,
paragraphs: docx.paragraphs.length,
numberingDefinitions: docx.numberingDefinitions,
numberedParagraphs: docx.numberedParagraphs,
hiddenLeaks: hiddenLeaks(
docx.paragraphs.map((paragraph) => paragraph.text),
corpus.hiddenTokens,
),
},
],
};
}
function percentage(value: number): string {
return `${(value * 100).toFixed(1)}%`;
}
function reportMarkdown(results: readonly VariantResult[]): string {
const lines = [
"# ATS export evaluation",
"",
"Synthetic extraction measurements only. These are not vendor parsing accuracy claims.",
"",
"Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses pairs expected in the same authored field group and recovered in the same extracted paragraph/line.",
"",
"Corpus: one deterministic resume fixture per layout variant, covering header/contact, two roles, free-text dates, education, skills, project, custom section, long lines, links, hidden item, and CJK text. Hidden item tokens are intentionally excluded from expected recall and checked for leakage.",
"",
"| Variant | Format | Recall raw | Order raw | Duplicate raw | Grouping raw | Pages/paragraphs | Links | Numbering defs/paragraphs | Hidden leaks |",
"| --- | --- | --- | --- | --- | --- | ---: | ---: | ---: | --- |",
];
for (const result of results) {
for (const format of result.formats) {
const m = format.metrics;
lines.push(
`| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | ${format.links.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`,
);
}
}
lines.push(
"",
"PDF font objects and raw link targets are recorded in adjacent JSON output.",
"",
"## Concrete loss/order evidence",
"",
);
for (const result of results) {
for (const format of result.formats) {
const m = format.metrics;
lines.push(
`- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}.`,
);
}
}
lines.push("");
return lines.join("\n");
}
describe("current unchanged PDF and DOCX exports", () => {
it("measures two-column and full-width synthetic corpus without mutating input", { timeout: 120_000 }, async () => {
await mkdir(outputDirectory, { recursive: true });
const results = [await measureVariant("two-column"), await measureVariant("full-width")];
const report = {
claimBoundary: "Local extraction metrics only; no vendor parsing accuracy claim.",
tokenRules: "NFC, en-US case-folding, Unicode letter/number runs; punctuation separates tokens.",
corpus: {
variants: results.map((result) => result.variant),
expectedDistinctTokens: createSyntheticCorpus("full-width")
.tokens.flatMap((entry) => tokenize(entry.value))
.filter((value, index, values) => values.indexOf(value) === index).length,
},
results,
};
await writeFile(join(outputDirectory, "ats-export-report.json"), `${JSON.stringify(report, null, 2)}\n`);
await writeFile(join(outputDirectory, "ats-export-report.md"), reportMarkdown(results));
for (const result of results) {
const pdf = result.formats.find((format) => format.format === "pdf");
const docx = result.formats.find((format) => format.format === "docx");
if (!pdf || !docx) throw new Error(`Missing measured format for ${result.variant}`);
expect(pdf.metrics.recall.denominator).toBeGreaterThan(20);
expect(docx.metrics.recall.denominator).toBe(pdf.metrics.recall.denominator);
expect(pdf.hiddenLeaks).toEqual([]);
expect(docx.hiddenLeaks).toEqual([]);
expect(pdf.fontCount).toBeGreaterThan(0);
expect(docx.numberingDefinitions).toBeGreaterThan(0);
expect(docx.numberedParagraphs).toBe(0);
}
});
});
+154
View File
@@ -0,0 +1,154 @@
import type { ExtractedDocument, PdfDocumentLike, RawExtraction } from "@reactive-resume/resume/ats-pdf";
import { execFile } from "node:child_process";
import { mkdtemp, rm, writeFile } from "node:fs/promises";
import { tmpdir } from "node:os";
import { join } from "node:path";
import { promisify } from "node:util";
import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs";
import { buildExtractedDocument, harvestPdfDocument } from "@reactive-resume/resume/ats-pdf";
const execFileAsync = promisify(execFile);
export type PdfExtraction = {
raw: RawExtraction;
document: ExtractedDocument;
paragraphs: readonly string[];
links: readonly string[];
};
export type DocxParagraph = {
text: string;
numbering: { numId: string; level: string; format: string; marker: string } | null;
};
export type DocxExtraction = {
paragraphs: readonly DocxParagraph[];
links: readonly string[];
numberingDefinitions: number;
numberedParagraphs: number;
};
const unescapeXml = (value: string): string =>
value
.replaceAll("&lt;", "<")
.replaceAll("&gt;", ">")
.replaceAll("&quot;", '"')
.replaceAll("&apos;", "'")
.replaceAll("&amp;", "&");
const attribute = (attributes: string, name: string): string | null => {
const match = attributes.match(new RegExp(`(?:^|\\s)(?:[A-Za-z][\\w-]*:)?${name}="([^"]*)"`));
return match ? unescapeXml(match[1] ?? "") : null;
};
const zipEntry = async (archivePath: string, entry: string): Promise<string> => {
const { stdout } = await execFileAsync("unzip", ["-p", archivePath, entry]);
return stdout;
};
export async function extractPdf(bytes: Uint8Array): Promise<PdfExtraction> {
const loadingTask = getDocument({ data: new Uint8Array(bytes), fontExtraProperties: true });
try {
const document = (await loadingTask.promise) as unknown as PdfDocumentLike;
const raw = await harvestPdfDocument(document, {
file: { name: "synthetic-resume.pdf", sizeBytes: bytes.byteLength, magicBytesOk: true },
});
const extracted = buildExtractedDocument(raw);
return {
raw,
document: extracted,
paragraphs: extracted.lines.map((line) => line.text),
links: raw.links.flatMap((link) => (link.url ? [link.url] : [])),
};
} finally {
await loadingTask.destroy();
}
}
function parseNumbering(numberingXml: string): Map<string, { format: string; marker: string }> {
const formats = new Map<string, { format: string; marker: string }>();
const abstractDefinitions = new Map<string, { format: string; marker: string }>();
for (const abstract of numberingXml.matchAll(/<w:abstractNum\b([^>]*)>([\s\S]*?)<\/w:abstractNum>/g)) {
const abstractId = attribute(abstract[1] ?? "", "abstractNumId");
if (!abstractId) continue;
for (const level of (abstract[2] ?? "").matchAll(/<w:lvl\b([^>]*)>([\s\S]*?)<\/w:lvl>/g)) {
const levelId = attribute(level[1] ?? "", "ilvl") ?? "0";
const format = attribute(level[2] ?? "", "val") ?? "unknown";
const marker = attribute((level[2] ?? "").match(/<w:lvlText\b([^>]*)\/>/)?.[1] ?? "", "val") ?? "";
abstractDefinitions.set(`${abstractId}:${levelId}`, { format, marker });
}
}
for (const numbering of numberingXml.matchAll(/<w:num\b([^>]*)>([\s\S]*?)<\/w:num>/g)) {
const numId = attribute(numbering[1] ?? "", "numId");
const abstractId = attribute((numbering[2] ?? "").match(/<w:abstractNumId\b([^>]*)\/>/)?.[1] ?? "", "val");
if (!numId || !abstractId) continue;
for (const level of ["0", "1", "2", "3", "4", "5", "6", "7", "8"]) {
const definition = abstractDefinitions.get(`${abstractId}:${level}`);
if (definition) formats.set(`${numId}:${level}`, definition);
}
}
return formats;
}
function parseDocxParagraphs(documentXml: string, numbering: Map<string, { format: string; marker: string }>) {
const paragraphs: DocxParagraph[] = [];
for (const paragraph of documentXml.matchAll(/<w:p\b[^>]*>([\s\S]*?)<\/w:p>/g)) {
const body = paragraph[1] ?? "";
const text = [...body.matchAll(/<w:t\b[^>]*>([\s\S]*?)<\/w:t>/g)]
.map((match) => unescapeXml(match[1] ?? ""))
.join("");
const numPr = body.match(/<w:numPr\b[^>]*>([\s\S]*?)<\/w:numPr>/)?.[1];
const numId = numPr ? numPr.match(/<w:numId\b([^>]*)\/>/) : null;
const level = numPr ? numPr.match(/<w:ilvl\b([^>]*)\/>/) : null;
const numIdValue = numId ? attribute(numId[1] ?? "", "val") : null;
const levelValue = level ? (attribute(level[1] ?? "", "val") ?? "0") : null;
const definition = numIdValue && levelValue ? numbering.get(`${numIdValue}:${levelValue}`) : undefined;
paragraphs.push({
text,
numbering:
numIdValue && levelValue && definition
? { numId: numIdValue, level: levelValue, format: definition.format, marker: definition.marker }
: null,
});
}
return paragraphs;
}
function parseDocxLinks(documentXml: string, relationshipsXml: string): string[] {
const relationships = new Map<string, string>();
for (const relationship of relationshipsXml.matchAll(/<Relationship\b([^>]*)\/>/g)) {
const id = attribute(relationship[1] ?? "", "Id");
const target = attribute(relationship[1] ?? "", "Target");
if (id && target) relationships.set(id, target);
}
const links: string[] = [];
for (const hyperlink of documentXml.matchAll(/<w:hyperlink\b([^>]*)>/g)) {
const id = attribute(hyperlink[1] ?? "", "id");
const target = id ? relationships.get(id) : undefined;
if (target) links.push(target);
}
return links;
}
export async function extractDocx(bytes: Uint8Array): Promise<DocxExtraction> {
const tempDirectory = await mkdtemp(join(tmpdir(), "reactive-resume-ats-"));
const archivePath = join(tempDirectory, "resume.docx");
try {
await writeFile(archivePath, bytes);
const [documentXml, numberingXml, relationshipsXml] = await Promise.all([
zipEntry(archivePath, "word/document.xml"),
zipEntry(archivePath, "word/numbering.xml"),
zipEntry(archivePath, "word/_rels/document.xml.rels"),
]);
const numbering = parseNumbering(numberingXml);
const paragraphs = parseDocxParagraphs(documentXml, numbering);
return {
paragraphs,
links: parseDocxLinks(documentXml, relationshipsXml),
numberingDefinitions: numbering.size,
numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length,
};
} finally {
await rm(tempDirectory, { recursive: true, force: true });
}
}
+250
View File
@@ -0,0 +1,250 @@
import type { ResumeData } from "@reactive-resume/schema/resume/data";
import type { EvaluationCorpus, ExpectedToken } from "./metrics";
import { sampleResumeData } from "@reactive-resume/schema/resume/sample";
export type ExportVariant = "two-column" | "full-width";
export type SyntheticCorpus = EvaluationCorpus & {
data: ResumeData;
hiddenTokens: readonly string[];
links: readonly string[];
};
const token = (value: string, group: string): ExpectedToken => ({ value, group });
const text = (html: string) => html.replace(/<[^>]+>/g, " ");
const SUMMARY_HTML =
"<p>summary-signal-alpha leads longline-calibration with multilingual 東京大学 context and durable systems.</p>";
const EXPERIENCE_ROLE_ONE_HTML =
"<ul><li><p>role-bullet-alpha shipped resilient pipeline-observability under longline-pressure.</p></li><li><p>role-bullet-beta measured queue-latency and improved release-safety.</p></li></ul>";
const EXPERIENCE_ROLE_TWO_HTML =
"<p>role-bullet-gamma guided distributed-runtime migration for platform-reliability.</p>";
const EDUCATION_HTML = "<p>education-signal-delta researched multilingual retrieval and evaluation.</p>";
const PROJECT_HTML = "<p>project-signal-epsilon demonstrates export-fixture determinism.</p>";
const CUSTOM_HTML = "<p>custom-signal-zeta preserves authored custom content and free-text dates.</p>";
const expectedTokens: readonly ExpectedToken[] = [
...[
"Mira Kova",
"Principal Systems Architect",
"mira.kova@example.com",
"+49 30 555 0142",
"Berlin 東京",
"mirakova.dev",
"orbit-field-omega",
].map((value) => token(value, "header")),
...[text(SUMMARY_HTML)].map((value) => token(value, "summary")),
...[
"Northstar Robotics",
"2018-02 — Present",
"Berlin",
"Staff Platform Engineer",
"2018-02 to 2020-12",
text(EXPERIENCE_ROLE_ONE_HTML),
"Principal Reliability Engineer",
"2021 / Present",
text(EXPERIENCE_ROLE_TWO_HTML),
].map((value) => token(value, "experience")),
...[
"東京大学",
"Master of Computer Science",
"Distributed Systems",
"3.98 GPA",
"2014 — 2018 (long academic period)",
"東京",
text(EDUCATION_HTML),
].map((value) => token(value, "education")),
...["TypeScript", "skill-keyword-alpha", "Kubernetes", "skill-keyword-beta"].map((value) => token(value, "skills")),
...["Export Observatory", "2022 to Winter 2024", text(PROJECT_HTML), "project.example/observatory"].map((value) =>
token(value, "projects"),
),
...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")),
];
export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta"] as const;
const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false });
export function createSyntheticCorpus(variant: ExportVariant): SyntheticCorpus {
const data = structuredClone(sampleResumeData);
data.picture.hidden = true;
data.basics = {
name: "Mira Kova",
headline: "Principal Systems Architect",
email: "mira.kova@example.com",
phone: "+49 30 555 0142",
location: "Berlin 東京",
website: resolveWebsite("https://mirakova.dev", "mirakova.dev"),
customFields: [
{
id: "synthetic-field-omega",
icon: "",
text: "orbit-field-omega",
link: "https://orbit.example/omega",
},
],
};
data.summary = {
...data.summary,
title: "Summary",
content: SUMMARY_HTML,
};
data.sections.profiles = {
...data.sections.profiles,
title: "Profiles",
items: [
{
id: "synthetic-profile-omega",
hidden: false,
icon: "",
iconColor: "",
network: "OrbitNet",
username: "orbit-profile-omega",
website: resolveWebsite("https://orbit.example/profile", "orbit.example/profile"),
},
],
};
data.sections.experience = {
...data.sections.experience,
title: "Experience",
items: [
{
id: "synthetic-experience-northstar",
hidden: false,
company: "Northstar Robotics",
position: "",
location: "Berlin",
period: "2018-02 — Present",
website: resolveWebsite("https://northstar.example/jobs", "northstar.example/jobs"),
roles: [
{
id: "synthetic-role-staff",
position: "Staff Platform Engineer",
period: "2018-02 to 2020-12",
description: EXPERIENCE_ROLE_ONE_HTML,
},
{
id: "synthetic-role-principal",
position: "Principal Reliability Engineer",
period: "2021 / Present",
description: EXPERIENCE_ROLE_TWO_HTML,
},
],
description: "",
},
{
id: "synthetic-hidden-experience",
hidden: true,
company: HIDDEN_TOKENS[0],
position: "",
location: "",
period: "",
website: resolveWebsite("https://hidden.example", "hidden.example"),
roles: [],
description: `<p>${HIDDEN_TOKENS[1]}</p>`,
},
],
};
data.sections.education = {
...data.sections.education,
title: "Education",
items: [
{
id: "synthetic-education-tokyo",
hidden: false,
school: "東京大学",
degree: "Master of Computer Science",
area: "Distributed Systems",
grade: "3.98 GPA",
location: "東京",
period: "2014 — 2018 (long academic period)",
website: resolveWebsite("https://u-tokyo.example/program", "u-tokyo.example/program"),
description: EDUCATION_HTML,
},
],
};
data.sections.skills = {
...data.sections.skills,
title: "Skills",
items: [
{
id: "synthetic-skill-typescript",
hidden: false,
icon: "",
iconColor: "",
name: "TypeScript",
proficiency: "Advanced",
level: 4,
keywords: ["skill-keyword-alpha"],
},
{
id: "synthetic-skill-kubernetes",
hidden: false,
icon: "",
iconColor: "",
name: "Kubernetes",
proficiency: "Expert",
level: 5,
keywords: ["skill-keyword-beta"],
},
],
};
data.sections.projects = {
...data.sections.projects,
title: "Projects",
items: [
{
id: "synthetic-project-observatory",
hidden: false,
name: "Export Observatory",
period: "2022 to Winter 2024",
website: resolveWebsite("https://project.example/observatory", "project.example/observatory"),
description: PROJECT_HTML,
},
],
};
data.customSections = [
{
id: "custom-ats-evidence",
type: "summary",
title: "Custom Evidence",
icon: "",
columns: 1,
hidden: false,
showHeading: true,
keepTogether: false,
startOnNewPage: false,
items: [{ id: "synthetic-custom-zeta", hidden: false, content: CUSTOM_HTML }],
},
];
const mainSections = ["profiles", "summary", "experience", "education", "projects"];
const sidebarSections = ["skills", "custom-ats-evidence"];
const page = {
fullWidth: variant === "full-width",
main: variant === "full-width" ? [...mainSections, ...sidebarSections] : mainSections,
sidebar: variant === "full-width" ? [] : sidebarSections,
};
data.metadata = {
...data.metadata,
template: variant === "full-width" ? "onyx" : "gengar",
layout: { ...data.metadata.layout, pages: [page] },
page: { ...data.metadata.page, locale: "en-US", hideIcons: true, hideSectionIcons: true },
};
return {
name: `ats-${variant}`,
tokens: expectedTokens,
data,
hiddenTokens: HIDDEN_TOKENS,
links: [
"https://mirakova.dev",
"https://orbit.example/omega",
"https://orbit.example/profile",
"https://northstar.example/jobs",
"https://u-tokyo.example/program",
"https://project.example/observatory",
],
};
}
@@ -0,0 +1,43 @@
import { describe, expect, it } from "vitest";
import { evaluateExport } from "./metrics";
const corpus = {
name: "synthetic",
tokens: [
{ value: "Alpha", group: "experience" },
{ value: "Bravo", group: "experience" },
{ value: "Charlie", group: "education" },
] as const,
};
describe("evaluateExport", () => {
it("counts distinct recall, order, duplicate, and grouping losses from raw tokens", () => {
const result = evaluateExport(corpus, {
paragraphs: ["Alpha Bravo Bravo", "Charlie"],
links: [],
});
expect(result.recall).toEqual({ numerator: 3, denominator: 3, value: 1 });
expect(result.order).toEqual({ numerator: 2, denominator: 2, value: 1 });
expect(result.duplicates).toEqual({ expected: 3, observed: 4, extra: 1 });
expect(result.missingTokens).toEqual([]);
expect(result.outOfOrderPairs).toEqual([]);
expect(result.grouping).toEqual({ numerator: 1, denominator: 2, value: 0.5 });
});
it("detects a dropped token and an inverted pair", () => {
const result = evaluateExport(corpus, {
paragraphs: ["Bravo", "Alpha"],
links: [],
});
expect(result.recall).toEqual({ numerator: 2, denominator: 3, value: 2 / 3 });
expect(result.order).toEqual({ numerator: 0, denominator: 2, value: 0 });
expect(result.duplicates).toEqual({ expected: 3, observed: 2, extra: 0 });
expect(result.missingTokens).toEqual(["charlie"]);
expect(result.outOfOrderPairs).toEqual([
["alpha", "bravo"],
["bravo", "charlie"],
]);
});
});
+118
View File
@@ -0,0 +1,118 @@
export type ExpectedToken = {
value: string;
group: string;
};
export type EvaluationCorpus = {
name: string;
tokens: readonly ExpectedToken[];
};
export type ExtractedExport = {
/** Paragraphs/lines in the extractor's returned order. */
paragraphs: readonly string[];
links: readonly string[];
};
export type ExportMetrics = {
recall: { numerator: number; denominator: number; value: number };
order: { numerator: number; denominator: number; value: number };
duplicates: { expected: number; observed: number; extra: number };
grouping: { numerator: number; denominator: number; value: number };
missingTokens: readonly string[];
outOfOrderPairs: readonly (readonly [string, string])[];
observedTokens: number;
};
/**
* Tokenization used by this evaluation only. It is deliberately transparent and locale-neutral:
* Unicode letters/numbers stay intact, punctuation is a separator, and matching is case-folded.
* This is a corpus metric, not a claim about any vendor parser.
*/
export function tokenize(value: string): string[] {
return (
value
.normalize("NFC")
.match(/[\p{L}\p{N}]+/gu)
?.map((token) => token.toLocaleLowerCase("en-US")) ?? []
);
}
const metric = (numerator: number, denominator: number) => ({
numerator,
denominator,
value: denominator === 0 ? 1 : numerator / denominator,
});
/**
* Computes raw extraction measurements from corpus tokens and extractor paragraphs.
*
* Recall uses distinct expected tokens. Duplicate accounting separately reports all matching
* occurrences, so dropping a token cannot be hidden by duplicate output. Order and grouping use
* the first occurrence of each distinct expected token, keeping those measures interpretable when
* an export repeats a heading or bullet.
*/
export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExport): ExportMetrics {
const expected = corpus.tokens.flatMap((entry) =>
tokenize(entry.value).map((value) => ({ value, group: entry.group })),
);
const expectedByValue = new Map<string, { group: string; index: number }>();
for (const [index, token] of expected.entries()) {
if (!expectedByValue.has(token.value)) expectedByValue.set(token.value, { group: token.group, index });
}
const observedByParagraph = extracted.paragraphs.map(tokenize);
const observed = observedByParagraph.flat();
const expectedValues = [...expectedByValue.keys()];
const observedPositions = new Map<string, number>();
for (const [index, token] of observed.entries()) {
if (!observedPositions.has(token)) observedPositions.set(token, index);
}
const recoveredDistinct = expectedValues.filter((value) => observedPositions.has(value)).length;
const expectedTokenPairs = expectedValues.flatMap((value, index) => {
const next = expectedValues[index + 1];
return next ? ([[value, next]] as const) : [];
});
const outOfOrderPairs = expectedTokenPairs.filter(([left, right]) => {
const leftPosition = observedPositions.get(left);
const rightPosition = observedPositions.get(right);
return leftPosition === undefined || rightPosition === undefined || leftPosition >= rightPosition;
});
const observedParagraphPositions = new Map<string, number>();
for (const [paragraphIndex, paragraphTokens] of observedByParagraph.entries()) {
for (const token of paragraphTokens) {
if (!observedParagraphPositions.has(token)) observedParagraphPositions.set(token, paragraphIndex);
}
}
const groupedPairs = expectedTokenPairs.filter(([left, right]) => {
const leftEntry = expectedByValue.get(left);
const rightEntry = expectedByValue.get(right);
const leftParagraph = observedParagraphPositions.get(left);
const rightParagraph = observedParagraphPositions.get(right);
return (
leftEntry?.group === rightEntry?.group &&
leftParagraph !== undefined &&
rightParagraph !== undefined &&
leftParagraph === rightParagraph
);
}).length;
const expectedValuesSet = new Set(expectedValues);
const observedExpectedOccurrences = observed.filter((token) => expectedValuesSet.has(token)).length;
return {
recall: metric(recoveredDistinct, expectedValues.length),
order: metric(expectedTokenPairs.length - outOfOrderPairs.length, expectedTokenPairs.length),
duplicates: {
expected: expected.length,
observed: observedExpectedOccurrences,
extra: Math.max(0, observedExpectedOccurrences - expected.length),
},
grouping: metric(groupedPairs, expectedTokenPairs.length),
missingTokens: expectedValues.filter((value) => !observedPositions.has(value)),
outOfOrderPairs,
observedTokens: observed.length,
};
}
+3
View File
@@ -18,9 +18,12 @@
"devDependencies": {
"@lingui/format-po": "^6.6.0",
"@reactive-resume/config": "workspace:*",
"@reactive-resume/docx": "workspace:*",
"@reactive-resume/env": "workspace:*",
"@reactive-resume/pdf": "workspace:*",
"@reactive-resume/resume": "workspace:*",
"@types/pg": "^8.23.1",
"pdfjs-dist": "6.3.289",
"@typescript/native-preview": "7.0.0-dev.20260707.2",
"drizzle-orm": "1.0.0-rc.4",
"pg": "^8.23.0",
+14
View File
@@ -0,0 +1,14 @@
import { fileURLToPath } from "node:url";
// @boundaries-ignore root shared Vitest config
import { createVitestProjectConfig } from "../vitest.shared.mts";
const config = createVitestProjectConfig({
name: "@reactive-resume/tooling",
dirname: fileURLToPath(new URL(".", import.meta.url)),
});
export default {
...config,
test: { ...config.test, include: ["**/*.{test,spec}.?(c|m)[jt]s?(x)"] },
oxc: { jsx: { runtime: "automatic" as const } },
};