mirror of
https://github.com/AmruthPillai/Reactive-Resume.git
synced 2026-10-03 10:13:47 +10:00
test: evaluate ATS PDF and DOCX extraction
This commit is contained in:
Generated
+9
@@ -1349,9 +1349,15 @@ importers:
|
||||
'@reactive-resume/config':
|
||||
specifier: workspace:*
|
||||
version: link:../packages/config
|
||||
'@reactive-resume/docx':
|
||||
specifier: workspace:*
|
||||
version: link:../packages/docx
|
||||
'@reactive-resume/env':
|
||||
specifier: workspace:*
|
||||
version: link:../packages/env
|
||||
'@reactive-resume/pdf':
|
||||
specifier: workspace:*
|
||||
version: link:../packages/pdf
|
||||
'@reactive-resume/resume':
|
||||
specifier: workspace:*
|
||||
version: link:../packages/resume
|
||||
@@ -1364,6 +1370,9 @@ importers:
|
||||
drizzle-orm:
|
||||
specifier: 1.0.0-rc.4
|
||||
version: 1.0.0-rc.4(@types/pg@8.23.1)(pg@8.23.0)(zod@4.5.4)
|
||||
pdfjs-dist:
|
||||
specifier: 6.3.289
|
||||
version: 6.3.289
|
||||
pg:
|
||||
specifier: ^8.23.0
|
||||
version: 8.23.0
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
# ATS export evaluation
|
||||
|
||||
`evaluation.integration.test.ts` renders deterministic synthetic resume data through current
|
||||
`ResumeDocument` and `buildDocx`, then measures extraction with installed PDF.js and DOCX XML.
|
||||
`metrics.test.ts` locks raw distinct-token recall, order, duplicate, and semantic-grouping behavior,
|
||||
including deliberate drop/duplicate regressions.
|
||||
|
||||
Run from repository root:
|
||||
|
||||
```sh
|
||||
pnpm --filter @reactive-resume/tooling test
|
||||
```
|
||||
|
||||
The integration test writes PDF/DOCX fixtures and raw-count reports to
|
||||
`tooling/ats-export-evaluation/test-results/` (ignored test output). Results explicitly distinguish
|
||||
local extraction measurements from vendor parser accuracy. Steps 1–2 add no ATS preset; a later
|
||||
product decision can use measured deficiencies from `ats-export-report.md`.
|
||||
@@ -0,0 +1,171 @@
|
||||
// @vitest-environment happy-dom
|
||||
|
||||
import type { SectionTitleResolver } from "@reactive-resume/pdf/section-title";
|
||||
import type { ExportVariant } from "./fixture";
|
||||
import type { ExportMetrics } from "./metrics";
|
||||
import { mkdir, writeFile } from "node:fs/promises";
|
||||
import { join, resolve } from "node:path";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { buildDocx } from "@reactive-resume/docx";
|
||||
import { createResumePdfFile } from "@reactive-resume/pdf/server";
|
||||
import { extractDocx, extractPdf } from "./extract";
|
||||
import { createSyntheticCorpus } from "./fixture";
|
||||
import { evaluateExport, tokenize } from "./metrics";
|
||||
|
||||
const outputDirectory = resolve(process.cwd(), "ats-export-evaluation/test-results");
|
||||
|
||||
const pdfTitleResolver: SectionTitleResolver = ({ defaultEnglishTitle, sectionId }) => defaultEnglishTitle ?? sectionId;
|
||||
const docxTitleResolver = (sectionId: string) => sectionId;
|
||||
|
||||
type FormatResult = {
|
||||
format: "pdf" | "docx";
|
||||
metrics: ExportMetrics;
|
||||
links: readonly string[];
|
||||
paragraphs: number;
|
||||
pageCount?: number;
|
||||
fontCount?: number;
|
||||
numberingDefinitions?: number;
|
||||
numberedParagraphs?: number;
|
||||
hiddenLeaks: readonly string[];
|
||||
};
|
||||
|
||||
type VariantResult = {
|
||||
variant: ExportVariant;
|
||||
formats: FormatResult[];
|
||||
};
|
||||
|
||||
function hiddenLeaks(paragraphs: readonly string[], hiddenTokens: readonly string[]): string[] {
|
||||
const observed = paragraphs.flatMap(tokenize);
|
||||
return hiddenTokens.filter((value) => {
|
||||
const expected = tokenize(value);
|
||||
return observed.some((_, index) => expected.every((token, offset) => observed[index + offset] === token));
|
||||
});
|
||||
}
|
||||
|
||||
async function measureVariant(variant: ExportVariant): Promise<VariantResult> {
|
||||
const corpus = createSyntheticCorpus(variant);
|
||||
const before = JSON.stringify(corpus.data);
|
||||
const pdfFile = await createResumePdfFile({
|
||||
data: corpus.data,
|
||||
filename: `${corpus.name}.pdf`,
|
||||
template: corpus.data.metadata.template,
|
||||
resolveSectionTitle: pdfTitleResolver,
|
||||
});
|
||||
const pdfBytes = new Uint8Array(await pdfFile.arrayBuffer());
|
||||
const pdf = await extractPdf(pdfBytes);
|
||||
await writeFile(join(outputDirectory, `${corpus.name}.pdf`), pdfBytes);
|
||||
|
||||
const docxBlob = await buildDocx(corpus.data, docxTitleResolver);
|
||||
const docxBytes = new Uint8Array(await docxBlob.arrayBuffer());
|
||||
const docx = await extractDocx(docxBytes);
|
||||
await writeFile(join(outputDirectory, `${corpus.name}.docx`), docxBytes);
|
||||
|
||||
expect(JSON.stringify(corpus.data)).toBe(before);
|
||||
|
||||
return {
|
||||
variant,
|
||||
formats: [
|
||||
{
|
||||
format: "pdf",
|
||||
metrics: evaluateExport(corpus, { paragraphs: pdf.paragraphs, links: pdf.links }),
|
||||
links: pdf.links,
|
||||
paragraphs: pdf.paragraphs.length,
|
||||
pageCount: pdf.raw.pageCount,
|
||||
fontCount: pdf.raw.fonts.length,
|
||||
hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens),
|
||||
},
|
||||
{
|
||||
format: "docx",
|
||||
metrics: evaluateExport(corpus, {
|
||||
paragraphs: docx.paragraphs.map((paragraph) => paragraph.text),
|
||||
links: docx.links,
|
||||
}),
|
||||
links: docx.links,
|
||||
paragraphs: docx.paragraphs.length,
|
||||
numberingDefinitions: docx.numberingDefinitions,
|
||||
numberedParagraphs: docx.numberedParagraphs,
|
||||
hiddenLeaks: hiddenLeaks(
|
||||
docx.paragraphs.map((paragraph) => paragraph.text),
|
||||
corpus.hiddenTokens,
|
||||
),
|
||||
},
|
||||
],
|
||||
};
|
||||
}
|
||||
|
||||
function percentage(value: number): string {
|
||||
return `${(value * 100).toFixed(1)}%`;
|
||||
}
|
||||
|
||||
function reportMarkdown(results: readonly VariantResult[]): string {
|
||||
const lines = [
|
||||
"# ATS export evaluation",
|
||||
"",
|
||||
"Synthetic extraction measurements only. These are not vendor parsing accuracy claims.",
|
||||
"",
|
||||
"Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses pairs expected in the same authored field group and recovered in the same extracted paragraph/line.",
|
||||
"",
|
||||
"Corpus: one deterministic resume fixture per layout variant, covering header/contact, two roles, free-text dates, education, skills, project, custom section, long lines, links, hidden item, and CJK text. Hidden item tokens are intentionally excluded from expected recall and checked for leakage.",
|
||||
"",
|
||||
"| Variant | Format | Recall raw | Order raw | Duplicate raw | Grouping raw | Pages/paragraphs | Links | Numbering defs/paragraphs | Hidden leaks |",
|
||||
"| --- | --- | --- | --- | --- | --- | ---: | ---: | ---: | --- |",
|
||||
];
|
||||
for (const result of results) {
|
||||
for (const format of result.formats) {
|
||||
const m = format.metrics;
|
||||
lines.push(
|
||||
`| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | ${format.links.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`,
|
||||
);
|
||||
}
|
||||
}
|
||||
lines.push(
|
||||
"",
|
||||
"PDF font objects and raw link targets are recorded in adjacent JSON output.",
|
||||
"",
|
||||
"## Concrete loss/order evidence",
|
||||
"",
|
||||
);
|
||||
for (const result of results) {
|
||||
for (const format of result.formats) {
|
||||
const m = format.metrics;
|
||||
lines.push(
|
||||
`- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
lines.push("");
|
||||
return lines.join("\n");
|
||||
}
|
||||
|
||||
describe("current unchanged PDF and DOCX exports", () => {
|
||||
it("measures two-column and full-width synthetic corpus without mutating input", { timeout: 120_000 }, async () => {
|
||||
await mkdir(outputDirectory, { recursive: true });
|
||||
const results = [await measureVariant("two-column"), await measureVariant("full-width")];
|
||||
const report = {
|
||||
claimBoundary: "Local extraction metrics only; no vendor parsing accuracy claim.",
|
||||
tokenRules: "NFC, en-US case-folding, Unicode letter/number runs; punctuation separates tokens.",
|
||||
corpus: {
|
||||
variants: results.map((result) => result.variant),
|
||||
expectedDistinctTokens: createSyntheticCorpus("full-width")
|
||||
.tokens.flatMap((entry) => tokenize(entry.value))
|
||||
.filter((value, index, values) => values.indexOf(value) === index).length,
|
||||
},
|
||||
results,
|
||||
};
|
||||
await writeFile(join(outputDirectory, "ats-export-report.json"), `${JSON.stringify(report, null, 2)}\n`);
|
||||
await writeFile(join(outputDirectory, "ats-export-report.md"), reportMarkdown(results));
|
||||
|
||||
for (const result of results) {
|
||||
const pdf = result.formats.find((format) => format.format === "pdf");
|
||||
const docx = result.formats.find((format) => format.format === "docx");
|
||||
if (!pdf || !docx) throw new Error(`Missing measured format for ${result.variant}`);
|
||||
expect(pdf.metrics.recall.denominator).toBeGreaterThan(20);
|
||||
expect(docx.metrics.recall.denominator).toBe(pdf.metrics.recall.denominator);
|
||||
expect(pdf.hiddenLeaks).toEqual([]);
|
||||
expect(docx.hiddenLeaks).toEqual([]);
|
||||
expect(pdf.fontCount).toBeGreaterThan(0);
|
||||
expect(docx.numberingDefinitions).toBeGreaterThan(0);
|
||||
expect(docx.numberedParagraphs).toBe(0);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,154 @@
|
||||
import type { ExtractedDocument, PdfDocumentLike, RawExtraction } from "@reactive-resume/resume/ats-pdf";
|
||||
import { execFile } from "node:child_process";
|
||||
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { promisify } from "node:util";
|
||||
import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs";
|
||||
import { buildExtractedDocument, harvestPdfDocument } from "@reactive-resume/resume/ats-pdf";
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
export type PdfExtraction = {
|
||||
raw: RawExtraction;
|
||||
document: ExtractedDocument;
|
||||
paragraphs: readonly string[];
|
||||
links: readonly string[];
|
||||
};
|
||||
|
||||
export type DocxParagraph = {
|
||||
text: string;
|
||||
numbering: { numId: string; level: string; format: string; marker: string } | null;
|
||||
};
|
||||
|
||||
export type DocxExtraction = {
|
||||
paragraphs: readonly DocxParagraph[];
|
||||
links: readonly string[];
|
||||
numberingDefinitions: number;
|
||||
numberedParagraphs: number;
|
||||
};
|
||||
|
||||
const unescapeXml = (value: string): string =>
|
||||
value
|
||||
.replaceAll("<", "<")
|
||||
.replaceAll(">", ">")
|
||||
.replaceAll(""", '"')
|
||||
.replaceAll("'", "'")
|
||||
.replaceAll("&", "&");
|
||||
|
||||
const attribute = (attributes: string, name: string): string | null => {
|
||||
const match = attributes.match(new RegExp(`(?:^|\\s)(?:[A-Za-z][\\w-]*:)?${name}="([^"]*)"`));
|
||||
return match ? unescapeXml(match[1] ?? "") : null;
|
||||
};
|
||||
|
||||
const zipEntry = async (archivePath: string, entry: string): Promise<string> => {
|
||||
const { stdout } = await execFileAsync("unzip", ["-p", archivePath, entry]);
|
||||
return stdout;
|
||||
};
|
||||
|
||||
export async function extractPdf(bytes: Uint8Array): Promise<PdfExtraction> {
|
||||
const loadingTask = getDocument({ data: new Uint8Array(bytes), fontExtraProperties: true });
|
||||
try {
|
||||
const document = (await loadingTask.promise) as unknown as PdfDocumentLike;
|
||||
const raw = await harvestPdfDocument(document, {
|
||||
file: { name: "synthetic-resume.pdf", sizeBytes: bytes.byteLength, magicBytesOk: true },
|
||||
});
|
||||
const extracted = buildExtractedDocument(raw);
|
||||
return {
|
||||
raw,
|
||||
document: extracted,
|
||||
paragraphs: extracted.lines.map((line) => line.text),
|
||||
links: raw.links.flatMap((link) => (link.url ? [link.url] : [])),
|
||||
};
|
||||
} finally {
|
||||
await loadingTask.destroy();
|
||||
}
|
||||
}
|
||||
|
||||
function parseNumbering(numberingXml: string): Map<string, { format: string; marker: string }> {
|
||||
const formats = new Map<string, { format: string; marker: string }>();
|
||||
const abstractDefinitions = new Map<string, { format: string; marker: string }>();
|
||||
for (const abstract of numberingXml.matchAll(/<w:abstractNum\b([^>]*)>([\s\S]*?)<\/w:abstractNum>/g)) {
|
||||
const abstractId = attribute(abstract[1] ?? "", "abstractNumId");
|
||||
if (!abstractId) continue;
|
||||
for (const level of (abstract[2] ?? "").matchAll(/<w:lvl\b([^>]*)>([\s\S]*?)<\/w:lvl>/g)) {
|
||||
const levelId = attribute(level[1] ?? "", "ilvl") ?? "0";
|
||||
const format = attribute(level[2] ?? "", "val") ?? "unknown";
|
||||
const marker = attribute((level[2] ?? "").match(/<w:lvlText\b([^>]*)\/>/)?.[1] ?? "", "val") ?? "";
|
||||
abstractDefinitions.set(`${abstractId}:${levelId}`, { format, marker });
|
||||
}
|
||||
}
|
||||
for (const numbering of numberingXml.matchAll(/<w:num\b([^>]*)>([\s\S]*?)<\/w:num>/g)) {
|
||||
const numId = attribute(numbering[1] ?? "", "numId");
|
||||
const abstractId = attribute((numbering[2] ?? "").match(/<w:abstractNumId\b([^>]*)\/>/)?.[1] ?? "", "val");
|
||||
if (!numId || !abstractId) continue;
|
||||
for (const level of ["0", "1", "2", "3", "4", "5", "6", "7", "8"]) {
|
||||
const definition = abstractDefinitions.get(`${abstractId}:${level}`);
|
||||
if (definition) formats.set(`${numId}:${level}`, definition);
|
||||
}
|
||||
}
|
||||
return formats;
|
||||
}
|
||||
|
||||
function parseDocxParagraphs(documentXml: string, numbering: Map<string, { format: string; marker: string }>) {
|
||||
const paragraphs: DocxParagraph[] = [];
|
||||
for (const paragraph of documentXml.matchAll(/<w:p\b[^>]*>([\s\S]*?)<\/w:p>/g)) {
|
||||
const body = paragraph[1] ?? "";
|
||||
const text = [...body.matchAll(/<w:t\b[^>]*>([\s\S]*?)<\/w:t>/g)]
|
||||
.map((match) => unescapeXml(match[1] ?? ""))
|
||||
.join("");
|
||||
const numPr = body.match(/<w:numPr\b[^>]*>([\s\S]*?)<\/w:numPr>/)?.[1];
|
||||
const numId = numPr ? numPr.match(/<w:numId\b([^>]*)\/>/) : null;
|
||||
const level = numPr ? numPr.match(/<w:ilvl\b([^>]*)\/>/) : null;
|
||||
const numIdValue = numId ? attribute(numId[1] ?? "", "val") : null;
|
||||
const levelValue = level ? (attribute(level[1] ?? "", "val") ?? "0") : null;
|
||||
const definition = numIdValue && levelValue ? numbering.get(`${numIdValue}:${levelValue}`) : undefined;
|
||||
paragraphs.push({
|
||||
text,
|
||||
numbering:
|
||||
numIdValue && levelValue && definition
|
||||
? { numId: numIdValue, level: levelValue, format: definition.format, marker: definition.marker }
|
||||
: null,
|
||||
});
|
||||
}
|
||||
return paragraphs;
|
||||
}
|
||||
|
||||
function parseDocxLinks(documentXml: string, relationshipsXml: string): string[] {
|
||||
const relationships = new Map<string, string>();
|
||||
for (const relationship of relationshipsXml.matchAll(/<Relationship\b([^>]*)\/>/g)) {
|
||||
const id = attribute(relationship[1] ?? "", "Id");
|
||||
const target = attribute(relationship[1] ?? "", "Target");
|
||||
if (id && target) relationships.set(id, target);
|
||||
}
|
||||
const links: string[] = [];
|
||||
for (const hyperlink of documentXml.matchAll(/<w:hyperlink\b([^>]*)>/g)) {
|
||||
const id = attribute(hyperlink[1] ?? "", "id");
|
||||
const target = id ? relationships.get(id) : undefined;
|
||||
if (target) links.push(target);
|
||||
}
|
||||
return links;
|
||||
}
|
||||
|
||||
export async function extractDocx(bytes: Uint8Array): Promise<DocxExtraction> {
|
||||
const tempDirectory = await mkdtemp(join(tmpdir(), "reactive-resume-ats-"));
|
||||
const archivePath = join(tempDirectory, "resume.docx");
|
||||
try {
|
||||
await writeFile(archivePath, bytes);
|
||||
const [documentXml, numberingXml, relationshipsXml] = await Promise.all([
|
||||
zipEntry(archivePath, "word/document.xml"),
|
||||
zipEntry(archivePath, "word/numbering.xml"),
|
||||
zipEntry(archivePath, "word/_rels/document.xml.rels"),
|
||||
]);
|
||||
const numbering = parseNumbering(numberingXml);
|
||||
const paragraphs = parseDocxParagraphs(documentXml, numbering);
|
||||
return {
|
||||
paragraphs,
|
||||
links: parseDocxLinks(documentXml, relationshipsXml),
|
||||
numberingDefinitions: numbering.size,
|
||||
numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length,
|
||||
};
|
||||
} finally {
|
||||
await rm(tempDirectory, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,250 @@
|
||||
import type { ResumeData } from "@reactive-resume/schema/resume/data";
|
||||
import type { EvaluationCorpus, ExpectedToken } from "./metrics";
|
||||
import { sampleResumeData } from "@reactive-resume/schema/resume/sample";
|
||||
|
||||
export type ExportVariant = "two-column" | "full-width";
|
||||
|
||||
export type SyntheticCorpus = EvaluationCorpus & {
|
||||
data: ResumeData;
|
||||
hiddenTokens: readonly string[];
|
||||
links: readonly string[];
|
||||
};
|
||||
|
||||
const token = (value: string, group: string): ExpectedToken => ({ value, group });
|
||||
|
||||
const text = (html: string) => html.replace(/<[^>]+>/g, " ");
|
||||
|
||||
const SUMMARY_HTML =
|
||||
"<p>summary-signal-alpha leads longline-calibration with multilingual 東京大学 context and durable systems.</p>";
|
||||
const EXPERIENCE_ROLE_ONE_HTML =
|
||||
"<ul><li><p>role-bullet-alpha shipped resilient pipeline-observability under longline-pressure.</p></li><li><p>role-bullet-beta measured queue-latency and improved release-safety.</p></li></ul>";
|
||||
const EXPERIENCE_ROLE_TWO_HTML =
|
||||
"<p>role-bullet-gamma guided distributed-runtime migration for platform-reliability.</p>";
|
||||
const EDUCATION_HTML = "<p>education-signal-delta researched multilingual retrieval and evaluation.</p>";
|
||||
const PROJECT_HTML = "<p>project-signal-epsilon demonstrates export-fixture determinism.</p>";
|
||||
const CUSTOM_HTML = "<p>custom-signal-zeta preserves authored custom content and free-text dates.</p>";
|
||||
|
||||
const expectedTokens: readonly ExpectedToken[] = [
|
||||
...[
|
||||
"Mira Kova",
|
||||
"Principal Systems Architect",
|
||||
"mira.kova@example.com",
|
||||
"+49 30 555 0142",
|
||||
"Berlin 東京",
|
||||
"mirakova.dev",
|
||||
"orbit-field-omega",
|
||||
].map((value) => token(value, "header")),
|
||||
...[text(SUMMARY_HTML)].map((value) => token(value, "summary")),
|
||||
...[
|
||||
"Northstar Robotics",
|
||||
"2018-02 — Present",
|
||||
"Berlin",
|
||||
"Staff Platform Engineer",
|
||||
"2018-02 to 2020-12",
|
||||
text(EXPERIENCE_ROLE_ONE_HTML),
|
||||
"Principal Reliability Engineer",
|
||||
"2021 / Present",
|
||||
text(EXPERIENCE_ROLE_TWO_HTML),
|
||||
].map((value) => token(value, "experience")),
|
||||
...[
|
||||
"東京大学",
|
||||
"Master of Computer Science",
|
||||
"Distributed Systems",
|
||||
"3.98 GPA",
|
||||
"2014 — 2018 (long academic period)",
|
||||
"東京",
|
||||
text(EDUCATION_HTML),
|
||||
].map((value) => token(value, "education")),
|
||||
...["TypeScript", "skill-keyword-alpha", "Kubernetes", "skill-keyword-beta"].map((value) => token(value, "skills")),
|
||||
...["Export Observatory", "2022 to Winter 2024", text(PROJECT_HTML), "project.example/observatory"].map((value) =>
|
||||
token(value, "projects"),
|
||||
),
|
||||
...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")),
|
||||
];
|
||||
|
||||
export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta"] as const;
|
||||
|
||||
const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false });
|
||||
|
||||
export function createSyntheticCorpus(variant: ExportVariant): SyntheticCorpus {
|
||||
const data = structuredClone(sampleResumeData);
|
||||
data.picture.hidden = true;
|
||||
data.basics = {
|
||||
name: "Mira Kova",
|
||||
headline: "Principal Systems Architect",
|
||||
email: "mira.kova@example.com",
|
||||
phone: "+49 30 555 0142",
|
||||
location: "Berlin 東京",
|
||||
website: resolveWebsite("https://mirakova.dev", "mirakova.dev"),
|
||||
customFields: [
|
||||
{
|
||||
id: "synthetic-field-omega",
|
||||
icon: "",
|
||||
text: "orbit-field-omega",
|
||||
link: "https://orbit.example/omega",
|
||||
},
|
||||
],
|
||||
};
|
||||
data.summary = {
|
||||
...data.summary,
|
||||
title: "Summary",
|
||||
content: SUMMARY_HTML,
|
||||
};
|
||||
data.sections.profiles = {
|
||||
...data.sections.profiles,
|
||||
title: "Profiles",
|
||||
items: [
|
||||
{
|
||||
id: "synthetic-profile-omega",
|
||||
hidden: false,
|
||||
icon: "",
|
||||
iconColor: "",
|
||||
network: "OrbitNet",
|
||||
username: "orbit-profile-omega",
|
||||
website: resolveWebsite("https://orbit.example/profile", "orbit.example/profile"),
|
||||
},
|
||||
],
|
||||
};
|
||||
data.sections.experience = {
|
||||
...data.sections.experience,
|
||||
title: "Experience",
|
||||
items: [
|
||||
{
|
||||
id: "synthetic-experience-northstar",
|
||||
hidden: false,
|
||||
company: "Northstar Robotics",
|
||||
position: "",
|
||||
location: "Berlin",
|
||||
period: "2018-02 — Present",
|
||||
website: resolveWebsite("https://northstar.example/jobs", "northstar.example/jobs"),
|
||||
roles: [
|
||||
{
|
||||
id: "synthetic-role-staff",
|
||||
position: "Staff Platform Engineer",
|
||||
period: "2018-02 to 2020-12",
|
||||
description: EXPERIENCE_ROLE_ONE_HTML,
|
||||
},
|
||||
{
|
||||
id: "synthetic-role-principal",
|
||||
position: "Principal Reliability Engineer",
|
||||
period: "2021 / Present",
|
||||
description: EXPERIENCE_ROLE_TWO_HTML,
|
||||
},
|
||||
],
|
||||
description: "",
|
||||
},
|
||||
{
|
||||
id: "synthetic-hidden-experience",
|
||||
hidden: true,
|
||||
company: HIDDEN_TOKENS[0],
|
||||
position: "",
|
||||
location: "",
|
||||
period: "",
|
||||
website: resolveWebsite("https://hidden.example", "hidden.example"),
|
||||
roles: [],
|
||||
description: `<p>${HIDDEN_TOKENS[1]}</p>`,
|
||||
},
|
||||
],
|
||||
};
|
||||
data.sections.education = {
|
||||
...data.sections.education,
|
||||
title: "Education",
|
||||
items: [
|
||||
{
|
||||
id: "synthetic-education-tokyo",
|
||||
hidden: false,
|
||||
school: "東京大学",
|
||||
degree: "Master of Computer Science",
|
||||
area: "Distributed Systems",
|
||||
grade: "3.98 GPA",
|
||||
location: "東京",
|
||||
period: "2014 — 2018 (long academic period)",
|
||||
website: resolveWebsite("https://u-tokyo.example/program", "u-tokyo.example/program"),
|
||||
description: EDUCATION_HTML,
|
||||
},
|
||||
],
|
||||
};
|
||||
data.sections.skills = {
|
||||
...data.sections.skills,
|
||||
title: "Skills",
|
||||
items: [
|
||||
{
|
||||
id: "synthetic-skill-typescript",
|
||||
hidden: false,
|
||||
icon: "",
|
||||
iconColor: "",
|
||||
name: "TypeScript",
|
||||
proficiency: "Advanced",
|
||||
level: 4,
|
||||
keywords: ["skill-keyword-alpha"],
|
||||
},
|
||||
{
|
||||
id: "synthetic-skill-kubernetes",
|
||||
hidden: false,
|
||||
icon: "",
|
||||
iconColor: "",
|
||||
name: "Kubernetes",
|
||||
proficiency: "Expert",
|
||||
level: 5,
|
||||
keywords: ["skill-keyword-beta"],
|
||||
},
|
||||
],
|
||||
};
|
||||
data.sections.projects = {
|
||||
...data.sections.projects,
|
||||
title: "Projects",
|
||||
items: [
|
||||
{
|
||||
id: "synthetic-project-observatory",
|
||||
hidden: false,
|
||||
name: "Export Observatory",
|
||||
period: "2022 to Winter 2024",
|
||||
website: resolveWebsite("https://project.example/observatory", "project.example/observatory"),
|
||||
description: PROJECT_HTML,
|
||||
},
|
||||
],
|
||||
};
|
||||
data.customSections = [
|
||||
{
|
||||
id: "custom-ats-evidence",
|
||||
type: "summary",
|
||||
title: "Custom Evidence",
|
||||
icon: "",
|
||||
columns: 1,
|
||||
hidden: false,
|
||||
showHeading: true,
|
||||
keepTogether: false,
|
||||
startOnNewPage: false,
|
||||
items: [{ id: "synthetic-custom-zeta", hidden: false, content: CUSTOM_HTML }],
|
||||
},
|
||||
];
|
||||
|
||||
const mainSections = ["profiles", "summary", "experience", "education", "projects"];
|
||||
const sidebarSections = ["skills", "custom-ats-evidence"];
|
||||
const page = {
|
||||
fullWidth: variant === "full-width",
|
||||
main: variant === "full-width" ? [...mainSections, ...sidebarSections] : mainSections,
|
||||
sidebar: variant === "full-width" ? [] : sidebarSections,
|
||||
};
|
||||
data.metadata = {
|
||||
...data.metadata,
|
||||
template: variant === "full-width" ? "onyx" : "gengar",
|
||||
layout: { ...data.metadata.layout, pages: [page] },
|
||||
page: { ...data.metadata.page, locale: "en-US", hideIcons: true, hideSectionIcons: true },
|
||||
};
|
||||
|
||||
return {
|
||||
name: `ats-${variant}`,
|
||||
tokens: expectedTokens,
|
||||
data,
|
||||
hiddenTokens: HIDDEN_TOKENS,
|
||||
links: [
|
||||
"https://mirakova.dev",
|
||||
"https://orbit.example/omega",
|
||||
"https://orbit.example/profile",
|
||||
"https://northstar.example/jobs",
|
||||
"https://u-tokyo.example/program",
|
||||
"https://project.example/observatory",
|
||||
],
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,43 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { evaluateExport } from "./metrics";
|
||||
|
||||
const corpus = {
|
||||
name: "synthetic",
|
||||
tokens: [
|
||||
{ value: "Alpha", group: "experience" },
|
||||
{ value: "Bravo", group: "experience" },
|
||||
{ value: "Charlie", group: "education" },
|
||||
] as const,
|
||||
};
|
||||
|
||||
describe("evaluateExport", () => {
|
||||
it("counts distinct recall, order, duplicate, and grouping losses from raw tokens", () => {
|
||||
const result = evaluateExport(corpus, {
|
||||
paragraphs: ["Alpha Bravo Bravo", "Charlie"],
|
||||
links: [],
|
||||
});
|
||||
|
||||
expect(result.recall).toEqual({ numerator: 3, denominator: 3, value: 1 });
|
||||
expect(result.order).toEqual({ numerator: 2, denominator: 2, value: 1 });
|
||||
expect(result.duplicates).toEqual({ expected: 3, observed: 4, extra: 1 });
|
||||
expect(result.missingTokens).toEqual([]);
|
||||
expect(result.outOfOrderPairs).toEqual([]);
|
||||
expect(result.grouping).toEqual({ numerator: 1, denominator: 2, value: 0.5 });
|
||||
});
|
||||
|
||||
it("detects a dropped token and an inverted pair", () => {
|
||||
const result = evaluateExport(corpus, {
|
||||
paragraphs: ["Bravo", "Alpha"],
|
||||
links: [],
|
||||
});
|
||||
|
||||
expect(result.recall).toEqual({ numerator: 2, denominator: 3, value: 2 / 3 });
|
||||
expect(result.order).toEqual({ numerator: 0, denominator: 2, value: 0 });
|
||||
expect(result.duplicates).toEqual({ expected: 3, observed: 2, extra: 0 });
|
||||
expect(result.missingTokens).toEqual(["charlie"]);
|
||||
expect(result.outOfOrderPairs).toEqual([
|
||||
["alpha", "bravo"],
|
||||
["bravo", "charlie"],
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,118 @@
|
||||
export type ExpectedToken = {
|
||||
value: string;
|
||||
group: string;
|
||||
};
|
||||
|
||||
export type EvaluationCorpus = {
|
||||
name: string;
|
||||
tokens: readonly ExpectedToken[];
|
||||
};
|
||||
|
||||
export type ExtractedExport = {
|
||||
/** Paragraphs/lines in the extractor's returned order. */
|
||||
paragraphs: readonly string[];
|
||||
links: readonly string[];
|
||||
};
|
||||
|
||||
export type ExportMetrics = {
|
||||
recall: { numerator: number; denominator: number; value: number };
|
||||
order: { numerator: number; denominator: number; value: number };
|
||||
duplicates: { expected: number; observed: number; extra: number };
|
||||
grouping: { numerator: number; denominator: number; value: number };
|
||||
missingTokens: readonly string[];
|
||||
outOfOrderPairs: readonly (readonly [string, string])[];
|
||||
observedTokens: number;
|
||||
};
|
||||
|
||||
/**
|
||||
* Tokenization used by this evaluation only. It is deliberately transparent and locale-neutral:
|
||||
* Unicode letters/numbers stay intact, punctuation is a separator, and matching is case-folded.
|
||||
* This is a corpus metric, not a claim about any vendor parser.
|
||||
*/
|
||||
export function tokenize(value: string): string[] {
|
||||
return (
|
||||
value
|
||||
.normalize("NFC")
|
||||
.match(/[\p{L}\p{N}]+/gu)
|
||||
?.map((token) => token.toLocaleLowerCase("en-US")) ?? []
|
||||
);
|
||||
}
|
||||
|
||||
const metric = (numerator: number, denominator: number) => ({
|
||||
numerator,
|
||||
denominator,
|
||||
value: denominator === 0 ? 1 : numerator / denominator,
|
||||
});
|
||||
|
||||
/**
|
||||
* Computes raw extraction measurements from corpus tokens and extractor paragraphs.
|
||||
*
|
||||
* Recall uses distinct expected tokens. Duplicate accounting separately reports all matching
|
||||
* occurrences, so dropping a token cannot be hidden by duplicate output. Order and grouping use
|
||||
* the first occurrence of each distinct expected token, keeping those measures interpretable when
|
||||
* an export repeats a heading or bullet.
|
||||
*/
|
||||
export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExport): ExportMetrics {
|
||||
const expected = corpus.tokens.flatMap((entry) =>
|
||||
tokenize(entry.value).map((value) => ({ value, group: entry.group })),
|
||||
);
|
||||
const expectedByValue = new Map<string, { group: string; index: number }>();
|
||||
for (const [index, token] of expected.entries()) {
|
||||
if (!expectedByValue.has(token.value)) expectedByValue.set(token.value, { group: token.group, index });
|
||||
}
|
||||
|
||||
const observedByParagraph = extracted.paragraphs.map(tokenize);
|
||||
const observed = observedByParagraph.flat();
|
||||
const expectedValues = [...expectedByValue.keys()];
|
||||
const observedPositions = new Map<string, number>();
|
||||
for (const [index, token] of observed.entries()) {
|
||||
if (!observedPositions.has(token)) observedPositions.set(token, index);
|
||||
}
|
||||
|
||||
const recoveredDistinct = expectedValues.filter((value) => observedPositions.has(value)).length;
|
||||
const expectedTokenPairs = expectedValues.flatMap((value, index) => {
|
||||
const next = expectedValues[index + 1];
|
||||
return next ? ([[value, next]] as const) : [];
|
||||
});
|
||||
const outOfOrderPairs = expectedTokenPairs.filter(([left, right]) => {
|
||||
const leftPosition = observedPositions.get(left);
|
||||
const rightPosition = observedPositions.get(right);
|
||||
return leftPosition === undefined || rightPosition === undefined || leftPosition >= rightPosition;
|
||||
});
|
||||
|
||||
const observedParagraphPositions = new Map<string, number>();
|
||||
for (const [paragraphIndex, paragraphTokens] of observedByParagraph.entries()) {
|
||||
for (const token of paragraphTokens) {
|
||||
if (!observedParagraphPositions.has(token)) observedParagraphPositions.set(token, paragraphIndex);
|
||||
}
|
||||
}
|
||||
const groupedPairs = expectedTokenPairs.filter(([left, right]) => {
|
||||
const leftEntry = expectedByValue.get(left);
|
||||
const rightEntry = expectedByValue.get(right);
|
||||
const leftParagraph = observedParagraphPositions.get(left);
|
||||
const rightParagraph = observedParagraphPositions.get(right);
|
||||
return (
|
||||
leftEntry?.group === rightEntry?.group &&
|
||||
leftParagraph !== undefined &&
|
||||
rightParagraph !== undefined &&
|
||||
leftParagraph === rightParagraph
|
||||
);
|
||||
}).length;
|
||||
|
||||
const expectedValuesSet = new Set(expectedValues);
|
||||
const observedExpectedOccurrences = observed.filter((token) => expectedValuesSet.has(token)).length;
|
||||
|
||||
return {
|
||||
recall: metric(recoveredDistinct, expectedValues.length),
|
||||
order: metric(expectedTokenPairs.length - outOfOrderPairs.length, expectedTokenPairs.length),
|
||||
duplicates: {
|
||||
expected: expected.length,
|
||||
observed: observedExpectedOccurrences,
|
||||
extra: Math.max(0, observedExpectedOccurrences - expected.length),
|
||||
},
|
||||
grouping: metric(groupedPairs, expectedTokenPairs.length),
|
||||
missingTokens: expectedValues.filter((value) => !observedPositions.has(value)),
|
||||
outOfOrderPairs,
|
||||
observedTokens: observed.length,
|
||||
};
|
||||
}
|
||||
@@ -18,9 +18,12 @@
|
||||
"devDependencies": {
|
||||
"@lingui/format-po": "^6.6.0",
|
||||
"@reactive-resume/config": "workspace:*",
|
||||
"@reactive-resume/docx": "workspace:*",
|
||||
"@reactive-resume/env": "workspace:*",
|
||||
"@reactive-resume/pdf": "workspace:*",
|
||||
"@reactive-resume/resume": "workspace:*",
|
||||
"@types/pg": "^8.23.1",
|
||||
"pdfjs-dist": "6.3.289",
|
||||
"@typescript/native-preview": "7.0.0-dev.20260707.2",
|
||||
"drizzle-orm": "1.0.0-rc.4",
|
||||
"pg": "^8.23.0",
|
||||
|
||||
@@ -0,0 +1,14 @@
|
||||
import { fileURLToPath } from "node:url";
|
||||
// @boundaries-ignore root shared Vitest config
|
||||
import { createVitestProjectConfig } from "../vitest.shared.mts";
|
||||
|
||||
const config = createVitestProjectConfig({
|
||||
name: "@reactive-resume/tooling",
|
||||
dirname: fileURLToPath(new URL(".", import.meta.url)),
|
||||
});
|
||||
|
||||
export default {
|
||||
...config,
|
||||
test: { ...config.test, include: ["**/*.{test,spec}.?(c|m)[jt]s?(x)"] },
|
||||
oxc: { jsx: { runtime: "automatic" as const } },
|
||||
};
|
||||
Reference in New Issue
Block a user