mirror of
https://github.com/AmruthPillai/Reactive-Resume.git
synced 2026-10-03 10:13:47 +10:00
test(tooling): harden ATS export evaluation
This commit is contained in:
Generated
+3
@@ -1370,6 +1370,9 @@ importers:
|
||||
drizzle-orm:
|
||||
specifier: 1.0.0-rc.4
|
||||
version: 1.0.0-rc.4(@types/pg@8.23.1)(pg@8.23.0)(zod@4.5.4)
|
||||
jszip:
|
||||
specifier: 3.10.1
|
||||
version: 3.10.1
|
||||
pdfjs-dist:
|
||||
specifier: 6.3.289
|
||||
version: 6.3.289
|
||||
|
||||
@@ -3,7 +3,16 @@
|
||||
`evaluation.integration.test.ts` renders deterministic synthetic resume data through current
|
||||
`ResumeDocument` and `buildDocx`, then measures extraction with installed PDF.js and DOCX XML.
|
||||
`metrics.test.ts` locks raw distinct-token recall, order, duplicate, and semantic-grouping behavior,
|
||||
including deliberate drop/duplicate regressions.
|
||||
including deliberate drop/duplicate regressions. Grouping compares eligible same-group expected
|
||||
occurrences, preserving repeated values' authored field identity. Link metrics normalize expected
|
||||
and extracted targets and report dropped or changed targets; hidden-leak checks scan paragraph text
|
||||
and link-target channels, including hidden URLs.
|
||||
|
||||
Current DOCX output intentionally records its known missing `tel:` target as a measured loss; any
|
||||
additional dropped or changed target fails integration assertions.
|
||||
|
||||
`extract.test.ts` supplies a positive XML numbering fixture covering paragraph order, `numId`, level,
|
||||
format, marker, and link-target extraction, plus valid archives with optional entries omitted.
|
||||
|
||||
Run from repository root:
|
||||
|
||||
|
||||
@@ -34,8 +34,12 @@ type VariantResult = {
|
||||
formats: FormatResult[];
|
||||
};
|
||||
|
||||
function hiddenLeaks(paragraphs: readonly string[], hiddenTokens: readonly string[]): string[] {
|
||||
const observed = paragraphs.flatMap(tokenize);
|
||||
function hiddenLeaks(
|
||||
paragraphs: readonly string[],
|
||||
hiddenTokens: readonly string[],
|
||||
links: readonly string[] = [],
|
||||
): string[] {
|
||||
const observed = [...paragraphs, ...links].flatMap(tokenize);
|
||||
return hiddenTokens.filter((value) => {
|
||||
const expected = tokenize(value);
|
||||
return observed.some((_, index) => expected.every((token, offset) => observed[index + offset] === token));
|
||||
@@ -72,7 +76,7 @@ async function measureVariant(variant: ExportVariant): Promise<VariantResult> {
|
||||
paragraphs: pdf.paragraphs.length,
|
||||
pageCount: pdf.raw.pageCount,
|
||||
fontCount: pdf.raw.fonts.length,
|
||||
hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens),
|
||||
hiddenLeaks: hiddenLeaks(pdf.paragraphs, corpus.hiddenTokens, pdf.links),
|
||||
},
|
||||
{
|
||||
format: "docx",
|
||||
@@ -87,6 +91,7 @@ async function measureVariant(variant: ExportVariant): Promise<VariantResult> {
|
||||
hiddenLeaks: hiddenLeaks(
|
||||
docx.paragraphs.map((paragraph) => paragraph.text),
|
||||
corpus.hiddenTokens,
|
||||
docx.links,
|
||||
),
|
||||
},
|
||||
],
|
||||
@@ -103,7 +108,7 @@ function reportMarkdown(results: readonly VariantResult[]): string {
|
||||
"",
|
||||
"Synthetic extraction measurements only. These are not vendor parsing accuracy claims.",
|
||||
"",
|
||||
"Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses pairs expected in the same authored field group and recovered in the same extracted paragraph/line.",
|
||||
"Token rules: NFC normalization, case-folding with `en-US`, and Unicode letter/number runs; punctuation separates tokens. Recall numerator is distinct expected tokens recovered; duplicate counts report matching occurrences separately. Order uses adjacent distinct expected-token pairs; grouping uses same-group expected occurrence pairs and recovered in the same extracted paragraph/line. Link targets are normalized before expected/observed set comparison.",
|
||||
"",
|
||||
"Corpus: one deterministic resume fixture per layout variant, covering header/contact, two roles, free-text dates, education, skills, project, custom section, long lines, links, hidden item, and CJK text. Hidden item tokens are intentionally excluded from expected recall and checked for leakage.",
|
||||
"",
|
||||
@@ -114,7 +119,7 @@ function reportMarkdown(results: readonly VariantResult[]): string {
|
||||
for (const format of result.formats) {
|
||||
const m = format.metrics;
|
||||
lines.push(
|
||||
`| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | ${format.links.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`,
|
||||
`| ${result.variant} | ${format.format} | ${m.recall.numerator}/${m.recall.denominator} (${percentage(m.recall.value)}) | ${m.order.numerator}/${m.order.denominator} (${percentage(m.order.value)}) | expected ${m.duplicates.expected}, observed ${m.duplicates.observed}, extra ${m.duplicates.extra} | ${m.grouping.numerator}/${m.grouping.denominator} (${percentage(m.grouping.value)}) | ${format.pageCount ?? "—"}/${format.paragraphs} | expected ${m.links.expected.length}, observed ${m.links.observed.length}, dropped ${m.links.missing.length}, changed/extra ${m.links.unexpected.length} | ${format.numberingDefinitions ?? "—"}/${format.numberedParagraphs ?? "—"} | ${format.hiddenLeaks.length === 0 ? "none" : format.hiddenLeaks.join(", ")} |`,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -129,7 +134,7 @@ function reportMarkdown(results: readonly VariantResult[]): string {
|
||||
for (const format of result.formats) {
|
||||
const m = format.metrics;
|
||||
lines.push(
|
||||
`- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}.`,
|
||||
`- ${result.variant} ${format.format}: missing tokens = ${m.missingTokens.length === 0 ? "none" : m.missingTokens.join(", ")}; out-of-order adjacent pairs = ${m.outOfOrderPairs.length === 0 ? "none" : m.outOfOrderPairs.map(([left, right]) => `${left} → ${right}`).join(", ")}; dropped links = ${m.links.missing.length === 0 ? "none" : m.links.missing.join(", ")}; changed/extra links = ${m.links.unexpected.length === 0 ? "none" : m.links.unexpected.join(", ")}.`,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -138,6 +143,12 @@ function reportMarkdown(results: readonly VariantResult[]): string {
|
||||
}
|
||||
|
||||
describe("current unchanged PDF and DOCX exports", () => {
|
||||
it("detects hidden content emitted through link targets", () => {
|
||||
expect(hiddenLeaks(["Visible"], ["https://hidden.example"], ["https://hidden.example"])).toEqual([
|
||||
"https://hidden.example",
|
||||
]);
|
||||
});
|
||||
|
||||
it("measures two-column and full-width synthetic corpus without mutating input", { timeout: 120_000 }, async () => {
|
||||
await mkdir(outputDirectory, { recursive: true });
|
||||
const results = [await measureVariant("two-column"), await measureVariant("full-width")];
|
||||
@@ -161,6 +172,11 @@ describe("current unchanged PDF and DOCX exports", () => {
|
||||
if (!pdf || !docx) throw new Error(`Missing measured format for ${result.variant}`);
|
||||
expect(pdf.metrics.recall.denominator).toBeGreaterThan(20);
|
||||
expect(docx.metrics.recall.denominator).toBe(pdf.metrics.recall.denominator);
|
||||
expect(pdf.metrics.links.missing).toEqual([]);
|
||||
expect(pdf.metrics.links.unexpected).toEqual([]);
|
||||
// DOCX currently emits email but not telephone hyperlinks; retain this known loss in the measurement.
|
||||
expect(docx.metrics.links.missing).toEqual(["tel:+49 30 555 0142"]);
|
||||
expect(docx.metrics.links.unexpected).toEqual([]);
|
||||
expect(pdf.hiddenLeaks).toEqual([]);
|
||||
expect(docx.hiddenLeaks).toEqual([]);
|
||||
expect(pdf.fontCount).toBeGreaterThan(0);
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import JSZip from "jszip";
|
||||
import { extractDocx } from "./extract";
|
||||
|
||||
const numberedDocument = `<?xml version="1.0" encoding="UTF-8"?>
|
||||
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships">
|
||||
<w:body>
|
||||
<w:p><w:pPr><w:numPr><w:ilvl w:val="1"/><w:numId w:val="42"/></w:numPr></w:pPr><w:r><w:t>First item</w:t></w:r><w:hyperlink r:id="rId1"><w:r><w:t>link</w:t></w:r></w:hyperlink></w:p>
|
||||
<w:p><w:pPr><w:numPr><w:ilvl w:val="0"/><w:numId w:val="42"/></w:numPr></w:pPr><w:r><w:t>Second item</w:t></w:r></w:p>
|
||||
</w:body>
|
||||
</w:document>`;
|
||||
|
||||
const numberedDefinitions = `<?xml version="1.0" encoding="UTF-8"?>
|
||||
<w:numbering xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:abstractNum w:abstractNumId="7">
|
||||
<w:lvl w:ilvl="0"><w:numFmt w:val="decimal"/><w:lvlText w:val="%1."/></w:lvl>
|
||||
<w:lvl w:ilvl="1"><w:numFmt w:val="lowerLetter"/><w:lvlText w:val="%2)"/></w:lvl>
|
||||
</w:abstractNum>
|
||||
<w:num w:numId="42"><w:abstractNumId w:val="7"/></w:num>
|
||||
</w:numbering>`;
|
||||
|
||||
const relationships = `<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships"><Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="https://example.com/item" TargetMode="External"/></Relationships>`;
|
||||
|
||||
function createDocxZip(entries: Record<string, string>): Promise<Uint8Array> {
|
||||
const zip = new JSZip();
|
||||
for (const [name, content] of Object.entries(entries)) zip.file(name, content);
|
||||
return zip.generateAsync({ type: "uint8array" });
|
||||
}
|
||||
|
||||
describe("DOCX extraction", () => {
|
||||
it("extracts numbering identity and link targets in paragraph order", async () => {
|
||||
const bytes = await createDocxZip({
|
||||
"word/document.xml": numberedDocument,
|
||||
"word/numbering.xml": numberedDefinitions,
|
||||
"word/_rels/document.xml.rels": relationships,
|
||||
});
|
||||
|
||||
const result = await extractDocx(bytes);
|
||||
|
||||
expect(result.paragraphs).toEqual([
|
||||
{ text: "First itemlink", numbering: { numId: "42", level: "1", format: "lowerLetter", marker: "%2)" } },
|
||||
{ text: "Second item", numbering: { numId: "42", level: "0", format: "decimal", marker: "%1." } },
|
||||
]);
|
||||
expect(result.links).toEqual(["https://example.com/item"]);
|
||||
expect(result.numberedParagraphs).toBe(2);
|
||||
});
|
||||
|
||||
it("handles valid DOCX archives without optional XML entries", async () => {
|
||||
const bytes = await createDocxZip({ "word/document.xml": numberedDocument });
|
||||
|
||||
const result = await extractDocx(bytes);
|
||||
|
||||
expect(result.paragraphs.map((paragraph) => paragraph.numbering)).toEqual([null, null]);
|
||||
expect(result.links).toEqual([]);
|
||||
expect(result.numberingDefinitions).toBe(0);
|
||||
expect(result.numberedParagraphs).toBe(0);
|
||||
});
|
||||
});
|
||||
@@ -1,14 +1,8 @@
|
||||
import type { ExtractedDocument, PdfDocumentLike, RawExtraction } from "@reactive-resume/resume/ats-pdf";
|
||||
import { execFile } from "node:child_process";
|
||||
import { mkdtemp, rm, writeFile } from "node:fs/promises";
|
||||
import { tmpdir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { promisify } from "node:util";
|
||||
import JSZip from "jszip";
|
||||
import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs";
|
||||
import { buildExtractedDocument, harvestPdfDocument } from "@reactive-resume/resume/ats-pdf";
|
||||
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
export type PdfExtraction = {
|
||||
raw: RawExtraction;
|
||||
document: ExtractedDocument;
|
||||
@@ -41,9 +35,9 @@ const attribute = (attributes: string, name: string): string | null => {
|
||||
return match ? unescapeXml(match[1] ?? "") : null;
|
||||
};
|
||||
|
||||
const zipEntry = async (archivePath: string, entry: string): Promise<string> => {
|
||||
const { stdout } = await execFileAsync("unzip", ["-p", archivePath, entry]);
|
||||
return stdout;
|
||||
const zipEntry = (archive: JSZip, entry: string): Promise<string | null> => {
|
||||
const file = archive.file(entry);
|
||||
return Promise.resolve(file ? file.async("string") : null);
|
||||
};
|
||||
|
||||
export async function extractPdf(bytes: Uint8Array): Promise<PdfExtraction> {
|
||||
@@ -131,24 +125,19 @@ function parseDocxLinks(documentXml: string, relationshipsXml: string): string[]
|
||||
}
|
||||
|
||||
export async function extractDocx(bytes: Uint8Array): Promise<DocxExtraction> {
|
||||
const tempDirectory = await mkdtemp(join(tmpdir(), "reactive-resume-ats-"));
|
||||
const archivePath = join(tempDirectory, "resume.docx");
|
||||
try {
|
||||
await writeFile(archivePath, bytes);
|
||||
const [documentXml, numberingXml, relationshipsXml] = await Promise.all([
|
||||
zipEntry(archivePath, "word/document.xml"),
|
||||
zipEntry(archivePath, "word/numbering.xml"),
|
||||
zipEntry(archivePath, "word/_rels/document.xml.rels"),
|
||||
]);
|
||||
const numbering = parseNumbering(numberingXml);
|
||||
const paragraphs = parseDocxParagraphs(documentXml, numbering);
|
||||
return {
|
||||
paragraphs,
|
||||
links: parseDocxLinks(documentXml, relationshipsXml),
|
||||
numberingDefinitions: numbering.size,
|
||||
numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length,
|
||||
};
|
||||
} finally {
|
||||
await rm(tempDirectory, { recursive: true, force: true });
|
||||
}
|
||||
const archive = await JSZip.loadAsync(bytes);
|
||||
const [documentXml, numberingXml, relationshipsXml] = await Promise.all([
|
||||
zipEntry(archive, "word/document.xml"),
|
||||
zipEntry(archive, "word/numbering.xml"),
|
||||
zipEntry(archive, "word/_rels/document.xml.rels"),
|
||||
]);
|
||||
if (!documentXml) throw new Error("DOCX archive is missing word/document.xml");
|
||||
const numbering = parseNumbering(numberingXml ?? "");
|
||||
const paragraphs = parseDocxParagraphs(documentXml, numbering);
|
||||
return {
|
||||
paragraphs,
|
||||
links: parseDocxLinks(documentXml, relationshipsXml ?? ""),
|
||||
numberingDefinitions: numbering.size,
|
||||
numberedParagraphs: paragraphs.filter((paragraph) => paragraph.numbering !== null).length,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -34,7 +34,10 @@ const expectedTokens: readonly ExpectedToken[] = [
|
||||
"mirakova.dev",
|
||||
"orbit-field-omega",
|
||||
].map((value) => token(value, "header")),
|
||||
...["Profiles", "OrbitNet", "orbit-profile-omega"].map((value) => token(value, "profiles")),
|
||||
...["Summary"].map((value) => token(value, "summary")),
|
||||
...[text(SUMMARY_HTML)].map((value) => token(value, "summary")),
|
||||
...["Experience"].map((value) => token(value, "experience")),
|
||||
...[
|
||||
"Northstar Robotics",
|
||||
"2018-02 — Present",
|
||||
@@ -46,6 +49,7 @@ const expectedTokens: readonly ExpectedToken[] = [
|
||||
"2021 / Present",
|
||||
text(EXPERIENCE_ROLE_TWO_HTML),
|
||||
].map((value) => token(value, "experience")),
|
||||
...["Education"].map((value) => token(value, "education")),
|
||||
...[
|
||||
"東京大学",
|
||||
"Master of Computer Science",
|
||||
@@ -55,14 +59,16 @@ const expectedTokens: readonly ExpectedToken[] = [
|
||||
"東京",
|
||||
text(EDUCATION_HTML),
|
||||
].map((value) => token(value, "education")),
|
||||
...["TypeScript", "skill-keyword-alpha", "Kubernetes", "skill-keyword-beta"].map((value) => token(value, "skills")),
|
||||
...["Skills", "TypeScript", "Advanced", "skill-keyword-alpha", "Kubernetes", "Expert", "skill-keyword-beta"].map(
|
||||
(value) => token(value, "skills"),
|
||||
),
|
||||
...["Export Observatory", "2022 to Winter 2024", text(PROJECT_HTML), "project.example/observatory"].map((value) =>
|
||||
token(value, "projects"),
|
||||
),
|
||||
...["Custom Evidence", text(CUSTOM_HTML)].map((value) => token(value, "custom")),
|
||||
];
|
||||
|
||||
export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta"] as const;
|
||||
export const HIDDEN_TOKENS = ["Hidden Confidential", "hidden-signal-theta", "https://hidden.example"] as const;
|
||||
|
||||
const resolveWebsite = (url: string, label: string) => ({ url, label, inlineLink: false });
|
||||
|
||||
@@ -240,6 +246,8 @@ export function createSyntheticCorpus(variant: ExportVariant): SyntheticCorpus {
|
||||
hiddenTokens: HIDDEN_TOKENS,
|
||||
links: [
|
||||
"https://mirakova.dev",
|
||||
"mailto:mira.kova@example.com",
|
||||
"tel:+49 30 555 0142",
|
||||
"https://orbit.example/omega",
|
||||
"https://orbit.example/profile",
|
||||
"https://northstar.example/jobs",
|
||||
|
||||
@@ -22,7 +22,42 @@ describe("evaluateExport", () => {
|
||||
expect(result.duplicates).toEqual({ expected: 3, observed: 4, extra: 1 });
|
||||
expect(result.missingTokens).toEqual([]);
|
||||
expect(result.outOfOrderPairs).toEqual([]);
|
||||
expect(result.grouping).toEqual({ numerator: 1, denominator: 2, value: 0.5 });
|
||||
expect(result.grouping).toEqual({ numerator: 1, denominator: 1, value: 1 });
|
||||
});
|
||||
|
||||
it("keeps repeated expected values tied to their authored groups", () => {
|
||||
const repeatedCorpus = {
|
||||
name: "repeated",
|
||||
tokens: [
|
||||
{ value: "Alpha Bravo", group: "first" },
|
||||
{ value: "Alpha Charlie", group: "second" },
|
||||
],
|
||||
} as const;
|
||||
|
||||
const result = evaluateExport(repeatedCorpus, {
|
||||
paragraphs: ["Alpha Bravo", "Alpha Charlie"],
|
||||
links: [],
|
||||
});
|
||||
|
||||
expect(result.grouping).toEqual({ numerator: 2, denominator: 2, value: 1 });
|
||||
});
|
||||
|
||||
it("reports dropped and changed link targets", () => {
|
||||
const result = evaluateExport(
|
||||
{
|
||||
name: "links",
|
||||
tokens: [],
|
||||
links: ["HTTPS://Example.com/profile/", "https://example.com/jobs"],
|
||||
},
|
||||
{ paragraphs: [], links: ["https://example.com/profile", "https://changed.example/jobs"] },
|
||||
);
|
||||
|
||||
expect(result.links).toEqual({
|
||||
expected: ["https://example.com/profile", "https://example.com/jobs"],
|
||||
observed: ["https://example.com/profile", "https://changed.example/jobs"],
|
||||
missing: ["https://example.com/jobs"],
|
||||
unexpected: ["https://changed.example/jobs"],
|
||||
});
|
||||
});
|
||||
|
||||
it("detects a dropped token and an inverted pair", () => {
|
||||
|
||||
@@ -6,6 +6,7 @@ export type ExpectedToken = {
|
||||
export type EvaluationCorpus = {
|
||||
name: string;
|
||||
tokens: readonly ExpectedToken[];
|
||||
links?: readonly string[];
|
||||
};
|
||||
|
||||
export type ExtractedExport = {
|
||||
@@ -19,6 +20,12 @@ export type ExportMetrics = {
|
||||
order: { numerator: number; denominator: number; value: number };
|
||||
duplicates: { expected: number; observed: number; extra: number };
|
||||
grouping: { numerator: number; denominator: number; value: number };
|
||||
links: {
|
||||
expected: readonly string[];
|
||||
observed: readonly string[];
|
||||
missing: readonly string[];
|
||||
unexpected: readonly string[];
|
||||
};
|
||||
missingTokens: readonly string[];
|
||||
outOfOrderPairs: readonly (readonly [string, string])[];
|
||||
observedTokens: number;
|
||||
@@ -44,29 +51,59 @@ const metric = (numerator: number, denominator: number) => ({
|
||||
value: denominator === 0 ? 1 : numerator / denominator,
|
||||
});
|
||||
|
||||
function normalizeLinkTarget(target: string): string {
|
||||
const trimmed = target.trim();
|
||||
try {
|
||||
const url = new URL(trimmed);
|
||||
url.protocol = url.protocol.toLowerCase();
|
||||
url.hostname = url.hostname.toLowerCase();
|
||||
if (url.pathname.length > 1) url.pathname = url.pathname.replace(/\/+$/, "");
|
||||
return url.toString();
|
||||
} catch {
|
||||
return trimmed;
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeLinks(links: readonly string[]): string[] {
|
||||
return [...new Set(links.map(normalizeLinkTarget))];
|
||||
}
|
||||
|
||||
/**
|
||||
* Computes raw extraction measurements from corpus tokens and extractor paragraphs.
|
||||
*
|
||||
* Recall uses distinct expected tokens. Duplicate accounting separately reports all matching
|
||||
* occurrences, so dropping a token cannot be hidden by duplicate output. Order and grouping use
|
||||
* the first occurrence of each distinct expected token, keeping those measures interpretable when
|
||||
* an export repeats a heading or bullet.
|
||||
* the first occurrence of each distinct expected token, keeping recall/order measures interpretable
|
||||
* when an export repeats a heading or bullet. Grouping retains every expected token occurrence so
|
||||
* repeated values keep their authored field group and occurrence identity.
|
||||
*/
|
||||
export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExport): ExportMetrics {
|
||||
const expected = corpus.tokens.flatMap((entry) =>
|
||||
tokenize(entry.value).map((value) => ({ value, group: entry.group })),
|
||||
);
|
||||
const expectedByValue = new Map<string, { group: string; index: number }>();
|
||||
for (const [index, token] of expected.entries()) {
|
||||
if (!expectedByValue.has(token.value)) expectedByValue.set(token.value, { group: token.group, index });
|
||||
const expectedOccurrenceOrdinals: number[] = [];
|
||||
const expectedValues: string[] = [];
|
||||
const expectedValuesSet = new Set<string>();
|
||||
const expectedCounts = new Map<string, number>();
|
||||
for (const token of expected) {
|
||||
const ordinal = expectedCounts.get(token.value) ?? 0;
|
||||
expectedOccurrenceOrdinals.push(ordinal);
|
||||
expectedCounts.set(token.value, ordinal + 1);
|
||||
if (!expectedValuesSet.has(token.value)) {
|
||||
expectedValuesSet.add(token.value);
|
||||
expectedValues.push(token.value);
|
||||
}
|
||||
}
|
||||
|
||||
const observedByParagraph = extracted.paragraphs.map(tokenize);
|
||||
const observed = observedByParagraph.flat();
|
||||
const expectedValues = [...expectedByValue.keys()];
|
||||
const observedPositions = new Map<string, number>();
|
||||
const observedPositionsByValue = new Map<string, number[]>();
|
||||
for (const [index, token] of observed.entries()) {
|
||||
if (!observedPositions.has(token)) observedPositions.set(token, index);
|
||||
const positions = observedPositionsByValue.get(token) ?? [];
|
||||
positions.push(index);
|
||||
observedPositionsByValue.set(token, positions);
|
||||
}
|
||||
|
||||
const recoveredDistinct = expectedValues.filter((value) => observedPositions.has(value)).length;
|
||||
@@ -80,27 +117,39 @@ export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExp
|
||||
return leftPosition === undefined || rightPosition === undefined || leftPosition >= rightPosition;
|
||||
});
|
||||
|
||||
const observedParagraphPositions = new Map<string, number>();
|
||||
const observedParagraphPositionsByValue = new Map<string, number[]>();
|
||||
for (const [paragraphIndex, paragraphTokens] of observedByParagraph.entries()) {
|
||||
for (const token of paragraphTokens) {
|
||||
if (!observedParagraphPositions.has(token)) observedParagraphPositions.set(token, paragraphIndex);
|
||||
const positions = observedParagraphPositionsByValue.get(token) ?? [];
|
||||
positions.push(paragraphIndex);
|
||||
observedParagraphPositionsByValue.set(token, positions);
|
||||
}
|
||||
}
|
||||
const groupedPairs = expectedTokenPairs.filter(([left, right]) => {
|
||||
const leftEntry = expectedByValue.get(left);
|
||||
const rightEntry = expectedByValue.get(right);
|
||||
const leftParagraph = observedParagraphPositions.get(left);
|
||||
const rightParagraph = observedParagraphPositions.get(right);
|
||||
return (
|
||||
leftEntry?.group === rightEntry?.group &&
|
||||
leftParagraph !== undefined &&
|
||||
rightParagraph !== undefined &&
|
||||
leftParagraph === rightParagraph
|
||||
);
|
||||
const eligibleExpectedPairs = expected.flatMap((token, index) => {
|
||||
const next = expected[index + 1];
|
||||
return next && token.group === next.group ? [[index, index + 1] as const] : [];
|
||||
});
|
||||
const groupedPairs = eligibleExpectedPairs.filter(([leftIndex, rightIndex]) => {
|
||||
const left = expected[leftIndex];
|
||||
const right = expected[rightIndex];
|
||||
if (!left || !right) return false;
|
||||
const leftPosition = observedPositionsByValue.get(left.value)?.[expectedOccurrenceOrdinals[leftIndex] ?? 0];
|
||||
const rightPosition = observedPositionsByValue.get(right.value)?.[expectedOccurrenceOrdinals[rightIndex] ?? 0];
|
||||
const leftParagraph = observedParagraphPositionsByValue.get(left.value)?.[
|
||||
expectedOccurrenceOrdinals[leftIndex] ?? 0
|
||||
];
|
||||
const rightParagraph = observedParagraphPositionsByValue.get(right.value)?.[
|
||||
expectedOccurrenceOrdinals[rightIndex] ?? 0
|
||||
];
|
||||
if (leftPosition === undefined || rightPosition === undefined) return false;
|
||||
return leftParagraph !== undefined && rightParagraph !== undefined && leftParagraph === rightParagraph;
|
||||
}).length;
|
||||
|
||||
const expectedValuesSet = new Set(expectedValues);
|
||||
const observedExpectedOccurrences = observed.filter((token) => expectedValuesSet.has(token)).length;
|
||||
const expectedLinks = normalizeLinks(corpus.links ?? []);
|
||||
const observedLinks = normalizeLinks(extracted.links);
|
||||
const observedLinksSet = new Set(observedLinks);
|
||||
const expectedLinksSet = new Set(expectedLinks);
|
||||
|
||||
return {
|
||||
recall: metric(recoveredDistinct, expectedValues.length),
|
||||
@@ -110,7 +159,13 @@ export function evaluateExport(corpus: EvaluationCorpus, extracted: ExtractedExp
|
||||
observed: observedExpectedOccurrences,
|
||||
extra: Math.max(0, observedExpectedOccurrences - expected.length),
|
||||
},
|
||||
grouping: metric(groupedPairs, expectedTokenPairs.length),
|
||||
grouping: metric(groupedPairs, eligibleExpectedPairs.length),
|
||||
links: {
|
||||
expected: expectedLinks,
|
||||
observed: observedLinks,
|
||||
missing: expectedLinks.filter((link) => !observedLinksSet.has(link)),
|
||||
unexpected: observedLinks.filter((link) => !expectedLinksSet.has(link)),
|
||||
},
|
||||
missingTokens: expectedValues.filter((value) => !observedPositions.has(value)),
|
||||
outOfOrderPairs,
|
||||
observedTokens: observed.length,
|
||||
|
||||
@@ -24,6 +24,7 @@
|
||||
"@reactive-resume/resume": "workspace:*",
|
||||
"@types/pg": "^8.23.1",
|
||||
"pdfjs-dist": "6.3.289",
|
||||
"jszip": "3.10.1",
|
||||
"@typescript/native-preview": "7.0.0-dev.20260707.2",
|
||||
"drizzle-orm": "1.0.0-rc.4",
|
||||
"pg": "^8.23.0",
|
||||
|
||||
Reference in New Issue
Block a user