fix(editor): preserve literal rich-text whitespace (#3472)

* fix(web): preserve imported rich-text tables

* fix(web): harden imported table preservation

* fix(web): fail closed on lossy table markup

* chore: remove plan 16 evidence reports

* fix(web): close imported table preservation gaps

* fix(editor): preserve literal rich-text whitespace

* chore: remove plan 19 evidence report

* fix: preserve literal whitespace through layout and editor transforms

* fix: preserve whitespace in bare table cells

* chore: keep orchestration evidence untracked
This commit is contained in:
Amruth Pillai
2026-09-05 19:57:50 -07:00
committed by GitHub
parent a6057abd79
commit ea97de5ec4
16 changed files with 1141 additions and 36 deletions
+11 -1
View File
@@ -23,6 +23,16 @@ interface InlineStyle {
type InlineChild = TextRun | ExternalHyperlink;
const preservesWhitespace = (node: Node) => {
let ancestor = node.parentElement;
while (ancestor) {
if (/^(P|H[1-6])$/.test(ancestor.tagName) && ancestor.getAttribute("data-resume-whitespace") === "preserve")
return true;
ancestor = ancestor.parentElement;
}
return false;
};
/** Module-level link color, set per htmlToParagraphs invocation. */
let currentLinkColor = "0563C1";
@@ -87,7 +97,7 @@ function collectInlineChildren(node: Node, style: InlineStyle): InlineChild[] {
for (const child of node.childNodes) {
if (child.nodeType === Node.TEXT_NODE) {
const text = child.textContent ?? "";
const text = (child.textContent ?? "").replace(/\t/g, preservesWhitespace(child) ? " " : "\t");
if (text) {
children.push(new TextRun({ text, ...style }));
}
@@ -0,0 +1,49 @@
// @vitest-environment happy-dom
import { describe, expect, it } from "vitest";
import { Document } from "docx";
import { htmlToParagraphs } from "./html-to-docx";
function paragraphXml(html: string) {
const paragraphs = htmlToParagraphs(html);
const file = new Document({ sections: [{ children: paragraphs }] });
return JSON.stringify(
paragraphs.map((paragraph) => paragraph.prepForXml({ file, viewWrapper: file.Document, stack: [] })),
);
}
describe("DOCX literal whitespace (#3397)", () => {
it.each(["p", "h2"])("emits exact marked %s spaces with XML preservation", (tag) => {
const xml = paragraphXml(`<${tag} data-resume-whitespace="preserve"> Lead middle end </${tag}>`);
expect(xml).toContain(" Lead middle end ");
expect(xml).toContain('"xml:space":"preserve"');
});
it("expands marked tabs to four ordinary spaces while leaving unmarked legacy tabs unchanged", () => {
const marked = paragraphXml('<p data-resume-whitespace="preserve">A\tB\t\tC</p>');
expect(marked).toContain("A B C");
expect(marked).not.toContain("\\t");
expect(marked).toContain('"xml:space":"preserve"');
const legacy = paragraphXml("<p>A\tB</p>");
expect(legacy).toContain("A\\tB");
});
it("preserves marked tabs through marks, line breaks, lists, quotes, and table cells", () => {
const html = [
'<p data-resume-whitespace="preserve"><strong> Bold</strong><br>\tNext</p>',
'<blockquote><p data-resume-whitespace="preserve">\tQuoted</p></blockquote>',
'<ul><li><p data-resume-whitespace="preserve">\tListed</p></li></ul>',
'<table><tbody><tr><td><p data-resume-whitespace="preserve">\tCell</p></td></tr></tbody></table>',
].join("");
const xml = paragraphXml(html);
expect(xml).toContain(" Bold");
expect(xml).toContain(" Next");
expect(xml).toContain(" Quoted");
expect(xml).toContain(" Listed");
expect(xml).toContain(" Cell");
expect(xml).not.toContain("\\t");
expect(xml).toContain('"w:b"');
expect(xml).toContain('"w:br"');
});
});
@@ -0,0 +1,186 @@
import { createRequire } from "node:module";
import { dirname, join } from "node:path";
import { describe, expect, it } from "vitest";
import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs";
import { act } from "react";
import { defaultResumeData } from "@reactive-resume/schema/resume/default";
import { createResumePdfFile } from "./server";
const require = createRequire(import.meta.url);
const standardFontDataUrl = `${join(dirname(require.resolve("pdfjs-dist/package.json")), "standard_fonts")}/`;
const preserve = 'data-resume-whitespace="preserve"';
type TextItem = {
text: string;
x: number;
y: number;
width: number;
};
async function renderItems(content: string, locale = "en-US", narrow = false): Promise<TextItem[]> {
const data = structuredClone(defaultResumeData);
data.picture.hidden = true;
data.basics.name = "";
data.metadata.template = narrow ? "chikorita" : "onyx";
data.metadata.page.locale = locale;
data.metadata.typography.body.fontFamily = "Helvetica";
data.metadata.typography.heading.fontFamily = "Helvetica";
data.metadata.layout.pages = [
narrow ? { fullWidth: false, main: [], sidebar: ["summary"] } : { fullWidth: true, main: ["summary"], sidebar: [] },
];
if (narrow) data.metadata.layout.sidebarWidth = 25;
data.summary.content = content;
let file: File | undefined;
await act(async () => {
file = await createResumePdfFile({ data, filename: "literal-whitespace.pdf" });
});
if (!file) throw new Error("PDF generation failed");
const task = getDocument({ data: new Uint8Array(await file.arrayBuffer()), standardFontDataUrl });
try {
const document = await task.promise;
const items: TextItem[] = [];
for (let pageIndex = 0; pageIndex < document.numPages; pageIndex++) {
const page = await document.getPage(pageIndex + 1);
const text = await page.getTextContent();
for (const item of text.items) {
if ("str" in item)
items.push({ text: item.str, x: item.transform[4], y: item.transform[5], width: item.width });
}
}
return items;
} finally {
await task.destroy();
}
}
function lineMetrics(items: TextItem[]) {
const anchor = items.find((item) => item.text.includes("LIT"));
if (!anchor) throw new Error(`Expected LIT anchor in: ${items.map((item) => item.text).join("|")}`);
const line = items.filter((item) => Math.abs(item.y - anchor.y) < 0.01);
const start = Math.min(...line.map((item) => item.x));
const end = Math.max(...line.map((item) => item.x + item.width));
return { start, width: end - start, text: line.map((item) => item.text).join("") };
}
async function line(content: string, locale = "en-US") {
return lineMetrics(await renderItems(content, locale));
}
describe("actual PDF literal whitespace (#3397)", () => {
it.each(["en-US", "he-IL", "ar-SA"])(
"advances first content for marked leading spaces and tabs in %s",
async (locale) => {
const rtl = locale !== "en-US";
const firstContent = async (prefix: string, marked = true) => {
const items = await renderItems(
`<p ${marked ? preserve : ""}>${prefix}<strong>LIT</strong> AB END</p>`,
locale,
);
const anchor = items.find((item) => item.text.includes("LIT"));
if (!anchor) throw new Error("Missing first-content anchor");
expect(
items
.filter((item) => Math.abs(item.y - anchor.y) < 0.01)
.map((item) => item.text)
.join("")
.replace(/\s/g, ""),
).toBe("LITABEND");
return rtl ? anchor.x + anchor.width : anchor.x;
};
const compact = await firstContent("");
const spaces = await firstContent(" ");
const tab = await firstContent("\t");
const sign = rtl ? -1 : 1;
// Helvetica body is 10pt, with an ordinary-space advance of 2.78pt.
expect(sign * (spaces - compact)).toBeCloseTo(5.56, 2);
expect(sign * (tab - compact)).toBeCloseTo(11.12, 2);
expect(await firstContent(" ", false)).toBeCloseTo(await firstContent("", false), 2);
},
);
it("keeps literal layout local to marked siblings in the same PDF", async () => {
const items = await renderItems(`<p ${preserve}>\tLIT AB END</p><p> LIT AB END</p><p>LIT AB END</p>`);
const anchors = items.filter((item) => item.text.includes("LIT"));
expect(anchors).toHaveLength(3);
const [marked, legacy, compact] = anchors;
if (!marked || !legacy || !compact) throw new Error("Missing mixed-block anchors");
expect(marked.x - legacy.x).toBeCloseTo(11.12, 2);
expect(legacy.x).toBeCloseTo(compact.x, 2);
for (const anchor of anchors) {
expect(
items
.filter((item) => Math.abs(item.y - anchor.y) < 0.01)
.map((item) => item.text)
.join("")
.replace(/\s/g, ""),
).toBe("LITABEND");
}
});
it("renders one tab as exactly four ordinary-space advances", async () => {
const compact = await line(`<p ${preserve}>LIT AB END</p>`);
const oneSpace = await line(`<p ${preserve}>LIT A B END</p>`);
const fourSpaces = await line(`<p ${preserve}>LIT A B END</p>`);
const tab = await line(`<p ${preserve}>LIT A\tB END</p>`);
const spaceAdvance = oneSpace.width - compact.width;
expect(spaceAdvance).toBeGreaterThan(0);
expect(fourSpaces.width - compact.width).toBeCloseTo(spaceAdvance * 4, 2);
expect(tab.width).toBeCloseTo(fourSpaces.width, 2);
});
it("renders two tabs as eight spaces independent of current x", async () => {
const twoTabs = await line(`<p ${preserve}>LIT A\t\tB END</p>`);
const eightSpaces = await line(`<p ${preserve}>LIT A B END</p>`);
const oneLetter = await line(`<p ${preserve}>LIT A\tB END</p>`);
const oneLetterCompact = await line(`<p ${preserve}>LIT AB END</p>`);
const threeLetters = await line(`<p ${preserve}>LIT ABC\tD END</p>`);
const threeLettersCompact = await line(`<p ${preserve}>LIT ABCD END</p>`);
expect(twoTabs.width).toBeCloseTo(eightSpaces.width, 2);
expect(oneLetter.width - oneLetterCompact.width).toBeCloseTo(threeLetters.width - threeLettersCompact.width, 2);
});
it.each(["en-US", "he-IL", "ar-SA"])("keeps marked tab geometry in %s", async (locale) => {
const tab = await line(`<p ${preserve}>LIT A\tB END</p>`, locale);
const spaces = await line(`<p ${preserve}>LIT A B END</p>`, locale);
expect(tab.width).toBeCloseTo(spaces.width, 2);
});
it("keeps unmarked collapse node-local and marked narrow content complete", async () => {
const unmarked = await line("<p>LIT A B END</p>");
const collapsed = await line("<p>LIT A B END</p>");
expect(unmarked).toEqual(collapsed);
const sample = "LIT START\tSome breakable words continue through narrow content without disappearing END";
const items = await renderItems(`<p ${preserve}>${sample}</p>`, "en-US", true);
const text = items
.map((item) => item.text)
.join("")
.replace(/\s/g, "");
expect(text.slice(text.indexOf("LIT"))).toBe(sample.replace(/\s/g, ""));
});
it.each([
[
"list",
'<ul><li><p data-resume-whitespace="preserve">LIT A\tB END</p></li></ul>',
'<ul><li><p data-resume-whitespace="preserve">LIT A B END</p></li></ul>',
],
[
"quote",
'<blockquote><p data-resume-whitespace="preserve">LIT A\tB END</p></blockquote>',
'<blockquote><p data-resume-whitespace="preserve">LIT A B END</p></blockquote>',
],
[
"table cell",
'<table><tbody><tr><td><p data-resume-whitespace="preserve">LIT A\tB END</p></td></tr></tbody></table>',
'<table><tbody><tr><td><p data-resume-whitespace="preserve">LIT A B END</p></td></tr></tbody></table>',
],
] as const)("preserves marked tab geometry in a %s", async (_name, tabHtml, spacesHtml) => {
const tab = await line(tabHtml);
const spaces = await line(spacesHtml);
expect(tab.width).toBeCloseTo(spaces.width, 2);
});
});
@@ -9,6 +9,29 @@ type PdfElement = ReactElement<{ children?: unknown; element?: { tag: string } }
const getPdfElementProps = (element: unknown) => (element as PdfElement).props;
describe("normalizeRichTextHtml", () => {
it("expands tabs only inside marked paragraphs and headings", () => {
expect(
normalizeRichTextHtml(
'<p data-resume-whitespace="preserve">A\tB</p><h2 data-resume-whitespace="preserve">\tC</h2><p>D\tE</p>',
),
).toBe(
'<p data-resume-whitespace="preserve">A B</p><h2 data-resume-whitespace="preserve"> C</h2><p>D\tE</p>',
);
});
it("retains marked paragraphs inside lists so preservation stays node-local", () => {
expect(normalizeRichTextHtml('<ul><li><p data-resume-whitespace="preserve"> Listed\ttext </p></li></ul>')).toBe(
'<ul><li><p data-resume-whitespace="preserve"> Listed text </p></li></ul>',
);
});
it("does not reinterpret marked RTL line breaks as pseudo-bullet lists", () => {
const html = '<p data-resume-whitespace="preserve"> - First<br> - Second</p>';
expect(normalizeRichTextHtml(html, { direction: "rtl" })).toBe(
'<p data-resume-whitespace="preserve">‏ - First<br> - Second</p>',
);
});
it("decodes opted-in soft hyphens in text without changing links or escaped literals", () => {
const html =
'<p title="&shy;">Soft&shy;ware &#173; &#xAD; &amp;shy; <a href="https://example.com/&shy;">link</a></p>';
@@ -104,6 +104,7 @@ const unwrapSingleParagraphListItems = (root: ReturnType<typeof parse>) => {
const child = meaningfulChildren[0];
if (!child || !isElement(child) || getTagName(child) !== "p") continue;
if (child.getAttribute("data-resume-whitespace") === "preserve") continue;
listItem.innerHTML = child.innerHTML;
}
@@ -127,6 +128,18 @@ const normalizeParagraphIndentation = (root: ReturnType<typeof parse>, direction
}
};
const expandPreservedTabs = (root: ReturnType<typeof parse>) => {
for (const element of root.querySelectorAll(
'p[data-resume-whitespace="preserve"],h1[data-resume-whitespace="preserve"],h2[data-resume-whitespace="preserve"],h3[data-resume-whitespace="preserve"],h4[data-resume-whitespace="preserve"],h5[data-resume-whitespace="preserve"],h6[data-resume-whitespace="preserve"]',
)) {
const visit = (node: Node): void => {
if (node.nodeType === NodeType.TEXT_NODE) node.rawText = node.rawText.replace(/\t/g, " ");
for (const child of node.childNodes) visit(child);
};
visit(element);
}
};
const isInlineNode = (node: Node): boolean => {
if (node.nodeType === NodeType.TEXT_NODE || node.nodeType === NodeType.COMMENT_NODE) return true;
if (node.nodeType !== NodeType.ELEMENT_NODE) return false;
@@ -165,9 +178,11 @@ const tryConvertPseudoBulletParagraph = (paragraphInnerHtml: string): string | n
export const convertPseudoBulletParagraphs = (html: string, direction: "ltr" | "rtl" = "ltr"): string =>
html.replace(/<p\b([^>]*)>([\s\S]*?)<\/p>/gi, (full, _attrs, inner) => {
const paragraph = parse(full).querySelector("p");
if (paragraph?.getAttribute("data-resume-whitespace") === "preserve") return full;
const converted = tryConvertPseudoBulletParagraph(inner);
if (!converted) return full;
const level = Number(parse(full).querySelector("p")?.getAttribute("data-indent"));
const level = Number(paragraph?.getAttribute("data-indent"));
if (!Number.isInteger(level) || level <= 0 || level > 8) return converted;
// Keep the original paragraph's offset around the entire generated list.
return converted.replace(
@@ -200,6 +215,7 @@ export const normalizeRichTextHtml = (
normalizeBoldBoundaryWhitespace(root);
normalizeMarkElements(root);
normalizeParagraphIndentation(root, direction);
expandPreservedTabs(root);
unwrapSingleParagraphListItems(root);
const flushInlineNodes = () => {
@@ -137,7 +137,11 @@ export const RichText = ({ children, semanticField }: RichTextProps) => {
const visible = isNodeVisible(nodeKey);
if (!visible) return null;
const text = (
<PdfText {...resolvedPdfTextProps(resolved)} style={composeStyles(style, resolved.style, safeTextStyle)}>
<PdfText
{...resolvedPdfTextProps(resolved)}
data-resume-whitespace={element.getAttribute("data-resume-whitespace")}
style={composeStyles(style, resolved.style, safeTextStyle)}
>
{textChildren}
</PdfText>
);
@@ -230,7 +234,11 @@ export const RichText = ({ children, semanticField }: RichTextProps) => {
style: props.style,
indent: Number(props.element.getAttribute("data-indent")),
semanticStyle: resolved.style,
textProps: { ...resolvedPdfTextProps(resolved), hyphenationCallback },
textProps: {
...resolvedPdfTextProps(resolved),
hyphenationCallback,
"data-resume-whitespace": props.element.getAttribute("data-resume-whitespace"),
},
rtl,
...(rtlTextWrapStyle ? { rtlTextWrapStyle } : {}),
...(rtl ? { applyRtlDirection: applyRtlDirectionRecursively } : {}),