mirror of
https://github.com/AmruthPillai/Reactive-Resume.git
synced 2026-10-03 10:13:47 +10:00
fix(editor): preserve literal rich-text whitespace (#3472)
* fix(web): preserve imported rich-text tables * fix(web): harden imported table preservation * fix(web): fail closed on lossy table markup * chore: remove plan 16 evidence reports * fix(web): close imported table preservation gaps * fix(editor): preserve literal rich-text whitespace * chore: remove plan 19 evidence report * fix: preserve literal whitespace through layout and editor transforms * fix: preserve whitespace in bare table cells * chore: keep orchestration evidence untracked
This commit is contained in:
@@ -23,6 +23,16 @@ interface InlineStyle {
|
||||
|
||||
type InlineChild = TextRun | ExternalHyperlink;
|
||||
|
||||
const preservesWhitespace = (node: Node) => {
|
||||
let ancestor = node.parentElement;
|
||||
while (ancestor) {
|
||||
if (/^(P|H[1-6])$/.test(ancestor.tagName) && ancestor.getAttribute("data-resume-whitespace") === "preserve")
|
||||
return true;
|
||||
ancestor = ancestor.parentElement;
|
||||
}
|
||||
return false;
|
||||
};
|
||||
|
||||
/** Module-level link color, set per htmlToParagraphs invocation. */
|
||||
let currentLinkColor = "0563C1";
|
||||
|
||||
@@ -87,7 +97,7 @@ function collectInlineChildren(node: Node, style: InlineStyle): InlineChild[] {
|
||||
|
||||
for (const child of node.childNodes) {
|
||||
if (child.nodeType === Node.TEXT_NODE) {
|
||||
const text = child.textContent ?? "";
|
||||
const text = (child.textContent ?? "").replace(/\t/g, preservesWhitespace(child) ? " " : "\t");
|
||||
if (text) {
|
||||
children.push(new TextRun({ text, ...style }));
|
||||
}
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
// @vitest-environment happy-dom
|
||||
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { Document } from "docx";
|
||||
import { htmlToParagraphs } from "./html-to-docx";
|
||||
|
||||
function paragraphXml(html: string) {
|
||||
const paragraphs = htmlToParagraphs(html);
|
||||
const file = new Document({ sections: [{ children: paragraphs }] });
|
||||
return JSON.stringify(
|
||||
paragraphs.map((paragraph) => paragraph.prepForXml({ file, viewWrapper: file.Document, stack: [] })),
|
||||
);
|
||||
}
|
||||
|
||||
describe("DOCX literal whitespace (#3397)", () => {
|
||||
it.each(["p", "h2"])("emits exact marked %s spaces with XML preservation", (tag) => {
|
||||
const xml = paragraphXml(`<${tag} data-resume-whitespace="preserve"> Lead middle end </${tag}>`);
|
||||
expect(xml).toContain(" Lead middle end ");
|
||||
expect(xml).toContain('"xml:space":"preserve"');
|
||||
});
|
||||
|
||||
it("expands marked tabs to four ordinary spaces while leaving unmarked legacy tabs unchanged", () => {
|
||||
const marked = paragraphXml('<p data-resume-whitespace="preserve">A\tB\t\tC</p>');
|
||||
expect(marked).toContain("A B C");
|
||||
expect(marked).not.toContain("\\t");
|
||||
expect(marked).toContain('"xml:space":"preserve"');
|
||||
|
||||
const legacy = paragraphXml("<p>A\tB</p>");
|
||||
expect(legacy).toContain("A\\tB");
|
||||
});
|
||||
|
||||
it("preserves marked tabs through marks, line breaks, lists, quotes, and table cells", () => {
|
||||
const html = [
|
||||
'<p data-resume-whitespace="preserve"><strong> Bold</strong><br>\tNext</p>',
|
||||
'<blockquote><p data-resume-whitespace="preserve">\tQuoted</p></blockquote>',
|
||||
'<ul><li><p data-resume-whitespace="preserve">\tListed</p></li></ul>',
|
||||
'<table><tbody><tr><td><p data-resume-whitespace="preserve">\tCell</p></td></tr></tbody></table>',
|
||||
].join("");
|
||||
const xml = paragraphXml(html);
|
||||
expect(xml).toContain(" Bold");
|
||||
expect(xml).toContain(" Next");
|
||||
expect(xml).toContain(" Quoted");
|
||||
expect(xml).toContain(" Listed");
|
||||
expect(xml).toContain(" Cell");
|
||||
expect(xml).not.toContain("\\t");
|
||||
expect(xml).toContain('"w:b"');
|
||||
expect(xml).toContain('"w:br"');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,186 @@
|
||||
import { createRequire } from "node:module";
|
||||
import { dirname, join } from "node:path";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs";
|
||||
import { act } from "react";
|
||||
import { defaultResumeData } from "@reactive-resume/schema/resume/default";
|
||||
import { createResumePdfFile } from "./server";
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const standardFontDataUrl = `${join(dirname(require.resolve("pdfjs-dist/package.json")), "standard_fonts")}/`;
|
||||
const preserve = 'data-resume-whitespace="preserve"';
|
||||
|
||||
type TextItem = {
|
||||
text: string;
|
||||
x: number;
|
||||
y: number;
|
||||
width: number;
|
||||
};
|
||||
|
||||
async function renderItems(content: string, locale = "en-US", narrow = false): Promise<TextItem[]> {
|
||||
const data = structuredClone(defaultResumeData);
|
||||
data.picture.hidden = true;
|
||||
data.basics.name = "";
|
||||
data.metadata.template = narrow ? "chikorita" : "onyx";
|
||||
data.metadata.page.locale = locale;
|
||||
data.metadata.typography.body.fontFamily = "Helvetica";
|
||||
data.metadata.typography.heading.fontFamily = "Helvetica";
|
||||
data.metadata.layout.pages = [
|
||||
narrow ? { fullWidth: false, main: [], sidebar: ["summary"] } : { fullWidth: true, main: ["summary"], sidebar: [] },
|
||||
];
|
||||
if (narrow) data.metadata.layout.sidebarWidth = 25;
|
||||
data.summary.content = content;
|
||||
|
||||
let file: File | undefined;
|
||||
await act(async () => {
|
||||
file = await createResumePdfFile({ data, filename: "literal-whitespace.pdf" });
|
||||
});
|
||||
if (!file) throw new Error("PDF generation failed");
|
||||
const task = getDocument({ data: new Uint8Array(await file.arrayBuffer()), standardFontDataUrl });
|
||||
try {
|
||||
const document = await task.promise;
|
||||
const items: TextItem[] = [];
|
||||
for (let pageIndex = 0; pageIndex < document.numPages; pageIndex++) {
|
||||
const page = await document.getPage(pageIndex + 1);
|
||||
const text = await page.getTextContent();
|
||||
for (const item of text.items) {
|
||||
if ("str" in item)
|
||||
items.push({ text: item.str, x: item.transform[4], y: item.transform[5], width: item.width });
|
||||
}
|
||||
}
|
||||
return items;
|
||||
} finally {
|
||||
await task.destroy();
|
||||
}
|
||||
}
|
||||
|
||||
function lineMetrics(items: TextItem[]) {
|
||||
const anchor = items.find((item) => item.text.includes("LIT"));
|
||||
if (!anchor) throw new Error(`Expected LIT anchor in: ${items.map((item) => item.text).join("|")}`);
|
||||
const line = items.filter((item) => Math.abs(item.y - anchor.y) < 0.01);
|
||||
const start = Math.min(...line.map((item) => item.x));
|
||||
const end = Math.max(...line.map((item) => item.x + item.width));
|
||||
return { start, width: end - start, text: line.map((item) => item.text).join("") };
|
||||
}
|
||||
|
||||
async function line(content: string, locale = "en-US") {
|
||||
return lineMetrics(await renderItems(content, locale));
|
||||
}
|
||||
|
||||
describe("actual PDF literal whitespace (#3397)", () => {
|
||||
it.each(["en-US", "he-IL", "ar-SA"])(
|
||||
"advances first content for marked leading spaces and tabs in %s",
|
||||
async (locale) => {
|
||||
const rtl = locale !== "en-US";
|
||||
const firstContent = async (prefix: string, marked = true) => {
|
||||
const items = await renderItems(
|
||||
`<p ${marked ? preserve : ""}>${prefix}<strong>LIT</strong> AB END</p>`,
|
||||
locale,
|
||||
);
|
||||
const anchor = items.find((item) => item.text.includes("LIT"));
|
||||
if (!anchor) throw new Error("Missing first-content anchor");
|
||||
expect(
|
||||
items
|
||||
.filter((item) => Math.abs(item.y - anchor.y) < 0.01)
|
||||
.map((item) => item.text)
|
||||
.join("")
|
||||
.replace(/\s/g, ""),
|
||||
).toBe("LITABEND");
|
||||
return rtl ? anchor.x + anchor.width : anchor.x;
|
||||
};
|
||||
const compact = await firstContent("");
|
||||
const spaces = await firstContent(" ");
|
||||
const tab = await firstContent("\t");
|
||||
const sign = rtl ? -1 : 1;
|
||||
// Helvetica body is 10pt, with an ordinary-space advance of 2.78pt.
|
||||
expect(sign * (spaces - compact)).toBeCloseTo(5.56, 2);
|
||||
expect(sign * (tab - compact)).toBeCloseTo(11.12, 2);
|
||||
expect(await firstContent(" ", false)).toBeCloseTo(await firstContent("", false), 2);
|
||||
},
|
||||
);
|
||||
|
||||
it("keeps literal layout local to marked siblings in the same PDF", async () => {
|
||||
const items = await renderItems(`<p ${preserve}>\tLIT AB END</p><p> LIT AB END</p><p>LIT AB END</p>`);
|
||||
const anchors = items.filter((item) => item.text.includes("LIT"));
|
||||
expect(anchors).toHaveLength(3);
|
||||
const [marked, legacy, compact] = anchors;
|
||||
if (!marked || !legacy || !compact) throw new Error("Missing mixed-block anchors");
|
||||
expect(marked.x - legacy.x).toBeCloseTo(11.12, 2);
|
||||
expect(legacy.x).toBeCloseTo(compact.x, 2);
|
||||
for (const anchor of anchors) {
|
||||
expect(
|
||||
items
|
||||
.filter((item) => Math.abs(item.y - anchor.y) < 0.01)
|
||||
.map((item) => item.text)
|
||||
.join("")
|
||||
.replace(/\s/g, ""),
|
||||
).toBe("LITABEND");
|
||||
}
|
||||
});
|
||||
|
||||
it("renders one tab as exactly four ordinary-space advances", async () => {
|
||||
const compact = await line(`<p ${preserve}>LIT AB END</p>`);
|
||||
const oneSpace = await line(`<p ${preserve}>LIT A B END</p>`);
|
||||
const fourSpaces = await line(`<p ${preserve}>LIT A B END</p>`);
|
||||
const tab = await line(`<p ${preserve}>LIT A\tB END</p>`);
|
||||
const spaceAdvance = oneSpace.width - compact.width;
|
||||
|
||||
expect(spaceAdvance).toBeGreaterThan(0);
|
||||
expect(fourSpaces.width - compact.width).toBeCloseTo(spaceAdvance * 4, 2);
|
||||
expect(tab.width).toBeCloseTo(fourSpaces.width, 2);
|
||||
});
|
||||
|
||||
it("renders two tabs as eight spaces independent of current x", async () => {
|
||||
const twoTabs = await line(`<p ${preserve}>LIT A\t\tB END</p>`);
|
||||
const eightSpaces = await line(`<p ${preserve}>LIT A B END</p>`);
|
||||
const oneLetter = await line(`<p ${preserve}>LIT A\tB END</p>`);
|
||||
const oneLetterCompact = await line(`<p ${preserve}>LIT AB END</p>`);
|
||||
const threeLetters = await line(`<p ${preserve}>LIT ABC\tD END</p>`);
|
||||
const threeLettersCompact = await line(`<p ${preserve}>LIT ABCD END</p>`);
|
||||
|
||||
expect(twoTabs.width).toBeCloseTo(eightSpaces.width, 2);
|
||||
expect(oneLetter.width - oneLetterCompact.width).toBeCloseTo(threeLetters.width - threeLettersCompact.width, 2);
|
||||
});
|
||||
|
||||
it.each(["en-US", "he-IL", "ar-SA"])("keeps marked tab geometry in %s", async (locale) => {
|
||||
const tab = await line(`<p ${preserve}>LIT A\tB END</p>`, locale);
|
||||
const spaces = await line(`<p ${preserve}>LIT A B END</p>`, locale);
|
||||
expect(tab.width).toBeCloseTo(spaces.width, 2);
|
||||
});
|
||||
|
||||
it("keeps unmarked collapse node-local and marked narrow content complete", async () => {
|
||||
const unmarked = await line("<p>LIT A B END</p>");
|
||||
const collapsed = await line("<p>LIT A B END</p>");
|
||||
expect(unmarked).toEqual(collapsed);
|
||||
|
||||
const sample = "LIT START\tSome breakable words continue through narrow content without disappearing END";
|
||||
const items = await renderItems(`<p ${preserve}>${sample}</p>`, "en-US", true);
|
||||
const text = items
|
||||
.map((item) => item.text)
|
||||
.join("")
|
||||
.replace(/\s/g, "");
|
||||
expect(text.slice(text.indexOf("LIT"))).toBe(sample.replace(/\s/g, ""));
|
||||
});
|
||||
|
||||
it.each([
|
||||
[
|
||||
"list",
|
||||
'<ul><li><p data-resume-whitespace="preserve">LIT A\tB END</p></li></ul>',
|
||||
'<ul><li><p data-resume-whitespace="preserve">LIT A B END</p></li></ul>',
|
||||
],
|
||||
[
|
||||
"quote",
|
||||
'<blockquote><p data-resume-whitespace="preserve">LIT A\tB END</p></blockquote>',
|
||||
'<blockquote><p data-resume-whitespace="preserve">LIT A B END</p></blockquote>',
|
||||
],
|
||||
[
|
||||
"table cell",
|
||||
'<table><tbody><tr><td><p data-resume-whitespace="preserve">LIT A\tB END</p></td></tr></tbody></table>',
|
||||
'<table><tbody><tr><td><p data-resume-whitespace="preserve">LIT A B END</p></td></tr></tbody></table>',
|
||||
],
|
||||
] as const)("preserves marked tab geometry in a %s", async (_name, tabHtml, spacesHtml) => {
|
||||
const tab = await line(tabHtml);
|
||||
const spaces = await line(spacesHtml);
|
||||
expect(tab.width).toBeCloseTo(spaces.width, 2);
|
||||
});
|
||||
});
|
||||
@@ -9,6 +9,29 @@ type PdfElement = ReactElement<{ children?: unknown; element?: { tag: string } }
|
||||
const getPdfElementProps = (element: unknown) => (element as PdfElement).props;
|
||||
|
||||
describe("normalizeRichTextHtml", () => {
|
||||
it("expands tabs only inside marked paragraphs and headings", () => {
|
||||
expect(
|
||||
normalizeRichTextHtml(
|
||||
'<p data-resume-whitespace="preserve">A\tB</p><h2 data-resume-whitespace="preserve">\tC</h2><p>D\tE</p>',
|
||||
),
|
||||
).toBe(
|
||||
'<p data-resume-whitespace="preserve">A B</p><h2 data-resume-whitespace="preserve"> C</h2><p>D\tE</p>',
|
||||
);
|
||||
});
|
||||
|
||||
it("retains marked paragraphs inside lists so preservation stays node-local", () => {
|
||||
expect(normalizeRichTextHtml('<ul><li><p data-resume-whitespace="preserve"> Listed\ttext </p></li></ul>')).toBe(
|
||||
'<ul><li><p data-resume-whitespace="preserve"> Listed text </p></li></ul>',
|
||||
);
|
||||
});
|
||||
|
||||
it("does not reinterpret marked RTL line breaks as pseudo-bullet lists", () => {
|
||||
const html = '<p data-resume-whitespace="preserve"> - First<br> - Second</p>';
|
||||
expect(normalizeRichTextHtml(html, { direction: "rtl" })).toBe(
|
||||
'<p data-resume-whitespace="preserve"> - First<br> - Second</p>',
|
||||
);
|
||||
});
|
||||
|
||||
it("decodes opted-in soft hyphens in text without changing links or escaped literals", () => {
|
||||
const html =
|
||||
'<p title="­">Soft­ware ­ ­ &shy; <a href="https://example.com/­">link</a></p>';
|
||||
|
||||
@@ -104,6 +104,7 @@ const unwrapSingleParagraphListItems = (root: ReturnType<typeof parse>) => {
|
||||
|
||||
const child = meaningfulChildren[0];
|
||||
if (!child || !isElement(child) || getTagName(child) !== "p") continue;
|
||||
if (child.getAttribute("data-resume-whitespace") === "preserve") continue;
|
||||
|
||||
listItem.innerHTML = child.innerHTML;
|
||||
}
|
||||
@@ -127,6 +128,18 @@ const normalizeParagraphIndentation = (root: ReturnType<typeof parse>, direction
|
||||
}
|
||||
};
|
||||
|
||||
const expandPreservedTabs = (root: ReturnType<typeof parse>) => {
|
||||
for (const element of root.querySelectorAll(
|
||||
'p[data-resume-whitespace="preserve"],h1[data-resume-whitespace="preserve"],h2[data-resume-whitespace="preserve"],h3[data-resume-whitespace="preserve"],h4[data-resume-whitespace="preserve"],h5[data-resume-whitespace="preserve"],h6[data-resume-whitespace="preserve"]',
|
||||
)) {
|
||||
const visit = (node: Node): void => {
|
||||
if (node.nodeType === NodeType.TEXT_NODE) node.rawText = node.rawText.replace(/\t/g, " ");
|
||||
for (const child of node.childNodes) visit(child);
|
||||
};
|
||||
visit(element);
|
||||
}
|
||||
};
|
||||
|
||||
const isInlineNode = (node: Node): boolean => {
|
||||
if (node.nodeType === NodeType.TEXT_NODE || node.nodeType === NodeType.COMMENT_NODE) return true;
|
||||
if (node.nodeType !== NodeType.ELEMENT_NODE) return false;
|
||||
@@ -165,9 +178,11 @@ const tryConvertPseudoBulletParagraph = (paragraphInnerHtml: string): string | n
|
||||
|
||||
export const convertPseudoBulletParagraphs = (html: string, direction: "ltr" | "rtl" = "ltr"): string =>
|
||||
html.replace(/<p\b([^>]*)>([\s\S]*?)<\/p>/gi, (full, _attrs, inner) => {
|
||||
const paragraph = parse(full).querySelector("p");
|
||||
if (paragraph?.getAttribute("data-resume-whitespace") === "preserve") return full;
|
||||
const converted = tryConvertPseudoBulletParagraph(inner);
|
||||
if (!converted) return full;
|
||||
const level = Number(parse(full).querySelector("p")?.getAttribute("data-indent"));
|
||||
const level = Number(paragraph?.getAttribute("data-indent"));
|
||||
if (!Number.isInteger(level) || level <= 0 || level > 8) return converted;
|
||||
// Keep the original paragraph's offset around the entire generated list.
|
||||
return converted.replace(
|
||||
@@ -200,6 +215,7 @@ export const normalizeRichTextHtml = (
|
||||
normalizeBoldBoundaryWhitespace(root);
|
||||
normalizeMarkElements(root);
|
||||
normalizeParagraphIndentation(root, direction);
|
||||
expandPreservedTabs(root);
|
||||
unwrapSingleParagraphListItems(root);
|
||||
|
||||
const flushInlineNodes = () => {
|
||||
|
||||
@@ -137,7 +137,11 @@ export const RichText = ({ children, semanticField }: RichTextProps) => {
|
||||
const visible = isNodeVisible(nodeKey);
|
||||
if (!visible) return null;
|
||||
const text = (
|
||||
<PdfText {...resolvedPdfTextProps(resolved)} style={composeStyles(style, resolved.style, safeTextStyle)}>
|
||||
<PdfText
|
||||
{...resolvedPdfTextProps(resolved)}
|
||||
data-resume-whitespace={element.getAttribute("data-resume-whitespace")}
|
||||
style={composeStyles(style, resolved.style, safeTextStyle)}
|
||||
>
|
||||
{textChildren}
|
||||
</PdfText>
|
||||
);
|
||||
@@ -230,7 +234,11 @@ export const RichText = ({ children, semanticField }: RichTextProps) => {
|
||||
style: props.style,
|
||||
indent: Number(props.element.getAttribute("data-indent")),
|
||||
semanticStyle: resolved.style,
|
||||
textProps: { ...resolvedPdfTextProps(resolved), hyphenationCallback },
|
||||
textProps: {
|
||||
...resolvedPdfTextProps(resolved),
|
||||
hyphenationCallback,
|
||||
"data-resume-whitespace": props.element.getAttribute("data-resume-whitespace"),
|
||||
},
|
||||
rtl,
|
||||
...(rtlTextWrapStyle ? { rtlTextWrapStyle } : {}),
|
||||
...(rtl ? { applyRtlDirection: applyRtlDirectionRecursively } : {}),
|
||||
|
||||
Reference in New Issue
Block a user