From 05e48a7cbcfb072526ea459a7c1cf976bd84fc35 Mon Sep 17 00:00:00 2001 From: Amruth Pillai Date: Sat, 5 Sep 2026 11:31:03 -0700 Subject: [PATCH] fix(pdf): preserve authored Unicode spaces in rich text (#3451) --- .../templates/shared/rich-text-html.test.ts | 13 ++ .../src/templates/shared/rich-text-html.ts | 9 +- .../src/unicode-spaces.integration.test.tsx | 125 ++++++++++++++++++ patches/react-pdf-html@2.1.5.patch | 50 +++++++ pnpm-lock.yaml | 7 +- pnpm-workspace.yaml | 1 + 6 files changed, 199 insertions(+), 6 deletions(-) create mode 100644 packages/pdf/src/unicode-spaces.integration.test.tsx create mode 100644 patches/react-pdf-html@2.1.5.patch diff --git a/packages/pdf/src/templates/shared/rich-text-html.test.ts b/packages/pdf/src/templates/shared/rich-text-html.test.ts index d40a31ce1..c9459251c 100644 --- a/packages/pdf/src/templates/shared/rich-text-html.test.ts +++ b/packages/pdf/src/templates/shared/rich-text-html.test.ts @@ -152,6 +152,19 @@ describe("normalizeRichTextHtml", () => { expect(normalizeRichTextHtml(" text ")).toBe("

text

"); }); + it("preserves authored Unicode spaces around bare rich text", () => { + expect(normalizeRichTextHtml("\u3000text\u00a0")).toBe("

\u3000text\u00a0

"); + }); + + it("retains an inline ideographic-space paragraph", () => { + expect(normalizeRichTextHtml("\u3000")).toBe("

\u3000

"); + }); + + it("does not discard a Unicode-space sibling when unwrapping a list paragraph", () => { + const html = ""; + expect(normalizeRichTextHtml(html)).toBe(html); + }); + it("returns empty string for empty input", () => { expect(normalizeRichTextHtml("")).toBe(""); }); diff --git a/packages/pdf/src/templates/shared/rich-text-html.ts b/packages/pdf/src/templates/shared/rich-text-html.ts index 2083edaf0..8c415bf3c 100644 --- a/packages/pdf/src/templates/shared/rich-text-html.ts +++ b/packages/pdf/src/templates/shared/rich-text-html.ts @@ -64,8 +64,11 @@ const normalizeMarkElements = (root: ReturnType) => { } }; +// Match HTML document whitespace, not Unicode spaces authored as visible content. +const trimHtmlWhitespace = (text: string): string => text.replace(/^[\t\n\f\r ]+|[\t\n\f\r ]+$/g, ""); + const isMeaningfulNode = (node: Node): boolean => - node.nodeType !== NodeType.TEXT_NODE || node.toString().trim().length > 0; + node.nodeType !== NodeType.TEXT_NODE || trimHtmlWhitespace(node.toString()).length > 0; const isElement = (node: Node): node is HTMLElement => node.nodeType === NodeType.ELEMENT_NODE; @@ -189,7 +192,7 @@ export const normalizeRichTextHtml = ( html: string, { direction = "ltr", softHyphens = false }: NormalizeRichTextHtmlOptions = {}, ): string => { - const root = parse(html.trim(), { comment: false }); + const root = parse(trimHtmlWhitespace(html), { comment: false }); const normalized: string[] = []; let inlineNodes: string[] = []; @@ -202,7 +205,7 @@ export const normalizeRichTextHtml = ( const flushInlineNodes = () => { const inlineHtml = inlineNodes.join(""); - if (inlineHtml.trim()) normalized.push(`

${inlineHtml}

`); + if (trimHtmlWhitespace(inlineHtml)) normalized.push(`

${inlineHtml}

`); inlineNodes = []; }; diff --git a/packages/pdf/src/unicode-spaces.integration.test.tsx b/packages/pdf/src/unicode-spaces.integration.test.tsx new file mode 100644 index 000000000..a5aafc324 --- /dev/null +++ b/packages/pdf/src/unicode-spaces.integration.test.tsx @@ -0,0 +1,125 @@ +import type { ResumeData } from "@reactive-resume/schema/resume/data"; +import { describe, expect, it } from "vitest"; +import { renderToBuffer } from "@react-pdf/renderer"; +import { getDocument } from "pdfjs-dist/legacy/build/pdf.mjs"; +import { act, createElement } from "react"; +import { defaultResumeData } from "@reactive-resume/schema/resume/default"; +import { ResumeDocument } from "./document"; + +type Line = { text: string; x: number; y: number; right: number }; + +function resume(plain: string, html: string, family = "Noto Serif SC", locale = "zh-CN"): ResumeData { + const data = structuredClone(defaultResumeData); + data.picture.hidden = true; + data.basics.name = "Probe"; + data.basics.headline = plain; + data.metadata.page.locale = locale; + data.metadata.typography.body.fontFamily = family; + data.metadata.typography.heading.fontFamily = family; + data.metadata.typography.body.fontSize = 10; + data.metadata.typography.body.fontWeights = ["400", "700"]; + data.metadata.typography.heading.fontWeights = ["400", "700"]; + data.metadata.layout.pages = [{ fullWidth: true, main: ["summary"], sidebar: [] }]; + data.summary.title = "Whitespace"; + data.summary.hidden = false; + data.summary.content = html; + return data; +} + +async function pdfLines(data: ResumeData) { + const element = createElement(ResumeDocument, { data, template: "onyx" }) as unknown as Parameters< + typeof renderToBuffer + >[0]; + const loading = getDocument({ data: new Uint8Array(await act(() => renderToBuffer(element))) }); + try { + const document = await loading.promise; + expect(document.numPages).toBe(1); + const page = await document.getPage(1); + const content = await page.getTextContent({ disableNormalization: true }); + const lines = new Map(); + for (const item of content.items) { + if (!("str" in item) || !item.str) continue; + const x = item.transform[4]; + const y = item.transform[5]; + const line = lines.get(y) ?? { text: "", x, y, right: x }; + line.text += item.str; + line.x = Math.min(line.x, x); + line.right = Math.max(line.right, x + item.width); + lines.set(y, line); + } + const ordered = [...lines.values()].sort((a, b) => b.y - a.y); + const title = ordered.find((line) => line.text === "Whitespace"); + if (!title) throw new Error("Missing summary title in PDF"); + const plain = ordered[1]; + if (!plain) throw new Error("Missing plain headline control in PDF"); + return { plain, body: ordered.filter((line) => line.y < title.y) }; + } finally { + await loading.destroy(); + } +} + +function width(line: Line | undefined) { + if (!line) throw new Error("Missing expected PDF text line"); + return line.right - line.x; +} + +describe("Unicode spaces in exported rich text", () => { + it.each([ + ["Noto Serif SC", "zh-CN"], + ["Noto Serif SC", "en-US"], + ["Noto Sans SC", "zh-CN"], + ["IBM Plex Serif", "zh-CN"], + ["Noto Serif SC", "ar-SA"], + ])("retains ideographic-space advances with %s / %s", { timeout: 60_000 }, async (family, locale) => { + const { plain, body } = await pdfLines(resume("中\u3000文\u3000字", "

中\u3000文\u3000字

", family, locale)); + expect(body).toHaveLength(1); + // Three full-width glyphs plus two ideographic spaces at 10pt. + expect(width(plain)).toBeCloseTo(50, 2); + expect(width(body[0])).toBeCloseTo(50, 2); + }); + + it.each([ + ["mixed Latin/CJK", "中\u3000A\u3000\u3000文", "

中\u3000A\u3000\u3000文

"], + ["inline leading spaces", "中\u3000\u3000文", "

中\u3000\u3000文

"], + ["marked spaces", "中\u3000\u3000文", "

中\u3000\u3000文

"], + ["literal nonbreaking spaces", "中\u00a0\u00a0文", "

中\u00a0\u00a0文

"], + ["named nonbreaking-space count", "中\u00a0\u00a0文", "

中  文

"], + ["ordinary ASCII whitespace", "中 文", "

中 \t\n\r\f 文

"], + ["preformatted spaces", "中\u3000\u3000文", '
中\u3000\u3000文
'], + ])("preserves %s", { timeout: 60_000 }, async (_name, text, html) => { + const { plain, body } = await pdfLines(resume(text, html)); + expect(body).toHaveLength(1); + expect(body[0]?.text.replaceAll(/\s/g, "")).toBe(text.replaceAll(/\s/g, "")); + expect(width(body[0])).toBeCloseTo(width(plain), 2); + }); + + it("retains ideographic spaces at the start of a paragraph", { timeout: 60_000 }, async () => { + const { plain, body } = await pdfLines(resume("中 文", "

\u3000中 文

")); + expect(body).toHaveLength(1); + expect(body[0]?.right).toBeCloseTo(plain.right + 10, 2); + }); + + it("retains ideographic spaces at the start of bare rich text", { timeout: 60_000 }, async () => { + const { plain, body } = await pdfLines(resume("中 文", "\u3000中 文")); + expect(body).toHaveLength(1); + expect(body[0]?.right).toBeCloseTo(plain.right + 10, 2); + }); + + it("keeps repeated ASCII spaces and line breaks in preformatted text", { timeout: 60_000 }, async () => { + const { plain, body } = await pdfLines(resume("A B", '
A  B\nA  B
')); + expect(body).toHaveLength(2); + for (const line of body) expect(width(line)).toBeCloseTo(width(plain), 2); + }); + + it("retains a literal nonbreaking space's word grouping", { timeout: 60_000 }, async () => { + const data = resume("a hello world", "

a hello\u00a0world

", "Helvetica", "en-US"); + data.metadata.stylesheet = { + mode: "semantic", + source: { languageVersion: 1, text: "@version 1; section { width: 50pt; }" }, + }; + const ascii = await pdfLines({ ...data, summary: { ...data.summary, content: "

a hello world

" } }); + expect(ascii.body.map((line) => line.text.trim())).toEqual(["a hello", "world"]); + const { body } = await pdfLines(data); + expect(body.map((line) => line.text.trim().replaceAll("\u00a0", " "))).toEqual(["a", "hello world"]); + }); +}); diff --git a/patches/react-pdf-html@2.1.5.patch b/patches/react-pdf-html@2.1.5.patch new file mode 100644 index 000000000..c57644995 --- /dev/null +++ b/patches/react-pdf-html@2.1.5.patch @@ -0,0 +1,50 @@ +diff --git a/dist/cjs/render.js b/dist/cjs/render.js +index cea3de37aceb8e98b7698087639a6e98205d40e3..a1e34ea0bf60f3bca2d7e79e0354eda5245b927d 100644 +--- a/dist/cjs/render.js ++++ b/dist/cjs/render.js +@@ -72,8 +72,9 @@ const hasBlockContent = (element) => { + return true; + }; + exports.hasBlockContent = hasBlockContent; +-const ltrim = (text) => text.replace(/^\s+/, ''); +-const rtrim = (text) => text.replace(/\s+$/, ''); ++// Collapse HTML document whitespace, preserving authored Unicode spaces. ++const ltrim = (text) => text.replace(/^[\t\n\f\r ]+/, ''); ++const rtrim = (text) => text.replace(/[\t\n\f\r ]+$/, ''); + const isCustomElement = (element) => { + if (!element || typeof element === 'string') + return false; +@@ -160,7 +161,7 @@ const renderElement = (element, stylesheets, renderers, children, index) => { + return (React.createElement(Element, { key: index, style: element.style, children: children, element: element, stylesheets: stylesheets })); + }; + exports.renderElement = renderElement; +-const collapseWhitespace = (string) => string.replace(/(\s+)/g, ' '); ++const collapseWhitespace = (string) => string.replace(/[\t\n\f\r ]+/g, ' '); + exports.collapseWhitespace = collapseWhitespace; + const renderBucketElement = (element, options, index) => { + if (typeof element === 'string') { +diff --git a/dist/esm/render.js b/dist/esm/render.js +index 30365527b67332d0a29f684152cad1d8417ae90d..18b24f488e81404df50d772bf5868bfcb61b810d 100644 +--- a/dist/esm/render.js ++++ b/dist/esm/render.js +@@ -40,8 +40,9 @@ export const hasBlockContent = (element) => { + } + return true; + }; +-const ltrim = (text) => text.replace(/^\s+/, ''); +-const rtrim = (text) => text.replace(/\s+$/, ''); ++// Collapse HTML document whitespace, preserving authored Unicode spaces. ++const ltrim = (text) => text.replace(/^[\t\n\f\r ]+/, ''); ++const rtrim = (text) => text.replace(/[\t\n\f\r ]+$/, ''); + const isCustomElement = (element) => { + if (!element || typeof element === 'string') + return false; +@@ -126,7 +127,7 @@ export const renderElement = (element, stylesheets, renderers, children, index) + } + return (React.createElement(Element, { key: index, style: element.style, children: children, element: element, stylesheets: stylesheets })); + }; +-export const collapseWhitespace = (string) => string.replace(/(\s+)/g, ' '); ++export const collapseWhitespace = (string) => string.replace(/[\t\n\f\r ]+/g, ' '); + export const renderBucketElement = (element, options, index) => { + if (typeof element === 'string') { + return renderElement(options.collapse ? collapseWhitespace(element) : element, options.stylesheets, options.renderers, undefined, index); diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 7e9162c57..f82dcbf15 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -15,6 +15,7 @@ patchedDependencies: '@react-pdf/layout@5.2.0': 75dac7cb8260f9c96cd9ab5522d095a3aa047881a9e9d1d2a597409b45944494 '@react-pdf/textkit': 092a6fe8baf3c472a81cdf2ab52a00ec98ef3364e1b8e744005dbcf219a8a213 fontkit@2.0.4: 90f4c51c676a88b91dcc397d60a677c77b5e6dc8e15ffb3538310965ef5c05a4 + react-pdf-html@2.1.5: f0618f9c9f654758495189f6c15e1ff2d45fbd1a4645c4a18885573aeadbbd10 importers: @@ -271,7 +272,7 @@ importers: version: 6.9.3(react-dom@19.2.8(react@19.2.8))(react@19.2.8)(supports-color@7.2.0) react-pdf-html: specifier: ^2.1.5 - version: 2.1.5(@react-pdf/renderer@4.9.0(react@19.2.8))(react@19.2.8) + version: 2.1.5(patch_hash=f0618f9c9f654758495189f6c15e1ff2d45fbd1a4645c4a18885573aeadbbd10)(@react-pdf/renderer@4.9.0(react@19.2.8))(react@19.2.8) resumable-stream: specifier: ^2.2.12 version: 2.2.12 @@ -1140,7 +1141,7 @@ importers: version: 19.2.8 react-pdf-html: specifier: ^2.1.5 - version: 2.1.5(@react-pdf/renderer@4.9.0(react@19.2.8))(react@19.2.8) + version: 2.1.5(patch_hash=f0618f9c9f654758495189f6c15e1ff2d45fbd1a4645c4a18885573aeadbbd10)(@react-pdf/renderer@4.9.0(react@19.2.8))(react@19.2.8) devDependencies: '@napi-rs/canvas': specifier: 1.0.8 @@ -15209,7 +15210,7 @@ snapshots: transitivePeerDependencies: - supports-color - react-pdf-html@2.1.5(@react-pdf/renderer@4.9.0(react@19.2.8))(react@19.2.8): + react-pdf-html@2.1.5(patch_hash=f0618f9c9f654758495189f6c15e1ff2d45fbd1a4645c4a18885573aeadbbd10)(@react-pdf/renderer@4.9.0(react@19.2.8))(react@19.2.8): dependencies: '@react-pdf/renderer': 4.9.0(react@19.2.8) css-tree: 1.1.3 diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index 621308968..4d83485c3 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -20,3 +20,4 @@ patchedDependencies: '@react-pdf/layout@5.2.0': patches/@react-pdf__layout@5.2.0.patch '@react-pdf/textkit': patches/@react-pdf__textkit.patch fontkit@2.0.4: patches/fontkit@2.0.4.patch + react-pdf-html@2.1.5: patches/react-pdf-html@2.1.5.patch