fix(pdf): render emoji via a Noto Emoji script fallback (#3351)

* fix(pdf): render emoji via a Noto Emoji script fallback

Emoji in resume content (flags, globe, pictographs) rendered as mojibake
in the preview and PDF export because the per-codepoint fallback chain
registered no emoji-capable font: every font in the stack lacked the
glyphs, so layout fell through to single-byte standard-font encoding —
each UTF-16 code unit truncated to its low byte (#3321).

Follows the #2986/#3190 script-fallback pattern: detect emoji content
(regional indicators unioned with Extended_Pictographic), map it to the
monochrome Noto Emoji web font (TrueType glyf outlines, PDF-embeddable),
and register it in the fallback stack for both serif and sans stacks.
Out-of-range weight requests alias to the nearest served weight (300-700)
so registration never falls back to the preview subset.

* fix(pdf): detect keycap emoji via the combining enclosing keycap

Greptile review on #3351: keycap sequences like 1\uFE0F\u20E3 carry no
regional indicator and no Extended_Pictographic codepoint, so they
bypassed the emoji detector and rendered garbled — the exact class of
bug #3321 fixes. Union U+20E3 into the detector; every valid keycap
sequence contains it.

* Update packages/utils/src/locale.ts

Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com>

* [autofix.ci] apply automated fixes

---------

Co-authored-by: Amruth Pillai <im.amruth@gmail.com>
Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com>
Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
This commit is contained in:
Santhi Prakash
2026-08-27 03:15:34 +02:00
committed by GitHub
co-authored by greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> Amruth Pillai autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
parent 2761bd6715
commit b47f805321
6 changed files with 93 additions and 2 deletions
+6
View File
@@ -200,9 +200,15 @@ describe("getPdfFallbackFontFamilies", () => {
expect(getPdfFallbackFontFamilies("Helvetica", { locale: "th-TH" })).toEqual(["Noto Sans Thai", "Noto Sans"]);
});
it("uses Noto Emoji for detected emoji content, for serif and sans stacks alike (#3321)", () => {
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["emoji"] })).toEqual(["Noto Emoji", "Noto Sans"]);
expect(getPdfFallbackFontFamilies("Times-Roman", { scripts: ["emoji"] })).toEqual(["Noto Emoji", "Noto Serif"]);
});
it("does not append the Simplified Chinese safety net for non-CJK scripts", () => {
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["arabic"] })).toEqual(["Noto Sans Arabic", "Noto Sans"]);
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["thai"] })).not.toContain("Noto Sans SC");
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["emoji"] })).not.toContain("Noto Sans SC");
});
it("orders the locale script first, then content scripts (mixed RTL + CJK resume)", () => {
+3
View File
@@ -82,6 +82,9 @@ const scriptFonts: Record<Script, { serif: string; sansSerif: string }> = {
arabic: { serif: "Noto Naskh Arabic", sansSerif: "Noto Sans Arabic" },
hebrew: { serif: "Noto Sans Hebrew", sansSerif: "Noto Sans Hebrew" },
thai: { serif: "Noto Sans Thai", sansSerif: "Noto Sans Thai" },
// Monochrome outlines (TrueType glyf, not CBDT bitmaps) so react-pdf can
// embed them; the serif/sans distinction is meaningless for emoji (#3321).
emoji: { serif: "Noto Emoji", sansSerif: "Noto Emoji" },
};
// Covers General Punctuation (U+2000–U+206F) and other symbols missing from
+18
View File
@@ -5861,6 +5861,24 @@
"900": "https://fonts.gstatic.com/s/notosanssymbols/v47/rP2up3q65FkAtHfwd-eIS2brbDN6gxP34F9jRRCe4W3g1AggavVFRkzrbQ.ttf"
}
},
{
"type": "web",
"category": "sans-serif",
"family": "Noto Emoji",
"weights": ["300", "400", "500", "600", "700"],
"preview": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob-r0jwvS-dGJQ.ttf",
"files": {
"100": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob_10jwvS-dGJQ.ttf",
"200": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob_10jwvS-dGJQ.ttf",
"300": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob_10jwvS-dGJQ.ttf",
"400": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob-r0jwvS-dGJQ.ttf",
"500": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob-Z0jwvS-dGJQ.ttf",
"600": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob911TwvS-dGJQ.ttf",
"700": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob9M1TwvS-dGJQ.ttf",
"800": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob9M1TwvS-dGJQ.ttf",
"900": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob9M1TwvS-dGJQ.ttf"
}
},
{
"type": "web",
"category": "handwriting",
@@ -165,6 +165,27 @@ describe("registerFonts", () => {
expect(registerSpy).toHaveBeenCalledWith(expect.objectContaining({ family: "Noto Sans Thai" }));
});
it("registers the Noto Emoji fallback when content contains emoji (#3321)", async () => {
const registerSpy = vi.spyOn(Font, "register").mockImplementation(() => {});
vi.spyOn(Font, "registerHyphenationCallback").mockImplementation(() => {});
const emojiSource = getWebFontSource("Noto Emoji", "400", false);
const { registerFonts } = await import("./use-register-fonts");
const pdfTypography = registerFonts(typography, "en-US", false, new Set(["emoji"]));
expect(pdfTypography.body.fontFamily).toEqual(["IBM Plex Serif", "Noto Emoji", "Noto Serif"]);
expect(registerSpy).toHaveBeenCalledWith(
expect.objectContaining({
family: "Noto Emoji",
fontWeight: 400,
fontStyle: "normal",
src: emojiSource,
}),
);
// Emoji is not CJK: no Simplified-Chinese safety net, no per-character breaking.
expect(registerSpy).not.toHaveBeenCalledWith(expect.objectContaining({ family: "Noto Serif SC" }));
});
it("does NOT enable CJK per-character line breaking for non-CJK fallback scripts", async () => {
const registerHyphenationSpy = vi.spyOn(Font, "registerHyphenationCallback").mockImplementation(() => {});
vi.spyOn(Font, "register").mockImplementation(() => {});
@@ -378,6 +399,38 @@ describe("resumeContentScripts", () => {
expect([...resumeContentScripts(withSummary("สวัสดี"))]).toEqual(["thai"]);
});
it("detects emoji flags and pictographs (#3321)", async () => {
const { resumeContentScripts } = await import("./use-register-fonts");
const data = {
...defaultResumeData,
basics: { ...defaultResumeData.basics, location: "Berlin \u{1F1E9}\u{1F1EA} \u{1F310} \u{2B50}" },
} satisfies ResumeData;
const scripts = resumeContentScripts(data);
expect(scripts.has("emoji")).toBe(true);
// Regional indicators are not Extended_Pictographic and pictographs are
// not any other script — the emoji detector must catch both alone.
expect(scripts.size).toBe(1);
});
it("detects keycap emoji without pictographs (#3321)", async () => {
const { resumeContentScripts } = await import("./use-register-fonts");
// "1\uFE0F\u20E3" (1\u20e3) and "#\uFE0F\u20E3" (#\u20e3) hold no regional
// indicator and no Extended_Pictographic codepoint, so the detector must
// catch the combining enclosing keycap on its own.
const data = {
...defaultResumeData,
basics: {
...defaultResumeData.basics,
location: "Steps \u0031\uFE0F\u20E3 and \u0023\uFE0F\u20E3",
},
} satisfies ResumeData;
const scripts = resumeContentScripts(data);
expect(scripts.has("emoji")).toBe(true);
expect(scripts.size).toBe(1);
});
it("detects multiple scripts in mixed content", async () => {
const { resumeContentScripts } = await import("./use-register-fonts");
const scripts = resumeContentScripts(withSummary("안녕 翠翠 سلام"));
@@ -159,6 +159,14 @@ const arabicRegex = /[؀-ۿݐ-ݿࢠ-ࣿﭐ-﷿ﹰ-ﻼ]/;
const hebrewRegex = /[֐-׿יִ-ﭏ]/;
const thaiRegex = /[฀-๿]/;
// Emoji: regional indicators (flags) are NOT Extended_Pictographic, so union
// them explicitly with the pictographic property (#3321). Keycap sequences
// (e.g. "1\uFE0F\u20E3") carry no pictographic codepoint either — their
// discriminator is the combining enclosing keycap U+20E3, unioned here for the
// same reason: without it, keycap-only content falls back to a font without
// the enclosure mark and renders garbled.
const emojiRegex = /[\u{1F1E6}-\u{1F1FF}]|\u{20E3}|\p{Extended_Pictographic}/u;
const scriptDetectors: { script: Script; regex: RegExp }[] = [
{ script: "hangul", regex: hangulRegex },
{ script: "kana", regex: kanaRegex },
@@ -166,6 +174,7 @@ const scriptDetectors: { script: Script; regex: RegExp }[] = [
{ script: "arabic", regex: arabicRegex },
{ script: "hebrew", regex: hebrewRegex },
{ script: "thai", regex: thaiRegex },
{ script: "emoji", regex: emojiRegex },
];
const collectScripts = (value: unknown, scripts: Set<Script>): void => {
+4 -2
View File
@@ -75,8 +75,10 @@ export function isCJKLocale(locale: Locale): boolean {
// because react-pdf (unlike a browser) has no automatic system-font fallback:
// a glyph only renders if a registered font contains it. We pick the matching
// Noto font per script so e.g. Hangul → Noto KR, Arabic → Noto Arabic, instead
// of falling back to a Latin/Han-only font and producing tofu.
export type Script = "hangul" | "kana" | "han-traditional" | "han-simplified" | "arabic" | "hebrew" | "thai";
// of falling back to a Latin/Han-only font and producing tofu. "emoji" is
// content-detected only (never locale-derived) and resolves to Noto Emoji so
// pictographs and regional indicators render instead of mojibake (#3321).
export type Script = "hangul" | "kana" | "han-traditional" | "han-simplified" | "arabic" | "hebrew" | "thai" | "emoji";
// The CJK subset of `Script`. CJK needs extra per-character line breaking that
// must NOT be applied to Arabic (cursive, joined letters) or Thai (combining