mirror of
https://github.com/AmruthPillai/Reactive-Resume.git
synced 2026-10-03 18:23:47 +10:00
fix(pdf): render emoji via a Noto Emoji script fallback (#3351)
* fix(pdf): render emoji via a Noto Emoji script fallback Emoji in resume content (flags, globe, pictographs) rendered as mojibake in the preview and PDF export because the per-codepoint fallback chain registered no emoji-capable font: every font in the stack lacked the glyphs, so layout fell through to single-byte standard-font encoding — each UTF-16 code unit truncated to its low byte (#3321). Follows the #2986/#3190 script-fallback pattern: detect emoji content (regional indicators unioned with Extended_Pictographic), map it to the monochrome Noto Emoji web font (TrueType glyf outlines, PDF-embeddable), and register it in the fallback stack for both serif and sans stacks. Out-of-range weight requests alias to the nearest served weight (300-700) so registration never falls back to the preview subset. * fix(pdf): detect keycap emoji via the combining enclosing keycap Greptile review on #3351: keycap sequences like 1\uFE0F\u20E3 carry no regional indicator and no Extended_Pictographic codepoint, so they bypassed the emoji detector and rendered garbled — the exact class of bug #3321 fixes. Union U+20E3 into the detector; every valid keycap sequence contains it. * Update packages/utils/src/locale.ts Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> * [autofix.ci] apply automated fixes --------- Co-authored-by: Amruth Pillai <im.amruth@gmail.com> Co-authored-by: greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com> Co-authored-by: autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
This commit is contained in:
co-authored by
greptile-apps[bot] <165735046+greptile-apps[bot]@users.noreply.github.com>
Amruth Pillai
autofix-ci[bot] <114827586+autofix-ci[bot]@users.noreply.github.com>
parent
2761bd6715
commit
b47f805321
@@ -200,9 +200,15 @@ describe("getPdfFallbackFontFamilies", () => {
|
||||
expect(getPdfFallbackFontFamilies("Helvetica", { locale: "th-TH" })).toEqual(["Noto Sans Thai", "Noto Sans"]);
|
||||
});
|
||||
|
||||
it("uses Noto Emoji for detected emoji content, for serif and sans stacks alike (#3321)", () => {
|
||||
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["emoji"] })).toEqual(["Noto Emoji", "Noto Sans"]);
|
||||
expect(getPdfFallbackFontFamilies("Times-Roman", { scripts: ["emoji"] })).toEqual(["Noto Emoji", "Noto Serif"]);
|
||||
});
|
||||
|
||||
it("does not append the Simplified Chinese safety net for non-CJK scripts", () => {
|
||||
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["arabic"] })).toEqual(["Noto Sans Arabic", "Noto Sans"]);
|
||||
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["thai"] })).not.toContain("Noto Sans SC");
|
||||
expect(getPdfFallbackFontFamilies("Helvetica", { scripts: ["emoji"] })).not.toContain("Noto Sans SC");
|
||||
});
|
||||
|
||||
it("orders the locale script first, then content scripts (mixed RTL + CJK resume)", () => {
|
||||
|
||||
@@ -82,6 +82,9 @@ const scriptFonts: Record<Script, { serif: string; sansSerif: string }> = {
|
||||
arabic: { serif: "Noto Naskh Arabic", sansSerif: "Noto Sans Arabic" },
|
||||
hebrew: { serif: "Noto Sans Hebrew", sansSerif: "Noto Sans Hebrew" },
|
||||
thai: { serif: "Noto Sans Thai", sansSerif: "Noto Sans Thai" },
|
||||
// Monochrome outlines (TrueType glyf, not CBDT bitmaps) so react-pdf can
|
||||
// embed them; the serif/sans distinction is meaningless for emoji (#3321).
|
||||
emoji: { serif: "Noto Emoji", sansSerif: "Noto Emoji" },
|
||||
};
|
||||
|
||||
// Covers General Punctuation (U+2000–U+206F) and other symbols missing from
|
||||
|
||||
@@ -5861,6 +5861,24 @@
|
||||
"900": "https://fonts.gstatic.com/s/notosanssymbols/v47/rP2up3q65FkAtHfwd-eIS2brbDN6gxP34F9jRRCe4W3g1AggavVFRkzrbQ.ttf"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "web",
|
||||
"category": "sans-serif",
|
||||
"family": "Noto Emoji",
|
||||
"weights": ["300", "400", "500", "600", "700"],
|
||||
"preview": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob-r0jwvS-dGJQ.ttf",
|
||||
"files": {
|
||||
"100": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob_10jwvS-dGJQ.ttf",
|
||||
"200": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob_10jwvS-dGJQ.ttf",
|
||||
"300": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob_10jwvS-dGJQ.ttf",
|
||||
"400": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob-r0jwvS-dGJQ.ttf",
|
||||
"500": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob-Z0jwvS-dGJQ.ttf",
|
||||
"600": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob911TwvS-dGJQ.ttf",
|
||||
"700": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob9M1TwvS-dGJQ.ttf",
|
||||
"800": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob9M1TwvS-dGJQ.ttf",
|
||||
"900": "https://fonts.gstatic.com/s/notoemoji/v62/bMrnmSyK7YY-MEu6aWjPDs-ar6uWaGWuob9M1TwvS-dGJQ.ttf"
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "web",
|
||||
"category": "handwriting",
|
||||
|
||||
@@ -165,6 +165,27 @@ describe("registerFonts", () => {
|
||||
expect(registerSpy).toHaveBeenCalledWith(expect.objectContaining({ family: "Noto Sans Thai" }));
|
||||
});
|
||||
|
||||
it("registers the Noto Emoji fallback when content contains emoji (#3321)", async () => {
|
||||
const registerSpy = vi.spyOn(Font, "register").mockImplementation(() => {});
|
||||
vi.spyOn(Font, "registerHyphenationCallback").mockImplementation(() => {});
|
||||
const emojiSource = getWebFontSource("Noto Emoji", "400", false);
|
||||
const { registerFonts } = await import("./use-register-fonts");
|
||||
|
||||
const pdfTypography = registerFonts(typography, "en-US", false, new Set(["emoji"]));
|
||||
|
||||
expect(pdfTypography.body.fontFamily).toEqual(["IBM Plex Serif", "Noto Emoji", "Noto Serif"]);
|
||||
expect(registerSpy).toHaveBeenCalledWith(
|
||||
expect.objectContaining({
|
||||
family: "Noto Emoji",
|
||||
fontWeight: 400,
|
||||
fontStyle: "normal",
|
||||
src: emojiSource,
|
||||
}),
|
||||
);
|
||||
// Emoji is not CJK: no Simplified-Chinese safety net, no per-character breaking.
|
||||
expect(registerSpy).not.toHaveBeenCalledWith(expect.objectContaining({ family: "Noto Serif SC" }));
|
||||
});
|
||||
|
||||
it("does NOT enable CJK per-character line breaking for non-CJK fallback scripts", async () => {
|
||||
const registerHyphenationSpy = vi.spyOn(Font, "registerHyphenationCallback").mockImplementation(() => {});
|
||||
vi.spyOn(Font, "register").mockImplementation(() => {});
|
||||
@@ -378,6 +399,38 @@ describe("resumeContentScripts", () => {
|
||||
expect([...resumeContentScripts(withSummary("สวัสดี"))]).toEqual(["thai"]);
|
||||
});
|
||||
|
||||
it("detects emoji flags and pictographs (#3321)", async () => {
|
||||
const { resumeContentScripts } = await import("./use-register-fonts");
|
||||
const data = {
|
||||
...defaultResumeData,
|
||||
basics: { ...defaultResumeData.basics, location: "Berlin \u{1F1E9}\u{1F1EA} \u{1F310} \u{2B50}" },
|
||||
} satisfies ResumeData;
|
||||
|
||||
const scripts = resumeContentScripts(data);
|
||||
expect(scripts.has("emoji")).toBe(true);
|
||||
// Regional indicators are not Extended_Pictographic and pictographs are
|
||||
// not any other script — the emoji detector must catch both alone.
|
||||
expect(scripts.size).toBe(1);
|
||||
});
|
||||
|
||||
it("detects keycap emoji without pictographs (#3321)", async () => {
|
||||
const { resumeContentScripts } = await import("./use-register-fonts");
|
||||
// "1\uFE0F\u20E3" (1\u20e3) and "#\uFE0F\u20E3" (#\u20e3) hold no regional
|
||||
// indicator and no Extended_Pictographic codepoint, so the detector must
|
||||
// catch the combining enclosing keycap on its own.
|
||||
const data = {
|
||||
...defaultResumeData,
|
||||
basics: {
|
||||
...defaultResumeData.basics,
|
||||
location: "Steps \u0031\uFE0F\u20E3 and \u0023\uFE0F\u20E3",
|
||||
},
|
||||
} satisfies ResumeData;
|
||||
|
||||
const scripts = resumeContentScripts(data);
|
||||
expect(scripts.has("emoji")).toBe(true);
|
||||
expect(scripts.size).toBe(1);
|
||||
});
|
||||
|
||||
it("detects multiple scripts in mixed content", async () => {
|
||||
const { resumeContentScripts } = await import("./use-register-fonts");
|
||||
const scripts = resumeContentScripts(withSummary("안녕 翠翠 سلام"));
|
||||
|
||||
@@ -159,6 +159,14 @@ const arabicRegex = /[-ۿݐ-ݿࢠ-ࣿﭐ-﷿ﹰ-ﻼ]/;
|
||||
const hebrewRegex = /[-יִ-ﭏ]/;
|
||||
const thaiRegex = /[-]/;
|
||||
|
||||
// Emoji: regional indicators (flags) are NOT Extended_Pictographic, so union
|
||||
// them explicitly with the pictographic property (#3321). Keycap sequences
|
||||
// (e.g. "1\uFE0F\u20E3") carry no pictographic codepoint either — their
|
||||
// discriminator is the combining enclosing keycap U+20E3, unioned here for the
|
||||
// same reason: without it, keycap-only content falls back to a font without
|
||||
// the enclosure mark and renders garbled.
|
||||
const emojiRegex = /[\u{1F1E6}-\u{1F1FF}]|\u{20E3}|\p{Extended_Pictographic}/u;
|
||||
|
||||
const scriptDetectors: { script: Script; regex: RegExp }[] = [
|
||||
{ script: "hangul", regex: hangulRegex },
|
||||
{ script: "kana", regex: kanaRegex },
|
||||
@@ -166,6 +174,7 @@ const scriptDetectors: { script: Script; regex: RegExp }[] = [
|
||||
{ script: "arabic", regex: arabicRegex },
|
||||
{ script: "hebrew", regex: hebrewRegex },
|
||||
{ script: "thai", regex: thaiRegex },
|
||||
{ script: "emoji", regex: emojiRegex },
|
||||
];
|
||||
|
||||
const collectScripts = (value: unknown, scripts: Set<Script>): void => {
|
||||
|
||||
@@ -75,8 +75,10 @@ export function isCJKLocale(locale: Locale): boolean {
|
||||
// because react-pdf (unlike a browser) has no automatic system-font fallback:
|
||||
// a glyph only renders if a registered font contains it. We pick the matching
|
||||
// Noto font per script so e.g. Hangul → Noto KR, Arabic → Noto Arabic, instead
|
||||
// of falling back to a Latin/Han-only font and producing tofu.
|
||||
export type Script = "hangul" | "kana" | "han-traditional" | "han-simplified" | "arabic" | "hebrew" | "thai";
|
||||
// of falling back to a Latin/Han-only font and producing tofu. "emoji" is
|
||||
// content-detected only (never locale-derived) and resolves to Noto Emoji so
|
||||
// pictographs and regional indicators render instead of mojibake (#3321).
|
||||
export type Script = "hangul" | "kana" | "han-traditional" | "han-simplified" | "arabic" | "hebrew" | "thai" | "emoji";
|
||||
|
||||
// The CJK subset of `Script`. CJK needs extra per-character line breaking that
|
||||
// must NOT be applied to Arabic (cursive, joined letters) or Thai (combining
|
||||
|
||||
Reference in New Issue
Block a user