Files
Reactive-Resume/vendor/forme/unicode-extraction.patch
T

1202 lines
51 KiB
Diff

diff --git a/engine/src/font/fallback.rs b/engine/src/font/fallback.rs
index 339241f..3f17171 100644
--- a/engine/src/font/fallback.rs
+++ b/engine/src/font/fallback.rs
@@ -19,6 +19,28 @@ pub struct FontRun {
pub family: String,
}
+/// Resolve controls and combining marks with the preceding base font so
+/// fallback does not split an OpenType shaping cluster.
+pub(crate) fn resolve_family(
+ ch: char,
+ families: &str,
+ weight: u32,
+ italic: bool,
+ previous: Option<&str>,
+ registry: &FontRegistry,
+) -> String {
+ if let Some(family) = previous {
+ let shaping_control = matches!(ch, '\u{200c}' | '\u{200d}' | '\u{fe00}'..='\u{fe0f}' | '\u{e0100}'..='\u{e01ef}');
+ let combining_mark = unicode_bidi::bidi_class(ch) == unicode_bidi::BidiClass::NSM;
+ if shaping_control
+ || (combining_mark && registry.resolve(family, weight, italic).has_char(ch))
+ {
+ return family.to_string();
+ }
+ }
+ registry.resolve_for_char(families, ch, weight, italic).1
+}
+
/// Segment characters into runs by font coverage.
///
/// **Fast path:** when `families` contains no comma, returns a single run
@@ -62,12 +84,19 @@ pub fn segment_by_font(
// Slow path: per-character font resolution
let mut runs = Vec::new();
- let (_, first_family) = registry.resolve_for_char(families, chars[0], weight, italic);
+ let first_family = resolve_family(chars[0], families, weight, italic, None, registry);
let mut current_family = first_family;
let mut run_start = 0;
for (i, &ch) in chars.iter().enumerate().skip(1) {
- let (_, family) = registry.resolve_for_char(families, ch, weight, italic);
+ let family = resolve_family(
+ ch,
+ families,
+ weight,
+ italic,
+ Some(&current_family),
+ registry,
+ );
if family != current_family {
runs.push(FontRun {
start: run_start,
@@ -104,6 +133,16 @@ mod tests {
assert_eq!(runs[0].end, 11);
}
+ #[test]
+ fn shaping_controls_stay_with_resolved_base_font() {
+ let registry = FontRegistry::new();
+ let chars: Vec<char> = "П\u{200d}П\u{fe0f}".chars().collect();
+ let runs = segment_by_font(&chars, "Helvetica, Noto Sans", 400, false, &registry);
+ assert_eq!(runs.len(), 1);
+ assert_eq!(runs[0].family, "Noto Sans");
+ assert_eq!(runs[0].end, chars.len());
+ }
+
#[test]
fn test_empty_input() {
let registry = FontRegistry::new();
diff --git a/engine/src/font/mod.rs b/engine/src/font/mod.rs
index 627c4ea..5efcbc9 100644
--- a/engine/src/font/mod.rs
+++ b/engine/src/font/mod.rs
@@ -74,17 +74,24 @@ impl CustomFontMetrics {
let mut glyph_ids = HashMap::new();
let mut default_advance = 0u16;
- // Sample common characters to build width and glyph ID maps
- for code in 32u32..=0xFFFF {
- if let Some(ch) = char::from_u32(code) {
- if let Some(glyph_id) = face.glyph_index(ch) {
- let advance = face.glyph_hor_advance(glyph_id).unwrap_or(0);
- advance_widths.insert(ch, advance);
- glyph_ids.insert(ch, glyph_id.0);
- if ch == ' ' {
- default_advance = advance;
- }
+ // Enumerate the font's Unicode cmap, including supplementary-plane emoji.
+ if let Some(cmap) = face.tables().cmap {
+ for subtable in cmap.subtables {
+ if !subtable.is_unicode() {
+ continue;
}
+ subtable.codepoints(|code| {
+ if let Some(ch) = char::from_u32(code) {
+ if let Some(glyph_id) = face.glyph_index(ch) {
+ let advance = face.glyph_hor_advance(glyph_id).unwrap_or(0);
+ advance_widths.insert(ch, advance);
+ glyph_ids.insert(ch, glyph_id.0);
+ if ch == ' ' {
+ default_advance = advance;
+ }
+ }
+ }
+ });
}
}
diff --git a/engine/src/layout/audit.rs b/engine/src/layout/audit.rs
index fc716ce..b89b8ca 100644
--- a/engine/src/layout/audit.rs
+++ b/engine/src/layout/audit.rs
@@ -532,6 +532,8 @@ mod tests {
text_decoration: TextDecoration::None,
letter_spacing: 0.0,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
})
.collect();
diff --git a/engine/src/layout/mod.rs b/engine/src/layout/mod.rs
index 47bc59d..0af4a14 100644
--- a/engine/src/layout/mod.rs
+++ b/engine/src/layout/mod.rs
@@ -891,6 +891,9 @@ pub struct PositionedGlyph {
/// For glyphs of a cluster spanning several chars, the full cluster text
/// (e.g., "fi" for an fi ligature). `None` for 1:1 char-to-glyph mappings.
pub cluster_text: Option<String>,
+ /// Text emitted once per source cluster; separate from visual glyph identity.
+ pub extraction_text: Option<String>,
+ pub extraction_advance: Option<f64>,
/// True when this glyph ALONE stands for every char of `cluster_text`: a
/// many-to-one substitution such as the "ffi" ligature. The PDF writer
/// maps such a glyph to its whole cluster in the ToUnicode CMap, so text
@@ -939,12 +942,20 @@ fn shaped_glyph_x(
/// long-standing behaviour: every one of them carries the cluster text when
/// the run has fewer glyphs than chars (so `LayoutInfo` and the render audit
/// see those chars), and none is a ligature.
-fn cluster_texts(shaped: &[shaping::ShapedGlyph], chars: &[char]) -> Vec<(Option<String>, bool)> {
+fn cluster_texts(
+ shaped: &[shaping::ShapedGlyph],
+ chars: &[char],
+) -> Vec<(Option<String>, bool, Option<String>, Option<i32>)> {
let num_chars = chars.len();
let fewer_glyphs_than_chars = shaped.len() < num_chars;
let mut starts: Vec<u32> = shaped.iter().map(|g| g.cluster).collect();
starts.sort_unstable();
+ let mut advances = HashMap::<u32, i32>::new();
+ for glyph in shaped {
+ *advances.entry(glyph.cluster).or_default() += glyph.x_advance;
+ }
+ let mut emitted = std::collections::HashSet::new();
shaped
.iter()
.map(|sg| {
@@ -956,14 +967,35 @@ fn cluster_texts(shaped: &[shaping::ShapedGlyph], chars: &[char]) -> Vec<(Option
.get(past)
.map_or(num_chars, |&c| c as usize)
.min(num_chars);
+ let extraction = if glyphs_in_cluster > 1 || end > start + 1 {
+ Some(if emitted.insert(sg.cluster) {
+ chars[start..end].iter().collect()
+ } else {
+ String::new()
+ })
+ } else {
+ None
+ };
+ let extraction_advance = extraction.as_ref().map(|text| {
+ if text.is_empty() {
+ 0
+ } else {
+ advances[&sg.cluster]
+ }
+ });
if end <= start + 1 {
- return (None, false);
+ return (None, false, extraction, extraction_advance);
}
let ligature = glyphs_in_cluster == 1;
if ligature || fewer_glyphs_than_chars {
- (Some(chars[start..end].iter().collect()), ligature)
+ (
+ Some(chars[start..end].iter().collect()),
+ ligature,
+ extraction,
+ extraction_advance,
+ )
} else {
- (None, false)
+ (None, false, extraction, extraction_advance)
}
})
.collect()
@@ -5394,7 +5426,11 @@ impl LayoutEngine {
let scale = style.font_size / units_per_em as f64;
let clusters = cluster_texts(&shaped, &sub_chars);
- for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) {
+ for (
+ sg,
+ (cluster_text, ligature, extraction_text, extraction_advance),
+ ) in shaped.iter().zip(clusters)
+ {
let cluster = sg.cluster as usize;
let char_value = sub_chars.get(cluster).copied().unwrap_or(' ');
@@ -5417,6 +5453,9 @@ impl LayoutEngine {
text_decoration: style.text_decoration,
letter_spacing: style.letter_spacing,
cluster_text,
+ extraction_text,
+ extraction_advance: extraction_advance
+ .map(|a| a as f64 * scale),
ligature,
});
bidi_levels.push(bidi_run.level);
@@ -5453,6 +5492,8 @@ impl LayoutEngine {
text_decoration: style.text_decoration,
letter_spacing: style.letter_spacing,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
});
bidi_levels.push(bidi_run.level);
@@ -5492,7 +5533,9 @@ impl LayoutEngine {
shaping::shape_text_with_direction(&run_text, font_data, run.is_rtl)
{
let clusters = cluster_texts(&shaped, &run_chars);
- for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) {
+ for (sg, (cluster_text, ligature, extraction_text, extraction_advance)) in
+ shaped.iter().zip(clusters)
+ {
let cluster = sg.cluster as usize;
let char_value = run_chars.get(cluster).copied().unwrap_or(' ');
@@ -5515,6 +5558,8 @@ impl LayoutEngine {
text_decoration: style.text_decoration,
letter_spacing: style.letter_spacing,
cluster_text,
+ extraction_text,
+ extraction_advance: extraction_advance.map(|a| a as f64 * scale),
ligature,
});
bidi_levels.push(run.level);
@@ -5581,6 +5626,8 @@ impl LayoutEngine {
text_decoration: style.text_decoration,
letter_spacing: style.letter_spacing,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
}
})
@@ -5624,40 +5671,8 @@ impl LayoutEngine {
return vec![];
}
- // Pre-resolve per-char font families — the same rule as
- // segment_by_font (the single-style path) and char_width
- // (measurement): the declared family when it covers the char,
- // per-char resolution otherwise. This path used to skip per-char
- // resolution entirely for comma-less families, so a non-WinAnsi
- // char in a TextRun rendered "?" on the base-14 path while the
- // identical char in single-style Text reached builtin Noto Sans —
- // measurement and rendering disagreeing about the char's font.
- let resolved_families: Vec<String> = chars
- .iter()
- .map(|sc| {
- let italic = matches!(sc.font_style, FontStyle::Italic | FontStyle::Oblique);
- if !sc.font_family.contains(',') {
- let primary =
- font_context
- .registry()
- .resolve(&sc.font_family, sc.font_weight, italic);
- if sc.ch.is_whitespace()
- || sc.ch == PAGE_NUMBER_SENTINEL
- || sc.ch == TOTAL_PAGES_SENTINEL
- || primary.has_char(sc.ch)
- {
- return sc.font_family.clone();
- }
- }
- let (_, family) = font_context.registry().resolve_for_char(
- &sc.font_family,
- sc.ch,
- sc.font_weight,
- italic,
- );
- family
- })
- .collect();
+ // Fallback must match line measurement, including shaping controls.
+ let resolved_families = crate::text::resolved_style_families(chars, font_context);
let line_text: String = chars.iter().map(|c| c.ch).collect();
let has_bidi = !bidi::is_pure_ltr(&line_text, direction);
@@ -5776,6 +5791,8 @@ impl LayoutEngine {
text_decoration: sc.text_decoration,
letter_spacing: sc.letter_spacing,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
});
bidi_levels.push(if is_rtl {
@@ -5817,7 +5834,9 @@ impl LayoutEngine {
let mut prev_cluster: Option<usize> = None;
let clusters = cluster_texts(shaped, chars);
- for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) {
+ for (sg, (cluster_text, ligature, extraction_text, extraction_advance)) in
+ shaped.iter().zip(clusters)
+ {
let cluster = sg.cluster as usize;
let char_value = chars.get(cluster).copied().unwrap_or(' ');
@@ -5844,6 +5863,8 @@ impl LayoutEngine {
text_decoration,
letter_spacing,
cluster_text,
+ extraction_text,
+ extraction_advance: extraction_advance.map(|a| a as f64 * scale),
ligature,
});
@@ -5871,7 +5892,9 @@ impl LayoutEngine {
let mut prev_cluster: Option<usize> = None;
let clusters = cluster_texts(shaped, chars);
- for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) {
+ for (sg, (cluster_text, ligature, extraction_text, extraction_advance)) in
+ shaped.iter().zip(clusters)
+ {
let cluster = sg.cluster as usize;
let sc = styled_chars.get(cluster).unwrap_or(&styled_chars[0]);
let char_value = chars.get(cluster).copied().unwrap_or(' ');
@@ -5903,6 +5926,8 @@ impl LayoutEngine {
text_decoration: sc.text_decoration,
letter_spacing: sc.letter_spacing,
cluster_text,
+ extraction_text,
+ extraction_advance: extraction_advance.map(|a| a as f64 * scale),
ligature,
});
@@ -7718,7 +7743,10 @@ impl LayoutEngine {
) as f64;
let clusters = cluster_texts(&shaped_glyphs, &text_chars);
- for (sg, (cluster_text, ligature)) in shaped_glyphs.iter().zip(clusters)
+ for (
+ sg,
+ (cluster_text, ligature, extraction_text, extraction_advance),
+ ) in shaped_glyphs.iter().zip(clusters)
{
let advance = sg.x_advance as f64 / units_per_em * *font_size;
let cluster_idx = sg.cluster as usize;
@@ -7738,6 +7766,9 @@ impl LayoutEngine {
text_decoration: TextDecoration::None,
letter_spacing: style.letter_spacing,
cluster_text,
+ extraction_text,
+ extraction_advance: extraction_advance
+ .map(|a| a as f64 / units_per_em * *font_size),
ligature,
});
x_pos += advance + style.letter_spacing;
@@ -7767,6 +7798,8 @@ impl LayoutEngine {
text_decoration: TextDecoration::None,
letter_spacing: style.letter_spacing,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
});
x_pos += w + style.letter_spacing;
@@ -8204,10 +8237,15 @@ mod tests {
assert_eq!(
got,
vec![
- (None, false),
- (Some("ffi".to_string()), true),
- (None, false),
- (None, false)
+ (None, false, None, None),
+ (
+ Some("ffi".to_string()),
+ true,
+ Some("ffi".to_string()),
+ Some(500)
+ ),
+ (None, false, None, None),
+ (None, false, None, None)
]
);
}
@@ -8223,7 +8261,16 @@ mod tests {
let got = cluster_texts(&shaped, &chars);
assert_eq!(
got,
- vec![(None, false), (Some("bc".to_string()), true), (None, false)]
+ vec![
+ (None, false, None, None),
+ (
+ Some("bc".to_string()),
+ true,
+ Some("bc".to_string()),
+ Some(500)
+ ),
+ (None, false, None, None)
+ ]
);
}
@@ -8236,10 +8283,18 @@ mod tests {
let chars: Vec<char> = "abcd".chars().collect();
let shaped = [sg(1, 0), sg(2, 2), sg(3, 3), sg(4, 3)];
let got = cluster_texts(&shaped, &chars);
- assert_eq!(got[0], (Some("ab".to_string()), true));
- assert_eq!(got[1], (None, false));
- assert_eq!(got[2], (None, false));
- assert_eq!(got[3], (None, false));
+ assert_eq!(
+ got[0],
+ (
+ Some("ab".to_string()),
+ true,
+ Some("ab".to_string()),
+ Some(500)
+ )
+ );
+ assert_eq!(got[1], (None, false, None, None));
+ assert_eq!(got[2], (None, false, Some("d".to_string()), Some(1000)));
+ assert_eq!(got[3], (None, false, Some(String::new()), Some(0)));
}
/// Several glyphs sharing a multi-char cluster: none is a ligature; each
@@ -8251,9 +8306,20 @@ mod tests {
let chars: Vec<char> = "kixy".chars().collect();
let shaped = [sg(1, 0), sg(2, 0), sg(3, 3)];
let got = cluster_texts(&shaped, &chars);
- assert_eq!(got[0], (Some("kix".to_string()), false));
- assert_eq!(got[1], (Some("kix".to_string()), false));
- assert_eq!(got[2], (None, false));
+ assert_eq!(
+ got[0],
+ (
+ Some("kix".to_string()),
+ false,
+ Some("kix".to_string()),
+ Some(1000)
+ )
+ );
+ assert_eq!(
+ got[1],
+ (Some("kix".to_string()), false, Some(String::new()), Some(0))
+ );
+ assert_eq!(got[2], (None, false, None, None));
}
fn make_text(content: &str, font_size: f64) -> Node {
diff --git a/engine/src/pdf/mod.rs b/engine/src/pdf/mod.rs
index 9824651..5da2635 100644
--- a/engine/src/pdf/mod.rs
+++ b/engine/src/pdf/mod.rs
@@ -137,6 +137,113 @@ fn record_glyph_text(glyph_to_text: &mut HashMap<u16, String>, glyph: &Positione
}
}
+fn extraction_cluster(glyph: &PositionedGlyph) -> String {
+ glyph.extraction_text.clone().unwrap_or_else(|| {
+ if glyph.ligature {
+ glyph
+ .cluster_text
+ .clone()
+ .unwrap_or_else(|| glyph.char_value.to_string())
+ } else {
+ glyph.char_value.to_string()
+ }
+ })
+}
+
+fn rtl_cluster(text: &str) -> bool {
+ text.chars().next().is_some_and(|ch| {
+ matches!(
+ unicode_bidi::bidi_class(ch),
+ unicode_bidi::BidiClass::R | unicode_bidi::BidiClass::AL
+ )
+ })
+}
+
+fn extraction_suffix(glyph: &PositionedGlyph) -> Option<String> {
+ let text = extraction_cluster(glyph);
+ if text.chars().nth(1).is_some()
+ && (rtl_cluster(&text)
+ || text
+ .chars()
+ .any(|ch| unicode_bidi::bidi_class(ch) == unicode_bidi::BidiClass::NSM))
+ {
+ Some(text.chars().skip(1).collect())
+ } else {
+ None
+ }
+}
+
+fn suffix_chunks(text: &str) -> Vec<String> {
+ // Keep a joiner with its following source character. Extractors
+ // otherwise discard a CID whose sole Unicode value is a format mark.
+ let mut chunks = Vec::new();
+ let mut pending = String::new();
+ for ch in text.chars() {
+ pending.push(ch);
+ if ch != '\u{200D}' && ch != '\u{200C}' {
+ chunks.push(std::mem::take(&mut pending));
+ }
+ }
+ if !pending.is_empty() {
+ chunks.push(pending);
+ }
+ chunks
+}
+
+#[derive(Default)]
+struct GlyphOutline {
+ path: String,
+ current: (f32, f32),
+}
+
+impl ttf_parser::OutlineBuilder for GlyphOutline {
+ fn move_to(&mut self, x: f32, y: f32) {
+ let _ = writeln!(
+ self.path,
+ "{} {} m",
+ pdf_number(x as f64),
+ pdf_number(y as f64)
+ );
+ self.current = (x, y);
+ }
+ fn line_to(&mut self, x: f32, y: f32) {
+ let _ = writeln!(
+ self.path,
+ "{} {} l",
+ pdf_number(x as f64),
+ pdf_number(y as f64)
+ );
+ self.current = (x, y);
+ }
+ fn quad_to(&mut self, x1: f32, y1: f32, x: f32, y: f32) {
+ let (x0, y0) = self.current;
+ self.curve_to(
+ x0 + (x1 - x0) * 2.0 / 3.0,
+ y0 + (y1 - y0) * 2.0 / 3.0,
+ x + (x1 - x) * 2.0 / 3.0,
+ y + (y1 - y) * 2.0 / 3.0,
+ x,
+ y,
+ );
+ }
+ fn curve_to(&mut self, x1: f32, y1: f32, x2: f32, y2: f32, x: f32, y: f32) {
+ let _ = writeln!(
+ self.path,
+ "{} {} {} {} {} {} c",
+ pdf_number(x1 as f64),
+ pdf_number(y1 as f64),
+ pdf_number(x2 as f64),
+ pdf_number(y2 as f64),
+ pdf_number(x as f64),
+ pdf_number(y as f64)
+ );
+ self.current = (x, y);
+ }
+ fn close(&mut self) {
+ self.path.push_str("h\n");
+ }
+}
+
/// Embedding data for a custom TrueType font.
#[allow(dead_code)]
struct CustomFontEmbedData {
@@ -145,6 +252,8 @@ struct CustomFontEmbedData {
gid_remap: HashMap<u16, u16>,
/// Maps original glyph IDs to the text each stands for (ToUnicode CMap).
glyph_to_text: HashMap<u16, String>,
+ glyph_to_cid: HashMap<(u16, String), u16>,
+ glyph_to_subset: HashMap<u16, u16>,
/// Legacy fallback: maps chars to subset GIDs (for page number placeholders).
char_to_gid: HashMap<char, u16>,
/// The /W widths written for each subset GID (thousandths of an em,
@@ -166,6 +275,8 @@ struct FontUsage {
/// Maps glyph ID → the text it stands for (for ToUnicode CMap): one char
/// for an ordinary glyph, the whole cluster for a ligature ("ffi").
glyph_to_text: HashMap<u16, String>,
+ glyph_texts: HashSet<(u16, String)>,
+ text_widths: HashMap<(u16, String), f64>,
}
/// Tracks allocated PDF objects during writing.
@@ -468,6 +579,8 @@ impl PdfWriter {
tag_builder.as_mut(),
flatten_forms,
);
+ let (content, text_forms) =
+ self.isolate_text_forms(content, page.width, page.height, &mut builder);
let compressed = compress_to_vec_zlib(content.as_bytes(), 6);
let content_obj_id = builder.objects.len();
@@ -503,7 +616,10 @@ impl PdfWriter {
// Build resource dict for this page
let font_resources = self.build_font_resource_dict(&builder.font_objects);
- let xobject_resources = self.build_xobject_resource_dict(page_idx, &builder);
+ let mut xobject_resources = self.build_xobject_resource_dict(page_idx, &builder);
+ for (name, id) in text_forms {
+ let _ = write!(xobject_resources, " /{} {} 0 R", name, id);
+ }
let ext_gstate_resources = self.build_ext_gstate_resource_dict(&builder);
let shading_resources = self.build_shading_resource_dict(page_idx, &builder);
let mut resources = format!("/Font << {} >>", font_resources);
@@ -1575,6 +1691,43 @@ impl PdfWriter {
stream
}
+ /// Use standard Form XObjects for lines that start with RTL combining
+ /// carriers. Readers process each form as one text context, preserving
+ /// the base/mark ordering across unrelated text baselines.
+ fn isolate_text_forms(
+ &self,
+ content: String,
+ width: f64,
+ height: f64,
+ builder: &mut PdfBuilder,
+ ) -> (String, Vec<(String, usize)>) {
+ const START: &str = "%FORME_TEXT_START\n";
+ const END: &str = "%FORME_TEXT_END\n";
+ let mut output = String::new();
+ let mut remaining = content.as_str();
+ let mut forms = Vec::new();
+ while let Some(start) = remaining.find(START) {
+ output.push_str(&remaining[..start]);
+ let after = &remaining[start + START.len()..];
+ let end = after.find(END).expect("text form end marker");
+ let compressed = compress_to_vec_zlib(after[..end].as_bytes(), 6);
+ let id = builder.objects.len();
+ let name = format!("Tx{}", id);
+ let mut data = format!(
+ "<< /Type /XObject /Subtype /Form /BBox [0 0 {:.2} {:.2}] /Resources << /Font << {} >> >> /Length {} /Filter /FlateDecode >>\nstream\n",
+ width, height, self.build_font_resource_dict(&builder.font_objects), compressed.len()
+ ).into_bytes();
+ data.extend_from_slice(&compressed);
+ data.extend_from_slice(b"\nendstream");
+ builder.objects.push(PdfObject { id, data });
+ let _ = writeln!(output, "q\n/{} Do\nQ", name);
+ forms.push((name, id));
+ remaining = &after[end + END.len()..];
+ }
+ output.push_str(remaining);
+ (output, forms)
+ }
+
/// Write a single layout element as PDF operators.
#[allow(clippy::too_many_arguments)]
#[allow(clippy::too_many_arguments)]
@@ -1949,9 +2102,71 @@ impl PdfWriter {
continue;
}
+ // A leading zero-width RTL mark must share an extraction
+ // context with its base. A Form isolates the line from the
+ // previous baseline without adding characters or changing ink.
+ let isolate = !tag_links
+ && line
+ .glyphs
+ .iter()
+ .find(|g| g.extraction_text.as_deref() != Some(""))
+ .is_some_and(|g| {
+ rtl_cluster(&extraction_cluster(g))
+ && extraction_suffix(g).is_some()
+ });
+ if isolate {
+ stream.push_str("%FORME_TEXT_START\n");
+ }
+
// Group consecutive glyphs by (font_family, font_weight, font_style, font_size, color)
// to support multi-font text runs
- let groups = Self::group_glyphs(&line.glyphs, tag_links);
+ for glyph in &line.glyphs {
+ if glyph.extraction_text.as_deref() != Some("") {
+ continue;
+ }
+ let key = FontKey {
+ family: glyph.font_family.to_string(),
+ weight: glyph.font_weight,
+ italic: matches!(
+ glyph.font_style,
+ FontStyle::Italic | FontStyle::Oblique
+ ),
+ };
+ if let Some(embed) = builder.custom_font_data.get(&key) {
+ if let (Ok(face), Some(&gid)) = (
+ ttf_parser::Face::parse(&embed.ttf_data, 0),
+ embed.glyph_to_subset.get(&glyph.glyph_id),
+ ) {
+ let mut outline = GlyphOutline::default();
+ if face
+ .outline_glyph(ttf_parser::GlyphId(gid), &mut outline)
+ .is_some()
+ {
+ let scale = glyph.font_size / face.units_per_em() as f64;
+ let paint = glyph.color.unwrap_or(*color);
+ let _ = writeln!(
+ stream,
+ "q\n{:.3} {:.3} {:.3} rg\n{} 0 0 {} {} {} cm\n{}f\nQ",
+ paint.r,
+ paint.g,
+ paint.b,
+ format!("{:.9}", scale),
+ format!("{:.9}", scale),
+ pdf_number(line.x + glyph.x_offset),
+ pdf_number(page_height - line.y + glyph.y_offset),
+ outline.path
+ );
+ }
+ }
+ }
+ }
+ let text_glyphs: Vec<_> = line
+ .glyphs
+ .iter()
+ .filter(|g| g.extraction_text.as_deref() != Some(""))
+ .cloned()
+ .collect();
+ let groups = Self::group_glyphs(&text_glyphs, tag_links);
let group_links = if tag_links {
Self::group_link_runs(&groups)
} else {
@@ -2077,8 +2292,8 @@ impl PdfWriter {
.iter()
.map(|g| {
embed_data
- .gid_remap
- .get(&g.glyph_id)
+ .glyph_to_cid
+ .get(&(g.glyph_id, extraction_cluster(g)))
.copied()
.unwrap_or_else(|| {
// Fallback: try char→gid
@@ -2224,6 +2439,9 @@ impl PdfWriter {
);
}
}
+ if isolate {
+ stream.push_str("%FORME_TEXT_END\n");
+ }
}
// A paragraph whose last line here is stretched (it goes on
@@ -2566,8 +2784,11 @@ impl PdfWriter {
if let Some(embed_data) = builder.custom_font_data.get(&fk) {
let mut hex = String::new();
for g in group.iter() {
- let gid =
- embed_data.gid_remap.get(&g.glyph_id).copied().unwrap_or(0);
+ let gid = embed_data
+ .glyph_to_cid
+ .get(&(g.glyph_id, extraction_cluster(g)))
+ .copied()
+ .unwrap_or(0);
let _ = write!(hex, "{:04X}", gid);
}
let _ = writeln!(stream, "<{}> Tj", hex);
@@ -3350,6 +3571,8 @@ impl PdfWriter {
let used_glyph_ids = usage.map(|u| &u.glyph_ids);
let used_chars = usage.map(|u| &u.chars);
let glyph_to_text = usage.map(|u| &u.glyph_to_text);
+ let glyph_texts = usage.map(|u| &u.glyph_texts);
+ let text_widths = usage.map(|u| &u.text_widths);
let type0_obj_id = Self::write_custom_font_objects(
builder,
key,
@@ -3357,6 +3580,8 @@ impl PdfWriter {
used_glyph_ids.cloned().unwrap_or_default(),
used_chars.cloned().unwrap_or_default(),
glyph_to_text.cloned().unwrap_or_default(),
+ glyph_texts.cloned().unwrap_or_default(),
+ text_widths.cloned().unwrap_or_default(),
)?;
builder.font_objects.push((key.clone(), type0_obj_id));
}
@@ -3391,6 +3616,8 @@ impl PdfWriter {
chars: HashSet::new(),
glyph_ids: HashSet::new(),
glyph_to_text: HashMap::new(),
+ glyph_texts: HashSet::new(),
+ text_widths: HashMap::new(),
});
usage.chars.insert(glyph.char_value);
// A page-number sentinel becomes digits at write
@@ -3404,6 +3631,23 @@ impl PdfWriter {
}
usage.glyph_ids.insert(glyph.glyph_id);
record_glyph_text(&mut usage.glyph_to_text, glyph);
+ usage.chars.extend(extraction_cluster(glyph).chars());
+ let pair = (glyph.glyph_id, extraction_cluster(glyph));
+ if !pair.1.is_empty() {
+ usage.glyph_texts.insert(pair.clone());
+ }
+ if let Some(suffix) = extraction_suffix(glyph) {
+ for text in suffix_chunks(&suffix) {
+ usage.glyph_texts.insert((0, text.clone()));
+ usage.text_widths.insert((0, text), 0.0);
+ }
+ }
+ if let Some(advance) = glyph.extraction_advance {
+ usage
+ .text_widths
+ .entry(pair)
+ .or_insert(advance / glyph.font_size * 1000.0);
+ }
}
}
}
@@ -4011,6 +4255,8 @@ impl PdfWriter {
used_glyph_ids: HashSet<u16>,
used_chars: HashSet<char>,
glyph_to_text_map: HashMap<u16, String>,
+ mut glyph_texts: HashSet<(u16, String)>,
+ text_widths: HashMap<(u16, String), f64>,
) -> Result<usize, FormeError> {
let face = ttf_parser::Face::parse(ttf_data, 0).map_err(|e| {
FormeError::FontError(format!(
@@ -4050,31 +4296,55 @@ impl PdfWriter {
}
};
- // Build char→new_gid mapping (for placeholder fallback in content stream)
- let char_to_gid: HashMap<char, u16> = char_to_orig_gid
- .iter()
- .filter_map(|(&ch, &orig_gid)| gid_remap.get(&orig_gid).map(|&new_gid| (ch, new_gid)))
- .collect();
-
- // Build glyph_id→new_gid mapping (for shaped content stream)
- let gid_remap_for_embed = gid_remap.clone();
-
- // Build new_gid→text mapping for ToUnicode CMap
- let mut new_gid_to_text: HashMap<u16, String> = HashMap::new();
- // From shaped glyph→text mapping
- for (orig_gid, text) in &glyph_to_text_map {
- if let Some(&new_gid) = gid_remap.get(orig_gid) {
- new_gid_to_text
- .entry(new_gid)
- .or_insert_with(|| text.clone());
- }
+ for (&ch, &gid) in &char_to_orig_gid {
+ glyph_texts.insert((gid, ch.to_string()));
}
- // Fill in from char→gid mapping too
- for (&ch, &new_gid) in &char_to_gid {
- new_gid_to_text
- .entry(new_gid)
- .or_insert_with(|| ch.to_string());
+ let mut pairs: Vec<_> = glyph_texts.into_iter().collect();
+ pairs.sort();
+ let mut glyph_to_cid = HashMap::new();
+ let mut cid_to_gid = HashMap::new();
+ let mut new_gid_to_text = HashMap::new();
+ let mut cid_widths = HashMap::new();
+ let mut gid_remap_for_embed = HashMap::new();
+ for (index, (orig_gid, text)) in pairs.into_iter().enumerate() {
+ let cid = u16::try_from(index + 1)
+ .map_err(|_| FormeError::FontError("Too many character mappings".to_string()))?;
+ let physical_gid = if orig_gid == 0 {
+ text.chars()
+ .last()
+ .and_then(|ch| char_to_orig_gid.get(&ch))
+ .copied()
+ .unwrap_or(0)
+ } else {
+ orig_gid
+ };
+ let new_gid = gid_remap.get(&physical_gid).copied().unwrap_or(0);
+ cid_to_gid.insert(cid, new_gid);
+ if let Some(&width) = text_widths.get(&(orig_gid, text.clone())) {
+ cid_widths.insert(cid, width);
+ }
+ let unicode = if text.chars().nth(1).is_some()
+ && (rtl_cluster(&text)
+ || text
+ .chars()
+ .any(|ch| unicode_bidi::bidi_class(ch) == unicode_bidi::BidiClass::NSM))
+ {
+ text.chars().take(1).collect()
+ } else {
+ text.clone()
+ };
+ new_gid_to_text.insert(cid, unicode);
+ gid_remap_for_embed.entry(orig_gid).or_insert(cid);
+ glyph_to_cid.insert((orig_gid, text), cid);
}
+ let char_to_gid = char_to_orig_gid
+ .iter()
+ .filter_map(|(&ch, &gid)| {
+ glyph_to_cid
+ .get(&(gid, ch.to_string()))
+ .map(|&cid| (ch, cid))
+ })
+ .collect();
let pdf_font_name = Self::sanitize_font_name(&key.family, key.weight, key.italic);
@@ -4136,10 +4406,33 @@ impl PdfWriter {
});
// 3. CIDFont dictionary (DescendantFont)
+ let map_id = builder.objects.len();
+ let mut cid_bytes =
+ vec![0u8; (cid_to_gid.keys().copied().max().unwrap_or(0) as usize + 1) * 2];
+ for (&cid, &gid) in &cid_to_gid {
+ cid_bytes[cid as usize * 2..cid as usize * 2 + 2].copy_from_slice(&gid.to_be_bytes());
+ }
+ let compressed = compress_to_vec_zlib(&cid_bytes, 6);
+ let mut map_data = format!(
+ "<< /Length {} /Filter /FlateDecode >>\nstream\n",
+ compressed.len()
+ )
+ .into_bytes();
+ map_data.extend_from_slice(&compressed);
+ map_data.extend_from_slice(b"\nendstream");
+ builder.objects.push(PdfObject {
+ id: map_id,
+ data: map_data,
+ });
let cidfont_id = builder.objects.len();
// Build /W array using new_gid→width from subset face
- let (w_array, pdf_widths) =
- Self::build_w_array_from_gids(&gid_remap, &subset_face, subset_upem);
+ let (w_array, pdf_widths) = Self::build_w_array_from_gids(
+ &cid_to_gid,
+ &subset_face,
+ subset_upem,
+ &new_gid_to_text,
+ &cid_widths,
+ );
let default_width = subset_face
.glyph_hor_advance(ttf_parser::GlyphId(0))
.map(|adv| (adv as f64 * 1000.0 / subset_upem as f64) as u32)
@@ -4148,8 +4441,8 @@ impl PdfWriter {
"<< /Type /Font /Subtype /CIDFontType2 /BaseFont /{} \
/CIDSystemInfo << /Registry (Adobe) /Ordering (Identity) /Supplement 0 >> \
/FontDescriptor {} 0 R /DW {} /W {} \
- /CIDToGIDMap /Identity >>",
- pdf_font_name, font_descriptor_id, default_width, w_array,
+ /CIDToGIDMap {} 0 R >>",
+ pdf_font_name, font_descriptor_id, default_width, w_array, map_id,
);
builder.objects.push(PdfObject {
id: cidfont_id,
@@ -4194,6 +4487,8 @@ impl PdfWriter {
ttf_data: embed_ttf,
gid_remap: gid_remap_for_embed,
glyph_to_text: glyph_to_text_map,
+ glyph_to_cid,
+ glyph_to_subset: gid_remap,
char_to_gid,
pdf_widths,
default_width,
@@ -4211,25 +4506,33 @@ impl PdfWriter {
gid_remap: &HashMap<u16, u16>,
face: &ttf_parser::Face,
units_per_em: u16,
+ texts: &HashMap<u16, String>,
+ widths: &HashMap<u16, f64>,
) -> (String, HashMap<u16, f64>) {
let scale = 1000.0 / units_per_em as f64;
let mut entries: Vec<(u16, f64)> = Vec::new();
let mut seen_gids: HashSet<u16> = HashSet::new();
- for &new_gid in gid_remap.values() {
- if seen_gids.contains(&new_gid) {
+ for (&cid, &new_gid) in gid_remap {
+ if seen_gids.contains(&cid) {
continue;
}
- seen_gids.insert(new_gid);
+ seen_gids.insert(cid);
let advance = face
.glyph_hor_advance(ttf_parser::GlyphId(new_gid))
.unwrap_or(0);
// Exact, not truncated: a truncated width drew every glyph up to
// 1/1000 em narrower than layout placed it, a drift that grew
// along the line.
- let width = advance as f64 * scale;
- entries.push((new_gid, width));
+ let width = widths.get(&cid).copied().unwrap_or_else(|| {
+ if texts.get(&cid).is_some_and(String::is_empty) {
+ 0.0
+ } else {
+ advance as f64 * scale
+ }
+ });
+ entries.push((cid, width));
}
entries.sort_by_key(|(gid, _)| *gid);
@@ -4255,8 +4558,8 @@ impl PdfWriter {
) -> String {
let mut gid_to_unicode: Vec<(u16, String)> = gid_to_text
.iter()
- .filter(|(_, text)| !text.is_empty())
.map(|(&gid, text)| {
+ let text = text.as_str();
let hex: String = text
.encode_utf16()
.map(|unit| format!("{:04X}", unit))
@@ -4416,7 +4719,10 @@ impl PdfWriter {
}
})
.collect();
- if !any && rises.iter().all(|r| *r == 0.0) {
+ if !any
+ && rises.iter().all(|r| *r == 0.0)
+ && group.iter().all(|g| extraction_suffix(g).is_none())
+ {
return None;
}
// One TJ per run of equal rise; a change of rise (a mark the shaper
@@ -4439,7 +4745,29 @@ impl PdfWriter {
out.push('[');
open = true;
}
+ if rtl_cluster(&extraction_cluster(group[i])) {
+ if let Some(suffix) = extraction_suffix(group[i]) {
+ out.push_str("] TJ\n3 Tr\n0 Tc\n<");
+ for text in suffix_chunks(&suffix).into_iter().rev() {
+ if let Some(cid) = embed.glyph_to_cid.get(&(0, text)) {
+ let _ = write!(out, "{:04X}", cid);
+ }
+ }
+ let _ = write!(out, "> Tj\n0 Tr\n{} Tc\n[", pdf_number(char_spacing));
+ }
+ }
let _ = write!(out, "<{:04X}>", gid);
+ if let Some(suffix) =
+ extraction_suffix(group[i]).filter(|_| !rtl_cluster(&extraction_cluster(group[i])))
+ {
+ out.push_str("] TJ\n3 Tr\n0 Tc\n<");
+ for text in suffix_chunks(&suffix) {
+ if let Some(cid) = embed.glyph_to_cid.get(&(0, text)) {
+ let _ = write!(out, "{:04X}", cid);
+ }
+ }
+ let _ = write!(out, "> Tj\n0 Tr\n{} Tc\n[", pdf_number(char_spacing));
+ }
if adjustments[i] != 0.0 && i + 1 < gids.len() {
let _ = write!(out, " {:.2} ", adjustments[i]);
}
@@ -5542,6 +5870,8 @@ mod tests {
text_decoration: TextDecoration::None,
letter_spacing: 0.0,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
}],
word_spacing: 0.0,
@@ -5590,6 +5920,8 @@ mod tests {
text_decoration: TextDecoration::None,
letter_spacing: 0.0,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
}],
word_spacing: 0.0,
@@ -5748,6 +6080,8 @@ mod tests {
text_decoration: TextDecoration::None,
letter_spacing: 0.0,
cluster_text: cluster.map(str::to_string),
+ extraction_text: None,
+ extraction_advance: None,
ligature,
}
}
@@ -5855,6 +6189,8 @@ mod tests {
text_decoration: TextDecoration::None,
letter_spacing: 0.0,
cluster_text: None,
+ extraction_text: None,
+ extraction_advance: None,
ligature: false,
}],
word_spacing: 0.0,
diff --git a/engine/src/text/mod.rs b/engine/src/text/mod.rs
index d99df7d..3096f64 100644
--- a/engine/src/text/mod.rs
+++ b/engine/src/text/mod.rs
@@ -46,6 +46,35 @@ pub struct StyledChar {
pub word_spacing: f64,
}
+/// The same fallback families are used to measure and position styled text.
+pub(crate) fn resolved_style_families(
+ chars: &[StyledChar],
+ font_context: &FontContext,
+) -> Vec<String> {
+ let mut families: Vec<String> = Vec::with_capacity(chars.len());
+ for (i, sc) in chars.iter().enumerate() {
+ let italic = matches!(sc.font_style, FontStyle::Italic | FontStyle::Oblique);
+ let previous = i
+ .checked_sub(1)
+ .filter(|&j| {
+ let prev = &chars[j];
+ prev.font_family == sc.font_family
+ && prev.font_weight == sc.font_weight
+ && matches!(prev.font_style, FontStyle::Italic | FontStyle::Oblique) == italic
+ })
+ .map(|j| families[j].as_str());
+ families.push(crate::font::fallback::resolve_family(
+ sc.ch,
+ &sc.font_family,
+ sc.font_weight,
+ italic,
+ previous,
+ font_context.registry(),
+ ));
+ }
+ families
+}
+
/// A line of text from multi-style (runs) line breaking.
#[derive(Debug, Clone)]
pub struct RunBrokenLine {
@@ -681,6 +710,7 @@ impl TextLayout {
return vec![];
}
+ let families = resolved_style_families(chars, font_context);
let mut widths = vec![0.0_f64; chars.len()];
let mut i = 0;
@@ -689,8 +719,7 @@ impl TextLayout {
let italic = matches!(sc.font_style, FontStyle::Italic | FontStyle::Oblique);
// Check if this char's font is a custom font with shaping data
- if let Some(font_data) = font_context.font_data(&sc.font_family, sc.font_weight, italic)
- {
+ if let Some(font_data) = font_context.font_data(&families[i], sc.font_weight, italic) {
// Find the end of the contiguous run with the same font
let run_start = i;
let mut run_end = i + 1;
@@ -698,7 +727,7 @@ impl TextLayout {
let next = &chars[run_end];
let next_italic =
matches!(next.font_style, FontStyle::Italic | FontStyle::Oblique);
- if next.font_family == sc.font_family
+ if families[run_end] == families[i]
&& next.font_weight == sc.font_weight
&& next_italic == italic
&& (next.font_size - sc.font_size).abs() < 0.001
@@ -714,7 +743,7 @@ impl TextLayout {
if let Some(shaped) = shaping::shape_text(&run_text, font_data) {
let num_chars = run_end - run_start;
let units_per_em =
- font_context.units_per_em(&sc.font_family, sc.font_weight, italic);
+ font_context.units_per_em(&families[i], sc.font_weight, italic);
let cluster_w = shaping::cluster_widths(
&shaped,
num_chars,
@@ -731,7 +760,7 @@ impl TextLayout {
if ch == PAGE_NUMBER_SENTINEL || ch == TOTAL_PAGES_SENTINEL {
widths[j] = font_context.char_width(
ch,
- &chars[j].font_family,
+ &families[j],
chars[j].font_weight,
italic,
chars[j].font_size,
@@ -748,13 +777,9 @@ impl TextLayout {
}
// Fallback: per-char measurement
- widths[i] = font_context.char_width(
- sc.ch,
- &sc.font_family,
- sc.font_weight,
- italic,
- sc.font_size,
- ) + extra_advance(sc.ch, sc.letter_spacing, sc.word_spacing);
+ widths[i] =
+ font_context.char_width(sc.ch, &families[i], sc.font_weight, italic, sc.font_size)
+ + extra_advance(sc.ch, sc.letter_spacing, sc.word_spacing);
i += 1;
}