diff --git a/engine/src/font/fallback.rs b/engine/src/font/fallback.rs index 339241f..3f17171 100644 --- a/engine/src/font/fallback.rs +++ b/engine/src/font/fallback.rs @@ -19,6 +19,28 @@ pub struct FontRun { pub family: String, } +/// Resolve controls and combining marks with the preceding base font so +/// fallback does not split an OpenType shaping cluster. +pub(crate) fn resolve_family( + ch: char, + families: &str, + weight: u32, + italic: bool, + previous: Option<&str>, + registry: &FontRegistry, +) -> String { + if let Some(family) = previous { + let shaping_control = matches!(ch, '\u{200c}' | '\u{200d}' | '\u{fe00}'..='\u{fe0f}' | '\u{e0100}'..='\u{e01ef}'); + let combining_mark = unicode_bidi::bidi_class(ch) == unicode_bidi::BidiClass::NSM; + if shaping_control + || (combining_mark && registry.resolve(family, weight, italic).has_char(ch)) + { + return family.to_string(); + } + } + registry.resolve_for_char(families, ch, weight, italic).1 +} + /// Segment characters into runs by font coverage. /// /// **Fast path:** when `families` contains no comma, returns a single run @@ -62,12 +84,19 @@ pub fn segment_by_font( // Slow path: per-character font resolution let mut runs = Vec::new(); - let (_, first_family) = registry.resolve_for_char(families, chars[0], weight, italic); + let first_family = resolve_family(chars[0], families, weight, italic, None, registry); let mut current_family = first_family; let mut run_start = 0; for (i, &ch) in chars.iter().enumerate().skip(1) { - let (_, family) = registry.resolve_for_char(families, ch, weight, italic); + let family = resolve_family( + ch, + families, + weight, + italic, + Some(¤t_family), + registry, + ); if family != current_family { runs.push(FontRun { start: run_start, @@ -104,6 +133,16 @@ mod tests { assert_eq!(runs[0].end, 11); } + #[test] + fn shaping_controls_stay_with_resolved_base_font() { + let registry = FontRegistry::new(); + let chars: Vec = "П\u{200d}П\u{fe0f}".chars().collect(); + let runs = segment_by_font(&chars, "Helvetica, Noto Sans", 400, false, ®istry); + assert_eq!(runs.len(), 1); + assert_eq!(runs[0].family, "Noto Sans"); + assert_eq!(runs[0].end, chars.len()); + } + #[test] fn test_empty_input() { let registry = FontRegistry::new(); diff --git a/engine/src/font/mod.rs b/engine/src/font/mod.rs index 627c4ea..5efcbc9 100644 --- a/engine/src/font/mod.rs +++ b/engine/src/font/mod.rs @@ -74,17 +74,24 @@ impl CustomFontMetrics { let mut glyph_ids = HashMap::new(); let mut default_advance = 0u16; - // Sample common characters to build width and glyph ID maps - for code in 32u32..=0xFFFF { - if let Some(ch) = char::from_u32(code) { - if let Some(glyph_id) = face.glyph_index(ch) { - let advance = face.glyph_hor_advance(glyph_id).unwrap_or(0); - advance_widths.insert(ch, advance); - glyph_ids.insert(ch, glyph_id.0); - if ch == ' ' { - default_advance = advance; - } + // Enumerate the font's Unicode cmap, including supplementary-plane emoji. + if let Some(cmap) = face.tables().cmap { + for subtable in cmap.subtables { + if !subtable.is_unicode() { + continue; } + subtable.codepoints(|code| { + if let Some(ch) = char::from_u32(code) { + if let Some(glyph_id) = face.glyph_index(ch) { + let advance = face.glyph_hor_advance(glyph_id).unwrap_or(0); + advance_widths.insert(ch, advance); + glyph_ids.insert(ch, glyph_id.0); + if ch == ' ' { + default_advance = advance; + } + } + } + }); } } diff --git a/engine/src/layout/audit.rs b/engine/src/layout/audit.rs index fc716ce..b89b8ca 100644 --- a/engine/src/layout/audit.rs +++ b/engine/src/layout/audit.rs @@ -532,6 +532,8 @@ mod tests { text_decoration: TextDecoration::None, letter_spacing: 0.0, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }) .collect(); diff --git a/engine/src/layout/mod.rs b/engine/src/layout/mod.rs index 47bc59d..0af4a14 100644 --- a/engine/src/layout/mod.rs +++ b/engine/src/layout/mod.rs @@ -891,6 +891,9 @@ pub struct PositionedGlyph { /// For glyphs of a cluster spanning several chars, the full cluster text /// (e.g., "fi" for an fi ligature). `None` for 1:1 char-to-glyph mappings. pub cluster_text: Option, + /// Text emitted once per source cluster; separate from visual glyph identity. + pub extraction_text: Option, + pub extraction_advance: Option, /// True when this glyph ALONE stands for every char of `cluster_text`: a /// many-to-one substitution such as the "ffi" ligature. The PDF writer /// maps such a glyph to its whole cluster in the ToUnicode CMap, so text @@ -939,12 +942,20 @@ fn shaped_glyph_x( /// long-standing behaviour: every one of them carries the cluster text when /// the run has fewer glyphs than chars (so `LayoutInfo` and the render audit /// see those chars), and none is a ligature. -fn cluster_texts(shaped: &[shaping::ShapedGlyph], chars: &[char]) -> Vec<(Option, bool)> { +fn cluster_texts( + shaped: &[shaping::ShapedGlyph], + chars: &[char], +) -> Vec<(Option, bool, Option, Option)> { let num_chars = chars.len(); let fewer_glyphs_than_chars = shaped.len() < num_chars; let mut starts: Vec = shaped.iter().map(|g| g.cluster).collect(); starts.sort_unstable(); + let mut advances = HashMap::::new(); + for glyph in shaped { + *advances.entry(glyph.cluster).or_default() += glyph.x_advance; + } + let mut emitted = std::collections::HashSet::new(); shaped .iter() .map(|sg| { @@ -956,14 +967,35 @@ fn cluster_texts(shaped: &[shaping::ShapedGlyph], chars: &[char]) -> Vec<(Option .get(past) .map_or(num_chars, |&c| c as usize) .min(num_chars); + let extraction = if glyphs_in_cluster > 1 || end > start + 1 { + Some(if emitted.insert(sg.cluster) { + chars[start..end].iter().collect() + } else { + String::new() + }) + } else { + None + }; + let extraction_advance = extraction.as_ref().map(|text| { + if text.is_empty() { + 0 + } else { + advances[&sg.cluster] + } + }); if end <= start + 1 { - return (None, false); + return (None, false, extraction, extraction_advance); } let ligature = glyphs_in_cluster == 1; if ligature || fewer_glyphs_than_chars { - (Some(chars[start..end].iter().collect()), ligature) + ( + Some(chars[start..end].iter().collect()), + ligature, + extraction, + extraction_advance, + ) } else { - (None, false) + (None, false, extraction, extraction_advance) } }) .collect() @@ -5394,7 +5426,11 @@ impl LayoutEngine { let scale = style.font_size / units_per_em as f64; let clusters = cluster_texts(&shaped, &sub_chars); - for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) { + for ( + sg, + (cluster_text, ligature, extraction_text, extraction_advance), + ) in shaped.iter().zip(clusters) + { let cluster = sg.cluster as usize; let char_value = sub_chars.get(cluster).copied().unwrap_or(' '); @@ -5417,6 +5453,9 @@ impl LayoutEngine { text_decoration: style.text_decoration, letter_spacing: style.letter_spacing, cluster_text, + extraction_text, + extraction_advance: extraction_advance + .map(|a| a as f64 * scale), ligature, }); bidi_levels.push(bidi_run.level); @@ -5453,6 +5492,8 @@ impl LayoutEngine { text_decoration: style.text_decoration, letter_spacing: style.letter_spacing, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }); bidi_levels.push(bidi_run.level); @@ -5492,7 +5533,9 @@ impl LayoutEngine { shaping::shape_text_with_direction(&run_text, font_data, run.is_rtl) { let clusters = cluster_texts(&shaped, &run_chars); - for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) { + for (sg, (cluster_text, ligature, extraction_text, extraction_advance)) in + shaped.iter().zip(clusters) + { let cluster = sg.cluster as usize; let char_value = run_chars.get(cluster).copied().unwrap_or(' '); @@ -5515,6 +5558,8 @@ impl LayoutEngine { text_decoration: style.text_decoration, letter_spacing: style.letter_spacing, cluster_text, + extraction_text, + extraction_advance: extraction_advance.map(|a| a as f64 * scale), ligature, }); bidi_levels.push(run.level); @@ -5581,6 +5626,8 @@ impl LayoutEngine { text_decoration: style.text_decoration, letter_spacing: style.letter_spacing, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, } }) @@ -5624,40 +5671,8 @@ impl LayoutEngine { return vec![]; } - // Pre-resolve per-char font families — the same rule as - // segment_by_font (the single-style path) and char_width - // (measurement): the declared family when it covers the char, - // per-char resolution otherwise. This path used to skip per-char - // resolution entirely for comma-less families, so a non-WinAnsi - // char in a TextRun rendered "?" on the base-14 path while the - // identical char in single-style Text reached builtin Noto Sans — - // measurement and rendering disagreeing about the char's font. - let resolved_families: Vec = chars - .iter() - .map(|sc| { - let italic = matches!(sc.font_style, FontStyle::Italic | FontStyle::Oblique); - if !sc.font_family.contains(',') { - let primary = - font_context - .registry() - .resolve(&sc.font_family, sc.font_weight, italic); - if sc.ch.is_whitespace() - || sc.ch == PAGE_NUMBER_SENTINEL - || sc.ch == TOTAL_PAGES_SENTINEL - || primary.has_char(sc.ch) - { - return sc.font_family.clone(); - } - } - let (_, family) = font_context.registry().resolve_for_char( - &sc.font_family, - sc.ch, - sc.font_weight, - italic, - ); - family - }) - .collect(); + // Fallback must match line measurement, including shaping controls. + let resolved_families = crate::text::resolved_style_families(chars, font_context); let line_text: String = chars.iter().map(|c| c.ch).collect(); let has_bidi = !bidi::is_pure_ltr(&line_text, direction); @@ -5776,6 +5791,8 @@ impl LayoutEngine { text_decoration: sc.text_decoration, letter_spacing: sc.letter_spacing, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }); bidi_levels.push(if is_rtl { @@ -5817,7 +5834,9 @@ impl LayoutEngine { let mut prev_cluster: Option = None; let clusters = cluster_texts(shaped, chars); - for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) { + for (sg, (cluster_text, ligature, extraction_text, extraction_advance)) in + shaped.iter().zip(clusters) + { let cluster = sg.cluster as usize; let char_value = chars.get(cluster).copied().unwrap_or(' '); @@ -5844,6 +5863,8 @@ impl LayoutEngine { text_decoration, letter_spacing, cluster_text, + extraction_text, + extraction_advance: extraction_advance.map(|a| a as f64 * scale), ligature, }); @@ -5871,7 +5892,9 @@ impl LayoutEngine { let mut prev_cluster: Option = None; let clusters = cluster_texts(shaped, chars); - for (sg, (cluster_text, ligature)) in shaped.iter().zip(clusters) { + for (sg, (cluster_text, ligature, extraction_text, extraction_advance)) in + shaped.iter().zip(clusters) + { let cluster = sg.cluster as usize; let sc = styled_chars.get(cluster).unwrap_or(&styled_chars[0]); let char_value = chars.get(cluster).copied().unwrap_or(' '); @@ -5903,6 +5926,8 @@ impl LayoutEngine { text_decoration: sc.text_decoration, letter_spacing: sc.letter_spacing, cluster_text, + extraction_text, + extraction_advance: extraction_advance.map(|a| a as f64 * scale), ligature, }); @@ -7718,7 +7743,10 @@ impl LayoutEngine { ) as f64; let clusters = cluster_texts(&shaped_glyphs, &text_chars); - for (sg, (cluster_text, ligature)) in shaped_glyphs.iter().zip(clusters) + for ( + sg, + (cluster_text, ligature, extraction_text, extraction_advance), + ) in shaped_glyphs.iter().zip(clusters) { let advance = sg.x_advance as f64 / units_per_em * *font_size; let cluster_idx = sg.cluster as usize; @@ -7738,6 +7766,9 @@ impl LayoutEngine { text_decoration: TextDecoration::None, letter_spacing: style.letter_spacing, cluster_text, + extraction_text, + extraction_advance: extraction_advance + .map(|a| a as f64 / units_per_em * *font_size), ligature, }); x_pos += advance + style.letter_spacing; @@ -7767,6 +7798,8 @@ impl LayoutEngine { text_decoration: TextDecoration::None, letter_spacing: style.letter_spacing, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }); x_pos += w + style.letter_spacing; @@ -8204,10 +8237,15 @@ mod tests { assert_eq!( got, vec![ - (None, false), - (Some("ffi".to_string()), true), - (None, false), - (None, false) + (None, false, None, None), + ( + Some("ffi".to_string()), + true, + Some("ffi".to_string()), + Some(500) + ), + (None, false, None, None), + (None, false, None, None) ] ); } @@ -8223,7 +8261,16 @@ mod tests { let got = cluster_texts(&shaped, &chars); assert_eq!( got, - vec![(None, false), (Some("bc".to_string()), true), (None, false)] + vec![ + (None, false, None, None), + ( + Some("bc".to_string()), + true, + Some("bc".to_string()), + Some(500) + ), + (None, false, None, None) + ] ); } @@ -8236,10 +8283,18 @@ mod tests { let chars: Vec = "abcd".chars().collect(); let shaped = [sg(1, 0), sg(2, 2), sg(3, 3), sg(4, 3)]; let got = cluster_texts(&shaped, &chars); - assert_eq!(got[0], (Some("ab".to_string()), true)); - assert_eq!(got[1], (None, false)); - assert_eq!(got[2], (None, false)); - assert_eq!(got[3], (None, false)); + assert_eq!( + got[0], + ( + Some("ab".to_string()), + true, + Some("ab".to_string()), + Some(500) + ) + ); + assert_eq!(got[1], (None, false, None, None)); + assert_eq!(got[2], (None, false, Some("d".to_string()), Some(1000))); + assert_eq!(got[3], (None, false, Some(String::new()), Some(0))); } /// Several glyphs sharing a multi-char cluster: none is a ligature; each @@ -8251,9 +8306,20 @@ mod tests { let chars: Vec = "kixy".chars().collect(); let shaped = [sg(1, 0), sg(2, 0), sg(3, 3)]; let got = cluster_texts(&shaped, &chars); - assert_eq!(got[0], (Some("kix".to_string()), false)); - assert_eq!(got[1], (Some("kix".to_string()), false)); - assert_eq!(got[2], (None, false)); + assert_eq!( + got[0], + ( + Some("kix".to_string()), + false, + Some("kix".to_string()), + Some(1000) + ) + ); + assert_eq!( + got[1], + (Some("kix".to_string()), false, Some(String::new()), Some(0)) + ); + assert_eq!(got[2], (None, false, None, None)); } fn make_text(content: &str, font_size: f64) -> Node { diff --git a/engine/src/pdf/mod.rs b/engine/src/pdf/mod.rs index 9824651..5da2635 100644 --- a/engine/src/pdf/mod.rs +++ b/engine/src/pdf/mod.rs @@ -137,6 +137,113 @@ fn record_glyph_text(glyph_to_text: &mut HashMap, glyph: &Positione } } +fn extraction_cluster(glyph: &PositionedGlyph) -> String { + glyph.extraction_text.clone().unwrap_or_else(|| { + if glyph.ligature { + glyph + .cluster_text + .clone() + .unwrap_or_else(|| glyph.char_value.to_string()) + } else { + glyph.char_value.to_string() + } + }) +} + +fn rtl_cluster(text: &str) -> bool { + text.chars().next().is_some_and(|ch| { + matches!( + unicode_bidi::bidi_class(ch), + unicode_bidi::BidiClass::R | unicode_bidi::BidiClass::AL + ) + }) +} + +fn extraction_suffix(glyph: &PositionedGlyph) -> Option { + let text = extraction_cluster(glyph); + if text.chars().nth(1).is_some() + && (rtl_cluster(&text) + || text + .chars() + .any(|ch| unicode_bidi::bidi_class(ch) == unicode_bidi::BidiClass::NSM)) + { + Some(text.chars().skip(1).collect()) + } else { + None + } +} + +fn suffix_chunks(text: &str) -> Vec { + // Keep a joiner with its following source character. Extractors + // otherwise discard a CID whose sole Unicode value is a format mark. + let mut chunks = Vec::new(); + let mut pending = String::new(); + for ch in text.chars() { + pending.push(ch); + if ch != '\u{200D}' && ch != '\u{200C}' { + chunks.push(std::mem::take(&mut pending)); + } + } + if !pending.is_empty() { + chunks.push(pending); + } + chunks +} + +#[derive(Default)] +struct GlyphOutline { + path: String, + current: (f32, f32), +} + +impl ttf_parser::OutlineBuilder for GlyphOutline { + fn move_to(&mut self, x: f32, y: f32) { + let _ = writeln!( + self.path, + "{} {} m", + pdf_number(x as f64), + pdf_number(y as f64) + ); + self.current = (x, y); + } + fn line_to(&mut self, x: f32, y: f32) { + let _ = writeln!( + self.path, + "{} {} l", + pdf_number(x as f64), + pdf_number(y as f64) + ); + self.current = (x, y); + } + fn quad_to(&mut self, x1: f32, y1: f32, x: f32, y: f32) { + let (x0, y0) = self.current; + self.curve_to( + x0 + (x1 - x0) * 2.0 / 3.0, + y0 + (y1 - y0) * 2.0 / 3.0, + x + (x1 - x) * 2.0 / 3.0, + y + (y1 - y) * 2.0 / 3.0, + x, + y, + ); + } + fn curve_to(&mut self, x1: f32, y1: f32, x2: f32, y2: f32, x: f32, y: f32) { + let _ = writeln!( + self.path, + "{} {} {} {} {} {} c", + pdf_number(x1 as f64), + pdf_number(y1 as f64), + pdf_number(x2 as f64), + pdf_number(y2 as f64), + pdf_number(x as f64), + pdf_number(y as f64) + ); + self.current = (x, y); + } + fn close(&mut self) { + self.path.push_str("h\n"); + } +} + /// Embedding data for a custom TrueType font. #[allow(dead_code)] struct CustomFontEmbedData { @@ -145,6 +252,8 @@ struct CustomFontEmbedData { gid_remap: HashMap, /// Maps original glyph IDs to the text each stands for (ToUnicode CMap). glyph_to_text: HashMap, + glyph_to_cid: HashMap<(u16, String), u16>, + glyph_to_subset: HashMap, /// Legacy fallback: maps chars to subset GIDs (for page number placeholders). char_to_gid: HashMap, /// The /W widths written for each subset GID (thousandths of an em, @@ -166,6 +275,8 @@ struct FontUsage { /// Maps glyph ID → the text it stands for (for ToUnicode CMap): one char /// for an ordinary glyph, the whole cluster for a ligature ("ffi"). glyph_to_text: HashMap, + glyph_texts: HashSet<(u16, String)>, + text_widths: HashMap<(u16, String), f64>, } /// Tracks allocated PDF objects during writing. @@ -468,6 +579,8 @@ impl PdfWriter { tag_builder.as_mut(), flatten_forms, ); + let (content, text_forms) = + self.isolate_text_forms(content, page.width, page.height, &mut builder); let compressed = compress_to_vec_zlib(content.as_bytes(), 6); let content_obj_id = builder.objects.len(); @@ -503,7 +616,10 @@ impl PdfWriter { // Build resource dict for this page let font_resources = self.build_font_resource_dict(&builder.font_objects); - let xobject_resources = self.build_xobject_resource_dict(page_idx, &builder); + let mut xobject_resources = self.build_xobject_resource_dict(page_idx, &builder); + for (name, id) in text_forms { + let _ = write!(xobject_resources, " /{} {} 0 R", name, id); + } let ext_gstate_resources = self.build_ext_gstate_resource_dict(&builder); let shading_resources = self.build_shading_resource_dict(page_idx, &builder); let mut resources = format!("/Font << {} >>", font_resources); @@ -1575,6 +1691,43 @@ impl PdfWriter { stream } + /// Use standard Form XObjects for lines that start with RTL combining + /// carriers. Readers process each form as one text context, preserving + /// the base/mark ordering across unrelated text baselines. + fn isolate_text_forms( + &self, + content: String, + width: f64, + height: f64, + builder: &mut PdfBuilder, + ) -> (String, Vec<(String, usize)>) { + const START: &str = "%FORME_TEXT_START\n"; + const END: &str = "%FORME_TEXT_END\n"; + let mut output = String::new(); + let mut remaining = content.as_str(); + let mut forms = Vec::new(); + while let Some(start) = remaining.find(START) { + output.push_str(&remaining[..start]); + let after = &remaining[start + START.len()..]; + let end = after.find(END).expect("text form end marker"); + let compressed = compress_to_vec_zlib(after[..end].as_bytes(), 6); + let id = builder.objects.len(); + let name = format!("Tx{}", id); + let mut data = format!( + "<< /Type /XObject /Subtype /Form /BBox [0 0 {:.2} {:.2}] /Resources << /Font << {} >> >> /Length {} /Filter /FlateDecode >>\nstream\n", + width, height, self.build_font_resource_dict(&builder.font_objects), compressed.len() + ).into_bytes(); + data.extend_from_slice(&compressed); + data.extend_from_slice(b"\nendstream"); + builder.objects.push(PdfObject { id, data }); + let _ = writeln!(output, "q\n/{} Do\nQ", name); + forms.push((name, id)); + remaining = &after[end + END.len()..]; + } + output.push_str(remaining); + (output, forms) + } + /// Write a single layout element as PDF operators. #[allow(clippy::too_many_arguments)] #[allow(clippy::too_many_arguments)] @@ -1949,9 +2102,71 @@ impl PdfWriter { continue; } + // A leading zero-width RTL mark must share an extraction + // context with its base. A Form isolates the line from the + // previous baseline without adding characters or changing ink. + let isolate = !tag_links + && line + .glyphs + .iter() + .find(|g| g.extraction_text.as_deref() != Some("")) + .is_some_and(|g| { + rtl_cluster(&extraction_cluster(g)) + && extraction_suffix(g).is_some() + }); + if isolate { + stream.push_str("%FORME_TEXT_START\n"); + } + // Group consecutive glyphs by (font_family, font_weight, font_style, font_size, color) // to support multi-font text runs - let groups = Self::group_glyphs(&line.glyphs, tag_links); + for glyph in &line.glyphs { + if glyph.extraction_text.as_deref() != Some("") { + continue; + } + let key = FontKey { + family: glyph.font_family.to_string(), + weight: glyph.font_weight, + italic: matches!( + glyph.font_style, + FontStyle::Italic | FontStyle::Oblique + ), + }; + if let Some(embed) = builder.custom_font_data.get(&key) { + if let (Ok(face), Some(&gid)) = ( + ttf_parser::Face::parse(&embed.ttf_data, 0), + embed.glyph_to_subset.get(&glyph.glyph_id), + ) { + let mut outline = GlyphOutline::default(); + if face + .outline_glyph(ttf_parser::GlyphId(gid), &mut outline) + .is_some() + { + let scale = glyph.font_size / face.units_per_em() as f64; + let paint = glyph.color.unwrap_or(*color); + let _ = writeln!( + stream, + "q\n{:.3} {:.3} {:.3} rg\n{} 0 0 {} {} {} cm\n{}f\nQ", + paint.r, + paint.g, + paint.b, + format!("{:.9}", scale), + format!("{:.9}", scale), + pdf_number(line.x + glyph.x_offset), + pdf_number(page_height - line.y + glyph.y_offset), + outline.path + ); + } + } + } + } + let text_glyphs: Vec<_> = line + .glyphs + .iter() + .filter(|g| g.extraction_text.as_deref() != Some("")) + .cloned() + .collect(); + let groups = Self::group_glyphs(&text_glyphs, tag_links); let group_links = if tag_links { Self::group_link_runs(&groups) } else { @@ -2077,8 +2292,8 @@ impl PdfWriter { .iter() .map(|g| { embed_data - .gid_remap - .get(&g.glyph_id) + .glyph_to_cid + .get(&(g.glyph_id, extraction_cluster(g))) .copied() .unwrap_or_else(|| { // Fallback: try char→gid @@ -2224,6 +2439,9 @@ impl PdfWriter { ); } } + if isolate { + stream.push_str("%FORME_TEXT_END\n"); + } } // A paragraph whose last line here is stretched (it goes on @@ -2566,8 +2784,11 @@ impl PdfWriter { if let Some(embed_data) = builder.custom_font_data.get(&fk) { let mut hex = String::new(); for g in group.iter() { - let gid = - embed_data.gid_remap.get(&g.glyph_id).copied().unwrap_or(0); + let gid = embed_data + .glyph_to_cid + .get(&(g.glyph_id, extraction_cluster(g))) + .copied() + .unwrap_or(0); let _ = write!(hex, "{:04X}", gid); } let _ = writeln!(stream, "<{}> Tj", hex); @@ -3350,6 +3571,8 @@ impl PdfWriter { let used_glyph_ids = usage.map(|u| &u.glyph_ids); let used_chars = usage.map(|u| &u.chars); let glyph_to_text = usage.map(|u| &u.glyph_to_text); + let glyph_texts = usage.map(|u| &u.glyph_texts); + let text_widths = usage.map(|u| &u.text_widths); let type0_obj_id = Self::write_custom_font_objects( builder, key, @@ -3357,6 +3580,8 @@ impl PdfWriter { used_glyph_ids.cloned().unwrap_or_default(), used_chars.cloned().unwrap_or_default(), glyph_to_text.cloned().unwrap_or_default(), + glyph_texts.cloned().unwrap_or_default(), + text_widths.cloned().unwrap_or_default(), )?; builder.font_objects.push((key.clone(), type0_obj_id)); } @@ -3391,6 +3616,8 @@ impl PdfWriter { chars: HashSet::new(), glyph_ids: HashSet::new(), glyph_to_text: HashMap::new(), + glyph_texts: HashSet::new(), + text_widths: HashMap::new(), }); usage.chars.insert(glyph.char_value); // A page-number sentinel becomes digits at write @@ -3404,6 +3631,23 @@ impl PdfWriter { } usage.glyph_ids.insert(glyph.glyph_id); record_glyph_text(&mut usage.glyph_to_text, glyph); + usage.chars.extend(extraction_cluster(glyph).chars()); + let pair = (glyph.glyph_id, extraction_cluster(glyph)); + if !pair.1.is_empty() { + usage.glyph_texts.insert(pair.clone()); + } + if let Some(suffix) = extraction_suffix(glyph) { + for text in suffix_chunks(&suffix) { + usage.glyph_texts.insert((0, text.clone())); + usage.text_widths.insert((0, text), 0.0); + } + } + if let Some(advance) = glyph.extraction_advance { + usage + .text_widths + .entry(pair) + .or_insert(advance / glyph.font_size * 1000.0); + } } } } @@ -4011,6 +4255,8 @@ impl PdfWriter { used_glyph_ids: HashSet, used_chars: HashSet, glyph_to_text_map: HashMap, + mut glyph_texts: HashSet<(u16, String)>, + text_widths: HashMap<(u16, String), f64>, ) -> Result { let face = ttf_parser::Face::parse(ttf_data, 0).map_err(|e| { FormeError::FontError(format!( @@ -4050,31 +4296,55 @@ impl PdfWriter { } }; - // Build char→new_gid mapping (for placeholder fallback in content stream) - let char_to_gid: HashMap = char_to_orig_gid - .iter() - .filter_map(|(&ch, &orig_gid)| gid_remap.get(&orig_gid).map(|&new_gid| (ch, new_gid))) - .collect(); - - // Build glyph_id→new_gid mapping (for shaped content stream) - let gid_remap_for_embed = gid_remap.clone(); - - // Build new_gid→text mapping for ToUnicode CMap - let mut new_gid_to_text: HashMap = HashMap::new(); - // From shaped glyph→text mapping - for (orig_gid, text) in &glyph_to_text_map { - if let Some(&new_gid) = gid_remap.get(orig_gid) { - new_gid_to_text - .entry(new_gid) - .or_insert_with(|| text.clone()); - } + for (&ch, &gid) in &char_to_orig_gid { + glyph_texts.insert((gid, ch.to_string())); } - // Fill in from char→gid mapping too - for (&ch, &new_gid) in &char_to_gid { - new_gid_to_text - .entry(new_gid) - .or_insert_with(|| ch.to_string()); + let mut pairs: Vec<_> = glyph_texts.into_iter().collect(); + pairs.sort(); + let mut glyph_to_cid = HashMap::new(); + let mut cid_to_gid = HashMap::new(); + let mut new_gid_to_text = HashMap::new(); + let mut cid_widths = HashMap::new(); + let mut gid_remap_for_embed = HashMap::new(); + for (index, (orig_gid, text)) in pairs.into_iter().enumerate() { + let cid = u16::try_from(index + 1) + .map_err(|_| FormeError::FontError("Too many character mappings".to_string()))?; + let physical_gid = if orig_gid == 0 { + text.chars() + .last() + .and_then(|ch| char_to_orig_gid.get(&ch)) + .copied() + .unwrap_or(0) + } else { + orig_gid + }; + let new_gid = gid_remap.get(&physical_gid).copied().unwrap_or(0); + cid_to_gid.insert(cid, new_gid); + if let Some(&width) = text_widths.get(&(orig_gid, text.clone())) { + cid_widths.insert(cid, width); + } + let unicode = if text.chars().nth(1).is_some() + && (rtl_cluster(&text) + || text + .chars() + .any(|ch| unicode_bidi::bidi_class(ch) == unicode_bidi::BidiClass::NSM)) + { + text.chars().take(1).collect() + } else { + text.clone() + }; + new_gid_to_text.insert(cid, unicode); + gid_remap_for_embed.entry(orig_gid).or_insert(cid); + glyph_to_cid.insert((orig_gid, text), cid); } + let char_to_gid = char_to_orig_gid + .iter() + .filter_map(|(&ch, &gid)| { + glyph_to_cid + .get(&(gid, ch.to_string())) + .map(|&cid| (ch, cid)) + }) + .collect(); let pdf_font_name = Self::sanitize_font_name(&key.family, key.weight, key.italic); @@ -4136,10 +4406,33 @@ impl PdfWriter { }); // 3. CIDFont dictionary (DescendantFont) + let map_id = builder.objects.len(); + let mut cid_bytes = + vec![0u8; (cid_to_gid.keys().copied().max().unwrap_or(0) as usize + 1) * 2]; + for (&cid, &gid) in &cid_to_gid { + cid_bytes[cid as usize * 2..cid as usize * 2 + 2].copy_from_slice(&gid.to_be_bytes()); + } + let compressed = compress_to_vec_zlib(&cid_bytes, 6); + let mut map_data = format!( + "<< /Length {} /Filter /FlateDecode >>\nstream\n", + compressed.len() + ) + .into_bytes(); + map_data.extend_from_slice(&compressed); + map_data.extend_from_slice(b"\nendstream"); + builder.objects.push(PdfObject { + id: map_id, + data: map_data, + }); let cidfont_id = builder.objects.len(); // Build /W array using new_gid→width from subset face - let (w_array, pdf_widths) = - Self::build_w_array_from_gids(&gid_remap, &subset_face, subset_upem); + let (w_array, pdf_widths) = Self::build_w_array_from_gids( + &cid_to_gid, + &subset_face, + subset_upem, + &new_gid_to_text, + &cid_widths, + ); let default_width = subset_face .glyph_hor_advance(ttf_parser::GlyphId(0)) .map(|adv| (adv as f64 * 1000.0 / subset_upem as f64) as u32) @@ -4148,8 +4441,8 @@ impl PdfWriter { "<< /Type /Font /Subtype /CIDFontType2 /BaseFont /{} \ /CIDSystemInfo << /Registry (Adobe) /Ordering (Identity) /Supplement 0 >> \ /FontDescriptor {} 0 R /DW {} /W {} \ - /CIDToGIDMap /Identity >>", - pdf_font_name, font_descriptor_id, default_width, w_array, + /CIDToGIDMap {} 0 R >>", + pdf_font_name, font_descriptor_id, default_width, w_array, map_id, ); builder.objects.push(PdfObject { id: cidfont_id, @@ -4194,6 +4487,8 @@ impl PdfWriter { ttf_data: embed_ttf, gid_remap: gid_remap_for_embed, glyph_to_text: glyph_to_text_map, + glyph_to_cid, + glyph_to_subset: gid_remap, char_to_gid, pdf_widths, default_width, @@ -4211,25 +4506,33 @@ impl PdfWriter { gid_remap: &HashMap, face: &ttf_parser::Face, units_per_em: u16, + texts: &HashMap, + widths: &HashMap, ) -> (String, HashMap) { let scale = 1000.0 / units_per_em as f64; let mut entries: Vec<(u16, f64)> = Vec::new(); let mut seen_gids: HashSet = HashSet::new(); - for &new_gid in gid_remap.values() { - if seen_gids.contains(&new_gid) { + for (&cid, &new_gid) in gid_remap { + if seen_gids.contains(&cid) { continue; } - seen_gids.insert(new_gid); + seen_gids.insert(cid); let advance = face .glyph_hor_advance(ttf_parser::GlyphId(new_gid)) .unwrap_or(0); // Exact, not truncated: a truncated width drew every glyph up to // 1/1000 em narrower than layout placed it, a drift that grew // along the line. - let width = advance as f64 * scale; - entries.push((new_gid, width)); + let width = widths.get(&cid).copied().unwrap_or_else(|| { + if texts.get(&cid).is_some_and(String::is_empty) { + 0.0 + } else { + advance as f64 * scale + } + }); + entries.push((cid, width)); } entries.sort_by_key(|(gid, _)| *gid); @@ -4255,8 +4558,8 @@ impl PdfWriter { ) -> String { let mut gid_to_unicode: Vec<(u16, String)> = gid_to_text .iter() - .filter(|(_, text)| !text.is_empty()) .map(|(&gid, text)| { + let text = text.as_str(); let hex: String = text .encode_utf16() .map(|unit| format!("{:04X}", unit)) @@ -4416,7 +4719,10 @@ impl PdfWriter { } }) .collect(); - if !any && rises.iter().all(|r| *r == 0.0) { + if !any + && rises.iter().all(|r| *r == 0.0) + && group.iter().all(|g| extraction_suffix(g).is_none()) + { return None; } // One TJ per run of equal rise; a change of rise (a mark the shaper @@ -4439,7 +4745,29 @@ impl PdfWriter { out.push('['); open = true; } + if rtl_cluster(&extraction_cluster(group[i])) { + if let Some(suffix) = extraction_suffix(group[i]) { + out.push_str("] TJ\n3 Tr\n0 Tc\n<"); + for text in suffix_chunks(&suffix).into_iter().rev() { + if let Some(cid) = embed.glyph_to_cid.get(&(0, text)) { + let _ = write!(out, "{:04X}", cid); + } + } + let _ = write!(out, "> Tj\n0 Tr\n{} Tc\n[", pdf_number(char_spacing)); + } + } let _ = write!(out, "<{:04X}>", gid); + if let Some(suffix) = + extraction_suffix(group[i]).filter(|_| !rtl_cluster(&extraction_cluster(group[i]))) + { + out.push_str("] TJ\n3 Tr\n0 Tc\n<"); + for text in suffix_chunks(&suffix) { + if let Some(cid) = embed.glyph_to_cid.get(&(0, text)) { + let _ = write!(out, "{:04X}", cid); + } + } + let _ = write!(out, "> Tj\n0 Tr\n{} Tc\n[", pdf_number(char_spacing)); + } if adjustments[i] != 0.0 && i + 1 < gids.len() { let _ = write!(out, " {:.2} ", adjustments[i]); } @@ -5542,6 +5870,8 @@ mod tests { text_decoration: TextDecoration::None, letter_spacing: 0.0, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }], word_spacing: 0.0, @@ -5590,6 +5920,8 @@ mod tests { text_decoration: TextDecoration::None, letter_spacing: 0.0, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }], word_spacing: 0.0, @@ -5748,6 +6080,8 @@ mod tests { text_decoration: TextDecoration::None, letter_spacing: 0.0, cluster_text: cluster.map(str::to_string), + extraction_text: None, + extraction_advance: None, ligature, } } @@ -5855,6 +6189,8 @@ mod tests { text_decoration: TextDecoration::None, letter_spacing: 0.0, cluster_text: None, + extraction_text: None, + extraction_advance: None, ligature: false, }], word_spacing: 0.0, diff --git a/engine/src/text/mod.rs b/engine/src/text/mod.rs index d99df7d..3096f64 100644 --- a/engine/src/text/mod.rs +++ b/engine/src/text/mod.rs @@ -46,6 +46,35 @@ pub struct StyledChar { pub word_spacing: f64, } +/// The same fallback families are used to measure and position styled text. +pub(crate) fn resolved_style_families( + chars: &[StyledChar], + font_context: &FontContext, +) -> Vec { + let mut families: Vec = Vec::with_capacity(chars.len()); + for (i, sc) in chars.iter().enumerate() { + let italic = matches!(sc.font_style, FontStyle::Italic | FontStyle::Oblique); + let previous = i + .checked_sub(1) + .filter(|&j| { + let prev = &chars[j]; + prev.font_family == sc.font_family + && prev.font_weight == sc.font_weight + && matches!(prev.font_style, FontStyle::Italic | FontStyle::Oblique) == italic + }) + .map(|j| families[j].as_str()); + families.push(crate::font::fallback::resolve_family( + sc.ch, + &sc.font_family, + sc.font_weight, + italic, + previous, + font_context.registry(), + )); + } + families +} + /// A line of text from multi-style (runs) line breaking. #[derive(Debug, Clone)] pub struct RunBrokenLine { @@ -681,6 +710,7 @@ impl TextLayout { return vec![]; } + let families = resolved_style_families(chars, font_context); let mut widths = vec![0.0_f64; chars.len()]; let mut i = 0; @@ -689,8 +719,7 @@ impl TextLayout { let italic = matches!(sc.font_style, FontStyle::Italic | FontStyle::Oblique); // Check if this char's font is a custom font with shaping data - if let Some(font_data) = font_context.font_data(&sc.font_family, sc.font_weight, italic) - { + if let Some(font_data) = font_context.font_data(&families[i], sc.font_weight, italic) { // Find the end of the contiguous run with the same font let run_start = i; let mut run_end = i + 1; @@ -698,7 +727,7 @@ impl TextLayout { let next = &chars[run_end]; let next_italic = matches!(next.font_style, FontStyle::Italic | FontStyle::Oblique); - if next.font_family == sc.font_family + if families[run_end] == families[i] && next.font_weight == sc.font_weight && next_italic == italic && (next.font_size - sc.font_size).abs() < 0.001 @@ -714,7 +743,7 @@ impl TextLayout { if let Some(shaped) = shaping::shape_text(&run_text, font_data) { let num_chars = run_end - run_start; let units_per_em = - font_context.units_per_em(&sc.font_family, sc.font_weight, italic); + font_context.units_per_em(&families[i], sc.font_weight, italic); let cluster_w = shaping::cluster_widths( &shaped, num_chars, @@ -731,7 +760,7 @@ impl TextLayout { if ch == PAGE_NUMBER_SENTINEL || ch == TOTAL_PAGES_SENTINEL { widths[j] = font_context.char_width( ch, - &chars[j].font_family, + &families[j], chars[j].font_weight, italic, chars[j].font_size, @@ -748,13 +777,9 @@ impl TextLayout { } // Fallback: per-char measurement - widths[i] = font_context.char_width( - sc.ch, - &sc.font_family, - sc.font_weight, - italic, - sc.font_size, - ) + extra_advance(sc.ch, sc.letter_spacing, sc.word_spacing); + widths[i] = + font_context.char_width(sc.ch, &families[i], sc.font_weight, italic, sc.font_size) + + extra_advance(sc.ch, sc.letter_spacing, sc.word_spacing); i += 1; }