From 6a204850559a4401b0848573da668f527beb47a0 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:57:54 +0200 Subject: [PATCH 1/6] Pair PDF glyphs with their text through the font cmap and GSUB The ToUnicode map of a plain glyph run paired the i-th glyph with the i-th character of its text. A ligature draws several characters with one glyph, so every later glyph of the run got the character one place further, and the font-wide map, first seen kept, spread that wrong character to every run using the glyph. Carlito, the bundled Calibri, joins ti, fi, ft, ff, ffi and more, so the extracted text of every Calibri document read like "Locatoo Ratog Actoo fieeo ofce". A plain run carries no shaping clusters, and carrying them would change public structs of oxml-layout, whose GlyphRun is built with a struct literal in several crates. So the pairing is rebuilt inside oxml-pdf from the font. A glyph draws the next character when the cmap gives it that character, and the next few when a GSUB ligature joins their glyphs into it, which also splits adjacent ligatures such as the fi and ft of "fifteen". Glyphs the font explains neither way share the characters up to the next glyph it does explain, and when none comes within 16 glyphs and characters, a glyph takes one character by position. A glyph that draws none of its group's characters by itself, such as the mark a shaper adds for a character the font has no glyph for, gets no text, since a plain run has no ActualText and would extract that character twice. The map value becomes a string, written as a multi-character ToUnicode entry, and a glyph keeps its strongest pairing instead of the first one seen. A GSUB ligature comes before the cmap, because Carlito draws "fi" and U+FB01 with one glyph, and one literal U+FB01 must not turn every "fi" of the document into U+FB01. Widths are declared for every drawn glyph, including one that carries no text of its own. GitHub issue #171. --- crates/oxml-pdf/src/font.rs | 567 +++++++++++++++++++++++++++++++++- docs/hld/08-rendering-spec.md | 20 ++ 2 files changed, 572 insertions(+), 15 deletions(-) diff --git a/crates/oxml-pdf/src/font.rs b/crates/oxml-pdf/src/font.rs index 14fcb864e..3e1e6018e 100644 --- a/crates/oxml-pdf/src/font.rs +++ b/crates/oxml-pdf/src/font.rs @@ -1,21 +1,31 @@ //! Font subsetting and ToUnicode CMap generation for PDF embedding. -use std::collections::{BTreeMap, HashMap}; +use std::collections::btree_map::Entry; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use oxml_layout::{FontData, FontId, LayoutResult, PositionedElement, walk}; use pdf_writer::types::{SystemInfo, UnicodeCmap}; use pdf_writer::{Name, Str}; use subsetter::GlyphRemapper; +use ttf_parser::gsub::SubstitutionSubtable; +use ttf_parser::opentype_layout::Coverage; /// Per-font glyph usage collected across all pages. pub(crate) struct FontUsage { - /// Mapping from original glyph ID to the Unicode text it represents. - /// Multiple characters may map to one glyph, so we store the first seen. + /// Mapping from original glyph ID to the Unicode text it draws. + /// + /// A ligature draws several characters with one glyph, so the text is a + /// string. The CMap holds one entry per glyph for the whole font, so when + /// runs pair one glyph with different text, the strongest pairing wins, + /// and the first one seen among equals. /// /// Ordered by glyph ID, because this map is iterated to emit the ToUnicode /// CMap. A hashed order put the same pairs in a different order on every /// run, which made the written PDF differ from itself byte for byte. - pub glyph_to_unicode: BTreeMap, + pub glyph_to_unicode: BTreeMap, + /// Every glyph a run draws. Each declares its width, including a glyph + /// that carries no text of its own. + pub drawn_glyphs: BTreeSet, /// The GlyphRemapper for subsetting. pub remapper: GlyphRemapper, } @@ -29,9 +39,290 @@ pub(crate) struct PreparedFont { pub widths: Vec<(u16, f64)>, // (new_gid, width_in_font_units) } +impl FontUsage { + /// Subset a drawn glyph, declare its width, and map it to its text unless + /// another run paired it more strongly. + fn draw(&mut self, glyph: u16, text: Option<(Pairing, String)>) { + self.remapper.remap(glyph); + self.drawn_glyphs.insert(glyph); + let Some((pairing, text)) = text else { + return; + }; + match self.glyph_to_unicode.entry(glyph) { + Entry::Vacant(entry) => { + entry.insert((pairing, text)); + } + Entry::Occupied(mut entry) if entry.get().0 < pairing => { + entry.insert((pairing, text)); + } + Entry::Occupied(_) => {} + } + } +} + +/// How strongly a glyph was paired with the text it draws, weakest first. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub(crate) enum Pairing { + /// The first of several glyphs that draw some characters between them + /// in no order the font explains. It carries all of those characters. + Group, + /// Paired by position among as many unexplained glyphs as characters. + Position, + /// The one glyph left to draw the characters of its group that no other + /// glyph explains. + Whole, + /// The glyph the font's cmap gives the character. + Nominal, + /// The glyph a GSUB ligature joins the glyphs of several characters into. + /// It outranks the cmap because one glyph can be both. Carlito draws "fi" + /// and U+FB01 with one glyph, and one literal U+FB01 must not turn every + /// "fi" of the document into U+FB01. + Ligature, +} + +/// What a font says about the glyphs a run draws. +struct FontGlyphs<'a> { + face: Option>, + /// Each ligature glyph, with the glyph sequences GSUB joins into it. + ligatures: HashMap>>, +} + +impl<'a> FontGlyphs<'a> { + fn new(font: Option<&'a FontData>) -> Self { + let face = font.and_then(|font| ttf_parser::Face::parse(&font.data, font.face_index).ok()); + let ligatures = face.as_ref().map(ligature_components).unwrap_or_default(); + Self { face, ligatures } + } + + /// The glyph the cmap gives each character, `.notdef` for a character + /// the font lacks, which is also what the shaper draws for it. + fn nominal_glyphs(&self, characters: &[char]) -> Vec> { + characters + .iter() + .map(|&character| { + let face = self.face.as_ref()?; + Some(face.glyph_index(character).map_or(0, |glyph| glyph.0)) + }) + .collect() + } + + /// How many characters, from the start of `nominal`, the glyph draws when + /// it is the cmap glyph of the first or a GSUB ligature of the first few. + fn explain(&self, glyph: u16, nominal: &[Option]) -> Option<(usize, Pairing)> { + if nominal.first() == Some(&Some(glyph)) { + return Some((1, Pairing::Nominal)); + } + let length = self.ligature_length(glyph, nominal, 2)?; + Some((length, Pairing::Ligature)) + } + + fn ligature_length(&self, glyph: u16, nominal: &[Option], depth: u8) -> Option { + let sequences = self.ligatures.get(&glyph)?; + sequences + .iter() + .filter_map(|components| { + let mut length = 0; + for &component in components { + let rest = &nominal[length..]; + if rest.first() == Some(&Some(component)) { + length += 1; + } else if depth > 0 { + length += self.ligature_length(component, rest, depth - 1)?; + } else { + return None; + } + } + Some(length) + }) + .max() + } +} + +/// Every ligature glyph of the font's GSUB table, with the glyph sequences +/// it replaces. +fn ligature_components(face: &ttf_parser::Face<'_>) -> HashMap>> { + let mut ligatures: HashMap>> = HashMap::new(); + let Some(gsub) = face.tables().gsub else { + return ligatures; + }; + for lookup in gsub.lookups { + for subtable in lookup.subtables.into_iter::() { + let SubstitutionSubtable::Ligature(table) = subtable else { + continue; + }; + let first_glyphs: Vec = match table.coverage { + Coverage::Format1 { glyphs } => glyphs.into_iter().collect(), + Coverage::Format2 { records } => records + .into_iter() + .flat_map(|record| record.start.0..=record.end.0) + .map(ttf_parser::GlyphId) + .collect(), + }; + for first in first_glyphs { + let Some(set) = table + .coverage + .get(first) + .and_then(|index| table.ligature_sets.get(index)) + else { + continue; + }; + for ligature in set { + let mut components = vec![first.0]; + components.extend(ligature.components.into_iter().map(|glyph| glyph.0)); + ligatures + .entry(ligature.glyph.0) + .or_default() + .push(components); + } + } + } + } + ligatures +} + +/// How far past an unexplained glyph the pairing looks for the next glyph the +/// font explains. A ligature or a cluster spans a few characters, and the +/// bound keeps a run the font explains nowhere linear. +const RESYNC_WINDOW: usize = 16; + +/// Pair each glyph of a plain run with the characters it draws. +/// +/// A plain run carries its glyphs and its text but not the shaper's clusters, +/// and a ligature draws several characters with one glyph, so pairing them by +/// index shifts every later glyph. The pairing is rebuilt from the font +/// instead. A glyph draws the next character when the cmap gives it that +/// character, and the next few when GSUB joins their glyphs into it. Glyphs +/// the font explains neither way share the characters up to the next glyph +/// it does explain. When there is none within `RESYNC_WINDOW`, an unexplained +/// glyph takes one character by position. A plain run has no `ActualText`, so +/// a glyph that draws none of its group's characters by itself gets no text +/// rather than repeat a character another glyph carries, and so does a glyph +/// left after the last character. +fn pair_plain_run( + characters: &[char], + glyphs: &[u16], + font: &FontGlyphs<'_>, +) -> Vec> { + let nominal = font.nominal_glyphs(characters); + let mut texts = vec![None; glyphs.len()]; + let (mut char_at, mut glyph_at) = (0, 0); + while char_at < characters.len() && glyph_at < glyphs.len() { + if let Some((length, pairing)) = font.explain(glyphs[glyph_at], &nominal[char_at..]) { + let text = characters[char_at..char_at + length].iter().collect(); + texts[glyph_at] = Some((pairing, text)); + char_at += length; + glyph_at += 1; + continue; + } + + let char_limit = characters.len().min(char_at + 1 + RESYNC_WINDOW); + let glyph_limit = glyphs.len().min(glyph_at + 1 + RESYNC_WINDOW); + let next_explained = (glyph_at + 1..glyph_limit).find_map(|glyph_end| { + (char_at + 1..char_limit) + .find(|&char_end| { + font.explain(glyphs[glyph_end], &nominal[char_end..]) + .is_some() + }) + .map(|char_end| (char_end, glyph_end)) + }); + let (char_end, glyph_end) = match next_explained { + Some(end) => end, + None if characters.len() - char_at <= RESYNC_WINDOW + && glyphs.len() - glyph_at <= RESYNC_WINDOW => + { + (characters.len(), glyphs.len()) + } + None => { + texts[glyph_at] = Some((Pairing::Position, characters[char_at].to_string())); + char_at += 1; + glyph_at += 1; + continue; + } + }; + let group = pair_group( + &characters[char_at..char_end], + &glyphs[glyph_at..glyph_end], + &nominal[char_at..char_end], + font, + ); + texts.splice(glyph_at..glyph_end, group); + char_at = char_end; + glyph_at = glyph_end; + } + texts +} + +/// Pair the glyphs of a group with the characters they draw between them, +/// such as one shaping cluster, in which glyph order need not follow text +/// order. +/// +/// A glyph the font explains draws the characters it explains, when no other +/// glyph of the group has taken them. The characters left over go to the one +/// glyph left over, or pair with the glyphs left over by position when they +/// are as many. Otherwise the first glyph left over carries them all, so that +/// none is lost. A glyph still without text draws none of the characters by +/// itself and gets no text, so that no character is paired twice. +fn pair_group( + characters: &[char], + glyphs: &[u16], + nominal: &[Option], + font: &FontGlyphs<'_>, +) -> Vec> { + let mut texts = vec![None; glyphs.len()]; + let mut claimed = vec![false; characters.len()]; + for (text, &glyph) in texts.iter_mut().zip(glyphs) { + let explained = (0..characters.len()).find_map(|start| { + let (length, pairing) = font.explain(glyph, &nominal[start..])?; + let range = start..start + length; + claimed[range.clone()] + .iter() + .all(|taken| !taken) + .then_some((range, pairing)) + }); + if let Some((range, pairing)) = explained { + claimed[range.clone()].fill(true); + *text = Some((pairing, characters[range].iter().collect())); + } + } + let rest_characters = characters + .iter() + .zip(&claimed) + .filter(|(_, claimed)| !**claimed) + .map(|(character, _)| *character) + .collect::>(); + let rest_glyphs = (0..glyphs.len()) + .filter(|&index| texts[index].is_none()) + .collect::>(); + match rest_glyphs.as_slice() { + _ if rest_characters.is_empty() => {} + [] => {} + [only] => texts[*only] = Some((Pairing::Whole, rest_characters.into_iter().collect())), + rest if rest.len() == rest_characters.len() => { + for (&index, character) in rest.iter().zip(rest_characters) { + texts[index] = Some((Pairing::Position, character.to_string())); + } + } + [first, ..] => { + texts[*first] = Some((Pairing::Group, rest_characters.into_iter().collect())); + } + } + texts +} + +fn font_glyphs<'f, 'a>( + fonts: &'f mut HashMap>, + layout: &'a LayoutResult, + font_id: FontId, +) -> &'f FontGlyphs<'a> { + fonts + .entry(font_id) + .or_insert_with(|| FontGlyphs::new(layout.fonts.iter().find(|font| font.id == font_id))) +} + /// Collect glyph usage across all pages for each font. pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap { let mut usage: HashMap = HashMap::new(); + let mut fonts: HashMap> = HashMap::new(); for page in &layout.pages { walk(&page.elements, &mut |element, _| { @@ -40,16 +331,16 @@ pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap = run.text.chars().collect(); - for (i, &gid) in run.glyph_ids.iter().enumerate() { - entry.remapper.remap(gid); - if let Some(&ch) = chars.get(i) { - entry.glyph_to_unicode.entry(gid).or_insert(ch); - } + let font = font_glyphs(&mut fonts, layout, run.font_id); + let texts = pair_plain_run(&chars, &run.glyph_ids, font); + for (&gid, text) in run.glyph_ids.iter().zip(texts) { + entry.draw(gid, text); } } if let PositionedElement::MultilingualText(run) = element @@ -58,6 +349,7 @@ pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap>(); @@ -69,8 +361,7 @@ pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap Optio supplement: 0, }, ); - for (&old_gid, &ch) in &usage.glyph_to_unicode { + for (&old_gid, (_, text)) in &usage.glyph_to_unicode { if let Some(new_gid) = usage.remapper.get(old_gid) { - cmap.pair(new_gid, ch); + cmap.pair_with_multiple(new_gid, text.chars()); } } let cmap_bytes = cmap.finish().to_vec(); @@ -127,7 +418,7 @@ fn compute_glyph_widths(font_data: &FontData, usage: &FontUsage) -> Vec<(u16, f6 let units_per_em = face.units_per_em() as f64; let scale = 1000.0 / units_per_em; - for &old_gid in usage.glyph_to_unicode.keys() { + for &old_gid in &usage.drawn_glyphs { if let Some(new_gid) = usage.remapper.get(old_gid) { let advance = face .glyph_hor_advance(ttf_parser::GlyphId(old_gid)) @@ -186,3 +477,249 @@ pub(crate) fn get_font_metrics(font_data: &FontData) -> Option stem_v, }) } + +#[cfg(test)] +mod tests { + use super::*; + use oxml_layout::{Color, FontManager, GlyphRun, PageFrame, Point}; + + /// The bfchar entries of a ToUnicode CMap, by subset glyph id. + fn to_unicode_entries(cmap: &[u8]) -> HashMap { + let cmap = std::str::from_utf8(cmap).expect("the CMap is ASCII"); + let mut entries = HashMap::new(); + let mut in_bfchar = false; + for line in cmap.lines() { + if line.ends_with("beginbfchar") { + in_bfchar = true; + } else if line == "endbfchar" { + in_bfchar = false; + } else if in_bfchar { + let (glyph, text) = line.split_once(' ').expect("one bfchar pair per line"); + let glyph = u16::from_str_radix(glyph.trim_matches(['<', '>']), 16).unwrap(); + let units = text.trim_matches(['<', '>']).as_bytes().chunks(4); + let units = units + .map(|unit| { + u16::from_str_radix(std::str::from_utf8(unit).unwrap(), 16).unwrap() + }) + .collect::>(); + entries.insert(glyph, String::from_utf16(&units).unwrap()); + } + } + entries + } + + fn plain_run(fonts: &FontManager, font_id: FontId, text: &str) -> PositionedElement { + let shaped = fonts.shape_text(font_id, text, 11.0).unwrap(); + PositionedElement::Text(GlyphRun { + origin: Point { x: 0.0, y: 0.0 }, + font_id, + font_size: 11.0, + glyph_ids: shaped.glyph_ids, + advances: shaped.advances, + text: text.to_owned(), + source: None, + color: Color::BLACK, + bold: false, + italic: false, + field_kind: None, + field_source: None, + note: None, + }) + } + + fn layout_of(fonts: &FontManager, elements: Vec) -> LayoutResult { + LayoutResult::new( + vec![PageFrame::new(1, 612.0, 792.0, elements).into()], + fonts.all_font_data(), + None, + vec![], + ) + } + + /// Each font's subset remapper and ToUnicode entries, as a reader sees them. + fn embedded_text_maps( + layout: &LayoutResult, + ) -> HashMap)> { + let mut usage = collect_glyph_usage(layout); + layout + .fonts + .iter() + .filter_map(|font| { + let prepared = prepare_font(font, usage.get_mut(&font.id)?)?; + let entries = to_unicode_entries(&prepared.cmap_bytes); + Some((font.id, (prepared.remapper, entries))) + }) + .collect() + } + + /// The text a reader extracts from each glyph through the ToUnicode CMap. + fn glyph_texts( + maps: &HashMap)>, + font_id: FontId, + glyph_ids: &[u16], + ) -> Vec { + let (remapper, entries) = &maps[&font_id]; + glyph_ids + .iter() + .map(|glyph| { + let subset_glyph = remapper.get(*glyph).expect("every drawn glyph is subset"); + entries.get(&subset_glyph).cloned().unwrap_or_default() + }) + .collect() + } + + #[test] + fn a_ligature_glyph_maps_to_every_character_it_draws() { + let mut fonts = FontManager::new_deterministic().unwrap(); + for bold in [false, true] { + let font_id = fonts.resolve_font(Some("Calibri"), bold, false).unwrap(); + // Ligature runs come first. Pairing glyphs and characters by index + // shifted every later glyph of such a run, and the font-wide map + // then kept that wrong character in the runs without a ligature. + let words = [ + "Location ", + "fifteen ", + "office ", + "attitude ", + "affluent ", + "fjord ", + "staff ", + "shifting ", + "Rating ", + "Action ", + "Observation ", + "no ligature here", + ]; + let layout = layout_of( + &fonts, + words + .iter() + .map(|word| plain_run(&fonts, font_id, word)) + .collect(), + ); + let maps = embedded_text_maps(&layout); + for element in &layout.pages[0].elements { + let PositionedElement::Text(run) = element else { + unreachable!("the page holds plain runs only"); + }; + assert_eq!( + glyph_texts(&maps, font_id, &run.glyph_ids).concat(), + run.text, + "bold: {bold}" + ); + } + for ligature in ["ti", "fi", "ft", "ffi", "ffl", "tti", "fj", "ff"] { + let shaped = fonts.shape_text(font_id, ligature, 11.0).unwrap(); + assert_eq!(shaped.glyph_ids.len(), 1, "Carlito joins {ligature}"); + assert_eq!( + glyph_texts(&maps, font_id, &shaped.glyph_ids), + [ligature], + "bold: {bold}" + ); + } + } + } + + #[test] + fn a_ligature_outranks_the_presentation_form_sharing_its_glyph() { + let mut fonts = FontManager::new_deterministic().unwrap(); + let font_id = fonts.resolve_font(Some("Calibri"), false, false).unwrap(); + // Carlito draws its fi ligature with the cmap glyph of U+FB01, and the + // CMap keeps one text per glyph. A literal U+FB01 seen first used to + // claim that glyph and turn every "fi" of the document into U+FB01. + let fi = fonts.shape_text(font_id, "fi", 11.0).unwrap().glyph_ids; + let form = fonts.shape_text(font_id, "\u{fb01}", 11.0).unwrap(); + assert_eq!(form.glyph_ids, fi, "Carlito shares the glyph"); + let extracted = |texts: &[&str]| { + let runs = texts.iter().map(|text| plain_run(&fonts, font_id, text)); + let layout = layout_of(&fonts, runs.collect()); + let maps = embedded_text_maps(&layout); + let runs = layout.pages[0].elements.iter().map(|element| { + let PositionedElement::Text(run) = element else { + unreachable!("the page holds plain runs only"); + }; + glyph_texts(&maps, font_id, &run.glyph_ids).concat() + }); + runs.collect::>() + }; + assert_eq!(extracted(&["office fifteen"]), ["office fifteen"]); + assert_eq!( + extracted(&["\u{fb01}", "office fifteen"]), + ["fi", "office fifteen"] + ); + // Without a ligature to decompose, the glyph keeps the literal. + assert_eq!(extracted(&["\u{fb01}"]), ["\u{fb01}"]); + } + + #[test] + fn a_plain_run_extracts_each_character_once() { + let mut fonts = FontManager::new_deterministic().unwrap(); + // Liberation Sans has no "≮" or "≯" and Carlito no "Ѷ", so the shaper + // draws each as a base glyph and a combining mark, neither of them the + // cmap glyph of the character. A plain run has no ActualText, so the + // mark must not repeat the character its base glyph carries. + let mut elements = Vec::new(); + for (family, text) in [("Arial", "a ≮ b ≯ c"), ("Calibri", "Ѷ")] { + let font_id = fonts.resolve_font(Some(family), false, false).unwrap(); + elements.push(plain_run(&fonts, font_id, text)); + } + let layout = layout_of(&fonts, elements); + let maps = embedded_text_maps(&layout); + for element in &layout.pages[0].elements { + let PositionedElement::Text(run) = element else { + unreachable!("the page holds plain runs only"); + }; + let characters = run.text.chars().count(); + assert!( + run.glyph_ids.len() > characters, + "{:?} draws a mark", + run.text + ); + let texts = glyph_texts(&maps, run.font_id, &run.glyph_ids); + assert_eq!(texts.concat(), run.text); + } + } + + #[test] + fn a_cmap_pairing_outranks_one_inferred_earlier() { + let mut fonts = FontManager::new_deterministic().unwrap(); + let font_id = fonts.resolve_font(Some("Calibri"), false, false).unwrap(); + // Carlito has no no-break space, so the shaper draws the space glyph + // for it. Seen first, that pairing used to claim the glyph for the + // whole font. + let layout = layout_of( + &fonts, + vec![ + plain_run(&fonts, font_id, "a\u{a0}b"), + plain_run(&fonts, font_id, "a b"), + ], + ); + let space = fonts.shape_text(font_id, " ", 11.0).unwrap().glyph_ids; + let texts = glyph_texts(&embedded_text_maps(&layout), font_id, &space); + assert_eq!(texts, [" "]); + } + + #[test] + fn a_run_without_ligatures_keeps_one_character_per_glyph() { + let mut fonts = FontManager::new_deterministic().unwrap(); + let font_id = fonts.resolve_font(Some("Arial"), false, false).unwrap(); + let text = "Location Rating Action fifteen office"; + let layout = layout_of(&fonts, vec![plain_run(&fonts, font_id, text)]); + let PositionedElement::Text(run) = &layout.pages[0].elements[0] else { + unreachable!("the page holds one plain run"); + }; + assert_eq!(run.glyph_ids.len(), text.chars().count()); + let maps = embedded_text_maps(&layout); + let texts = glyph_texts(&maps, font_id, &run.glyph_ids); + let characters = text.chars().map(String::from).collect::>(); + assert_eq!(texts, characters); + + let mut usage = collect_glyph_usage(&layout); + let font = layout.fonts.iter().find(|font| font.id == font_id).unwrap(); + let prepared = prepare_font(font, usage.get_mut(&font_id).unwrap()).unwrap(); + let mut drawn = run.glyph_ids.clone(); + drawn.sort_unstable(); + drawn.dedup(); + assert_eq!(prepared.widths.len(), drawn.len()); + } +} diff --git a/docs/hld/08-rendering-spec.md b/docs/hld/08-rendering-spec.md index abd0629e2..9c2e6fe53 100644 --- a/docs/hld/08-rendering-spec.md +++ b/docs/hld/08-rendering-spec.md @@ -297,6 +297,26 @@ with no visible difference and no failing test. Two writes of one document cannot detect that, since they reuse the same map instances. The regression builds two documents and compares their bytes. +**The text layer follows what each glyph draws.** A `GlyphRun` carries its +glyphs and its text but not the shaper's clusters, and a ligature draws several +characters with one glyph, so pairing them by index shifts every later glyph. +The writer reads the pairing from the font instead. A glyph draws the next +character when the cmap gives it that character, and the next few when a GSUB +ligature joins their glyphs into it, so the Carlito `ti` or `ffi` of a Calibri +document extracts as its characters. Glyphs the font explains neither way share +the characters up to the next glyph it does explain, looked for within 16 +glyphs and characters. When there is none, an unexplained glyph takes one +character by position, which keeps the pairing linear in a run the font +explains nowhere. A plain run has no `ActualText`, so there a glyph that draws +none of its group's characters by itself adds no ToUnicode text rather than +repeat a character another glyph carries, and neither does a glyph left after +the run's last character. The ToUnicode CMap holds one entry per glyph for the +whole font, so a glyph that draws different text in different places keeps its +strongest pairing, a GSUB ligature first, then the cmap, then an inferred +pairing. The ligature comes first because one glyph can be both, as the Carlito +`fi` ligature and U+FB01 are, and one literal U+FB01 must not turn every `fi` +of the document into U+FB01. + When `LayoutResult::structure` is present, the writer emits deterministic `BDC` and `EMC` pairs with page-local MCIDs, `/StructParents`, one parent number tree, `/StructTreeRoot`, `/MarkInfo`, an undetermined `/Lang`, and accessible From 16d1c8d3bcd4e58fe14acc82b59d098f2838557d Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:58:21 +0200 Subject: [PATCH 2/6] Keep every character of a rich cluster in its ToUnicode entries A MultilingualText run carries the shaper's clusters, but the ToUnicode map took only the first character of each cluster and gave it to every glyph of the cluster. A Devanagari conjunct that draws three characters with one glyph lost two of them, and the mark glyph of a cluster took its base character, which could then stand for that base in the whole font. Readers that honour the run's ActualText did not see it, but the map is what every other reader and search index uses. Each cluster now goes through the pairing plain runs use, so its characters go to the glyphs the font says draw them, whatever their visual order. A glyph that draws none of them by itself, such as the dots an Arabic font draws apart from their letter, repeats the text of its cluster at the weakest pairing, so every glyph of a rich run still maps to Unicode as it did before. The run's ActualText covers that repetition. A plain run has none, which is why it leaves such a glyph without text. GitHub issue #171. --- crates/oxml-pdf/src/font.rs | 139 ++++++++++++++++++++++++++++++++-- docs/hld/08-rendering-spec.md | 17 +++-- 2 files changed, 143 insertions(+), 13 deletions(-) diff --git a/crates/oxml-pdf/src/font.rs b/crates/oxml-pdf/src/font.rs index 3e1e6018e..89976cf3a 100644 --- a/crates/oxml-pdf/src/font.rs +++ b/crates/oxml-pdf/src/font.rs @@ -63,6 +63,11 @@ impl FontUsage { /// How strongly a glyph was paired with the text it draws, weakest first. #[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] pub(crate) enum Pairing { + /// A glyph of a rich run that draws none of its cluster's characters by + /// itself, such as the dots an Arabic font draws apart from their letter. + /// It repeats the cluster's text, which the run's `ActualText` covers, so + /// that every glyph of a rich run maps to Unicode. + Shared, /// The first of several glyphs that draw some characters between them /// in no order the font explains. It carries all of those characters. Group, @@ -353,15 +358,25 @@ pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap>(); + let font = font_glyphs(&mut fonts, layout, run.font_id); for cluster in &run.clusters { - let Some(&character) = chars.get(cluster.char_start as usize) else { + let characters = + chars.get(cluster.char_start as usize..cluster.char_end as usize); + let glyphs = run + .glyph_ids + .get(cluster.glyph_start as usize..cluster.glyph_end as usize); + let (Some(characters), Some(glyphs)) = (characters, glyphs) else { continue; }; - for glyph in cluster.glyph_start..cluster.glyph_end { - let Some(&gid) = run.glyph_ids.get(glyph as usize) else { - continue; - }; - entry.draw(gid, Some((Pairing::Position, character.to_string()))); + let nominal = font.nominal_glyphs(characters); + let texts = pair_group(characters, glyphs, &nominal, font); + // The run's ActualText carries its text, so a glyph that + // draws none of the cluster's characters by itself repeats + // them all, and every glyph of the run maps to Unicode. + let cluster_text = characters.iter().collect::(); + for (&gid, text) in glyphs.iter().zip(texts) { + let text = text.unwrap_or_else(|| (Pairing::Shared, cluster_text.clone())); + entry.draw(gid, Some(text)); } } } @@ -481,7 +496,10 @@ pub(crate) fn get_font_metrics(font_data: &FontData) -> Option #[cfg(test)] mod tests { use super::*; - use oxml_layout::{Color, FontManager, GlyphRun, PageFrame, Point}; + use oxml_layout::{ + Color, FontManager, GlyphRun, MultilingualGlyphRun, PageFrame, Point, TextDirection, + TextSegment, + }; /// The bfchar entries of a ToUnicode CMap, by subset glyph id. fn to_unicode_entries(cmap: &[u8]) -> HashMap { @@ -527,6 +545,66 @@ mod tests { }) } + fn multilingual_runs(fonts: &mut FontManager, text: &str) -> Vec { + let font_id = fonts.resolve_font(Some("Calibri"), false, false).unwrap(); + let segment = TextSegment { + text: text.to_owned(), + direction: TextDirection::Auto, + source: None, + font_id, + font_size: 11.0, + glyph_ids: Vec::new(), + advances: Vec::new(), + width: 0.0, + ascent: 0.0, + descent: 0.0, + line_gap: 0.0, + color: Color::BLACK, + bold: false, + italic: false, + underline: None, + strike: false, + dstrike: false, + highlight: None, + baseline_offset: 0.0, + hyperlink_url: None, + field_kind: None, + field_source: None, + note: None, + }; + fonts + .shape_multilingual_text(segment, None, TextDirection::Auto, false) + .unwrap() + .into_iter() + .map(|span| { + PositionedElement::MultilingualText(MultilingualGlyphRun { + origin: Point { x: 0.0, y: 0.0 }, + font_id: span.font_id(), + font_size: 11.0, + glyph_ids: span.glyph_ids().to_vec(), + x_advances: span.x_advances().to_vec(), + y_advances: span.y_advances().to_vec(), + x_offsets: span.x_offsets().to_vec(), + y_offsets: span.y_offsets().to_vec(), + clusters: span.clusters().to_vec(), + logical_text: span.text().to_owned(), + logical_index: span.logical_index(), + source: None, + script: span.script(), + language: None, + direction: span.direction(), + bidi_level: span.bidi_level(), + color: Color::BLACK, + bold: false, + italic: false, + field_kind: None, + field_source: None, + note: None, + }) + }) + .collect() + } + fn layout_of(fonts: &FontManager, elements: Vec) -> LayoutResult { LayoutResult::new( vec![PageFrame::new(1, 612.0, 792.0, elements).into()], @@ -699,6 +777,53 @@ mod tests { assert_eq!(texts, [" "]); } + #[test] + fn a_multilingual_cluster_keeps_every_character() { + let mut fonts = FontManager::new_deterministic().unwrap(); + // The conjunct of "क्ष" draws three characters with one glyph. The + // cluster "त्रि" draws four with two, its vowel sign first and in a + // width variant the cmap does not give, and "कि" then pairs that + // variant with the vowel sign alone. The Arabic font draws the dot of + // "ب" apart from the letter, and draws contextual forms. + let mut elements = multilingual_runs(&mut fonts, "क्षत्रिय कि"); + elements.extend(multilingual_runs(&mut fonts, "سلام ب")); + let layout = layout_of(&fonts, elements); + let maps = embedded_text_maps(&layout); + let mut clusters = Vec::new(); + for element in &layout.pages[0].elements { + let PositionedElement::MultilingualText(run) = element else { + unreachable!("the page holds multilingual runs only"); + }; + let texts = glyph_texts(&maps, run.font_id, &run.glyph_ids); + let characters = run.logical_text.chars().collect::>(); + for cluster in &run.clusters { + let glyphs = cluster.glyph_start as usize..cluster.glyph_end as usize; + let source = &characters[cluster.char_start as usize..cluster.char_end as usize]; + let texts = texts[glyphs].to_vec(); + assert!(texts.iter().all(|text| !text.is_empty()), "{texts:?}"); + let mut extracted = texts.concat(); + for character in source { + let at = extracted.find(*character).unwrap_or_else(|| { + panic!("{character:?} of {source:?} is missing from {texts:?}") + }); + extracted.remove(at); + } + clusters.push((source.iter().collect::(), texts)); + } + } + let texts_of = |source: &str| { + clusters + .iter() + .find(|(cluster, _)| cluster == source) + .map(|(_, texts)| texts.clone()) + .unwrap_or_else(|| panic!("no cluster {source:?} in {clusters:?}")) + }; + assert_eq!(texts_of("क्ष"), ["क्ष"]); + assert_eq!(texts_of("त्रि"), ["ि", "त्र"]); + assert_eq!(texts_of("कि"), ["ि", "क"]); + assert_eq!(texts_of("ب"), ["ب", "ب"]); + } + #[test] fn a_run_without_ligatures_keeps_one_character_per_glyph() { let mut fonts = FontManager::new_deterministic().unwrap(); diff --git a/docs/hld/08-rendering-spec.md b/docs/hld/08-rendering-spec.md index 9c2e6fe53..d04b58233 100644 --- a/docs/hld/08-rendering-spec.md +++ b/docs/hld/08-rendering-spec.md @@ -310,12 +310,17 @@ character by position, which keeps the pairing linear in a run the font explains nowhere. A plain run has no `ActualText`, so there a glyph that draws none of its group's characters by itself adds no ToUnicode text rather than repeat a character another glyph carries, and neither does a glyph left after -the run's last character. The ToUnicode CMap holds one entry per glyph for the -whole font, so a glyph that draws different text in different places keeps its -strongest pairing, a GSUB ligature first, then the cmap, then an inferred -pairing. The ligature comes first because one glyph can be both, as the Carlito -`fi` ligature and U+FB01 are, and one literal U+FB01 must not turn every `fi` -of the document into U+FB01. +the run's last character. A rich run's clusters give their characters to their +glyphs the same way, so no character of a cluster is lost. In a rich run, a +glyph that draws none of them by itself, such as the dots an Arabic font draws +apart from their letter, repeats the text of its cluster, which the run's +`ActualText` covers, so every glyph of a rich run maps to Unicode. The +ToUnicode CMap holds one entry per glyph for the whole font, so a glyph that +draws different text in different places keeps its strongest pairing, a GSUB +ligature first, then the cmap, then an inferred pairing. The ligature comes +first because one glyph can be both, as the Carlito `fi` ligature and U+FB01 +are, and one literal U+FB01 must not turn every `fi` of the document into +U+FB01. When `LayoutResult::structure` is present, the writer emits deterministic `BDC` and `EMC` pairs with page-local MCIDs, `/StructParents`, one parent number From 524a7c2559e1aac47c59a44b08c56527271a4030 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:58:31 +0200 Subject: [PATCH 3/6] Test that rpptx plain runs keep their ligatures in PDF text The issue reported rpptx as unaffected, because its deck inherits rtl="0" from the master, which sends every paragraph down the rich path with clusters and ActualText. A paragraph with no inherited direction draws its Latin text as plain runs, and their Carlito ligatures garbled the PDF text as in rdocx. The test builds a text box in a deck whose presentation and master declare no rtl, checks that it takes the plain path, and decodes the PDF text with lopdf, and with pdftotext when it is installed. GitHub issue #171. --- crates/rpptx/tests/integration.rs | 65 +++++++++++++++++++++++++++++++ 1 file changed, 65 insertions(+) diff --git a/crates/rpptx/tests/integration.rs b/crates/rpptx/tests/integration.rs index cbf989468..20829956e 100644 --- a/crates/rpptx/tests/integration.rs +++ b/crates/rpptx/tests/integration.rs @@ -598,6 +598,71 @@ fn pdf_import_differential_rejects_geometry_text_link_and_pixel_perturbations() )); } +/// GitHub issue #171. Without an inherited `rtl` a paragraph draws its Latin +/// text as plain runs, which carry no shaping clusters, so a Carlito ligature +/// reaches the PDF text layer only through the ToUnicode map. +#[test] +#[cfg(feature = "render")] +fn plain_run_ligatures_keep_their_characters_in_pdf_text() { + const TEXT: &str = "Location Rating Action fifteen office Observation"; + let mut source = Presentation::new().unwrap(); + source.add_slide(0).unwrap(); + source + .slide_mut(0) + .unwrap() + .add_textbox(Emu(457_200), Emu(914_400), Emu(8_229_600), Emu(914_400)) + .unwrap() + .set_text(TEXT) + .unwrap(); + let mut package = open_opc(&source.to_bytes().unwrap(), "#171 ligature deck"); + for part in [ + "/ppt/presentation.xml", + "/ppt/slideMasters/slideMaster1.xml", + ] { + let xml = String::from_utf8(package.get_part(part).unwrap().to_vec()).unwrap(); + assert!(xml.contains(r#" rtl="0""#), "{part} sets the direction"); + package.set_part(part, xml.replace(r#" rtl="0""#, "").into_bytes()); + } + let presentation = Presentation::from_bytes(&package_bytes(package)).unwrap(); + + let (_, layout) = presentation.render_deterministic().unwrap(); + let mut plain_text = String::new(); + walk(&layout.pages[0].elements, &mut |element, _| match element { + PositionedElement::Text(run) => plain_text.push_str(&run.text), + PositionedElement::MultilingualText(run) => { + panic!("{:?} took the path with clusters", run.logical_text) + } + _ => {} + }); + assert_eq!(plain_text.trim_end(), TEXT); + + // lopdf decodes each drawn run through the ToUnicode map on its own line. + let pdf = presentation.to_pdf_deterministic().unwrap(); + let decoded = lopdf::Document::load_mem(&pdf) + .unwrap() + .extract_text(&[1]) + .unwrap(); + assert_eq!( + decoded.split_whitespace().collect::>().join(" "), + TEXT + ); + + let path = std::env::temp_dir().join(format!("rpptx-171-ligatures-{}.pdf", std::process::id())); + fs::write(&path, &pdf).unwrap(); + let extracted = Command::new("pdftotext").arg(&path).arg("-").output(); + fs::remove_file(&path).unwrap(); + match extracted { + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + eprintln!("#171 pdftotext check skipped because pdftotext is absent"); + } + extracted => { + let extracted = extracted.unwrap(); + assert!(extracted.status.success()); + assert_eq!(String::from_utf8(extracted.stdout).unwrap().trim(), TEXT); + } + } +} + #[test] #[cfg(feature = "default-template")] #[ignore = "requires Google Chrome 152.0.7977.65"] From cf5e8950b1ec510271837a93ed47b399b3a91568 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:58:31 +0200 Subject: [PATCH 4/6] Test PDF text of every bundled family, regular and bold The issue's acceptance is that pdftotext returns the text of a rendered document for every bundled family, regular and bold, ligatures included. Nothing tested the meaning of the ToUnicode maps, so the hash harness baselined the wrong ones. The test renders the issue's sentence in Calibri, Arial, Cambria, Times New Roman and Courier New, regular and bold, and decodes the PDF with the ToUnicode decoder of the header and footer PDF tests. When pdftotext is installed it also compares its output line by line, and otherwise says it skipped that half. Without the fix it prints the issue's rows, "Calibri regular: Locatoo Ratog Actoo fieeo ofce". GitHub issue #171. --- crates/rdocx/tests/integration_test.rs | 66 ++++++++++++++++++++++++++ 1 file changed, 66 insertions(+) diff --git a/crates/rdocx/tests/integration_test.rs b/crates/rdocx/tests/integration_test.rs index 74d019501..72c652738 100644 --- a/crates/rdocx/tests/integration_test.rs +++ b/crates/rdocx/tests/integration_test.rs @@ -13060,6 +13060,72 @@ mod header_footer_pdf { ); } + /// GitHub issue #171. Carlito, the bundled Calibri, joins ti, fi, ft, ffi + /// and more into one glyph. The PDF text layer paired glyphs with + /// characters by index, which garbled every Calibri paragraph. + #[test] + fn ligatures_keep_their_characters_in_pdf_text_of_every_bundled_family() { + const TEXT: &str = "Location Rating Action fifteen office Observation"; + let mut document = Document::new(); + let mut expected = Vec::new(); + for font in [ + "Calibri", + "Arial", + "Cambria", + "Times New Roman", + "Courier New", + ] { + for bold in [false, true] { + let weight = if bold { "bold" } else { "regular" }; + let line = format!("{font} {weight}: {TEXT}"); + document + .add_paragraph("") + .add_run(&line) + .font(font) + .size(11.0) + .bold(bold); + expected.push(line); + } + } + let pdf = document.to_pdf_deterministic().unwrap(); + + let page_text = pdf_page_text(&pdf).concat(); + for line in &expected { + assert!( + page_text.contains(line.as_str()), + "{line:?} in {page_text:?}" + ); + } + + let path = std::env::temp_dir().join(format!( + "rdocx-171-ligatures-{}-{:?}.pdf", + std::process::id(), + std::thread::current().id() + )); + std::fs::write(&path, &pdf).unwrap(); + let extracted = std::process::Command::new("pdftotext") + .arg(&path) + .arg("-") + .output(); + std::fs::remove_file(&path).unwrap(); + match extracted { + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + eprintln!("#171 pdftotext check skipped because pdftotext is absent"); + } + extracted => { + let extracted = extracted.unwrap(); + assert!(extracted.status.success()); + let lines = String::from_utf8(extracted.stdout).unwrap(); + let lines = lines + .lines() + .map(str::trim) + .filter(|line| !line.is_empty()) + .collect::>(); + assert_eq!(lines, expected); + } + } + } + #[test] fn word_semantics_reach_owned_multi_page_pdf_structure() { fn count_bytes(haystack: &[u8], needle: &[u8]) -> usize { From e129b5d104a80deef298d2405ba8d4ce5830f29d Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:59:12 +0200 Subject: [PATCH 5/6] Re-record the hash baseline for multi-character ToUnicode maps The ToUnicode CMaps of every Calibri sample change: ligature glyphs map to all their characters, and glyphs that index pairing had given a neighbour's character map to their own. The harness moves exactly the pdf/resources and pdf/bytes entries of the seven samples. No pdf/pages, PNG or OOXML entry moves, because the subset fonts, the widths and every content stream are byte-identical, which an object by object comparison of the old and new sample PDFs confirms. GitHub issue #171. --- scripts/hash_baseline.json | 30 +++++++++++++++--------------- 1 file changed, 15 insertions(+), 15 deletions(-) diff --git a/scripts/hash_baseline.json b/scripts/hash_baseline.json index 511ee7e8d..8390ca5df 100644 --- a/scripts/hash_baseline.json +++ b/scripts/hash_baseline.json @@ -1,54 +1,54 @@ { "entries": { "contract:page1.png": "4c4249d4efe7ea72590dfdeb3773d316da626f7189c7fc0606566f4bfb11a92f", - "contract:pdf/bytes": "0b2aa4c1031e4bc5a1a59a77b5a1a0a48ec3d933c5c89306971d7a4a30a78c89", + "contract:pdf/bytes": "2bb2a4344037ee8221cc0f330e1574551236d81e46f4999d0cccca658dd2b355", "contract:pdf/pages": "9b3ff34d7683336e1fea5ad3c42d733dc443ead52d645b79d30f531ff93e00aa", - "contract:pdf/resources": "1acfd604144a00109958c756bb0cb8894089040954cfbd573ceab9b40823d5fc", + "contract:pdf/resources": "1d23fd81e7aaa8ef154223e1574675613bcb428234ef1ded21ffc29ac2aedaf5", "contract:word/document.xml": "a6cebb1e5df54f1191494d9f54600494b7709e09b37402fef42eae06cd48c2fa", "contract:word/numbering.xml": "01d6cb0aa0ecc30b3d2a77c4df062b1c727d70ff9cbc0824910f049daf878a26", "contract:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "feature_showcase:page1.png": "6e88403fe63a789fd2ce6803029f67fcd45e00f2d13b9c13d5201a99eda0ada9", - "feature_showcase:pdf/bytes": "bf8a0c5bbd4ba70aba19b39e1a19bc1c78d19a12ac6f6324964628df5f78d410", + "feature_showcase:pdf/bytes": "38be2148f2e50ec969e0d652e20d6bf4927d90a24cc46e70ca801343e3014d4f", "feature_showcase:pdf/pages": "562b6508a015c227d598b408dfafaf2056cbba4e617e323b20ee39168ff64224", - "feature_showcase:pdf/resources": "e624cef80cac1f99ba398b0b0f8fc7986c0389cf12b6599f2a9bb187749bb797", + "feature_showcase:pdf/resources": "3573cc6e7183b5e912bfde97f11edea4ec44bb40e1b8f066eeb9b4e5e7726d62", "feature_showcase:word/document.xml": "7bfdbc2740871b1b38314dc95b353237d18e49b56ea64fa7a4c6499e4d21c8d3", "feature_showcase:word/numbering.xml": "2aa0599486d98573be0b7febb04ec63c08d3dfc5341b0f722b93cae113803bbc", "feature_showcase:word/styles.xml": "b815acd04cc89189d5b2b0caf822435d941c8cc70077f34932e4a2b3ce1b6595", "invoice:page1.png": "a209c0493bc90aae717e930b4466739212aaa4b646d1123a8a21c5e75b5c9644", - "invoice:pdf/bytes": "8e53fa760135722ab1ddd1593ee6fab2405068efec023e8c281438292fcdba03", + "invoice:pdf/bytes": "f2062c5bf095469f63d8c3b834150b8f7e1ef30ddd3a2fd1e77728f1928ccb87", "invoice:pdf/pages": "aadd561eb1aae1ff10748513e7cbdad0d9bdc6207baa0a286f934528bc3428a2", - "invoice:pdf/resources": "01d97372b5ad8c9fb06649d519f14311eb0428572af93c330c5fb393e74e323a", + "invoice:pdf/resources": "e48c1d7946b7e918931371e483bf48de7906358fd60213be521d49df2b83773c", "invoice:word/document.xml": "848426495d2a6f94e1c1514959938de96a13f9d2997d5ebfdb15ab0d18b7dbe7", "invoice:word/numbering.xml": null, "invoice:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "letter:page1.png": "79b23d965cd43bbdbb3cf2128df8f51b5d2d53d4a8934aaa34ed00b2aef8821e", - "letter:pdf/bytes": "634bf0bc38b2fe8ef5b55feadc3edef8fece56837effac278e4851a910055611", + "letter:pdf/bytes": "3893bd426a5e55eb13f0a4f581cc65da09b83a095605d836d1239a16f37a9667", "letter:pdf/pages": "92a27b62d2e6f40cee2eb742dba240e2b49c002fa2b351e60c7bd6952cd289ba", - "letter:pdf/resources": "0393c6de085093a2af078e708a4092bc1a55137616fd6fac30e01fdb32f732ee", + "letter:pdf/resources": "c63ba60143ff50ba08f981ec6eddde770d71205442441df18902a2c1857f591b", "letter:word/document.xml": "568e43d5433e4bd80d7c8a747a1c9b8b8f295dfb59464570fa37aa32668369a9", "letter:word/numbering.xml": "c6511604704117eb00ad2faffb9173e4f48d22f62557ca72a51d60b7907c8058", "letter:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "proposal:page1.png": "d92257f65b7f149b68f39ace59aec2613924112fe0dc1584ac9dbbd8d3beda34", - "proposal:pdf/bytes": "ccd3d55fea503f170955b5d5c427c873766c152bc3d37fe782fa167438bf92b0", + "proposal:pdf/bytes": "2f6d1bc76995ed9469dedcf022f1b95684ef6b9cb9a47da0ba5a83ed87882967", "proposal:pdf/pages": "1f9b047bfd0e39ff0e46d1fb6fc6cf04fcabac26a3f3735f70352d7b596eaefb", - "proposal:pdf/resources": "7e56e693e3e1fbaf829b2893172b76c7e63345b4b81d37ab5c76a35febc675be", + "proposal:pdf/resources": "775e228634aad44a60949777692e6381bdd3f5d754764c3dae276e241ea81569", "proposal:word/document.xml": "7235579097314e5cc8b07822884ebfadd9927b3ae34820a2673ba63097c5e02e", "proposal:word/numbering.xml": "061e20beb3409ce0f3feda1a99793956124a2967dd23afe1b528dd8fdeb82182", "proposal:word/styles.xml": "dae1d6c1083bba6bb4b29ba2ab748a713f5da60b8c2cd4d52224c6f0a8f21bef", "quote:page1.png": "0cb8e2ee78d96f816e490bd3830ccc6e7807c400c094872b42359284460c62af", - "quote:pdf/bytes": "b74bc8a08f1e66aabcb3334f2110e4da497cb2b467b773c9a61ac072d251db44", + "quote:pdf/bytes": "7cf33d2395adbe235271919ab35872300b230ba55d212adfed804c1f21f4a3e2", "quote:pdf/pages": "bb61084121bf416506cd62bd49683e0b1773f13f72c6d0e15322826d5c4b129b", - "quote:pdf/resources": "c0433753ef113d5aad02e5ce159bf67aded188fb7828d2dc8c99293a81608313", + "quote:pdf/resources": "7e9626233181d07e8d0bfa3c302f6a457133de1e9b5a124ae0e0d53ed7dbe03d", "quote:word/document.xml": "0c72970432f77927f6fb3bd6e980f9e12eebcf3a4c56b5260bf5902ef92ebdb5", "quote:word/numbering.xml": "c6511604704117eb00ad2faffb9173e4f48d22f62557ca72a51d60b7907c8058", "quote:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "report:page1.png": "792f1fdc032c7ec1b6a3e451aa7f94fc45e8ea810d12b9815ced82d2fa7fbd82", - "report:pdf/bytes": "960f31cf7cf44328783d793eb1a3e61a66f96463e298af4d499fc08f080590dc", + "report:pdf/bytes": "eea0e88d4a6f70a07ee62606e346aff26df9179c5e38ac0767f1bb29457aa694", "report:pdf/pages": "2e1bd079af84488e8c0c37f8d970e20cfe4c09cf586d293407069c3234ef9b39", - "report:pdf/resources": "cabc4cbad4abfbb595669566cc9ef98e9b96eb7c885f7fbce6a482a4e669b6bf", + "report:pdf/resources": "33cd91323ff09d9d7fad4cc4d83fc51954214304439eba29fb2a415752e1a669", "report:word/document.xml": "7c30b636bc89becd68f098a70127532cbbbe284b008e2ed7cbb31a48d342d5ed", "report:word/numbering.xml": "2aa0599486d98573be0b7febb04ec63c08d3dfc5341b0f722b93cae113803bbc", "report:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4" }, - "reason": "Paragraph spacing keeps the larger of space after and space before and horizontal table borders take their width between rows, which moves the rendered PDF and PNG of six samples and no OOXML part" + "reason": "ToUnicode CMaps pair each glyph with the characters it draws through the font cmap and GSUB, so Carlito ligatures map to all their characters (GitHub issue #171)" } From 2da774a518826670d4db217656ceaa647725d7c6 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 20:01:08 +0200 Subject: [PATCH 6/6] Re-record the archive measurements of oxml-pdf, rdocx and rpptx The oxml-pdf package grows with the new ToUnicode pairing and its unit tests, and the rdocx and rpptx packages carry their integration tests, which gain the ligature cases. The three rows in readme_doctests.py and in the crate READMEs now hold the measured sizes, dated 2026-09-27 through ARCHIVE_REMEASUREMENT_DATES, so the Docs job keeps passing. GitHub issue #171. --- README.md | 2 +- crates/oxml-pdf/README.md | 2 +- crates/rpptx/README.md | 2 +- scripts/readme_doctests.py | 11 ++++++----- 4 files changed, 9 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 479b462ef..cd99960c2 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,093,062 compressed bytes, 6,500,922 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/oxml-pdf/README.md b/crates/oxml-pdf/README.md index 56bb4a874..fcc49c6ab 100644 --- a/crates/oxml-pdf/README.md +++ b/crates/oxml-pdf/README.md @@ -19,7 +19,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: oxml-pdf | 66,015 compressed bytes, 304,432 member bytes, 14 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `oxml-pdf` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: oxml-pdf | 72,802 compressed bytes, 332,454 member bytes, 14 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `oxml-pdf` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rpptx/README.md b/crates/rpptx/README.md index 34eb7caa2..3f7d5298b 100644 --- a/crates/rpptx/README.md +++ b/crates/rpptx/README.md @@ -22,7 +22,7 @@ presentation, notes, handout, PDF, and animation outputs. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rpptx | 407,658 compressed bytes, 2,122,094 member bytes, 16 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rpptx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rpptx | 408,469 compressed bytes, 2,124,754 member bytes, 16 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rpptx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643be..133299639 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,9 +367,10 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", + "oxml-pdf": "2026-09-27", + "rdocx": "2026-09-27", "rdocx-layout": "2026-09-26", - "rpptx": "2026-09-26", + "rpptx": "2026-09-27", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" ARCHIVE_COMPRESSION_TOLERANCE_BYTES = 64 @@ -381,16 +382,16 @@ class ReadmeCase: "oxml-layout": (4_623_324, 9_227_483, 51), "oxml-media": (12_252, 50_992, 6), "oxml-opc": (92_122, 355_510, 12), - "oxml-pdf": (66_015, 304_432, 14), + "oxml-pdf": (72_802, 332_454, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_093_062, 6_500_922, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), "rdocx-oxml": (367_500, 2_380_047, 32), "rdocx-pdf": (8_111, 26_758, 6), - "rpptx": (407_658, 2_122_094, 16), + "rpptx": (408_469, 2_124_754, 16), "rpptx-chart": (6_648, 21_136, 6), "rpptx-cli": (36_709, 159_585, 8), "rpptx-layout": (79_109, 458_112, 11),