diff --git a/README.md b/README.md index 479b462e..cd99960c 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,093,062 compressed bytes, 6,500,922 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/oxml-pdf/README.md b/crates/oxml-pdf/README.md index 56bb4a87..fcc49c6a 100644 --- a/crates/oxml-pdf/README.md +++ b/crates/oxml-pdf/README.md @@ -19,7 +19,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: oxml-pdf | 66,015 compressed bytes, 304,432 member bytes, 14 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `oxml-pdf` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: oxml-pdf | 72,802 compressed bytes, 332,454 member bytes, 14 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `oxml-pdf` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/oxml-pdf/src/font.rs b/crates/oxml-pdf/src/font.rs index 14fcb864..89976cf3 100644 --- a/crates/oxml-pdf/src/font.rs +++ b/crates/oxml-pdf/src/font.rs @@ -1,21 +1,31 @@ //! Font subsetting and ToUnicode CMap generation for PDF embedding. -use std::collections::{BTreeMap, HashMap}; +use std::collections::btree_map::Entry; +use std::collections::{BTreeMap, BTreeSet, HashMap}; use oxml_layout::{FontData, FontId, LayoutResult, PositionedElement, walk}; use pdf_writer::types::{SystemInfo, UnicodeCmap}; use pdf_writer::{Name, Str}; use subsetter::GlyphRemapper; +use ttf_parser::gsub::SubstitutionSubtable; +use ttf_parser::opentype_layout::Coverage; /// Per-font glyph usage collected across all pages. pub(crate) struct FontUsage { - /// Mapping from original glyph ID to the Unicode text it represents. - /// Multiple characters may map to one glyph, so we store the first seen. + /// Mapping from original glyph ID to the Unicode text it draws. + /// + /// A ligature draws several characters with one glyph, so the text is a + /// string. The CMap holds one entry per glyph for the whole font, so when + /// runs pair one glyph with different text, the strongest pairing wins, + /// and the first one seen among equals. /// /// Ordered by glyph ID, because this map is iterated to emit the ToUnicode /// CMap. A hashed order put the same pairs in a different order on every /// run, which made the written PDF differ from itself byte for byte. - pub glyph_to_unicode: BTreeMap, + pub glyph_to_unicode: BTreeMap, + /// Every glyph a run draws. Each declares its width, including a glyph + /// that carries no text of its own. + pub drawn_glyphs: BTreeSet, /// The GlyphRemapper for subsetting. pub remapper: GlyphRemapper, } @@ -29,9 +39,295 @@ pub(crate) struct PreparedFont { pub widths: Vec<(u16, f64)>, // (new_gid, width_in_font_units) } +impl FontUsage { + /// Subset a drawn glyph, declare its width, and map it to its text unless + /// another run paired it more strongly. + fn draw(&mut self, glyph: u16, text: Option<(Pairing, String)>) { + self.remapper.remap(glyph); + self.drawn_glyphs.insert(glyph); + let Some((pairing, text)) = text else { + return; + }; + match self.glyph_to_unicode.entry(glyph) { + Entry::Vacant(entry) => { + entry.insert((pairing, text)); + } + Entry::Occupied(mut entry) if entry.get().0 < pairing => { + entry.insert((pairing, text)); + } + Entry::Occupied(_) => {} + } + } +} + +/// How strongly a glyph was paired with the text it draws, weakest first. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +pub(crate) enum Pairing { + /// A glyph of a rich run that draws none of its cluster's characters by + /// itself, such as the dots an Arabic font draws apart from their letter. + /// It repeats the cluster's text, which the run's `ActualText` covers, so + /// that every glyph of a rich run maps to Unicode. + Shared, + /// The first of several glyphs that draw some characters between them + /// in no order the font explains. It carries all of those characters. + Group, + /// Paired by position among as many unexplained glyphs as characters. + Position, + /// The one glyph left to draw the characters of its group that no other + /// glyph explains. + Whole, + /// The glyph the font's cmap gives the character. + Nominal, + /// The glyph a GSUB ligature joins the glyphs of several characters into. + /// It outranks the cmap because one glyph can be both. Carlito draws "fi" + /// and U+FB01 with one glyph, and one literal U+FB01 must not turn every + /// "fi" of the document into U+FB01. + Ligature, +} + +/// What a font says about the glyphs a run draws. +struct FontGlyphs<'a> { + face: Option>, + /// Each ligature glyph, with the glyph sequences GSUB joins into it. + ligatures: HashMap>>, +} + +impl<'a> FontGlyphs<'a> { + fn new(font: Option<&'a FontData>) -> Self { + let face = font.and_then(|font| ttf_parser::Face::parse(&font.data, font.face_index).ok()); + let ligatures = face.as_ref().map(ligature_components).unwrap_or_default(); + Self { face, ligatures } + } + + /// The glyph the cmap gives each character, `.notdef` for a character + /// the font lacks, which is also what the shaper draws for it. + fn nominal_glyphs(&self, characters: &[char]) -> Vec> { + characters + .iter() + .map(|&character| { + let face = self.face.as_ref()?; + Some(face.glyph_index(character).map_or(0, |glyph| glyph.0)) + }) + .collect() + } + + /// How many characters, from the start of `nominal`, the glyph draws when + /// it is the cmap glyph of the first or a GSUB ligature of the first few. + fn explain(&self, glyph: u16, nominal: &[Option]) -> Option<(usize, Pairing)> { + if nominal.first() == Some(&Some(glyph)) { + return Some((1, Pairing::Nominal)); + } + let length = self.ligature_length(glyph, nominal, 2)?; + Some((length, Pairing::Ligature)) + } + + fn ligature_length(&self, glyph: u16, nominal: &[Option], depth: u8) -> Option { + let sequences = self.ligatures.get(&glyph)?; + sequences + .iter() + .filter_map(|components| { + let mut length = 0; + for &component in components { + let rest = &nominal[length..]; + if rest.first() == Some(&Some(component)) { + length += 1; + } else if depth > 0 { + length += self.ligature_length(component, rest, depth - 1)?; + } else { + return None; + } + } + Some(length) + }) + .max() + } +} + +/// Every ligature glyph of the font's GSUB table, with the glyph sequences +/// it replaces. +fn ligature_components(face: &ttf_parser::Face<'_>) -> HashMap>> { + let mut ligatures: HashMap>> = HashMap::new(); + let Some(gsub) = face.tables().gsub else { + return ligatures; + }; + for lookup in gsub.lookups { + for subtable in lookup.subtables.into_iter::() { + let SubstitutionSubtable::Ligature(table) = subtable else { + continue; + }; + let first_glyphs: Vec = match table.coverage { + Coverage::Format1 { glyphs } => glyphs.into_iter().collect(), + Coverage::Format2 { records } => records + .into_iter() + .flat_map(|record| record.start.0..=record.end.0) + .map(ttf_parser::GlyphId) + .collect(), + }; + for first in first_glyphs { + let Some(set) = table + .coverage + .get(first) + .and_then(|index| table.ligature_sets.get(index)) + else { + continue; + }; + for ligature in set { + let mut components = vec![first.0]; + components.extend(ligature.components.into_iter().map(|glyph| glyph.0)); + ligatures + .entry(ligature.glyph.0) + .or_default() + .push(components); + } + } + } + } + ligatures +} + +/// How far past an unexplained glyph the pairing looks for the next glyph the +/// font explains. A ligature or a cluster spans a few characters, and the +/// bound keeps a run the font explains nowhere linear. +const RESYNC_WINDOW: usize = 16; + +/// Pair each glyph of a plain run with the characters it draws. +/// +/// A plain run carries its glyphs and its text but not the shaper's clusters, +/// and a ligature draws several characters with one glyph, so pairing them by +/// index shifts every later glyph. The pairing is rebuilt from the font +/// instead. A glyph draws the next character when the cmap gives it that +/// character, and the next few when GSUB joins their glyphs into it. Glyphs +/// the font explains neither way share the characters up to the next glyph +/// it does explain. When there is none within `RESYNC_WINDOW`, an unexplained +/// glyph takes one character by position. A plain run has no `ActualText`, so +/// a glyph that draws none of its group's characters by itself gets no text +/// rather than repeat a character another glyph carries, and so does a glyph +/// left after the last character. +fn pair_plain_run( + characters: &[char], + glyphs: &[u16], + font: &FontGlyphs<'_>, +) -> Vec> { + let nominal = font.nominal_glyphs(characters); + let mut texts = vec![None; glyphs.len()]; + let (mut char_at, mut glyph_at) = (0, 0); + while char_at < characters.len() && glyph_at < glyphs.len() { + if let Some((length, pairing)) = font.explain(glyphs[glyph_at], &nominal[char_at..]) { + let text = characters[char_at..char_at + length].iter().collect(); + texts[glyph_at] = Some((pairing, text)); + char_at += length; + glyph_at += 1; + continue; + } + + let char_limit = characters.len().min(char_at + 1 + RESYNC_WINDOW); + let glyph_limit = glyphs.len().min(glyph_at + 1 + RESYNC_WINDOW); + let next_explained = (glyph_at + 1..glyph_limit).find_map(|glyph_end| { + (char_at + 1..char_limit) + .find(|&char_end| { + font.explain(glyphs[glyph_end], &nominal[char_end..]) + .is_some() + }) + .map(|char_end| (char_end, glyph_end)) + }); + let (char_end, glyph_end) = match next_explained { + Some(end) => end, + None if characters.len() - char_at <= RESYNC_WINDOW + && glyphs.len() - glyph_at <= RESYNC_WINDOW => + { + (characters.len(), glyphs.len()) + } + None => { + texts[glyph_at] = Some((Pairing::Position, characters[char_at].to_string())); + char_at += 1; + glyph_at += 1; + continue; + } + }; + let group = pair_group( + &characters[char_at..char_end], + &glyphs[glyph_at..glyph_end], + &nominal[char_at..char_end], + font, + ); + texts.splice(glyph_at..glyph_end, group); + char_at = char_end; + glyph_at = glyph_end; + } + texts +} + +/// Pair the glyphs of a group with the characters they draw between them, +/// such as one shaping cluster, in which glyph order need not follow text +/// order. +/// +/// A glyph the font explains draws the characters it explains, when no other +/// glyph of the group has taken them. The characters left over go to the one +/// glyph left over, or pair with the glyphs left over by position when they +/// are as many. Otherwise the first glyph left over carries them all, so that +/// none is lost. A glyph still without text draws none of the characters by +/// itself and gets no text, so that no character is paired twice. +fn pair_group( + characters: &[char], + glyphs: &[u16], + nominal: &[Option], + font: &FontGlyphs<'_>, +) -> Vec> { + let mut texts = vec![None; glyphs.len()]; + let mut claimed = vec![false; characters.len()]; + for (text, &glyph) in texts.iter_mut().zip(glyphs) { + let explained = (0..characters.len()).find_map(|start| { + let (length, pairing) = font.explain(glyph, &nominal[start..])?; + let range = start..start + length; + claimed[range.clone()] + .iter() + .all(|taken| !taken) + .then_some((range, pairing)) + }); + if let Some((range, pairing)) = explained { + claimed[range.clone()].fill(true); + *text = Some((pairing, characters[range].iter().collect())); + } + } + let rest_characters = characters + .iter() + .zip(&claimed) + .filter(|(_, claimed)| !**claimed) + .map(|(character, _)| *character) + .collect::>(); + let rest_glyphs = (0..glyphs.len()) + .filter(|&index| texts[index].is_none()) + .collect::>(); + match rest_glyphs.as_slice() { + _ if rest_characters.is_empty() => {} + [] => {} + [only] => texts[*only] = Some((Pairing::Whole, rest_characters.into_iter().collect())), + rest if rest.len() == rest_characters.len() => { + for (&index, character) in rest.iter().zip(rest_characters) { + texts[index] = Some((Pairing::Position, character.to_string())); + } + } + [first, ..] => { + texts[*first] = Some((Pairing::Group, rest_characters.into_iter().collect())); + } + } + texts +} + +fn font_glyphs<'f, 'a>( + fonts: &'f mut HashMap>, + layout: &'a LayoutResult, + font_id: FontId, +) -> &'f FontGlyphs<'a> { + fonts + .entry(font_id) + .or_insert_with(|| FontGlyphs::new(layout.fonts.iter().find(|font| font.id == font_id))) +} + /// Collect glyph usage across all pages for each font. pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap { let mut usage: HashMap = HashMap::new(); + let mut fonts: HashMap> = HashMap::new(); for page in &layout.pages { walk(&page.elements, &mut |element, _| { @@ -40,16 +336,16 @@ pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap = run.text.chars().collect(); - for (i, &gid) in run.glyph_ids.iter().enumerate() { - entry.remapper.remap(gid); - if let Some(&ch) = chars.get(i) { - entry.glyph_to_unicode.entry(gid).or_insert(ch); - } + let font = font_glyphs(&mut fonts, layout, run.font_id); + let texts = pair_plain_run(&chars, &run.glyph_ids, font); + for (&gid, text) in run.glyph_ids.iter().zip(texts) { + entry.draw(gid, text); } } if let PositionedElement::MultilingualText(run) = element @@ -58,19 +354,29 @@ pub(crate) fn collect_glyph_usage(layout: &LayoutResult) -> HashMap>(); + let font = font_glyphs(&mut fonts, layout, run.font_id); for cluster in &run.clusters { - let Some(&character) = chars.get(cluster.char_start as usize) else { + let characters = + chars.get(cluster.char_start as usize..cluster.char_end as usize); + let glyphs = run + .glyph_ids + .get(cluster.glyph_start as usize..cluster.glyph_end as usize); + let (Some(characters), Some(glyphs)) = (characters, glyphs) else { continue; }; - for glyph in cluster.glyph_start..cluster.glyph_end { - let Some(&gid) = run.glyph_ids.get(glyph as usize) else { - continue; - }; - entry.remapper.remap(gid); - entry.glyph_to_unicode.entry(gid).or_insert(character); + let nominal = font.nominal_glyphs(characters); + let texts = pair_group(characters, glyphs, &nominal, font); + // The run's ActualText carries its text, so a glyph that + // draws none of the cluster's characters by itself repeats + // them all, and every glyph of the run maps to Unicode. + let cluster_text = characters.iter().collect::(); + for (&gid, text) in glyphs.iter().zip(texts) { + let text = text.unwrap_or_else(|| (Pairing::Shared, cluster_text.clone())); + entry.draw(gid, Some(text)); } } } @@ -95,9 +401,9 @@ pub(crate) fn prepare_font(font_data: &FontData, usage: &mut FontUsage) -> Optio supplement: 0, }, ); - for (&old_gid, &ch) in &usage.glyph_to_unicode { + for (&old_gid, (_, text)) in &usage.glyph_to_unicode { if let Some(new_gid) = usage.remapper.get(old_gid) { - cmap.pair(new_gid, ch); + cmap.pair_with_multiple(new_gid, text.chars()); } } let cmap_bytes = cmap.finish().to_vec(); @@ -127,7 +433,7 @@ fn compute_glyph_widths(font_data: &FontData, usage: &FontUsage) -> Vec<(u16, f6 let units_per_em = face.units_per_em() as f64; let scale = 1000.0 / units_per_em; - for &old_gid in usage.glyph_to_unicode.keys() { + for &old_gid in &usage.drawn_glyphs { if let Some(new_gid) = usage.remapper.get(old_gid) { let advance = face .glyph_hor_advance(ttf_parser::GlyphId(old_gid)) @@ -186,3 +492,359 @@ pub(crate) fn get_font_metrics(font_data: &FontData) -> Option stem_v, }) } + +#[cfg(test)] +mod tests { + use super::*; + use oxml_layout::{ + Color, FontManager, GlyphRun, MultilingualGlyphRun, PageFrame, Point, TextDirection, + TextSegment, + }; + + /// The bfchar entries of a ToUnicode CMap, by subset glyph id. + fn to_unicode_entries(cmap: &[u8]) -> HashMap { + let cmap = std::str::from_utf8(cmap).expect("the CMap is ASCII"); + let mut entries = HashMap::new(); + let mut in_bfchar = false; + for line in cmap.lines() { + if line.ends_with("beginbfchar") { + in_bfchar = true; + } else if line == "endbfchar" { + in_bfchar = false; + } else if in_bfchar { + let (glyph, text) = line.split_once(' ').expect("one bfchar pair per line"); + let glyph = u16::from_str_radix(glyph.trim_matches(['<', '>']), 16).unwrap(); + let units = text.trim_matches(['<', '>']).as_bytes().chunks(4); + let units = units + .map(|unit| { + u16::from_str_radix(std::str::from_utf8(unit).unwrap(), 16).unwrap() + }) + .collect::>(); + entries.insert(glyph, String::from_utf16(&units).unwrap()); + } + } + entries + } + + fn plain_run(fonts: &FontManager, font_id: FontId, text: &str) -> PositionedElement { + let shaped = fonts.shape_text(font_id, text, 11.0).unwrap(); + PositionedElement::Text(GlyphRun { + origin: Point { x: 0.0, y: 0.0 }, + font_id, + font_size: 11.0, + glyph_ids: shaped.glyph_ids, + advances: shaped.advances, + text: text.to_owned(), + source: None, + color: Color::BLACK, + bold: false, + italic: false, + field_kind: None, + field_source: None, + note: None, + }) + } + + fn multilingual_runs(fonts: &mut FontManager, text: &str) -> Vec { + let font_id = fonts.resolve_font(Some("Calibri"), false, false).unwrap(); + let segment = TextSegment { + text: text.to_owned(), + direction: TextDirection::Auto, + source: None, + font_id, + font_size: 11.0, + glyph_ids: Vec::new(), + advances: Vec::new(), + width: 0.0, + ascent: 0.0, + descent: 0.0, + line_gap: 0.0, + color: Color::BLACK, + bold: false, + italic: false, + underline: None, + strike: false, + dstrike: false, + highlight: None, + baseline_offset: 0.0, + hyperlink_url: None, + field_kind: None, + field_source: None, + note: None, + }; + fonts + .shape_multilingual_text(segment, None, TextDirection::Auto, false) + .unwrap() + .into_iter() + .map(|span| { + PositionedElement::MultilingualText(MultilingualGlyphRun { + origin: Point { x: 0.0, y: 0.0 }, + font_id: span.font_id(), + font_size: 11.0, + glyph_ids: span.glyph_ids().to_vec(), + x_advances: span.x_advances().to_vec(), + y_advances: span.y_advances().to_vec(), + x_offsets: span.x_offsets().to_vec(), + y_offsets: span.y_offsets().to_vec(), + clusters: span.clusters().to_vec(), + logical_text: span.text().to_owned(), + logical_index: span.logical_index(), + source: None, + script: span.script(), + language: None, + direction: span.direction(), + bidi_level: span.bidi_level(), + color: Color::BLACK, + bold: false, + italic: false, + field_kind: None, + field_source: None, + note: None, + }) + }) + .collect() + } + + fn layout_of(fonts: &FontManager, elements: Vec) -> LayoutResult { + LayoutResult::new( + vec![PageFrame::new(1, 612.0, 792.0, elements).into()], + fonts.all_font_data(), + None, + vec![], + ) + } + + /// Each font's subset remapper and ToUnicode entries, as a reader sees them. + fn embedded_text_maps( + layout: &LayoutResult, + ) -> HashMap)> { + let mut usage = collect_glyph_usage(layout); + layout + .fonts + .iter() + .filter_map(|font| { + let prepared = prepare_font(font, usage.get_mut(&font.id)?)?; + let entries = to_unicode_entries(&prepared.cmap_bytes); + Some((font.id, (prepared.remapper, entries))) + }) + .collect() + } + + /// The text a reader extracts from each glyph through the ToUnicode CMap. + fn glyph_texts( + maps: &HashMap)>, + font_id: FontId, + glyph_ids: &[u16], + ) -> Vec { + let (remapper, entries) = &maps[&font_id]; + glyph_ids + .iter() + .map(|glyph| { + let subset_glyph = remapper.get(*glyph).expect("every drawn glyph is subset"); + entries.get(&subset_glyph).cloned().unwrap_or_default() + }) + .collect() + } + + #[test] + fn a_ligature_glyph_maps_to_every_character_it_draws() { + let mut fonts = FontManager::new_deterministic().unwrap(); + for bold in [false, true] { + let font_id = fonts.resolve_font(Some("Calibri"), bold, false).unwrap(); + // Ligature runs come first. Pairing glyphs and characters by index + // shifted every later glyph of such a run, and the font-wide map + // then kept that wrong character in the runs without a ligature. + let words = [ + "Location ", + "fifteen ", + "office ", + "attitude ", + "affluent ", + "fjord ", + "staff ", + "shifting ", + "Rating ", + "Action ", + "Observation ", + "no ligature here", + ]; + let layout = layout_of( + &fonts, + words + .iter() + .map(|word| plain_run(&fonts, font_id, word)) + .collect(), + ); + let maps = embedded_text_maps(&layout); + for element in &layout.pages[0].elements { + let PositionedElement::Text(run) = element else { + unreachable!("the page holds plain runs only"); + }; + assert_eq!( + glyph_texts(&maps, font_id, &run.glyph_ids).concat(), + run.text, + "bold: {bold}" + ); + } + for ligature in ["ti", "fi", "ft", "ffi", "ffl", "tti", "fj", "ff"] { + let shaped = fonts.shape_text(font_id, ligature, 11.0).unwrap(); + assert_eq!(shaped.glyph_ids.len(), 1, "Carlito joins {ligature}"); + assert_eq!( + glyph_texts(&maps, font_id, &shaped.glyph_ids), + [ligature], + "bold: {bold}" + ); + } + } + } + + #[test] + fn a_ligature_outranks_the_presentation_form_sharing_its_glyph() { + let mut fonts = FontManager::new_deterministic().unwrap(); + let font_id = fonts.resolve_font(Some("Calibri"), false, false).unwrap(); + // Carlito draws its fi ligature with the cmap glyph of U+FB01, and the + // CMap keeps one text per glyph. A literal U+FB01 seen first used to + // claim that glyph and turn every "fi" of the document into U+FB01. + let fi = fonts.shape_text(font_id, "fi", 11.0).unwrap().glyph_ids; + let form = fonts.shape_text(font_id, "\u{fb01}", 11.0).unwrap(); + assert_eq!(form.glyph_ids, fi, "Carlito shares the glyph"); + let extracted = |texts: &[&str]| { + let runs = texts.iter().map(|text| plain_run(&fonts, font_id, text)); + let layout = layout_of(&fonts, runs.collect()); + let maps = embedded_text_maps(&layout); + let runs = layout.pages[0].elements.iter().map(|element| { + let PositionedElement::Text(run) = element else { + unreachable!("the page holds plain runs only"); + }; + glyph_texts(&maps, font_id, &run.glyph_ids).concat() + }); + runs.collect::>() + }; + assert_eq!(extracted(&["office fifteen"]), ["office fifteen"]); + assert_eq!( + extracted(&["\u{fb01}", "office fifteen"]), + ["fi", "office fifteen"] + ); + // Without a ligature to decompose, the glyph keeps the literal. + assert_eq!(extracted(&["\u{fb01}"]), ["\u{fb01}"]); + } + + #[test] + fn a_plain_run_extracts_each_character_once() { + let mut fonts = FontManager::new_deterministic().unwrap(); + // Liberation Sans has no "≮" or "≯" and Carlito no "Ѷ", so the shaper + // draws each as a base glyph and a combining mark, neither of them the + // cmap glyph of the character. A plain run has no ActualText, so the + // mark must not repeat the character its base glyph carries. + let mut elements = Vec::new(); + for (family, text) in [("Arial", "a ≮ b ≯ c"), ("Calibri", "Ѷ")] { + let font_id = fonts.resolve_font(Some(family), false, false).unwrap(); + elements.push(plain_run(&fonts, font_id, text)); + } + let layout = layout_of(&fonts, elements); + let maps = embedded_text_maps(&layout); + for element in &layout.pages[0].elements { + let PositionedElement::Text(run) = element else { + unreachable!("the page holds plain runs only"); + }; + let characters = run.text.chars().count(); + assert!( + run.glyph_ids.len() > characters, + "{:?} draws a mark", + run.text + ); + let texts = glyph_texts(&maps, run.font_id, &run.glyph_ids); + assert_eq!(texts.concat(), run.text); + } + } + + #[test] + fn a_cmap_pairing_outranks_one_inferred_earlier() { + let mut fonts = FontManager::new_deterministic().unwrap(); + let font_id = fonts.resolve_font(Some("Calibri"), false, false).unwrap(); + // Carlito has no no-break space, so the shaper draws the space glyph + // for it. Seen first, that pairing used to claim the glyph for the + // whole font. + let layout = layout_of( + &fonts, + vec![ + plain_run(&fonts, font_id, "a\u{a0}b"), + plain_run(&fonts, font_id, "a b"), + ], + ); + let space = fonts.shape_text(font_id, " ", 11.0).unwrap().glyph_ids; + let texts = glyph_texts(&embedded_text_maps(&layout), font_id, &space); + assert_eq!(texts, [" "]); + } + + #[test] + fn a_multilingual_cluster_keeps_every_character() { + let mut fonts = FontManager::new_deterministic().unwrap(); + // The conjunct of "क्ष" draws three characters with one glyph. The + // cluster "त्रि" draws four with two, its vowel sign first and in a + // width variant the cmap does not give, and "कि" then pairs that + // variant with the vowel sign alone. The Arabic font draws the dot of + // "ب" apart from the letter, and draws contextual forms. + let mut elements = multilingual_runs(&mut fonts, "क्षत्रिय कि"); + elements.extend(multilingual_runs(&mut fonts, "سلام ب")); + let layout = layout_of(&fonts, elements); + let maps = embedded_text_maps(&layout); + let mut clusters = Vec::new(); + for element in &layout.pages[0].elements { + let PositionedElement::MultilingualText(run) = element else { + unreachable!("the page holds multilingual runs only"); + }; + let texts = glyph_texts(&maps, run.font_id, &run.glyph_ids); + let characters = run.logical_text.chars().collect::>(); + for cluster in &run.clusters { + let glyphs = cluster.glyph_start as usize..cluster.glyph_end as usize; + let source = &characters[cluster.char_start as usize..cluster.char_end as usize]; + let texts = texts[glyphs].to_vec(); + assert!(texts.iter().all(|text| !text.is_empty()), "{texts:?}"); + let mut extracted = texts.concat(); + for character in source { + let at = extracted.find(*character).unwrap_or_else(|| { + panic!("{character:?} of {source:?} is missing from {texts:?}") + }); + extracted.remove(at); + } + clusters.push((source.iter().collect::(), texts)); + } + } + let texts_of = |source: &str| { + clusters + .iter() + .find(|(cluster, _)| cluster == source) + .map(|(_, texts)| texts.clone()) + .unwrap_or_else(|| panic!("no cluster {source:?} in {clusters:?}")) + }; + assert_eq!(texts_of("क्ष"), ["क्ष"]); + assert_eq!(texts_of("त्रि"), ["ि", "त्र"]); + assert_eq!(texts_of("कि"), ["ि", "क"]); + assert_eq!(texts_of("ب"), ["ب", "ب"]); + } + + #[test] + fn a_run_without_ligatures_keeps_one_character_per_glyph() { + let mut fonts = FontManager::new_deterministic().unwrap(); + let font_id = fonts.resolve_font(Some("Arial"), false, false).unwrap(); + let text = "Location Rating Action fifteen office"; + let layout = layout_of(&fonts, vec![plain_run(&fonts, font_id, text)]); + let PositionedElement::Text(run) = &layout.pages[0].elements[0] else { + unreachable!("the page holds one plain run"); + }; + assert_eq!(run.glyph_ids.len(), text.chars().count()); + let maps = embedded_text_maps(&layout); + let texts = glyph_texts(&maps, font_id, &run.glyph_ids); + let characters = text.chars().map(String::from).collect::>(); + assert_eq!(texts, characters); + + let mut usage = collect_glyph_usage(&layout); + let font = layout.fonts.iter().find(|font| font.id == font_id).unwrap(); + let prepared = prepare_font(font, usage.get_mut(&font_id).unwrap()).unwrap(); + let mut drawn = run.glyph_ids.clone(); + drawn.sort_unstable(); + drawn.dedup(); + assert_eq!(prepared.widths.len(), drawn.len()); + } +} diff --git a/crates/rdocx/tests/integration_test.rs b/crates/rdocx/tests/integration_test.rs index 74d01950..72c65273 100644 --- a/crates/rdocx/tests/integration_test.rs +++ b/crates/rdocx/tests/integration_test.rs @@ -13060,6 +13060,72 @@ mod header_footer_pdf { ); } + /// GitHub issue #171. Carlito, the bundled Calibri, joins ti, fi, ft, ffi + /// and more into one glyph. The PDF text layer paired glyphs with + /// characters by index, which garbled every Calibri paragraph. + #[test] + fn ligatures_keep_their_characters_in_pdf_text_of_every_bundled_family() { + const TEXT: &str = "Location Rating Action fifteen office Observation"; + let mut document = Document::new(); + let mut expected = Vec::new(); + for font in [ + "Calibri", + "Arial", + "Cambria", + "Times New Roman", + "Courier New", + ] { + for bold in [false, true] { + let weight = if bold { "bold" } else { "regular" }; + let line = format!("{font} {weight}: {TEXT}"); + document + .add_paragraph("") + .add_run(&line) + .font(font) + .size(11.0) + .bold(bold); + expected.push(line); + } + } + let pdf = document.to_pdf_deterministic().unwrap(); + + let page_text = pdf_page_text(&pdf).concat(); + for line in &expected { + assert!( + page_text.contains(line.as_str()), + "{line:?} in {page_text:?}" + ); + } + + let path = std::env::temp_dir().join(format!( + "rdocx-171-ligatures-{}-{:?}.pdf", + std::process::id(), + std::thread::current().id() + )); + std::fs::write(&path, &pdf).unwrap(); + let extracted = std::process::Command::new("pdftotext") + .arg(&path) + .arg("-") + .output(); + std::fs::remove_file(&path).unwrap(); + match extracted { + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + eprintln!("#171 pdftotext check skipped because pdftotext is absent"); + } + extracted => { + let extracted = extracted.unwrap(); + assert!(extracted.status.success()); + let lines = String::from_utf8(extracted.stdout).unwrap(); + let lines = lines + .lines() + .map(str::trim) + .filter(|line| !line.is_empty()) + .collect::>(); + assert_eq!(lines, expected); + } + } + } + #[test] fn word_semantics_reach_owned_multi_page_pdf_structure() { fn count_bytes(haystack: &[u8], needle: &[u8]) -> usize { diff --git a/crates/rpptx/README.md b/crates/rpptx/README.md index 34eb7caa..3f7d5298 100644 --- a/crates/rpptx/README.md +++ b/crates/rpptx/README.md @@ -22,7 +22,7 @@ presentation, notes, handout, PDF, and animation outputs. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rpptx | 407,658 compressed bytes, 2,122,094 member bytes, 16 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rpptx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rpptx | 408,469 compressed bytes, 2,124,754 member bytes, 16 members | 0.12.1 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rpptx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/crates/rpptx/tests/integration.rs b/crates/rpptx/tests/integration.rs index cbf98946..20829956 100644 --- a/crates/rpptx/tests/integration.rs +++ b/crates/rpptx/tests/integration.rs @@ -598,6 +598,71 @@ fn pdf_import_differential_rejects_geometry_text_link_and_pixel_perturbations() )); } +/// GitHub issue #171. Without an inherited `rtl` a paragraph draws its Latin +/// text as plain runs, which carry no shaping clusters, so a Carlito ligature +/// reaches the PDF text layer only through the ToUnicode map. +#[test] +#[cfg(feature = "render")] +fn plain_run_ligatures_keep_their_characters_in_pdf_text() { + const TEXT: &str = "Location Rating Action fifteen office Observation"; + let mut source = Presentation::new().unwrap(); + source.add_slide(0).unwrap(); + source + .slide_mut(0) + .unwrap() + .add_textbox(Emu(457_200), Emu(914_400), Emu(8_229_600), Emu(914_400)) + .unwrap() + .set_text(TEXT) + .unwrap(); + let mut package = open_opc(&source.to_bytes().unwrap(), "#171 ligature deck"); + for part in [ + "/ppt/presentation.xml", + "/ppt/slideMasters/slideMaster1.xml", + ] { + let xml = String::from_utf8(package.get_part(part).unwrap().to_vec()).unwrap(); + assert!(xml.contains(r#" rtl="0""#), "{part} sets the direction"); + package.set_part(part, xml.replace(r#" rtl="0""#, "").into_bytes()); + } + let presentation = Presentation::from_bytes(&package_bytes(package)).unwrap(); + + let (_, layout) = presentation.render_deterministic().unwrap(); + let mut plain_text = String::new(); + walk(&layout.pages[0].elements, &mut |element, _| match element { + PositionedElement::Text(run) => plain_text.push_str(&run.text), + PositionedElement::MultilingualText(run) => { + panic!("{:?} took the path with clusters", run.logical_text) + } + _ => {} + }); + assert_eq!(plain_text.trim_end(), TEXT); + + // lopdf decodes each drawn run through the ToUnicode map on its own line. + let pdf = presentation.to_pdf_deterministic().unwrap(); + let decoded = lopdf::Document::load_mem(&pdf) + .unwrap() + .extract_text(&[1]) + .unwrap(); + assert_eq!( + decoded.split_whitespace().collect::>().join(" "), + TEXT + ); + + let path = std::env::temp_dir().join(format!("rpptx-171-ligatures-{}.pdf", std::process::id())); + fs::write(&path, &pdf).unwrap(); + let extracted = Command::new("pdftotext").arg(&path).arg("-").output(); + fs::remove_file(&path).unwrap(); + match extracted { + Err(error) if error.kind() == std::io::ErrorKind::NotFound => { + eprintln!("#171 pdftotext check skipped because pdftotext is absent"); + } + extracted => { + let extracted = extracted.unwrap(); + assert!(extracted.status.success()); + assert_eq!(String::from_utf8(extracted.stdout).unwrap().trim(), TEXT); + } + } +} + #[test] #[cfg(feature = "default-template")] #[ignore = "requires Google Chrome 152.0.7977.65"] diff --git a/docs/hld/08-rendering-spec.md b/docs/hld/08-rendering-spec.md index abd0629e..d04b5823 100644 --- a/docs/hld/08-rendering-spec.md +++ b/docs/hld/08-rendering-spec.md @@ -297,6 +297,31 @@ with no visible difference and no failing test. Two writes of one document cannot detect that, since they reuse the same map instances. The regression builds two documents and compares their bytes. +**The text layer follows what each glyph draws.** A `GlyphRun` carries its +glyphs and its text but not the shaper's clusters, and a ligature draws several +characters with one glyph, so pairing them by index shifts every later glyph. +The writer reads the pairing from the font instead. A glyph draws the next +character when the cmap gives it that character, and the next few when a GSUB +ligature joins their glyphs into it, so the Carlito `ti` or `ffi` of a Calibri +document extracts as its characters. Glyphs the font explains neither way share +the characters up to the next glyph it does explain, looked for within 16 +glyphs and characters. When there is none, an unexplained glyph takes one +character by position, which keeps the pairing linear in a run the font +explains nowhere. A plain run has no `ActualText`, so there a glyph that draws +none of its group's characters by itself adds no ToUnicode text rather than +repeat a character another glyph carries, and neither does a glyph left after +the run's last character. A rich run's clusters give their characters to their +glyphs the same way, so no character of a cluster is lost. In a rich run, a +glyph that draws none of them by itself, such as the dots an Arabic font draws +apart from their letter, repeats the text of its cluster, which the run's +`ActualText` covers, so every glyph of a rich run maps to Unicode. The +ToUnicode CMap holds one entry per glyph for the whole font, so a glyph that +draws different text in different places keeps its strongest pairing, a GSUB +ligature first, then the cmap, then an inferred pairing. The ligature comes +first because one glyph can be both, as the Carlito `fi` ligature and U+FB01 +are, and one literal U+FB01 must not turn every `fi` of the document into +U+FB01. + When `LayoutResult::structure` is present, the writer emits deterministic `BDC` and `EMC` pairs with page-local MCIDs, `/StructParents`, one parent number tree, `/StructTreeRoot`, `/MarkInfo`, an undetermined `/Lang`, and accessible diff --git a/scripts/hash_baseline.json b/scripts/hash_baseline.json index 511ee7e8..8390ca5d 100644 --- a/scripts/hash_baseline.json +++ b/scripts/hash_baseline.json @@ -1,54 +1,54 @@ { "entries": { "contract:page1.png": "4c4249d4efe7ea72590dfdeb3773d316da626f7189c7fc0606566f4bfb11a92f", - "contract:pdf/bytes": "0b2aa4c1031e4bc5a1a59a77b5a1a0a48ec3d933c5c89306971d7a4a30a78c89", + "contract:pdf/bytes": "2bb2a4344037ee8221cc0f330e1574551236d81e46f4999d0cccca658dd2b355", "contract:pdf/pages": "9b3ff34d7683336e1fea5ad3c42d733dc443ead52d645b79d30f531ff93e00aa", - "contract:pdf/resources": "1acfd604144a00109958c756bb0cb8894089040954cfbd573ceab9b40823d5fc", + "contract:pdf/resources": "1d23fd81e7aaa8ef154223e1574675613bcb428234ef1ded21ffc29ac2aedaf5", "contract:word/document.xml": "a6cebb1e5df54f1191494d9f54600494b7709e09b37402fef42eae06cd48c2fa", "contract:word/numbering.xml": "01d6cb0aa0ecc30b3d2a77c4df062b1c727d70ff9cbc0824910f049daf878a26", "contract:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "feature_showcase:page1.png": "6e88403fe63a789fd2ce6803029f67fcd45e00f2d13b9c13d5201a99eda0ada9", - "feature_showcase:pdf/bytes": "bf8a0c5bbd4ba70aba19b39e1a19bc1c78d19a12ac6f6324964628df5f78d410", + "feature_showcase:pdf/bytes": "38be2148f2e50ec969e0d652e20d6bf4927d90a24cc46e70ca801343e3014d4f", "feature_showcase:pdf/pages": "562b6508a015c227d598b408dfafaf2056cbba4e617e323b20ee39168ff64224", - "feature_showcase:pdf/resources": "e624cef80cac1f99ba398b0b0f8fc7986c0389cf12b6599f2a9bb187749bb797", + "feature_showcase:pdf/resources": "3573cc6e7183b5e912bfde97f11edea4ec44bb40e1b8f066eeb9b4e5e7726d62", "feature_showcase:word/document.xml": "7bfdbc2740871b1b38314dc95b353237d18e49b56ea64fa7a4c6499e4d21c8d3", "feature_showcase:word/numbering.xml": "2aa0599486d98573be0b7febb04ec63c08d3dfc5341b0f722b93cae113803bbc", "feature_showcase:word/styles.xml": "b815acd04cc89189d5b2b0caf822435d941c8cc70077f34932e4a2b3ce1b6595", "invoice:page1.png": "a209c0493bc90aae717e930b4466739212aaa4b646d1123a8a21c5e75b5c9644", - "invoice:pdf/bytes": "8e53fa760135722ab1ddd1593ee6fab2405068efec023e8c281438292fcdba03", + "invoice:pdf/bytes": "f2062c5bf095469f63d8c3b834150b8f7e1ef30ddd3a2fd1e77728f1928ccb87", "invoice:pdf/pages": "aadd561eb1aae1ff10748513e7cbdad0d9bdc6207baa0a286f934528bc3428a2", - "invoice:pdf/resources": "01d97372b5ad8c9fb06649d519f14311eb0428572af93c330c5fb393e74e323a", + "invoice:pdf/resources": "e48c1d7946b7e918931371e483bf48de7906358fd60213be521d49df2b83773c", "invoice:word/document.xml": "848426495d2a6f94e1c1514959938de96a13f9d2997d5ebfdb15ab0d18b7dbe7", "invoice:word/numbering.xml": null, "invoice:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "letter:page1.png": "79b23d965cd43bbdbb3cf2128df8f51b5d2d53d4a8934aaa34ed00b2aef8821e", - "letter:pdf/bytes": "634bf0bc38b2fe8ef5b55feadc3edef8fece56837effac278e4851a910055611", + "letter:pdf/bytes": "3893bd426a5e55eb13f0a4f581cc65da09b83a095605d836d1239a16f37a9667", "letter:pdf/pages": "92a27b62d2e6f40cee2eb742dba240e2b49c002fa2b351e60c7bd6952cd289ba", - "letter:pdf/resources": "0393c6de085093a2af078e708a4092bc1a55137616fd6fac30e01fdb32f732ee", + "letter:pdf/resources": "c63ba60143ff50ba08f981ec6eddde770d71205442441df18902a2c1857f591b", "letter:word/document.xml": "568e43d5433e4bd80d7c8a747a1c9b8b8f295dfb59464570fa37aa32668369a9", "letter:word/numbering.xml": "c6511604704117eb00ad2faffb9173e4f48d22f62557ca72a51d60b7907c8058", "letter:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "proposal:page1.png": "d92257f65b7f149b68f39ace59aec2613924112fe0dc1584ac9dbbd8d3beda34", - "proposal:pdf/bytes": "ccd3d55fea503f170955b5d5c427c873766c152bc3d37fe782fa167438bf92b0", + "proposal:pdf/bytes": "2f6d1bc76995ed9469dedcf022f1b95684ef6b9cb9a47da0ba5a83ed87882967", "proposal:pdf/pages": "1f9b047bfd0e39ff0e46d1fb6fc6cf04fcabac26a3f3735f70352d7b596eaefb", - "proposal:pdf/resources": "7e56e693e3e1fbaf829b2893172b76c7e63345b4b81d37ab5c76a35febc675be", + "proposal:pdf/resources": "775e228634aad44a60949777692e6381bdd3f5d754764c3dae276e241ea81569", "proposal:word/document.xml": "7235579097314e5cc8b07822884ebfadd9927b3ae34820a2673ba63097c5e02e", "proposal:word/numbering.xml": "061e20beb3409ce0f3feda1a99793956124a2967dd23afe1b528dd8fdeb82182", "proposal:word/styles.xml": "dae1d6c1083bba6bb4b29ba2ab748a713f5da60b8c2cd4d52224c6f0a8f21bef", "quote:page1.png": "0cb8e2ee78d96f816e490bd3830ccc6e7807c400c094872b42359284460c62af", - "quote:pdf/bytes": "b74bc8a08f1e66aabcb3334f2110e4da497cb2b467b773c9a61ac072d251db44", + "quote:pdf/bytes": "7cf33d2395adbe235271919ab35872300b230ba55d212adfed804c1f21f4a3e2", "quote:pdf/pages": "bb61084121bf416506cd62bd49683e0b1773f13f72c6d0e15322826d5c4b129b", - "quote:pdf/resources": "c0433753ef113d5aad02e5ce159bf67aded188fb7828d2dc8c99293a81608313", + "quote:pdf/resources": "7e9626233181d07e8d0bfa3c302f6a457133de1e9b5a124ae0e0d53ed7dbe03d", "quote:word/document.xml": "0c72970432f77927f6fb3bd6e980f9e12eebcf3a4c56b5260bf5902ef92ebdb5", "quote:word/numbering.xml": "c6511604704117eb00ad2faffb9173e4f48d22f62557ca72a51d60b7907c8058", "quote:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4", "report:page1.png": "792f1fdc032c7ec1b6a3e451aa7f94fc45e8ea810d12b9815ced82d2fa7fbd82", - "report:pdf/bytes": "960f31cf7cf44328783d793eb1a3e61a66f96463e298af4d499fc08f080590dc", + "report:pdf/bytes": "eea0e88d4a6f70a07ee62606e346aff26df9179c5e38ac0767f1bb29457aa694", "report:pdf/pages": "2e1bd079af84488e8c0c37f8d970e20cfe4c09cf586d293407069c3234ef9b39", - "report:pdf/resources": "cabc4cbad4abfbb595669566cc9ef98e9b96eb7c885f7fbce6a482a4e669b6bf", + "report:pdf/resources": "33cd91323ff09d9d7fad4cc4d83fc51954214304439eba29fb2a415752e1a669", "report:word/document.xml": "7c30b636bc89becd68f098a70127532cbbbe284b008e2ed7cbb31a48d342d5ed", "report:word/numbering.xml": "2aa0599486d98573be0b7febb04ec63c08d3dfc5341b0f722b93cae113803bbc", "report:word/styles.xml": "0dc0b047b6019b798b83bfe4d66eb91d14b03caea3f6bdb0934d48fbb863fba4" }, - "reason": "Paragraph spacing keeps the larger of space after and space before and horizontal table borders take their width between rows, which moves the rendered PDF and PNG of six samples and no OOXML part" + "reason": "ToUnicode CMaps pair each glyph with the characters it draws through the font cmap and GSUB, so Carlito ligatures map to all their characters (GitHub issue #171)" } diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643b..13329963 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,9 +367,10 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", + "oxml-pdf": "2026-09-27", + "rdocx": "2026-09-27", "rdocx-layout": "2026-09-26", - "rpptx": "2026-09-26", + "rpptx": "2026-09-27", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" ARCHIVE_COMPRESSION_TOLERANCE_BYTES = 64 @@ -381,16 +382,16 @@ class ReadmeCase: "oxml-layout": (4_623_324, 9_227_483, 51), "oxml-media": (12_252, 50_992, 6), "oxml-opc": (92_122, 355_510, 12), - "oxml-pdf": (66_015, 304_432, 14), + "oxml-pdf": (72_802, 332_454, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_093_062, 6_500_922, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), "rdocx-oxml": (367_500, 2_380_047, 32), "rdocx-pdf": (8_111, 26_758, 6), - "rpptx": (407_658, 2_122_094, 16), + "rpptx": (408_469, 2_124_754, 16), "rpptx-chart": (6_648, 21_136, 6), "rpptx-cli": (36_709, 159_585, 8), "rpptx-layout": (79_109, 458_112, 11),