From acb82672039ea6b40de9403f0936703659a0b994 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 18:57:56 +0200 Subject: [PATCH 1/3] Parse the TOC instruction paragraph in the scope it inherits rebuild_toc and rdocx toc rebuild failed with "root attribute prefix `w` is unbound" as soon as a run of the TOC instruction paragraph, before separate, carried a prefixed attribute such as w:rsidR, w:rsidRPr or w:rsidDel. Word writes w:rsidR and w:rsidRPr on the runs it saves, and Google Docs exports write them on every run, so a table of contents from either producer could not be rebuilt, packed field runs and a TOC inside a block content control included. An attribute under any other prefix the part binds, such as w14, a foreign namespace or a second WordprocessingML alias, failed the same way. The rebuild cuts the instruction paragraph out of document.xml and validated it with CT_P::from_xml, which never sees the declarations of the part. The run root-attribute retention added with the fix for #130 resolves each attribute prefix through the explicit bindings of its scope, so it refused the first prefixed attribute on a run of that paragraph. The regression never shipped in a release. The field scan now keeps the bindings the instruction paragraph inherits, adds them to the start tag of the cut-out source and validates it with CT_P::from_xml_fragment, as the projected parse already does for each instruction run. Every run then resolves the prefixes it resolved when the document was read. The validation result is still discarded, so a table of contents that rebuilt before rebuilds to the same bytes. GitHub issue #159. --- crates/rdocx-py/tests/test_core.py | 24 +++++ crates/rdocx/src/field.rs | 35 +++++-- crates/rdocx/tests/regression_test.rs | 128 ++++++++++++++++++++++++++ docs/hld/04-opc-and-packaging.md | 5 + 4 files changed, 183 insertions(+), 9 deletions(-) diff --git a/crates/rdocx-py/tests/test_core.py b/crates/rdocx-py/tests/test_core.py index 0f8ce927..a48df4d6 100644 --- a/crates/rdocx-py/tests/test_core.py +++ b/crates/rdocx-py/tests/test_core.py @@ -939,6 +939,30 @@ def test_word_default_toc_switch_rebuilds_and_reports_ordered_diagnostics(): report.diagnostics = () +def test_rebuild_toc_accepts_identity_attributes_on_the_field_runs(): + import rdocx + + # Word writes w:rsidR and w:rsidRPr on the runs it saves, and Google Docs + # exports write them on every run, the runs of the TOC field code included. + source = rdocx.Document() + source.add_paragraph("placeholder") + document = _replace_document_body( + source, + """ + TOC \\o "1-1" + + Heading + """, + ) + + assert document.rebuild_toc() == rdocx.TocRebuildReport( + entry_count=1, bookmark_count=1, diagnostics=() + ) + saved = _document_xml(document) + for identity in (b'w:rsidR="00A1B2C3"', b'w:rsidRPr="00A1B2C3"', b'w:rsidDel="00A1B2C3"'): + assert identity in saved + + def test_update_page_fields_writes_layout_page_numbers(): import rdocx diff --git a/crates/rdocx/src/field.rs b/crates/rdocx/src/field.rs index 6d3f6089..d3bbb1d6 100644 --- a/crates/rdocx/src/field.rs +++ b/crates/rdocx/src/field.rs @@ -3298,6 +3298,7 @@ struct DynamicTocSpan { result_end_position: TocRunPosition, end_run_end: usize, start_paragraph_name: String, + start_paragraph_namespaces: BTreeMap, separator_wrapper_names: Vec, instruction_runs: Vec, end_paragraph_start: usize, @@ -3317,6 +3318,7 @@ struct DynamicFieldScan { result_start: Option, result_start_position: Option, start_paragraph_name: Option, + start_paragraph_namespaces: BTreeMap, separator_wrapper_names: Vec, instruction_runs: Vec, } @@ -4638,6 +4640,7 @@ fn update_dynamic_field_stack( result_start: None, result_start_position: None, start_paragraph_name: None, + start_paragraph_namespaces: BTreeMap::new(), separator_wrapper_names: Vec::new(), instruction_runs: Vec::new(), }), @@ -4687,6 +4690,7 @@ fn update_dynamic_field_stack( } }); field.start_paragraph_name = Some(para.qualified_name.clone()); + field.start_paragraph_namespaces = para.inherited_namespaces.clone(); let paragraph_position = elements .iter() .position(|element| std::ptr::eq(element, para)) @@ -4802,6 +4806,7 @@ fn update_dynamic_field_stack( start_paragraph_name: field.start_paragraph_name.ok_or_else(|| { Error::Other("table of contents field is missing its separator".to_owned()) })?, + start_paragraph_namespaces: field.start_paragraph_namespaces, separator_wrapper_names: field.separator_wrapper_names, instruction_runs: field.instruction_runs, end_paragraph_start: end_para.start, @@ -4851,7 +4856,15 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result { .start_paragraph_name .split_once(':') .map_or("w", |(prefix, _)| prefix); - let mut source = xml[span.instruction_paragraph_start..span.result_start].to_vec(); + // The instruction paragraph is cut out of its part, so its start tag gets + // the declarations it inherits there. Every run in it then resolves the + // prefixes it resolved when the document was read. + let mut source = Vec::new(); + append_with_inherited_namespaces( + &mut source, + &xml[span.instruction_paragraph_start..span.result_start], + &span.start_paragraph_namespaces, + )?; source.extend_from_slice( format!("<{prefix}:r><{prefix}:fldChar {prefix}:fldCharType=\"end\"/>") .as_bytes(), @@ -4881,11 +4894,15 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result { buffer.clear(); } }; - parse_paragraph(&source)?; + CT_P::from_xml_fragment(&source)?; let mut projected = format!("").into_bytes(); for run in &span.instruction_runs { - append_instruction_run_with_namespaces(&mut projected, &xml[run.start..run.end], run)?; + append_with_inherited_namespaces( + &mut projected, + &xml[run.start..run.end], + &run.inherited_namespaces, + )?; } projected.extend_from_slice(b""); let paragraph = parse_paragraph(&projected)?; @@ -4906,10 +4923,10 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result { }) } -fn append_instruction_run_with_namespaces( +fn append_with_inherited_namespaces( output: &mut Vec, raw: &[u8], - run: &DynamicInstructionRun, + inherited_namespaces: &BTreeMap, ) -> Result<()> { let mut reader = quick_xml::Reader::from_reader(raw); reader.config_mut().trim_text(false); @@ -4917,7 +4934,7 @@ fn append_instruction_run_with_namespaces( let (insertion, local_namespaces) = match reader.read_event_into(&mut buffer).map_err(|error| { Error::Other(format!( - "invalid table of contents instruction run: {error}" + "invalid table of contents instruction XML: {error}" )) })? { Event::Start(start) | Event::Empty(start) => { @@ -4925,7 +4942,7 @@ fn append_instruction_run_with_namespaces( for attribute in start.attributes() { let attribute = attribute.map_err(|error| { Error::Other(format!( - "invalid table of contents instruction run: {error}" + "invalid table of contents instruction XML: {error}" )) })?; let key = attribute.key.as_ref(); @@ -4945,12 +4962,12 @@ fn append_instruction_run_with_namespaces( } _ => { return Err(Error::Other( - "table of contents instruction run has no start tag".to_owned(), + "table of contents instruction XML has no start tag".to_owned(), )); } }; output.extend_from_slice(&raw[..insertion]); - for (prefix, namespace) in &run.inherited_namespaces { + for (prefix, namespace) in inherited_namespaces { if prefix == "xml" || local_namespaces.contains(prefix) { continue; } diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c7..8867b381 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -8618,6 +8618,134 @@ fn toc_rebuild_accepts_several_defaults_of_one_style_type_and_follows_the_layout ); } +#[test] +fn toc_rebuild_accepts_identity_attributes_on_the_instruction_paragraph_runs() { + // Word writes `w:rsidR` and `w:rsidRPr` on the runs it saves, Google Docs + // `w:rsidR`, `w:rsidDel` and `w:rsidRPr`. The rebuild parses the + // instruction paragraph out of its part, and the retained-attribute + // capture used to see its `w` prefix unbound and fail the whole rebuild. + // Any other prefix the part binds failed the same way. + let titles = [ + ("Heading1", "Chapter 1"), + ("Heading2", "Section 1.1"), + ("Heading1", "Chapter 2"), + ("Heading2", "Section 2.1"), + ("Heading1", "Chapter 3"), + ("Heading2", "Section 3.1"), + ]; + let body = |field: &str, entry: &str, lead: &str, packed: bool, control: bool| { + let begin = r#""#; + let instruction = + r#" TOC \o "1-3" \h \z \u "#; + let separate = r#""#; + let field_code = if packed { + format!("{begin}{instruction}{separate}") + } else { + [begin, instruction, separate] + .map(|child| format!("{child}")) + .concat() + }; + let mut xml = String::new(); + for (index, (_, title)) in titles.iter().enumerate() { + xml.push_str(""); + if index == 0 { + xml.push_str(lead); + xml.push_str(&field_code); + } + xml.push_str(&format!( + "{title}1" + )); + if index + 1 == titles.len() { + xml.push_str(r#""#); + } + xml.push_str(""); + } + if control { + xml = format!( + r#"{xml}"# + ); + } + for (style, title) in titles { + xml.push_str(&format!( + r#"{title}Body text."# + )); + } + xml + }; + let check = |label: &str, xml: &str, kept: usize| { + let mut document = document_with_field_parts(xml, None, None); + let report = document + .rebuild_toc() + .unwrap_or_else(|error| panic!("{label}: {error}")); + assert_eq!(report.entry_count, 6, "{label}"); + assert!(report.diagnostics.is_empty(), "{label}: {report:?}"); + + // The rebuilt entries replace the cached ones, so only the runs before + // `separate` still carry an identity, and each keeps it. The rebuild + // keeps their bytes, so no run start tag gains a declaration. + let saved = document_xml(&mut document); + assert_eq!( + saved.matches(r#"="00A1B2C3""#).count(), + kept, + "{label}: {saved}" + ); + assert!( + saved + .split("').unwrap()].contains("xmlns")), + "{label}: {saved}" + ); + assert_eq!( + saved.matches(r#"Contents "#; + for (label, field, entry, lead, packed, control, kept) in [ + ("no attribute", "", "", "", false, false, 0), + ("entry runs", "", rsid_r, "", false, false, 0), + ("w:rsidR field runs", rsid_r, "", "", false, false, 3), + ("w:rsidRPr field runs", rsid_rpr, "", "", false, false, 3), + ("w:rsidDel field runs", rsid_del, "", "", false, false, 3), + ("packed field run", rsid_r, "", "", true, false, 1), + ("block control", rsid_r, "", "", false, true, 3), + ("packed in a block control", rsid_r, "", "", true, true, 1), + ("run before begin", "", "", lead, false, false, 1), + ] { + let xml = wrap_word_body(&body(field, entry, lead, packed, control)); + check(label, &xml, kept); + } + for (label, declaration, field) in [ + ( + "foreign prefix", + r#"xmlns:x="urn:producer""#.to_owned(), + r#" x:id="00A1B2C3""#, + ), + ( + "w14 prefix", + r#"xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#.to_owned(), + r#" w14:id="00A1B2C3""#, + ), + ( + "second Word alias", + format!(r#"xmlns:wx="{W_NS}""#), + r#" wx:rsidRPr="00A1B2C3""#, + ), + ] { + let xml = wrap_word_body(&body(field, "", "", false, false)).replacen( + "TOC \o "1-1" \h stale diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index b9c29769..3bbf0d14 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -447,6 +447,11 @@ root that owns the element already declares it and the authored identity write makes the same assumption. Together these keep a reopened save byte identical to the save it was read from. +A paragraph cut out of its part and parsed on its own carries none of the +declarations of its part. The table-of-contents rebuild adds the bindings the +instruction paragraph inherits to its start tag before it parses it, so every +run attribute resolves as it does inside the part. + An unknown default namespace declared on the document root is classified by its effective lexical scope before canonical serialization. An unused root default may be omitted without blocking a typed mutation. An unprefixed element From af94eeaace4494711d49477188d0f8316fd61bfe Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 18:58:49 +0200 Subject: [PATCH 2/3] Resolve run attribute prefixes in text-box paragraphs Text-box paragraphs are also parsed out of their part, with the default scope of CT_P::from_xml, which names w as the Word prefix without binding it. Since the run root-attribute retention added with the fix for #130, a Word text box whose runs carried w:rsidR or w:rsidRPr lost its shape and text from layout inside mc:AlternateContent and failed the document open as a bare wp:anchor. try_replace_text skipped it without reporting anything and render_template failed on it. CT_Anchor now adds the bindings in scope to the start tag of each text-box paragraph before it parses it, on top of the default scope, so a run attribute under any prefix the part binds resolves for layout. The anchor keeps its raw XML, so saved bytes do not change. rewrite_text_boxes and text_box_sources walk the part with a plain reader that tracks no bindings. For them capture_root_attribute_record now binds a plain Word prefix of its scope to the WordprocessingML namespace when the scope has no explicit binding for it. A plain scope entry only ever names a Word prefix and an explicit binding still wins, so every input that parsed before records the same attributes and saves the same bytes. A run attribute under another prefix still fails those two walkers. GitHub issue #159. --- crates/rdocx-oxml/src/drawing.rs | 74 +++++++++++++++------- crates/rdocx-oxml/src/text.rs | 62 +++++++++++++++++- crates/rdocx/tests/regression_test.rs | 90 +++++++++++++++++++++++++++ docs/hld/04-opc-and-packaging.md | 9 ++- 4 files changed, 209 insertions(+), 26 deletions(-) diff --git a/crates/rdocx-oxml/src/drawing.rs b/crates/rdocx-oxml/src/drawing.rs index bcac872b..8fd49f92 100644 --- a/crates/rdocx-oxml/src/drawing.rs +++ b/crates/rdocx-oxml/src/drawing.rs @@ -1,7 +1,7 @@ //! Drawing elements for inline and anchor images: `CT_Drawing`, `CT_Inline`, `CT_Anchor`. use quick_xml::events::{BytesEnd, BytesStart, BytesText, Event}; -use quick_xml::name::{Namespace, ResolveResult}; +use quick_xml::name::{Namespace, PrefixDeclaration, ResolveResult}; use quick_xml::reader::NsReader; use quick_xml::{Reader, Writer, XmlVersion}; @@ -698,29 +698,7 @@ impl CT_Anchor { Ok(Event::Start(ref ie)) if matches_local_name(ie.name().as_ref(), b"p") => { - let raw = capture_ns_element(reader, ie)?; - let mut paragraph_reader = Reader::from_reader(raw.as_slice()); - let mut paragraph_buffer = Vec::new(); - loop { - match paragraph_reader - .read_event_into(&mut paragraph_buffer)? - { - Event::Start(ref paragraph_start) - if matches_local_name( - paragraph_start.name().as_ref(), - b"p", - ) => - { - paragraphs.push(crate::text::CT_P::from_xml( - &mut paragraph_reader, - )?); - break; - } - Event::Eof => break, - _ => {} - } - paragraph_buffer.clear(); - } + paragraphs.extend(text_box_paragraph(reader, ie)?); } Ok(Event::End(ref ie)) if matches_local_name(ie.name().as_ref(), b"txbxContent") => @@ -1384,6 +1362,54 @@ fn canonical_wp_element( namespace_matches(&namespace, drawing_ns::WP, b"wp") && local.as_ref() == expected_local } +/// Parse a text-box paragraph out of its anchor. +/// +/// The paragraph is parsed on its own, so its start tag gets the bindings in +/// scope here, on top of the scope `CT_P::from_xml` assumes. A run attribute +/// under any prefix the part binds then resolves as it does in the body. +fn text_box_paragraph( + reader: &mut NsReader<&[u8]>, + start: &BytesStart<'_>, +) -> Result> { + let mut bindings = Vec::new(); + for (prefix, namespace) in reader.resolver().bindings() { + let prefix = match prefix { + PrefixDeclaration::Default => "", + PrefixDeclaration::Named(prefix) => std::str::from_utf8(prefix)?, + }; + let namespace = quick_xml::escape::unescape(std::str::from_utf8(namespace.as_ref())?) + .map_err(quick_xml::Error::from)?; + bindings.push((prefix.to_owned(), namespace.into_owned())); + } + let raw = + crate::text::raw_with_external_bindings(&capture_ns_element(reader, start)?, &bindings)?; + let mut paragraph_reader = Reader::from_reader(raw.as_slice()); + let mut buffer = Vec::new(); + loop { + match paragraph_reader.read_event_into(&mut buffer)? { + Event::Start(ref paragraph_start) + if matches_local_name(paragraph_start.name().as_ref(), b"p") => + { + let prefixes = crate::numbering::word_prefixes_at( + paragraph_start, + &[ + "w".to_owned(), + format!("\0r\0{}", crate::namespace::R_NS), + format!("\0mc\0{}", crate::namespace::MC_NS), + ], + )?; + return Ok(Some(crate::text::CT_P::from_xml_with_prefixes( + &mut paragraph_reader, + &prefixes, + )?)); + } + Event::Eof => return Ok(None), + _ => {} + } + buffer.clear(); + } +} + fn capture_ns_element(reader: &mut NsReader<&[u8]>, start: &BytesStart<'_>) -> Result> { let mut writer = Writer::new(Vec::new()); writer.write_event(Event::Start(start.to_owned()))?; diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs index cd5501ca..00ab0eaf 100644 --- a/crates/rdocx-oxml/src/text.rs +++ b/crates/rdocx-oxml/src/text.rs @@ -53,7 +53,19 @@ pub(crate) fn capture_root_attribute_record( start: &BytesStart<'_>, prefixes: &[String], ) -> Result>> { - let bindings = namespace_bindings(prefixes); + let mut bindings = namespace_bindings(prefixes); + // A plain scope entry names a Word prefix. `word_prefixes_at` adds one + // only beside its binding, but the default scope of the public `from_xml` + // entrypoints, `CT_P::from_xml` among them, names `w` by convention and + // binds nothing. A caller that parses a paragraph cut out of its part in + // that scope, as the text-box replacement and template walkers do, reaches + // this capture with it, so the Word prefix resolves here instead of + // failing on the first `w:rsidR`. An explicit binding always wins. + for prefix in prefixes { + if !prefix.starts_with('\0') && !bindings.iter().any(|(bound, _)| bound == prefix) { + bindings.push((prefix.clone(), crate::namespace::W_NS.to_owned())); + } + } let mut attributes = Vec::new(); let mut expanded = HashSet::new(); let mut used_prefixes = Vec::new(); @@ -8722,6 +8734,54 @@ mod tests { assert!(record.is_none(), "{record:?}"); } + #[test] + fn a_run_identity_resolves_the_word_prefix_the_default_scope_names() { + // `CT_P::from_xml` names `w` as the Word prefix without binding it, + // which is the scope a paragraph cut out of its part can be parsed in. + // A run identity under it used to fail the paragraph as unbound. + let paragraph = parse_paragraph( + r#"entry"#, + ); + assert_eq!(paragraph.text(), "entry"); + let mut output = Vec::new(); + paragraph.to_xml(&mut Writer::new(&mut output)).unwrap(); + let output = String::from_utf8(output).unwrap(); + assert!( + output.contains(r#" Document { + let content = format!( + r#"{text}"# + ); + let drawing = format!( + r#"00{content}"# + ); + let shape = if compatibility_block { + format!( + r#"{drawing}{content}"# + ) + } else { + drawing + }; + super::document_with_content_controls(&format!( + r#"Host paragraph{shape}"# + )) + } + + #[test] + fn a_text_box_is_laid_out_when_its_runs_carry_identity_attributes() { + // The bare anchor used to fail the document open, and the + // compatibility block used to lose its shape and text from layout. + for compatibility_block in [true, false] { + let page_text = |run_attributes| { + let document = text_box_document(run_attributes, compatibility_block, "Boxed"); + super::f252_page_text(&document.layout_page(0).unwrap().unwrap()) + }; + let plain = page_text(""); + assert!(plain.contains("Boxed"), "{plain}"); + for run_attributes in [IDENTITY, FOREIGN] { + assert_eq!( + page_text(run_attributes), + plain, + "{compatibility_block}{run_attributes}" + ); + } + } + } + + #[test] + fn replacing_text_reaches_a_text_box_whose_runs_carry_identity_attributes() { + let replace = |run_attributes| { + let mut document = text_box_document(run_attributes, true, "Boxed text"); + let count = document.try_replace_text("Boxed", "Filled").unwrap(); + (count, super::document_xml(&mut document)) + }; + let (plain_count, _) = replace(""); + let (count, saved) = replace(IDENTITY); + assert_eq!(count, plain_count); + assert_eq!(saved.matches("Filled text").count(), 2, "{saved}"); + assert!(!saved.contains("Boxed"), "{saved}"); + assert_eq!( + saved.matches(r#"w:rsidRPr="00D4E5F6""#).count(), + 2, + "{saved}" + ); + } + + #[test] + fn a_template_fills_a_text_box_whose_runs_carry_identity_attributes() { + let data = serde_json::json!({"name": "Ada"}); + let render = |run_attributes| { + let mut document = text_box_document(run_attributes, true, "Dear {{ name }}"); + let count = document.render_template(&data).unwrap(); + (count, super::document_xml(&mut document)) + }; + let (plain_count, _) = render(""); + let (count, saved) = render(IDENTITY); + assert_eq!(count, plain_count); + assert_eq!(saved.matches("Dear Ada").count(), 2, "{saved}"); + assert!(!saved.contains("{{"), "{saved}"); + } +} + /// F-266a, script identity and font slot resolution. mod f266a_script_and_font_slot_regressions { use super::*; diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index 3bbf0d14..a15f6e2d 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -450,7 +450,14 @@ to the save it was read from. A paragraph cut out of its part and parsed on its own carries none of the declarations of its part. The table-of-contents rebuild adds the bindings the instruction paragraph inherits to its start tag before it parses it, so every -run attribute resolves as it does inside the part. +run attribute resolves as it does inside the part. The text-box anchor reader +does the same for each text-box paragraph. The text-box replacement and +template walkers still parse such a paragraph in the default scope, which names +`w` as the Word prefix without binding it. For that scope the capture resolves +a Word prefix that the scope names without a binding to the WordprocessingML +namespace, and an explicit binding always wins. A run attribute under any other +prefix, such as `w14`, a foreign namespace or a second WordprocessingML alias, +still fails those two walkers. An unknown default namespace declared on the document root is classified by its effective lexical scope before canonical serialization. An unused root From 1bed176b3256014bb360d0ee1b59cea1a21e7cee Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:05:52 +0200 Subject: [PATCH 3/3] Re-record the archive measurements of rdocx-oxml and rdocx The text-box paragraph scope, the Word prefix fallback and their unit tests grow the rdocx-oxml package, and the TOC instruction scope and the TOC and text-box regression tests grow the rdocx package, so the README archive rows of both crates and their ARCHIVE_MEASUREMENTS entries are re-measured, with today as their measurement date. GitHub issue #159. --- README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 7 ++++--- 3 files changed, 6 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 479b462e..a0520e66 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,095,349 compressed bytes, 6,509,769 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef241..867ca2e3 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 368,549 compressed bytes, 2,383,545 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643b..1c4a709b 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,8 +367,9 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", + "rdocx": "2026-09-27", "rdocx-layout": "2026-09-26", + "rdocx-oxml": "2026-09-27", "rpptx": "2026-09-26", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" @@ -383,12 +384,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_095_349, 6_509_769, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (368_549, 2_383_545, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),