From acb82672039ea6b40de9403f0936703659a0b994 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 18:57:56 +0200 Subject: [PATCH 1/7] Parse the TOC instruction paragraph in the scope it inherits rebuild_toc and rdocx toc rebuild failed with "root attribute prefix `w` is unbound" as soon as a run of the TOC instruction paragraph, before separate, carried a prefixed attribute such as w:rsidR, w:rsidRPr or w:rsidDel. Word writes w:rsidR and w:rsidRPr on the runs it saves, and Google Docs exports write them on every run, so a table of contents from either producer could not be rebuilt, packed field runs and a TOC inside a block content control included. An attribute under any other prefix the part binds, such as w14, a foreign namespace or a second WordprocessingML alias, failed the same way. The rebuild cuts the instruction paragraph out of document.xml and validated it with CT_P::from_xml, which never sees the declarations of the part. The run root-attribute retention added with the fix for #130 resolves each attribute prefix through the explicit bindings of its scope, so it refused the first prefixed attribute on a run of that paragraph. The regression never shipped in a release. The field scan now keeps the bindings the instruction paragraph inherits, adds them to the start tag of the cut-out source and validates it with CT_P::from_xml_fragment, as the projected parse already does for each instruction run. Every run then resolves the prefixes it resolved when the document was read. The validation result is still discarded, so a table of contents that rebuilt before rebuilds to the same bytes. GitHub issue #159. --- crates/rdocx-py/tests/test_core.py | 24 +++++ crates/rdocx/src/field.rs | 35 +++++-- crates/rdocx/tests/regression_test.rs | 128 ++++++++++++++++++++++++++ docs/hld/04-opc-and-packaging.md | 5 + 4 files changed, 183 insertions(+), 9 deletions(-) diff --git a/crates/rdocx-py/tests/test_core.py b/crates/rdocx-py/tests/test_core.py index 0f8ce9271..a48df4d62 100644 --- a/crates/rdocx-py/tests/test_core.py +++ b/crates/rdocx-py/tests/test_core.py @@ -939,6 +939,30 @@ def test_word_default_toc_switch_rebuilds_and_reports_ordered_diagnostics(): report.diagnostics = () +def test_rebuild_toc_accepts_identity_attributes_on_the_field_runs(): + import rdocx + + # Word writes w:rsidR and w:rsidRPr on the runs it saves, and Google Docs + # exports write them on every run, the runs of the TOC field code included. + source = rdocx.Document() + source.add_paragraph("placeholder") + document = _replace_document_body( + source, + """ + TOC \\o "1-1" + + Heading + """, + ) + + assert document.rebuild_toc() == rdocx.TocRebuildReport( + entry_count=1, bookmark_count=1, diagnostics=() + ) + saved = _document_xml(document) + for identity in (b'w:rsidR="00A1B2C3"', b'w:rsidRPr="00A1B2C3"', b'w:rsidDel="00A1B2C3"'): + assert identity in saved + + def test_update_page_fields_writes_layout_page_numbers(): import rdocx diff --git a/crates/rdocx/src/field.rs b/crates/rdocx/src/field.rs index 6d3f60896..d3bbb1d67 100644 --- a/crates/rdocx/src/field.rs +++ b/crates/rdocx/src/field.rs @@ -3298,6 +3298,7 @@ struct DynamicTocSpan { result_end_position: TocRunPosition, end_run_end: usize, start_paragraph_name: String, + start_paragraph_namespaces: BTreeMap, separator_wrapper_names: Vec, instruction_runs: Vec, end_paragraph_start: usize, @@ -3317,6 +3318,7 @@ struct DynamicFieldScan { result_start: Option, result_start_position: Option, start_paragraph_name: Option, + start_paragraph_namespaces: BTreeMap, separator_wrapper_names: Vec, instruction_runs: Vec, } @@ -4638,6 +4640,7 @@ fn update_dynamic_field_stack( result_start: None, result_start_position: None, start_paragraph_name: None, + start_paragraph_namespaces: BTreeMap::new(), separator_wrapper_names: Vec::new(), instruction_runs: Vec::new(), }), @@ -4687,6 +4690,7 @@ fn update_dynamic_field_stack( } }); field.start_paragraph_name = Some(para.qualified_name.clone()); + field.start_paragraph_namespaces = para.inherited_namespaces.clone(); let paragraph_position = elements .iter() .position(|element| std::ptr::eq(element, para)) @@ -4802,6 +4806,7 @@ fn update_dynamic_field_stack( start_paragraph_name: field.start_paragraph_name.ok_or_else(|| { Error::Other("table of contents field is missing its separator".to_owned()) })?, + start_paragraph_namespaces: field.start_paragraph_namespaces, separator_wrapper_names: field.separator_wrapper_names, instruction_runs: field.instruction_runs, end_paragraph_start: end_para.start, @@ -4851,7 +4856,15 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result { .start_paragraph_name .split_once(':') .map_or("w", |(prefix, _)| prefix); - let mut source = xml[span.instruction_paragraph_start..span.result_start].to_vec(); + // The instruction paragraph is cut out of its part, so its start tag gets + // the declarations it inherits there. Every run in it then resolves the + // prefixes it resolved when the document was read. + let mut source = Vec::new(); + append_with_inherited_namespaces( + &mut source, + &xml[span.instruction_paragraph_start..span.result_start], + &span.start_paragraph_namespaces, + )?; source.extend_from_slice( format!("<{prefix}:r><{prefix}:fldChar {prefix}:fldCharType=\"end\"/>") .as_bytes(), @@ -4881,11 +4894,15 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result { buffer.clear(); } }; - parse_paragraph(&source)?; + CT_P::from_xml_fragment(&source)?; let mut projected = format!("").into_bytes(); for run in &span.instruction_runs { - append_instruction_run_with_namespaces(&mut projected, &xml[run.start..run.end], run)?; + append_with_inherited_namespaces( + &mut projected, + &xml[run.start..run.end], + &run.inherited_namespaces, + )?; } projected.extend_from_slice(b""); let paragraph = parse_paragraph(&projected)?; @@ -4906,10 +4923,10 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result { }) } -fn append_instruction_run_with_namespaces( +fn append_with_inherited_namespaces( output: &mut Vec, raw: &[u8], - run: &DynamicInstructionRun, + inherited_namespaces: &BTreeMap, ) -> Result<()> { let mut reader = quick_xml::Reader::from_reader(raw); reader.config_mut().trim_text(false); @@ -4917,7 +4934,7 @@ fn append_instruction_run_with_namespaces( let (insertion, local_namespaces) = match reader.read_event_into(&mut buffer).map_err(|error| { Error::Other(format!( - "invalid table of contents instruction run: {error}" + "invalid table of contents instruction XML: {error}" )) })? { Event::Start(start) | Event::Empty(start) => { @@ -4925,7 +4942,7 @@ fn append_instruction_run_with_namespaces( for attribute in start.attributes() { let attribute = attribute.map_err(|error| { Error::Other(format!( - "invalid table of contents instruction run: {error}" + "invalid table of contents instruction XML: {error}" )) })?; let key = attribute.key.as_ref(); @@ -4945,12 +4962,12 @@ fn append_instruction_run_with_namespaces( } _ => { return Err(Error::Other( - "table of contents instruction run has no start tag".to_owned(), + "table of contents instruction XML has no start tag".to_owned(), )); } }; output.extend_from_slice(&raw[..insertion]); - for (prefix, namespace) in &run.inherited_namespaces { + for (prefix, namespace) in inherited_namespaces { if prefix == "xml" || local_namespaces.contains(prefix) { continue; } diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c74..8867b3810 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -8618,6 +8618,134 @@ fn toc_rebuild_accepts_several_defaults_of_one_style_type_and_follows_the_layout ); } +#[test] +fn toc_rebuild_accepts_identity_attributes_on_the_instruction_paragraph_runs() { + // Word writes `w:rsidR` and `w:rsidRPr` on the runs it saves, Google Docs + // `w:rsidR`, `w:rsidDel` and `w:rsidRPr`. The rebuild parses the + // instruction paragraph out of its part, and the retained-attribute + // capture used to see its `w` prefix unbound and fail the whole rebuild. + // Any other prefix the part binds failed the same way. + let titles = [ + ("Heading1", "Chapter 1"), + ("Heading2", "Section 1.1"), + ("Heading1", "Chapter 2"), + ("Heading2", "Section 2.1"), + ("Heading1", "Chapter 3"), + ("Heading2", "Section 3.1"), + ]; + let body = |field: &str, entry: &str, lead: &str, packed: bool, control: bool| { + let begin = r#""#; + let instruction = + r#" TOC \o "1-3" \h \z \u "#; + let separate = r#""#; + let field_code = if packed { + format!("{begin}{instruction}{separate}") + } else { + [begin, instruction, separate] + .map(|child| format!("{child}")) + .concat() + }; + let mut xml = String::new(); + for (index, (_, title)) in titles.iter().enumerate() { + xml.push_str(""); + if index == 0 { + xml.push_str(lead); + xml.push_str(&field_code); + } + xml.push_str(&format!( + "{title}1" + )); + if index + 1 == titles.len() { + xml.push_str(r#""#); + } + xml.push_str(""); + } + if control { + xml = format!( + r#"{xml}"# + ); + } + for (style, title) in titles { + xml.push_str(&format!( + r#"{title}Body text."# + )); + } + xml + }; + let check = |label: &str, xml: &str, kept: usize| { + let mut document = document_with_field_parts(xml, None, None); + let report = document + .rebuild_toc() + .unwrap_or_else(|error| panic!("{label}: {error}")); + assert_eq!(report.entry_count, 6, "{label}"); + assert!(report.diagnostics.is_empty(), "{label}: {report:?}"); + + // The rebuilt entries replace the cached ones, so only the runs before + // `separate` still carry an identity, and each keeps it. The rebuild + // keeps their bytes, so no run start tag gains a declaration. + let saved = document_xml(&mut document); + assert_eq!( + saved.matches(r#"="00A1B2C3""#).count(), + kept, + "{label}: {saved}" + ); + assert!( + saved + .split("').unwrap()].contains("xmlns")), + "{label}: {saved}" + ); + assert_eq!( + saved.matches(r#"Contents "#; + for (label, field, entry, lead, packed, control, kept) in [ + ("no attribute", "", "", "", false, false, 0), + ("entry runs", "", rsid_r, "", false, false, 0), + ("w:rsidR field runs", rsid_r, "", "", false, false, 3), + ("w:rsidRPr field runs", rsid_rpr, "", "", false, false, 3), + ("w:rsidDel field runs", rsid_del, "", "", false, false, 3), + ("packed field run", rsid_r, "", "", true, false, 1), + ("block control", rsid_r, "", "", false, true, 3), + ("packed in a block control", rsid_r, "", "", true, true, 1), + ("run before begin", "", "", lead, false, false, 1), + ] { + let xml = wrap_word_body(&body(field, entry, lead, packed, control)); + check(label, &xml, kept); + } + for (label, declaration, field) in [ + ( + "foreign prefix", + r#"xmlns:x="urn:producer""#.to_owned(), + r#" x:id="00A1B2C3""#, + ), + ( + "w14 prefix", + r#"xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#.to_owned(), + r#" w14:id="00A1B2C3""#, + ), + ( + "second Word alias", + format!(r#"xmlns:wx="{W_NS}""#), + r#" wx:rsidRPr="00A1B2C3""#, + ), + ] { + let xml = wrap_word_body(&body(field, "", "", false, false)).replacen( + "TOC \o "1-1" \h stale diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index b9c297692..3bbf0d14f 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -447,6 +447,11 @@ root that owns the element already declares it and the authored identity write makes the same assumption. Together these keep a reopened save byte identical to the save it was read from. +A paragraph cut out of its part and parsed on its own carries none of the +declarations of its part. The table-of-contents rebuild adds the bindings the +instruction paragraph inherits to its start tag before it parses it, so every +run attribute resolves as it does inside the part. + An unknown default namespace declared on the document root is classified by its effective lexical scope before canonical serialization. An unused root default may be omitted without blocking a typed mutation. An unprefixed element From af94eeaace4494711d49477188d0f8316fd61bfe Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 18:58:49 +0200 Subject: [PATCH 2/7] Resolve run attribute prefixes in text-box paragraphs Text-box paragraphs are also parsed out of their part, with the default scope of CT_P::from_xml, which names w as the Word prefix without binding it. Since the run root-attribute retention added with the fix for #130, a Word text box whose runs carried w:rsidR or w:rsidRPr lost its shape and text from layout inside mc:AlternateContent and failed the document open as a bare wp:anchor. try_replace_text skipped it without reporting anything and render_template failed on it. CT_Anchor now adds the bindings in scope to the start tag of each text-box paragraph before it parses it, on top of the default scope, so a run attribute under any prefix the part binds resolves for layout. The anchor keeps its raw XML, so saved bytes do not change. rewrite_text_boxes and text_box_sources walk the part with a plain reader that tracks no bindings. For them capture_root_attribute_record now binds a plain Word prefix of its scope to the WordprocessingML namespace when the scope has no explicit binding for it. A plain scope entry only ever names a Word prefix and an explicit binding still wins, so every input that parsed before records the same attributes and saves the same bytes. A run attribute under another prefix still fails those two walkers. GitHub issue #159. --- crates/rdocx-oxml/src/drawing.rs | 74 +++++++++++++++------- crates/rdocx-oxml/src/text.rs | 62 +++++++++++++++++- crates/rdocx/tests/regression_test.rs | 90 +++++++++++++++++++++++++++ docs/hld/04-opc-and-packaging.md | 9 ++- 4 files changed, 209 insertions(+), 26 deletions(-) diff --git a/crates/rdocx-oxml/src/drawing.rs b/crates/rdocx-oxml/src/drawing.rs index bcac872b5..8fd49f92e 100644 --- a/crates/rdocx-oxml/src/drawing.rs +++ b/crates/rdocx-oxml/src/drawing.rs @@ -1,7 +1,7 @@ //! Drawing elements for inline and anchor images: `CT_Drawing`, `CT_Inline`, `CT_Anchor`. use quick_xml::events::{BytesEnd, BytesStart, BytesText, Event}; -use quick_xml::name::{Namespace, ResolveResult}; +use quick_xml::name::{Namespace, PrefixDeclaration, ResolveResult}; use quick_xml::reader::NsReader; use quick_xml::{Reader, Writer, XmlVersion}; @@ -698,29 +698,7 @@ impl CT_Anchor { Ok(Event::Start(ref ie)) if matches_local_name(ie.name().as_ref(), b"p") => { - let raw = capture_ns_element(reader, ie)?; - let mut paragraph_reader = Reader::from_reader(raw.as_slice()); - let mut paragraph_buffer = Vec::new(); - loop { - match paragraph_reader - .read_event_into(&mut paragraph_buffer)? - { - Event::Start(ref paragraph_start) - if matches_local_name( - paragraph_start.name().as_ref(), - b"p", - ) => - { - paragraphs.push(crate::text::CT_P::from_xml( - &mut paragraph_reader, - )?); - break; - } - Event::Eof => break, - _ => {} - } - paragraph_buffer.clear(); - } + paragraphs.extend(text_box_paragraph(reader, ie)?); } Ok(Event::End(ref ie)) if matches_local_name(ie.name().as_ref(), b"txbxContent") => @@ -1384,6 +1362,54 @@ fn canonical_wp_element( namespace_matches(&namespace, drawing_ns::WP, b"wp") && local.as_ref() == expected_local } +/// Parse a text-box paragraph out of its anchor. +/// +/// The paragraph is parsed on its own, so its start tag gets the bindings in +/// scope here, on top of the scope `CT_P::from_xml` assumes. A run attribute +/// under any prefix the part binds then resolves as it does in the body. +fn text_box_paragraph( + reader: &mut NsReader<&[u8]>, + start: &BytesStart<'_>, +) -> Result> { + let mut bindings = Vec::new(); + for (prefix, namespace) in reader.resolver().bindings() { + let prefix = match prefix { + PrefixDeclaration::Default => "", + PrefixDeclaration::Named(prefix) => std::str::from_utf8(prefix)?, + }; + let namespace = quick_xml::escape::unescape(std::str::from_utf8(namespace.as_ref())?) + .map_err(quick_xml::Error::from)?; + bindings.push((prefix.to_owned(), namespace.into_owned())); + } + let raw = + crate::text::raw_with_external_bindings(&capture_ns_element(reader, start)?, &bindings)?; + let mut paragraph_reader = Reader::from_reader(raw.as_slice()); + let mut buffer = Vec::new(); + loop { + match paragraph_reader.read_event_into(&mut buffer)? { + Event::Start(ref paragraph_start) + if matches_local_name(paragraph_start.name().as_ref(), b"p") => + { + let prefixes = crate::numbering::word_prefixes_at( + paragraph_start, + &[ + "w".to_owned(), + format!("\0r\0{}", crate::namespace::R_NS), + format!("\0mc\0{}", crate::namespace::MC_NS), + ], + )?; + return Ok(Some(crate::text::CT_P::from_xml_with_prefixes( + &mut paragraph_reader, + &prefixes, + )?)); + } + Event::Eof => return Ok(None), + _ => {} + } + buffer.clear(); + } +} + fn capture_ns_element(reader: &mut NsReader<&[u8]>, start: &BytesStart<'_>) -> Result> { let mut writer = Writer::new(Vec::new()); writer.write_event(Event::Start(start.to_owned()))?; diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs index cd5501caa..00ab0eaf7 100644 --- a/crates/rdocx-oxml/src/text.rs +++ b/crates/rdocx-oxml/src/text.rs @@ -53,7 +53,19 @@ pub(crate) fn capture_root_attribute_record( start: &BytesStart<'_>, prefixes: &[String], ) -> Result>> { - let bindings = namespace_bindings(prefixes); + let mut bindings = namespace_bindings(prefixes); + // A plain scope entry names a Word prefix. `word_prefixes_at` adds one + // only beside its binding, but the default scope of the public `from_xml` + // entrypoints, `CT_P::from_xml` among them, names `w` by convention and + // binds nothing. A caller that parses a paragraph cut out of its part in + // that scope, as the text-box replacement and template walkers do, reaches + // this capture with it, so the Word prefix resolves here instead of + // failing on the first `w:rsidR`. An explicit binding always wins. + for prefix in prefixes { + if !prefix.starts_with('\0') && !bindings.iter().any(|(bound, _)| bound == prefix) { + bindings.push((prefix.clone(), crate::namespace::W_NS.to_owned())); + } + } let mut attributes = Vec::new(); let mut expanded = HashSet::new(); let mut used_prefixes = Vec::new(); @@ -8722,6 +8734,54 @@ mod tests { assert!(record.is_none(), "{record:?}"); } + #[test] + fn a_run_identity_resolves_the_word_prefix_the_default_scope_names() { + // `CT_P::from_xml` names `w` as the Word prefix without binding it, + // which is the scope a paragraph cut out of its part can be parsed in. + // A run identity under it used to fail the paragraph as unbound. + let paragraph = parse_paragraph( + r#"entry"#, + ); + assert_eq!(paragraph.text(), "entry"); + let mut output = Vec::new(); + paragraph.to_xml(&mut Writer::new(&mut output)).unwrap(); + let output = String::from_utf8(output).unwrap(); + assert!( + output.contains(r#" Document { + let content = format!( + r#"{text}"# + ); + let drawing = format!( + r#"00{content}"# + ); + let shape = if compatibility_block { + format!( + r#"{drawing}{content}"# + ) + } else { + drawing + }; + super::document_with_content_controls(&format!( + r#"Host paragraph{shape}"# + )) + } + + #[test] + fn a_text_box_is_laid_out_when_its_runs_carry_identity_attributes() { + // The bare anchor used to fail the document open, and the + // compatibility block used to lose its shape and text from layout. + for compatibility_block in [true, false] { + let page_text = |run_attributes| { + let document = text_box_document(run_attributes, compatibility_block, "Boxed"); + super::f252_page_text(&document.layout_page(0).unwrap().unwrap()) + }; + let plain = page_text(""); + assert!(plain.contains("Boxed"), "{plain}"); + for run_attributes in [IDENTITY, FOREIGN] { + assert_eq!( + page_text(run_attributes), + plain, + "{compatibility_block}{run_attributes}" + ); + } + } + } + + #[test] + fn replacing_text_reaches_a_text_box_whose_runs_carry_identity_attributes() { + let replace = |run_attributes| { + let mut document = text_box_document(run_attributes, true, "Boxed text"); + let count = document.try_replace_text("Boxed", "Filled").unwrap(); + (count, super::document_xml(&mut document)) + }; + let (plain_count, _) = replace(""); + let (count, saved) = replace(IDENTITY); + assert_eq!(count, plain_count); + assert_eq!(saved.matches("Filled text").count(), 2, "{saved}"); + assert!(!saved.contains("Boxed"), "{saved}"); + assert_eq!( + saved.matches(r#"w:rsidRPr="00D4E5F6""#).count(), + 2, + "{saved}" + ); + } + + #[test] + fn a_template_fills_a_text_box_whose_runs_carry_identity_attributes() { + let data = serde_json::json!({"name": "Ada"}); + let render = |run_attributes| { + let mut document = text_box_document(run_attributes, true, "Dear {{ name }}"); + let count = document.render_template(&data).unwrap(); + (count, super::document_xml(&mut document)) + }; + let (plain_count, _) = render(""); + let (count, saved) = render(IDENTITY); + assert_eq!(count, plain_count); + assert_eq!(saved.matches("Dear Ada").count(), 2, "{saved}"); + assert!(!saved.contains("{{"), "{saved}"); + } +} + /// F-266a, script identity and font slot resolution. mod f266a_script_and_font_slot_regressions { use super::*; diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index 3bbf0d14f..a15f6e2d4 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -450,7 +450,14 @@ to the save it was read from. A paragraph cut out of its part and parsed on its own carries none of the declarations of its part. The table-of-contents rebuild adds the bindings the instruction paragraph inherits to its start tag before it parses it, so every -run attribute resolves as it does inside the part. +run attribute resolves as it does inside the part. The text-box anchor reader +does the same for each text-box paragraph. The text-box replacement and +template walkers still parse such a paragraph in the default scope, which names +`w` as the Word prefix without binding it. For that scope the capture resolves +a Word prefix that the scope names without a binding to the WordprocessingML +namespace, and an explicit binding always wins. A run attribute under any other +prefix, such as `w14`, a foreign namespace or a second WordprocessingML alias, +still fails those two walkers. An unknown default namespace declared on the document root is classified by its effective lexical scope before canonical serialization. An unused root From 1bed176b3256014bb360d0ee1b59cea1a21e7cee Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:05:52 +0200 Subject: [PATCH 3/7] Re-record the archive measurements of rdocx-oxml and rdocx The text-box paragraph scope, the Word prefix fallback and their unit tests grow the rdocx-oxml package, and the TOC instruction scope and the TOC and text-box regression tests grow the rdocx package, so the README archive rows of both crates and their ARCHIVE_MEASUREMENTS entries are re-measured, with today as their measurement date. GitHub issue #159. --- README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 7 ++++--- 3 files changed, 6 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 479b462ef..a0520e66a 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,095,349 compressed bytes, 6,509,769 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef2416..867ca2e35 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 368,549 compressed bytes, 2,383,545 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643be..1c4a709b9 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,8 +367,9 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", + "rdocx": "2026-09-27", "rdocx-layout": "2026-09-26", + "rdocx-oxml": "2026-09-27", "rpptx": "2026-09-26", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" @@ -383,12 +384,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_095_349, 6_509_769, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (368_549, 2_383_545, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6), From 547c3a7430a0a24c88a5ec1165b84e1398c2355b Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:21:26 +0200 Subject: [PATCH 4/7] Keep identity attributes on table rows through modelled edits Word and Google Docs write w:rsidR, w:rsidTr, w:rsidRPr, w:rsidDel, w14:paraId and w14:textId on every table row. They survived a save with no edit only because the original bytes were written back. As soon as any modelled edit changed the document, even a one-word replacement outside the table, every row was written with a bare start tag and lost them, while the same attributes on paragraphs, runs and section properties are kept since the fix for #130. CT_Row now keeps the attributes of its start tag in the retention record that paragraphs use, at a raw-child position no cell boundary reaches, and writes them back on w:tr. Direct rows, self-closing rows and rows inside a table-level content control all capture it. The record carries no content, so every reader that treats raw row children as content skips it: the comparison row signatures and row boundary check, the retained table layout cache, RowRef::has_unsupported_content and the row diagnostics of the MHTML, ODT, RTF and EPUB writers, and the rich mail merge row-region markers. Without that, an identity-only row difference would refuse the comparison and a Word-saved row would turn off the table cache, raise a spurious export diagnostic or hide a merge region. The rich mail merge region-marker and whole-paragraph fragment checks skip the record of a paragraph the same way. Word writes w:rsidR and w14:paraId on the marker paragraph whenever it writes them on its row, and since the fix for #130 that record hid every region marker and fragment field of a document Word saved. The record writes w14:paraId without its declaration, on the assumption that the part root declares w14. A root rdocx wrote does not, and a row may declare the prefix itself, so an edit elsewhere could save a part with an unbound prefix, and a comparison against a copy saved by Word failed. The document, header, footer, note and comment part serializers and the comparison output of every story now declare w14 on the part root when the written content uses it and the root does not. Paragraphs, runs and section properties share the record and gain the same guarantee. GitHub issue #159. --- crates/rdocx-layout/src/engine.rs | 22 +- crates/rdocx-oxml/src/comments.rs | 19 +- crates/rdocx-oxml/src/content_control.rs | 3 +- crates/rdocx-oxml/src/document.rs | 7 +- crates/rdocx-oxml/src/footnotes.rs | 18 +- crates/rdocx-oxml/src/header_footer.rs | 20 +- crates/rdocx-oxml/src/table.rs | 93 +++++- crates/rdocx-oxml/src/text.rs | 70 ++++- crates/rdocx/src/comparison.rs | 30 +- crates/rdocx/src/epub.rs | 5 +- crates/rdocx/src/field.rs | 26 +- crates/rdocx/src/odt.rs | 5 +- crates/rdocx/src/rtf.rs | 5 +- crates/rdocx/src/table.rs | 9 +- crates/rdocx/tests/regression_test.rs | 360 +++++++++++++++++++++++ docs/hld/04-opc-and-packaging.md | 25 +- docs/hld/12-testing-strategy.md | 11 +- 17 files changed, 683 insertions(+), 45 deletions(-) diff --git a/crates/rdocx-layout/src/engine.rs b/crates/rdocx-layout/src/engine.rs index 497200c13..5e2cc8b5a 100644 --- a/crates/rdocx-layout/src/engine.rs +++ b/crates/rdocx-layout/src/engine.rs @@ -3617,7 +3617,9 @@ fn table_is_cache_safe(table: &CT_Tbl, styles: &CT_Styles) -> bool { properties.change.is_none() && properties.revision_xml.is_empty() }) && table.rows.iter().all(|row| { - row.extra_xml.is_empty() + row.extra_xml + .iter() + .all(|(position, raw)| CT_Row::raw_is_root_attributes(*position, raw)) && row.content_controls.is_empty() && row.properties.as_ref().is_none_or(|properties| { properties.revision_markers.is_empty() && properties.revision_xml.is_empty() @@ -14416,6 +14418,24 @@ mod tests { .extra_xml .push((0, br#""#.to_vec())); assert!(!table_is_cache_safe(&preserved_cell, &input.styles)); + + // Word writes revision-save and paragraph identities on every row. + // They carry no content, so the row stays cache safe, unlike a raw + // child of the row. + let document = rdocx_oxml::CT_Document::from_xml( + br#"identified row"#, + ) + .expect("identified row document parses"); + let Some(BodyContent::Table(mut identified_row)) = document.body.content.into_iter().next() + else { + panic!("identified row table parses"); + }; + assert!(!identified_row.rows[0].extra_xml.is_empty()); + assert!(table_is_cache_safe(&identified_row, &input.styles)); + identified_row.rows[0] + .extra_xml + .push((0, br#""#.to_vec())); + assert!(!table_is_cache_safe(&identified_row, &input.styles)); } fn restart_input() -> LayoutInput { diff --git a/crates/rdocx-oxml/src/comments.rs b/crates/rdocx-oxml/src/comments.rs index 05edd9237..85c83497f 100644 --- a/crates/rdocx-oxml/src/comments.rs +++ b/crates/rdocx-oxml/src/comments.rs @@ -10,7 +10,7 @@ use crate::namespace::{W_NS, matches_local_name}; use crate::numbering::word_prefixes_at; use crate::properties::is_word_element; use crate::raw_xml::{capture_element, capture_empty_element}; -use crate::text::CT_P; +use crate::text::{CT_P, declare_w14_on_part_root}; const W14_NS: &str = "http://schemas.microsoft.com/office/word/2010/wordml"; @@ -137,7 +137,9 @@ impl CT_Comments { write_raw_at(&mut writer, &self.extra_xml, index + 1)?; } writer.write_event(Event::End(BytesEnd::new("w:comments")))?; - Ok(writer.into_inner()) + let mut xml = writer.into_inner(); + declare_w14_on_part_root(&mut xml)?; + Ok(xml) } } @@ -447,6 +449,19 @@ mod tests { assert!(output.contains(r#"ext:item="kept""#)); } + #[test] + fn a_retained_text_identity_stays_bound_without_a_paragraph_identity() { + // The root declares `w14` for an authored paragraph identity only, so + // a paragraph that carries `w14:textId` alone relies on the check of + // the written part. + let xml = br#"thread"#; + let comments = CT_Comments::from_xml(xml).unwrap(); + let output = comments.to_xml().unwrap(); + let reparsed = CT_Comments::from_xml(&output) + .unwrap_or_else(|error| panic!("{error}: {}", String::from_utf8_lossy(&output))); + assert_eq!(reparsed.to_xml().unwrap(), output); + } + #[test] fn malformed_comment_id_is_rejected() { let xml = br#""#; diff --git a/crates/rdocx-oxml/src/content_control.rs b/crates/rdocx-oxml/src/content_control.rs index 8ff31cd52..3fbcfec1d 100644 --- a/crates/rdocx-oxml/src/content_control.rs +++ b/crates/rdocx-oxml/src/content_control.rs @@ -1242,6 +1242,7 @@ fn parse_content( reader, &prefixes, &owner_bindings, + Some(&child), )?, )); } else if is_word_element(child.name().as_ref(), b"tc", &prefixes) { @@ -1298,7 +1299,7 @@ fn parse_content( } else if is_word_element(child.name().as_ref(), b"tbl", &prefixes) { content.push(SdtContent::Table(CT_Tbl::new())); } else if is_word_element(child.name().as_ref(), b"tr", &prefixes) { - content.push(SdtContent::Row(CT_Row::new())); + content.push(SdtContent::Row(CT_Row::from_empty_root(&child, &prefixes)?)); } else if is_word_element(child.name().as_ref(), b"tc", &prefixes) { content.push(SdtContent::Cell(CT_Tc { properties: None, diff --git a/crates/rdocx-oxml/src/document.rs b/crates/rdocx-oxml/src/document.rs index 3fc302e93..00e823531 100644 --- a/crates/rdocx-oxml/src/document.rs +++ b/crates/rdocx-oxml/src/document.rs @@ -17,7 +17,8 @@ use crate::revision::CT_Revision; use crate::shared::{ST_PageOrientation, ST_SectionType}; use crate::table::{CT_Tbl, ST_VerticalJc}; use crate::text::{ - CT_P, capture_root_attribute_record, is_root_attribute_record, push_root_attribute_record, + CT_P, capture_root_attribute_record, declare_w14_on_part_root, is_root_attribute_record, + push_root_attribute_record, }; use crate::units::Twips; @@ -2998,7 +2999,9 @@ impl CT_Document { writer.write_event(Event::End(BytesEnd::new("w:document")))?; - Ok(writer.into_inner()) + let mut xml = writer.into_inner(); + declare_w14_on_part_root(&mut xml)?; + Ok(xml) } } diff --git a/crates/rdocx-oxml/src/footnotes.rs b/crates/rdocx-oxml/src/footnotes.rs index 035788b17..b0a574085 100644 --- a/crates/rdocx-oxml/src/footnotes.rs +++ b/crates/rdocx-oxml/src/footnotes.rs @@ -7,7 +7,7 @@ use crate::error::Result; use crate::namespace::{W_NS, matches_local_name}; use crate::numbering::word_prefixes_at; use crate::properties::is_word_element; -use crate::text::CT_P; +use crate::text::{CT_P, declare_w14_on_part_root}; /// `ST_FtnEdn` — what a note in the stream is for. /// @@ -212,7 +212,9 @@ impl CT_Footnotes { writer.write_event(Event::End(BytesEnd::new(root_tag)))?; - Ok(writer.into_inner()) + let mut xml = writer.into_inner(); + declare_w14_on_part_root(&mut xml)?; + Ok(xml) } } @@ -273,6 +275,18 @@ fn parse_footnote_content( mod tests { use super::*; + #[test] + fn a_note_paragraph_identity_stays_bound_under_the_written_root() { + // The written root declares only `w` and `r`, while Word declares + // `w14` on the root of the notes part it writes. + let xml = br#"note"#; + let footnotes = CT_Footnotes::from_xml(xml).unwrap(); + let output = footnotes.to_xml_footnotes().unwrap(); + let reparsed = CT_Footnotes::from_xml(&output) + .unwrap_or_else(|error| panic!("{error}: {}", String::from_utf8_lossy(&output))); + assert_eq!(reparsed.to_xml_footnotes().unwrap(), output); + } + #[test] fn parse_footnotes_xml() { let xml = br#" diff --git a/crates/rdocx-oxml/src/header_footer.rs b/crates/rdocx-oxml/src/header_footer.rs index 2ec2cb1b8..104b541ca 100644 --- a/crates/rdocx-oxml/src/header_footer.rs +++ b/crates/rdocx-oxml/src/header_footer.rs @@ -10,7 +10,7 @@ use crate::namespace::{W_NS, matches_local_name}; use crate::numbering::{namespace_bindings, word_prefixes_at}; use crate::properties::is_word_element; use crate::raw_xml::{capture_element, capture_empty_element}; -use crate::text::CT_P; +use crate::text::{CT_P, declare_w14_on_part_root}; const VML_NS: &str = "urn:schemas-microsoft-com:vml"; const OFFICE_NS: &str = "urn:schemas-microsoft-com:office:office"; @@ -352,7 +352,9 @@ impl CT_HdrFtr { writer.write_event(Event::End(BytesEnd::new(root_tag)))?; - Ok(writer.into_inner()) + let mut xml = writer.into_inner(); + declare_w14_on_part_root(&mut xml)?; + Ok(xml) } } @@ -1186,6 +1188,20 @@ pub struct HdrFtrRef { mod tests { use super::*; + #[test] + fn a_paragraph_identity_stays_bound_when_only_the_paragraph_declares_w14() { + let w14 = "http://schemas.microsoft.com/office/word/2010/wordml"; + let xml = format!( + r#"header"# + ); + let header = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + let output = String::from_utf8(header.to_xml_header().unwrap()).unwrap(); + let root = &output[output.find("').unwrap()]; + assert!(root.contains(&format!(r#"xmlns:w14="{w14}""#)), "{output}"); + assert!(output.contains(r#"w14:paraId="1A2B3C4D""#), "{output}"); + } + #[test] fn round_trip_header() { let mut hdr = CT_HdrFtr::new(); diff --git a/crates/rdocx-oxml/src/table.rs b/crates/rdocx-oxml/src/table.rs index 3bb9d9880..07ef32bc9 100644 --- a/crates/rdocx-oxml/src/table.rs +++ b/crates/rdocx-oxml/src/table.rs @@ -19,7 +19,10 @@ use crate::revision::{CT_Revision, RevisionKind}; #[cfg(test)] use crate::shared::ST_Border; use crate::shared::ST_Jc; -use crate::text::CT_P; +use crate::text::{ + CT_P, ROOT_ATTRIBUTES_POSITION, capture_root_attribute_record, is_root_attribute_record, + push_root_attribute_record, +}; use crate::units::Twips; const MAX_RECOGNIZED_TABLE_NESTING: usize = 32; @@ -2192,7 +2195,9 @@ pub struct CT_Row { pub properties: Option, pub cells: Vec, /// Raw XML for children we do not model, tagged with the cell index they - /// appeared before so they can be written back in place. + /// appeared before so they can be written back in place. The attributes + /// of the `w:tr` start tag, such as `w:rsidR` or `w14:paraId`, are kept in + /// one record at position `usize::MAX`, which no cell boundary reaches. pub extra_xml: Vec<(usize, Vec)>, /// Typed cell controls at `(cell index, raw children before, control)`. pub content_controls: Vec<(usize, usize, CT_Sdt)>, @@ -2210,20 +2215,36 @@ impl CT_Row { } } + /// Report whether one raw row carrier retains root attributes. + #[doc(hidden)] + pub fn raw_is_root_attributes(position: usize, raw: &[u8]) -> bool { + position == ROOT_ATTRIBUTES_POSITION && is_root_attribute_record(raw) + } + + pub(crate) fn from_empty_root(root: &BytesStart<'_>, word_prefixes: &[String]) -> Result { + let mut row = Self::new(); + if let Some(record) = capture_root_attribute_record(root, word_prefixes)? { + row.extra_xml.push((ROOT_ATTRIBUTES_POSITION, record)); + } + Ok(row) + } + pub fn from_xml(reader: &mut Reader<&[u8]>) -> Result { - Self::from_xml_with_prefixes_and_owner_bindings(reader, &["w".to_string()], &[]) + Self::from_xml_with_prefixes_and_owner_bindings(reader, &["w".to_string()], &[], None) } pub(crate) fn from_xml_with_prefixes_and_owner_bindings( reader: &mut Reader<&[u8]>, word_prefixes: &[String], owner_bindings: &[(String, String)], + root: Option<&BytesStart<'_>>, ) -> Result { Self::from_xml_with_prefixes_and_owner_bindings_at_depth( reader, word_prefixes, owner_bindings, 0, + root, ) } @@ -2232,6 +2253,7 @@ impl CT_Row { word_prefixes: &[String], owner_bindings: &[(String, String)], table_depth: usize, + root: Option<&BytesStart<'_>>, ) -> Result { let mut table_property_exception = None; let mut properties = None; @@ -2328,6 +2350,14 @@ impl CT_Row { buf.clear(); } + if let Some(record) = root + .map(|root| capture_root_attribute_record(root, word_prefixes)) + .transpose()? + .flatten() + { + extra_xml.push((ROOT_ATTRIBUTES_POSITION, record)); + } + Ok(CT_Row { table_property_exception, properties, @@ -2338,7 +2368,13 @@ impl CT_Row { } pub fn to_xml(&self, writer: &mut Writer) -> Result<()> { - writer.write_event(Event::Start(BytesStart::new("w:tr")))?; + let mut start = BytesStart::new("w:tr"); + for (position, raw) in &self.extra_xml { + if Self::raw_is_root_attributes(*position, raw) { + push_root_attribute_record(&mut start, raw, None)?; + } + } + writer.write_event(Event::Start(start))?; if let Some(raw) = &self.table_property_exception { writer.get_mut().write_all(raw)?; @@ -2517,6 +2553,7 @@ impl CT_Tbl { &prefixes, &row_bindings, table_depth, + Some(e), )?); } else if is_word_element(name.as_ref(), b"sdt", &prefixes) { let raw = crate::text::raw_with_external_bindings( @@ -2560,7 +2597,7 @@ impl CT_Tbl { if is_word_element(name.as_ref(), b"tblGrid", &prefixes) { grid = Some(CT_TblGrid::default()); } else if is_word_element(name.as_ref(), b"tr", &prefixes) { - rows.push(CT_Row::new()); + rows.push(CT_Row::from_empty_root(e, &prefixes)?); } else if matches_local_name(name.as_ref(), b"tblGrid") { let raw = capture_empty_element(e)?; let raw_bindings = merged_owner_bindings( @@ -4025,6 +4062,52 @@ mod tests { assert!(exception < properties, "tblPrEx must precede trPr: {xml}"); } + /// Word and Google Docs write revision-save and paragraph identities on + /// every row. A row used to be written back with a bare start tag, so any + /// save of an edited model lost them. + #[test] + fn row_root_attributes_survive_serialization_in_source_order() { + const IDENTITY: &str = + r#"w:rsidR="00A1B2C3" w:rsidTr="00A1B2C4" w14:paraId="1A2B3C4D" w14:textId="5E6F7A8B""#; + let cell = r#"x"#; + let xml = format!( + r#"{cell}{cell}"#, + crate::namespace::W_NS + ); + let mut table = parse_scoped_table(&xml).unwrap(); + let direct = &table.rows[0]; + assert_eq!(direct.cells.len(), 1); + assert!( + direct + .extra_xml + .iter() + .any(|(position, raw)| CT_Row::raw_is_root_attributes(*position, raw)) + ); + table.rows[0].cells.push(CT_Tc::new()); + + let written = table_to_xml(&table); + assert_eq!( + written.matches(&format!(""#), + "{written}" + ); + let reparsed = parse_scoped_table(&written.replacen( + "", + &format!( + r#""#, + crate::namespace::W_NS + ), + 1, + )) + .unwrap(); + assert_eq!(table_to_xml(&reparsed), written); + } + #[test] fn unmodelled_table_property_groups_are_retained() { let table = parse_table( diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs index 00ab0eaf7..0c87e502d 100644 --- a/crates/rdocx-oxml/src/text.rs +++ b/crates/rdocx-oxml/src/text.rs @@ -22,7 +22,7 @@ use crate::table::{CT_Row, CT_Tbl, CT_Tc, CellContent}; static NEXT_FIELD_SOURCE_ID: AtomicU64 = AtomicU64::new(1); const ROOT_ATTRIBUTES_ELEMENT: &[u8] = b"rdocxRootAttributes"; -const ROOT_ATTRIBUTES_POSITION: usize = usize::MAX; +pub(crate) const ROOT_ATTRIBUTES_POSITION: usize = usize::MAX; const W14_NS: &str = "http://schemas.microsoft.com/office/word/2010/wordml"; fn namespace_declaration(name: &[u8]) -> bool { @@ -188,6 +188,45 @@ pub(crate) fn push_root_attribute_record( Ok(()) } +/// Declare the canonical `w14` prefix on the root of a serialized part whose +/// content uses the prefix while the root does not bind it. +/// +/// A retained root-attribute record writes `w14:paraId` and `w14:textId` +/// without their declaration, because Word and python-docx declare `w14` on +/// the part root. A root rdocx wrote does not, and a producer may declare the +/// prefix on the element that uses it, so the part declares it here once. A +/// root that binds `w14` itself is left as it is. +#[doc(hidden)] +pub fn declare_w14_on_part_root(xml: &mut Vec) -> Result<()> { + let mut reader = Reader::from_reader(xml.as_slice()); + let mut buffer = Vec::new(); + let root_end = loop { + match reader.read_event_into(&mut buffer)? { + Event::Start(root) => { + for attribute in root.attributes() { + if attribute?.key.as_ref() == b"xmlns:w14" { + return Ok(()); + } + } + break reader.buffer_position() as usize; + } + Event::Empty(_) | Event::Eof => return Ok(()), + _ => {} + } + buffer.clear(); + }; + // A qualified name starts after `<`, `{body}"# + ) + .into_bytes() + }; + let identity = r#""#; + let mut unbound = part("", identity); + declare_w14_on_part_root(&mut unbound).unwrap(); + assert_eq!( + unbound, + part(&format!(r#" xmlns:w14="{W14_NS}""#), identity) + ); + crate::document::CT_Document::from_xml(&unbound).unwrap(); + + for unchanged in [ + part(&format!(r#" xmlns:w14="{W14_NS}""#), identity), + part(r#" xmlns:w14="urn:other""#, identity), + part("", r#"w14"#), + ] { + let mut xml = unchanged.clone(); + declare_w14_on_part_root(&mut xml).unwrap(); + assert_eq!(xml, unchanged); + } + } + #[test] fn a_root_with_only_namespace_declarations_records_nothing() { let start = BytesStart::from_content( diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 19c557ed0..057d88752 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -12,7 +12,7 @@ use rdocx_oxml::document::{BodyContent, CT_Document}; use rdocx_oxml::namespace::W_NS; use rdocx_oxml::properties::CT_PPr; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_TblPr, CT_Tc, CT_TrPr, CellContent}; -use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent}; +use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent, declare_w14_on_part_root}; use sha2::{Digest, Sha256}; use crate::revision::validate_revision_timestamp; @@ -357,8 +357,11 @@ impl Document { &mut diagnostics, )? }; - let tracked_xml = replace_body_inner(original_xml, &tracked_body)?; - let tracked_xml = crate::document::uniquify_drawing_ids_in_xml(tracked_xml.as_bytes())?; + let mut tracked_xml = replace_body_inner(original_xml, &tracked_body)?.into_bytes(); + // Content from the edited document can use the `w14` its own root + // declares, which the original root may not. + declare_w14_on_part_root(&mut tracked_xml)?; + let tracked_xml = crate::document::uniquify_drawing_ids_in_xml(&tracked_xml)?; let tracked_model_xml = close_drawing_namespaces(tracked_xml, true)?; let tracked = CT_Document::from_xml(&tracked_model_xml)?; tracked.to_xml()?; @@ -877,8 +880,10 @@ fn compare_story_part( tracked.replace_range(original_root, &tracked_inner); tracked }; - crate::revision::modeled_revision_count(tracked.as_bytes())?; - Ok(tracked.into_bytes()) + let mut tracked = tracked.into_bytes(); + declare_w14_on_part_root(&mut tracked)?; + crate::revision::modeled_revision_count(&tracked)?; + Ok(tracked) } #[allow(clippy::too_many_arguments)] @@ -4358,7 +4363,7 @@ fn compare_row( diagnostics: &mut Vec, ) -> Result { if original.cells.len() != edited.cells.len() - || original.extra_xml != edited.extra_xml + || row_raw_children(original) != row_raw_children(edited) || row_control_boundaries(original) != row_control_boundaries(edited) { return Err(Error::Other(format!( @@ -4415,6 +4420,15 @@ fn compare_row( replace_direct_word_elements(&output, "sdt", &control_replacements) } +/// Return the raw children of a row without the record of its start-tag +/// attributes, which carry producer identities and no content. +fn row_raw_children(row: &CT_Row) -> Vec<&(usize, Vec)> { + row.extra_xml + .iter() + .filter(|(position, raw)| !CT_Row::raw_is_root_attributes(*position, raw)) + .collect() +} + fn row_control_boundaries(row: &CT_Row) -> Vec<(usize, usize)> { row.content_controls .iter() @@ -5433,7 +5447,7 @@ fn row_signature(row: &CT_Row) -> String { format!( "{:?}:{:?}:{:?}", row.cells.iter().map(cell_signature).collect::>(), - row.extra_xml, + row_raw_children(row), row.content_controls .iter() .map(|(at, raw_before, control)| (at, raw_before, control_signature(control))) @@ -5674,7 +5688,7 @@ fn row_signature_with_options(row: &CT_Row, options: &ComparisonOptions) -> Stri .iter() .map(|cell| cell_signature_with_options(cell, options)) .collect::>(), - row.extra_xml, + row_raw_children(row), row.content_controls .iter() .map(|(at, raw_before, control)| ( diff --git a/crates/rdocx/src/epub.rs b/crates/rdocx/src/epub.rs index 0105065c4..d6b0bce41 100644 --- a/crates/rdocx/src/epub.rs +++ b/crates/rdocx/src/epub.rs @@ -557,7 +557,10 @@ impl<'a> EpubWriter<'a> { if let Some(properties) = &row.properties { self.scan_row_properties(properties, &row_path)?; } - for (raw_index, _) in row.extra_xml.iter().enumerate() { + for (raw_index, (position, raw)) in row.extra_xml.iter().enumerate() { + if CT_Row::raw_is_root_attributes(*position, raw) { + continue; + } self.diagnose( format!("{row_path}/xml[{raw_index}]"), "unmodelled table-row XML was dropped during EPUB export".to_owned(), diff --git a/crates/rdocx/src/field.rs b/crates/rdocx/src/field.rs index d3bbb1d67..a2b8da9fd 100644 --- a/crates/rdocx/src/field.rs +++ b/crates/rdocx/src/field.rs @@ -1643,10 +1643,7 @@ fn merge_field_name(field: &Field) -> Option { } fn rich_region_marker(paragraph: &CT_P) -> Option { - if paragraph - .extra_xml - .iter() - .any(|(_, raw)| !raw.iter().all(u8::is_ascii_whitespace)) + if paragraph_has_raw_content(paragraph) || paragraph.properties.is_some() || !paragraph.content_controls.is_empty() || !paragraph.revisions.is_empty() @@ -1686,6 +1683,14 @@ fn rich_region_marker(paragraph: &CT_P) -> Option { }) } +/// Whether a paragraph holds raw XML other than whitespace. The attributes +/// of its start tag, such as `w:rsidR` or `w14:paraId`, are not content. +fn paragraph_has_raw_content(paragraph: &CT_P) -> bool { + paragraph.extra_xml.iter().any(|(position, raw)| { + !CT_P::raw_is_root_attributes(*position, raw) && !raw.iter().all(u8::is_ascii_whitespace) + }) +} + fn find_body_region_end(items: &[BodyContent], start: usize, name: &str) -> Result { let mut stack = vec![name.to_owned()]; for (index, item) in items.iter().enumerate().skip(start) { @@ -1801,10 +1806,7 @@ fn whole_paragraph_fragment<'a>( paragraph: &CT_P, scopes: &[&'a MailMergeRecord], ) -> Result> { - if paragraph - .extra_xml - .iter() - .any(|(_, raw)| !raw.iter().all(u8::is_ascii_whitespace)) + if paragraph_has_raw_content(paragraph) || !paragraph.content_controls.is_empty() || !paragraph.revisions.is_empty() || !paragraph.hyperlinks.is_empty() @@ -3057,7 +3059,13 @@ fn expand_rich_rows( } fn rich_row_region_marker(row: &CT_Row) -> Option { - if !row.extra_xml.is_empty() || !row.content_controls.is_empty() || row.cells.len() != 1 { + if row + .extra_xml + .iter() + .any(|(position, raw)| !CT_Row::raw_is_root_attributes(*position, raw)) + || !row.content_controls.is_empty() + || row.cells.len() != 1 + { return None; } let cell = &row.cells[0]; diff --git a/crates/rdocx/src/odt.rs b/crates/rdocx/src/odt.rs index c40d96574..33199fed3 100644 --- a/crates/rdocx/src/odt.rs +++ b/crates/rdocx/src/odt.rs @@ -339,7 +339,10 @@ impl<'a> OdtWriter<'a> { "table-row properties were dropped during ODT export", )?; } - for (index, _) in &row.extra_xml { + for (index, raw) in &row.extra_xml { + if CT_Row::raw_is_root_attributes(*index, raw) { + continue; + } self.diagnose( &format!("{row_path}/raw[{index}]"), "unmodelled table-row XML was dropped during ODT export", diff --git a/crates/rdocx/src/rtf.rs b/crates/rdocx/src/rtf.rs index 615a470a0..dd836b4ca 100644 --- a/crates/rdocx/src/rtf.rs +++ b/crates/rdocx/src/rtf.rs @@ -247,7 +247,10 @@ impl<'a> RtfWriter<'a> { if let Some(properties) = &row.properties { self.scan_row_properties(properties, &format!("{location}/row[{row_index}]")); } - for (index, _) in &row.extra_xml { + for (index, raw) in &row.extra_xml { + if CT_Row::raw_is_root_attributes(*index, raw) { + continue; + } self.diagnose( &format!("{location}/row[{row_index}]/raw[{index}]"), "unmodelled table-row XML was dropped during RTF export", diff --git a/crates/rdocx/src/table.rs b/crates/rdocx/src/table.rs index 4101eafbd..b0b6489dc 100644 --- a/crates/rdocx/src/table.rs +++ b/crates/rdocx/src/table.rs @@ -2313,8 +2313,15 @@ impl<'a> RowRef<'a> { } /// Whether the row contains direct content other than cells. + /// + /// The attributes of the row start tag, such as `w:rsidR` or + /// `w14:paraId`, are not content. pub fn has_unsupported_content(&self) -> bool { - !self.inner.extra_xml.is_empty() || !self.inner.content_controls.is_empty() + self.inner + .extra_xml + .iter() + .any(|(position, raw)| !CT_Row::raw_is_root_attributes(*position, raw)) + || !self.inner.content_controls.is_empty() } /// Whether row properties retain facts outside the public reader model. diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 87e57f34f..264a67f2a 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -14538,6 +14538,366 @@ fn paragraph_run_and_section_identity_attributes_survive_noop_save() { assert_eq!(stable.to_bytes().unwrap(), reopened.to_bytes().unwrap()); } +/// Word and Google Docs write revision-save and paragraph identities on every +/// table row. They survived only a save with no edit, because the row model +/// had no carrier for them. They are now kept like the ones on paragraphs and +/// runs, and nothing that reads a row takes them for row content. +mod table_row_identity_attribute_regressions { + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + const W14_NS: &str = "http://schemas.microsoft.com/office/word/2010/wordml"; + /// Every identity Word writes on a row, in the order it writes them. + const ALL: &str = r#" w:rsidR="00A1B2C3" w:rsidTr="00A1B2C4" w:rsidRPr="00A1B2C5" w:rsidDel="00A1B2C6" w14:paraId="1A2B3C4D" w14:textId="5E6F7A8B""#; + + /// A paragraph to edit, then a table with two direct rows, a row inside + /// a table-level content control and a self-closing row, each carrying + /// `attributes`. The first cell holds `word`. + fn row_document_xml(attributes: &str, word: &str) -> String { + let row = |first: &str, second: &str| { + format!( + r#"{first}{second}"# + ) + }; + let rows = [ + row(word, "one"), + row("beta", "two"), + format!( + r#"{}"#, + row("gamma", "three") + ), + format!(""), + ] + .concat(); + format!( + r#"Body text of section 3.1.{rows}After the table."# + ) + } + + fn document_with_row_attributes(attributes: &str, word: &str) -> Document { + super::document_with_content_controls(&row_document_xml(attributes, word)) + } + + #[test] + fn row_identity_attributes_survive_an_edit_elsewhere() { + // The table-row rows of the identity matrix of #159, one attribute at + // a time, then every identity Word writes together. + for attributes in [ + r#" w:rsidR="00A1B2C3""#, + r#" w:rsidTr="00A1B2C4""#, + r#" w14:paraId="1A2B3C4D""#, + r#" w14:textId="5E6F7A8B""#, + r#" w:rsidRPr="00A1B2C5""#, + r#" w:rsidDel="00A1B2C6""#, + ALL, + ] { + let mut document = document_with_row_attributes(attributes, "alpha"); + assert_eq!( + document + .try_replace_text("Body text of section 3.1.", "Body text of section three.") + .unwrap(), + 1 + ); + let saved = super::document_xml(&mut document); + assert!(saved.contains("section three"), "{saved}"); + for attribute in attributes.split_whitespace() { + assert_eq!(saved.matches(attribute).count(), 4, "{attribute}: {saved}"); + } + assert!(!saved.contains("rdocxRootAttributes"), "{saved}"); + for tag in saved.split("').unwrap()]; + assert!( + tag.trim_end_matches('/') + .starts_with(attributes.trim_start()), + "row attribute order changed: {tag}" + ); + } + + let bytes = document.to_bytes().unwrap(); + let mut reopened = Document::from_bytes(&bytes).unwrap(); + assert_eq!(reopened.to_bytes().unwrap(), bytes); + } + } + + #[test] + fn row_identity_attributes_are_not_row_content() { + // A row without a cell has no grid position for the exporters. + let document = super::document_with_content_controls( + &row_document_xml(ALL, "alpha").replace(&format!(""), ""), + ); + let table = document.table(0).unwrap(); + for index in 0..table.row_count() { + assert!(!table.row(index).unwrap().has_unsupported_content()); + } + let messages = [ + document + .to_mhtml_bytes() + .unwrap() + .diagnostics + .into_iter() + .map(|diagnostic| diagnostic.message) + .collect::>(), + document + .to_odt_bytes() + .unwrap() + .diagnostics + .into_iter() + .map(|diagnostic| diagnostic.message) + .collect(), + document + .to_rtf_bytes() + .unwrap() + .diagnostics + .into_iter() + .map(|diagnostic| diagnostic.message) + .collect(), + document + .to_epub_bytes() + .unwrap() + .diagnostics + .into_iter() + .map(|diagnostic| diagnostic.message) + .collect(), + ] + .concat(); + assert!( + messages + .iter() + .all(|message| !message.contains("table-row XML") + && message != "dropped unsupported Word table row content"), + "{messages:?}" + ); + } + + #[test] + fn row_identity_differences_add_no_comparison_revision() { + let compare = |original: &str, edited: &str, word: &str| { + let mut compared = document_with_row_attributes(original, "alpha"); + compared + .compare( + &document_with_row_attributes(edited, word), + "R", + "2026-09-27T12:00:00Z", + ) + .unwrap_or_else(|error| panic!("{original} against {edited}: {error}")); + compared.revisions().len() + }; + let resaved = ALL + .replace("00A1B2C3", "00D4E5F6") + .replace("1A2B3C4D", "2B3C4D5E"); + // An identity-only difference, as between a file and its copy saved + // again by Word, records nothing. + assert_eq!(compare("", ALL, "alpha"), 0); + assert_eq!(compare(ALL, "", "alpha"), 0); + assert_eq!(compare(ALL, &resaved, "alpha"), 0); + // One changed word in a row records its deletion and insertion, with + // or without new identity values on the rows. + assert_eq!(compare(ALL, ALL, "delta"), 2); + assert_eq!(compare(ALL, &resaved, "delta"), 2); + } + + #[test] + fn w14_identities_stay_bound_when_only_their_element_declares_w14() { + // The retained identities leave the `w14` declaration to the part + // root, where Word and python-docx write it. A producer may declare + // it on the row or paragraph instead, and an edit elsewhere must + // still save a readable part. + let local = format!(r#" xmlns:w14="{W14_NS}" w:rsidR="00A1B2C3" w14:paraId="1A2B3C4D""#); + let xml = row_document_xml(&local, "alpha").replacen( + &format!(r#" xmlns:w14="{W14_NS}">"#), + &format!(">"), + 1, + ); + let mut document = super::document_with_content_controls(&xml); + assert_eq!( + document + .try_replace_text("After the table.", "After the rows.") + .unwrap(), + 1 + ); + let bytes = document.to_bytes().unwrap(); + let saved = super::document_xml(&mut document); + assert!(saved.contains("After the rows."), "{saved}"); + assert_eq!( + saved.matches(r#"w14:paraId="1A2B3C4D""#).count(), + 5, + "{saved}" + ); + let mut reopened = Document::from_bytes(&bytes) + .unwrap_or_else(|error| panic!("the saved part is unreadable: {error}: {saved}")); + assert_eq!(reopened.to_bytes().unwrap(), bytes); + } + + #[test] + fn comparing_with_a_copy_that_declares_w14_keeps_its_identities_bound() { + // The original part declares no `w14`, as a part rdocx wrote. Its + // copy saved again by Word declares it on the root and adds a row + // whose row and paragraphs carry w14 identities. + let original = + row_document_xml("", "alpha").replacen(&format!(r#" xmlns:w14="{W14_NS}">"#), ">", 1); + let cell = |text: &str, id: u8| { + format!( + r#"{text}"# + ) + }; + let inserted = format!( + r#"{}{}"#, + cell("inserted", 2), + cell("row", 3) + ); + let edited = row_document_xml("", "alpha").replacen( + "", + &format!("{inserted}"), + 1, + ); + assert!(edited.contains("inserted")); + let mut compared = super::document_with_content_controls(&original); + compared + .compare( + &super::document_with_content_controls(&edited), + "R", + "2026-09-27T12:00:00Z", + ) + .unwrap_or_else(|error| panic!("{error}")); + assert!(!compared.revisions().is_empty()); + let bytes = compared.to_bytes().unwrap(); + let saved = super::document_xml(&mut compared); + for attribute in [ + r#"w14:paraId="1A2B3C01""#, + r#"w14:paraId="3C4D5E02""#, + r#"w14:textId="7A8B9C03""#, + ] { + assert_eq!(saved.matches(attribute).count(), 1, "{attribute}: {saved}"); + } + Document::from_bytes(&bytes) + .unwrap_or_else(|error| panic!("the compared part is unreadable: {error}: {saved}")); + + // A header story is compared the same way. + let table = |rows: &str| { + format!( + r#"keptrow{rows}"# + ) + }; + let with_header = |header: &str| { + let mut document = super::document_with_comparison_stories("alpha"); + let mut package = oxml_opc::OpcPackage::from_reader(std::io::Cursor::new( + document.to_bytes().unwrap(), + )) + .unwrap(); + package.set_part("/word/header1.xml", header.as_bytes().to_vec()); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + }; + let header = |root: &str, rows: &str| { + format!( + r#"alpha header{}"#, + table(rows) + ) + }; + let mut compared = with_header(&header("", "")); + compared + .compare( + &with_header(&header(&format!(r#" xmlns:w14="{W14_NS}""#), &inserted)), + "R", + "2026-09-27T12:00:00Z", + ) + .unwrap_or_else(|error| panic!("header: {error}")); + let bytes = compared.to_bytes().unwrap(); + let header_xml = super::comparison_part_xml(&mut compared, "/word/header1.xml"); + assert!( + header_xml.contains(r#"w14:paraId="1A2B3C01""#), + "{header_xml}" + ); + assert!( + header_xml.contains(r#"w14:paraId="3C4D5E02""#), + "{header_xml}" + ); + Document::from_bytes(&bytes).unwrap_or_else(|error| { + panic!("the compared header is unreadable: {error}: {header_xml}") + }); + } + + #[test] + fn mail_merge_regions_and_fragments_are_found_in_word_paragraphs_and_rows() { + // Word writes distinct identities on every row and paragraph, and a + // revision-save identity on the runs it edits. None of them is + // content, so the region markers and a whole-paragraph fragment field + // are still found. + let paragraph = |instruction: &str, id: u8| { + format!( + r#" MERGEFIELD {instruction} stored"# + ) + }; + let row = |instruction: &str, id: u8| { + format!( + r#"{}"#, + paragraph(instruction, id + 10) + ) + }; + let body = format!( + r#"{}{}{}{}{}{}{}"#, + paragraph("TableStart:Lines", 1), + paragraph("Name", 2), + paragraph("Insert", 3), + paragraph("TableEnd:Lines", 4), + row("TableStart:Rows", 5), + row("RowValue", 6), + row("TableEnd:Rows", 7) + ); + let document = + super::document_with_content_controls(&super::wrap_word_body(&body).replacen( + ">(); + assert_eq!(paragraphs, ["alpha", "fragment", "beta", "fragment"]); + let table = output.table(0).unwrap(); + assert_eq!(table.row_count(), 2); + assert_eq!(table.cell(0, 0).unwrap().text(), "first"); + assert_eq!(table.cell(1, 0).unwrap().text(), "second"); + } +} + fn document_with_comment_paragraphs(paragraphs: &str) -> (Document, i32) { let mut document = Document::new(); document.add_paragraph("seed"); diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index a15f6e2d4..1ef9fa2dc 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -428,14 +428,20 @@ namespace URI escaping are resolved by the XML parser. Serialization fails closed when owner identity or a serializer prefix binding cannot be preserved safely, leaving the opened package bytes authoritative. -Modeled paragraph, run, and section-property owners retain every ordered root -attribute, including producer identity, revision-session, foreign, and -unqualified attributes. Retention uses the existing raw-preservation carriers +Modeled paragraph, run, table-row, and section-property owners retain every +ordered root attribute, including producer identity, revision-session, foreign, +and unqualified attributes. Retention uses the existing raw-preservation carriers without exposing the attribute record as child XML. Expanded names govern duplicate rejection and authored paragraph identity precedence, so an authored `paraId` replaces only the retained attribute with the same namespace and local name. Typed child mutation leaves all other retained root attributes in -source order. +source order. A table row keeps its record in its raw-child list at a position +no cell boundary reaches. Every reader that treats raw row children as content +skips it: comparison row signatures and boundaries, the retained table layout +cache, the row diagnostics of the MHTML, ODT, RTF and EPUB writers, and the +rich merge row-region markers. The rich merge region-marker and whole-paragraph +fragment checks skip the record of a paragraph the same way, so a paragraph +Word wrote still holds a region marker or a fragment field. Retention covers attributes, not namespace bindings. A declaration is recorded only when a retained attribute uses its prefix, because the alias machinery @@ -443,9 +449,14 @@ already materializes a binding onto every element that needs one, and recording a declaration a child carries for itself would emit it twice. A root carrying nothing but declarations retains no record at all. On the way back out, the canonical `w14` binding is not copied onto the written element, since the part -root that owns the element already declares it and the authored identity write -makes the same assumption. Together these keep a reopened save byte identical -to the save it was read from. +root that owns the element declares it in what Word and python-docx write, and +the authored identity write makes the same assumption. Together these keep a +reopened save byte identical to the save it was read from. A part root that +does not declare `w14`, such as one rdocx wrote or one under an element that +declared the prefix itself, gains the canonical declaration when the written +content uses the prefix. The serializers of the document, header, footer, +note and comment parts and the comparison output of every story add it, so the +written part stays namespace well formed. A paragraph cut out of its part and parsed on its own carries none of the declarations of its part. The table-of-contents rebuild adds the bindings the diff --git a/docs/hld/12-testing-strategy.md b/docs/hld/12-testing-strategy.md index cd55adaec..e4ef412eb 100644 --- a/docs/hld/12-testing-strategy.md +++ b/docs/hld/12-testing-strategy.md @@ -1841,7 +1841,16 @@ values, source attribute order, child schema order, and deterministic package bytes. Focused unit coverage rejects duplicate expanded names and proves that authored `paraId` replaces only its expanded-name match. The public run-shape regression prevents the internal retention record from changing the existing -`CT_R` struct literal surface. +`CT_R` struct literal surface. `table_row_identity_attribute_regressions` +extends the gate to table rows, direct, inside a table-level content control, +and self-closing. A one-word edit outside the table keeps every row identity +in source order and a reopened save is byte identical. Identity-only row +differences add no comparison revision, a changed word adds two, the rows +raise no unsupported-content or export diagnostic, and rich merge region +markers and whole-paragraph fragment fields still resolve in paragraphs and +rows that carry Word identities. A row or paragraph that declares `w14` itself +under a root that does not, and a comparison of main and header stories +against a copy whose roots declare `w14`, both write a readable part. The run-level page-break differential gate authors both the break-only paragraph written by python-docx and a break between two pieces of text. Its From ddb8f61e718ec7ae50f72dead8fdbf4c20d1655c Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:25:06 +0200 Subject: [PATCH 5/7] Keep the start tag of a text-box paragraph that replacement edits Replacement in text boxes walks the raw XML of a part, parses each w:p of a w:txbxContent without its start tag and writes the edited paragraphs back with CT_P::to_xml. A paragraph therefore lost the attributes of its start tag, the w:rsidR, w14:paraId and w14:textId Word writes on every text-box paragraph and any local declaration, as soon as a replacement reached its text box. The walker keeps the start tag each paragraph was read from and writes the edited paragraph under a start tag with the same attributes. The paragraph goes back where it was read, so every prefix those attributes use resolves as before. Parsing the start tag into the retained-attribute record instead would need the bindings in scope, which this walker does not track, and w14:paraId would then fail the whole part. GitHub issue #159. --- crates/rdocx-oxml/src/placeholder.rs | 64 +++++++++++++++++++++++++-- crates/rdocx/tests/regression_test.rs | 14 ++++++ docs/hld/04-opc-and-packaging.md | 5 ++- 3 files changed, 79 insertions(+), 4 deletions(-) diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 0bf078dd4..6d848431e 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -382,8 +382,10 @@ fn rewrite_text_boxes( let mut depth = 1u32; let mut inner_buf = Vec::new(); - // Collect paragraphs from inside txbxContent + // Collect paragraphs from inside txbxContent, with the start + // tag each one was read from. let mut paragraphs: Vec = Vec::new(); + let mut sources = Vec::new(); loop { match reader.read_event_into(&mut inner_buf) { @@ -436,6 +438,7 @@ fn rewrite_text_boxes( } let para = CT_P::from_xml(&mut para_reader)?; paragraphs.push(para); + sources.push(ie.to_owned()); } else { depth += 1; // Non-paragraph element inside txbxContent; skip it @@ -462,8 +465,8 @@ fn rewrite_text_boxes( total_count += edit(&mut paragraphs); // Re-serialize paragraphs into the writer - for p in ¶graphs { - p.to_xml(&mut writer)?; + for (p, source) in paragraphs.iter().zip(&sources) { + write_text_box_paragraph(&mut writer, p, source)?; } // Write closing txbxContent tag @@ -482,6 +485,38 @@ fn rewrite_text_boxes( Ok((writer.into_inner(), total_count)) } +/// Write a text-box paragraph that was parsed without its start tag. +/// +/// The start tag keeps the attributes it was read with, such as `w:rsidR`, +/// `w14:paraId` or a local namespace declaration. The paragraph goes back to +/// the place it was read from, so every prefix they use resolves as before. +fn write_text_box_paragraph( + writer: &mut quick_xml::Writer, + paragraph: &CT_P, + source: &quick_xml::events::BytesStart<'_>, +) -> crate::error::Result<()> { + use quick_xml::events::{BytesStart, Event}; + + let mut xml = Vec::new(); + paragraph.to_xml(&mut quick_xml::Writer::new(&mut xml))?; + let mut start = BytesStart::new("w:p"); + for attribute in source.attributes() { + start.push_attribute(attribute?); + } + // Parsed without its start tag, the paragraph records no attribute of its + // own, so it serializes with a bare start tag. Anything else is written as + // it serialized. + if xml == b"" { + writer.write_event(Event::Empty(start))?; + } else if let Some(content) = xml.strip_prefix(b"") { + writer.write_event(Event::Start(start))?; + writer.get_mut().write_all(content)?; + } else { + writer.get_mut().write_all(&xml)?; + } + Ok(()) +} + /// Replace placeholders in chart XML parts. /// /// Chart text uses DrawingML runs: `a:r` → `a:t` (not `w:r`/`w:t`). @@ -896,6 +931,29 @@ mod tests { assert!(!result_str.contains("{{name}}")); } + /// Word writes revision-save and paragraph identities on text-box + /// paragraphs too. An edited paragraph used to be written back with a + /// bare start tag, so a replacement dropped them. + #[test] + fn replace_in_textbox_keeps_the_paragraph_start_tag_attributes() { + const START: &str = r#""#; + let xml = format!( + r#"{START}Hello {{{{name}}}}"# + ); + let expected = format!("{START}Hello Alice"); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 1); + let result = String::from_utf8(result).unwrap(); + assert!(result.contains(&expected), "{result}"); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + let (result, count) = replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap(); + assert_eq!(count, 1); + let result = String::from_utf8(result).unwrap(); + assert!(result.contains(&expected), "{result}"); + } + #[test] fn replace_in_vml_textbox() { let xml = br#" diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 264a67f2a..5e0332961 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -32520,6 +32520,20 @@ mod text_box_identity_attribute_regressions { ); } + #[test] + fn replacing_text_keeps_the_identities_of_the_text_box_paragraph() { + // The replacement parses a text-box paragraph without its start tag, + // so the paragraph it edited used to lose its own identities. + let paragraph = r#""#; + let mut document = super::document_with_content_controls(&format!( + r#"Host paragraph{paragraph}Boxed text"# + )); + assert_eq!(document.try_replace_text("Boxed", "Filled").unwrap(), 1); + let saved = super::document_xml(&mut document); + assert!(saved.contains("Filled text"), "{saved}"); + assert!(saved.contains(paragraph), "{saved}"); + } + #[test] fn a_template_fills_a_text_box_whose_runs_carry_identity_attributes() { let data = serde_json::json!({"name": "Ada"}); diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index 1ef9fa2dc..18a2bb249 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -468,7 +468,10 @@ template walkers still parse such a paragraph in the default scope, which names a Word prefix that the scope names without a binding to the WordprocessingML namespace, and an explicit binding always wins. A run attribute under any other prefix, such as `w14`, a foreign namespace or a second WordprocessingML alias, -still fails those two walkers. +still fails those two walkers. The replacement walker parses a text-box +paragraph without its start tag. It writes an edited paragraph back under a +start tag that carries the attributes it was read with, identities and local +declarations included, since the paragraph returns to the scope it came from. An unknown default namespace declared on the document root is classified by its effective lexical scope before canonical serialization. An unused root From 5b64f1cf2dd06e820b4e3861253687399b1eda05 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:30:29 +0200 Subject: [PATCH 6/7] Drop w14 paragraph identities from copied paragraphs and rows A paragraph copy kept the w14:paraId and w14:textId of its source since the fix for #130, and table rows now keep them too. So clone_content, clone_table_row, the region rows of a rich mail merge, the records of a merge into sections and the loops of render_template wrote one w14:paraId several times in one document, and an imported fragment could bring one that the destination already used. Duplicate paragraph identities are invalid, and Word assigns new ones to an element that has none. The identity pass that freshens the bookmark, content-control and drawing identities of copied body content now also removes w14:paraId and w14:textId from every copied w:p and w:tr, text-box paragraphs included. The revision-save identities stay, since Word repeats them freely. Renaming passes, such as the bookmark rename of the TOC rebuild, leave them untouched. The template evaluator copies typed paragraphs and rows without that pass. It now removes the two identities from the retained record of every paragraph and table row a loop renders, through a doc-hidden rdocx-oxml helper that keeps every other retained attribute. GitHub issue #159. --- crates/rdocx-oxml/src/text.rs | 122 ++++++++++++++++++++++++ crates/rdocx/src/document.rs | 8 +- crates/rdocx/src/field.rs | 52 +++++++++- crates/rdocx/src/template.rs | 128 ++++++++++++++++++++++--- crates/rdocx/tests/regression_test.rs | 132 +++++++++++++++++++++++++- docs/hld/03-architecture.md | 24 +++-- docs/hld/04-opc-and-packaging.md | 17 ++-- docs/hld/12-testing-strategy.md | 4 +- 8 files changed, 453 insertions(+), 34 deletions(-) diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs index 0c87e502d..4c0ffa531 100644 --- a/crates/rdocx-oxml/src/text.rs +++ b/crates/rdocx-oxml/src/text.rs @@ -188,6 +188,99 @@ pub(crate) fn push_root_attribute_record( Ok(()) } +/// Remove `w14:paraId` and `w14:textId` from the retained start-tag +/// attributes of a paragraph or table row, given its `extra_xml`. +/// +/// A copy must not share them with its source, and Word assigns new ones to +/// an element that has none. Every other retained attribute stays, with the +/// declarations its prefix needs. +#[doc(hidden)] +pub fn drop_w14_paragraph_identities(extra_xml: &mut Vec<(usize, Vec)>) -> Result<()> { + let mut index = 0; + while index < extra_xml.len() { + let (position, raw) = &extra_xml[index]; + if *position == ROOT_ATTRIBUTES_POSITION && is_root_attribute_record(raw) { + match root_attribute_record_without_w14_identities(raw)? { + Some(record) => extra_xml[index].1 = record, + None => { + extra_xml.remove(index); + continue; + } + } + } + index += 1; + } + Ok(()) +} + +/// Return the record without its w14 identities, or `None` when nothing else +/// is left in it. +fn root_attribute_record_without_w14_identities(raw: &[u8]) -> Result>> { + let mut reader = NsReader::from_reader(raw); + let mut buffer = Vec::new(); + let element = loop { + match reader.read_resolved_event_into(&mut buffer)? { + (_, Event::Empty(element)) if element.name().as_ref() == ROOT_ATTRIBUTES_ELEMENT => { + break element.into_owned(); + } + (_, Event::Eof) => { + return Err(OxmlError::InvalidValue( + "invalid retained root-attribute record".to_owned(), + )); + } + _ => {} + } + buffer.clear(); + }; + let mut dropped = false; + let mut attributes = Vec::new(); + let mut declarations = Vec::new(); + for attribute in element.attributes() { + let attribute = attribute?; + let name = std::str::from_utf8(attribute.key.as_ref())?.to_owned(); + let value = attribute + .decoded_and_normalized_value(XmlVersion::Implicit1_0, element.decoder())? + .into_owned(); + if namespace_declaration(name.as_bytes()) { + declarations.push((name, value)); + continue; + } + let (namespace, local) = reader.resolver().resolve_attribute(attribute.key); + if matches!(namespace, ResolveResult::Bound(Namespace(uri)) if uri == W14_NS.as_bytes()) + && matches!(local.as_ref(), b"paraId" | b"textId") + { + dropped = true; + continue; + } + attributes.push((name, value)); + } + if !dropped { + return Ok(Some(raw.to_vec())); + } + if attributes.is_empty() { + return Ok(None); + } + let mut record = BytesStart::new(std::str::from_utf8(ROOT_ATTRIBUTES_ELEMENT)?); + for (name, value) in &attributes { + record.push_attribute((name.as_str(), value.as_str())); + } + // The record declares exactly the prefixes its attributes use, so a + // declaration only the dropped identities used goes with them. + for (name, value) in &declarations { + let prefix = name.strip_prefix("xmlns:").unwrap_or_default(); + if attributes.iter().any(|(attribute, _)| { + attribute + .split_once(':') + .is_some_and(|(used, _)| used == prefix) + }) { + record.push_attribute((name.as_str(), value.as_str())); + } + } + let mut writer = Writer::new(Vec::new()); + writer.write_event(Event::Empty(record))?; + Ok(Some(writer.into_inner())) +} + /// Declare the canonical `w14` prefix on the root of a serialized part whose /// content uses the prefix while the root does not bind it. /// @@ -8762,6 +8855,35 @@ mod tests { ); } + #[test] + fn dropping_w14_identities_keeps_every_other_retained_attribute() { + // An alias prefix goes with the identities it bound, and a record + // left with nothing else goes away. + let w_ns = crate::namespace::W_NS; + let source = format!( + r#""# + ); + let mut paragraph = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + drop_w14_paragraph_identities(&mut paragraph.extra_xml).unwrap(); + let mut output = Vec::new(); + paragraph.to_xml(&mut Writer::new(&mut output)).unwrap(); + let output = String::from_utf8(output).unwrap(); + assert!( + output.starts_with(r#""# + ); + let mut paragraph = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + drop_w14_paragraph_identities(&mut paragraph.extra_xml).unwrap(); + assert!(paragraph.extra_xml.is_empty(), "{:?}", paragraph.extra_xml); + } + #[test] fn a_part_root_declares_w14_only_when_its_content_needs_it() { let w_ns = crate::namespace::W_NS; diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 5c4b2b468..46947898b 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -14357,6 +14357,9 @@ impl Document { } /// Clone one checked direct child into a checked insertion boundary. + /// + /// Document-wide identities of the copy are made unique, and its + /// paragraphs and table rows drop `w14:paraId` and `w14:textId`. pub fn clone_content( &mut self, source_location: &ContentLocation, @@ -14736,8 +14739,9 @@ impl Document { /// `table_index` counts every table in document order, including nested /// tables. `insert_at` may equal the row count to append. The copy keeps /// row properties, cell formatting, nested content, relationships, and - /// preserved producer XML. Document-wide identities are made unique and - /// comment anchors are omitted. The document is unchanged on error. + /// preserved producer XML. Document-wide identities are made unique, the + /// row and its paragraphs drop `w14:paraId` and `w14:textId`, and comment + /// anchors are omitted. The document is unchanged on error. pub fn clone_table_row( &mut self, table_index: usize, diff --git a/crates/rdocx/src/field.rs b/crates/rdocx/src/field.rs index a2b8da9fd..a172bf4cd 100644 --- a/crates/rdocx/src/field.rs +++ b/crates/rdocx/src/field.rs @@ -7123,6 +7123,10 @@ struct BodyIdentityRemap { drawing_ids: BTreeMap, non_visual_drawing_ids: BTreeMap, bookmark_names: BTreeMap, + /// Remove `w14:paraId` and `w14:textId` from paragraphs and table rows. + /// A copy must not share them with its source, and Word assigns new ones + /// to an element that has none. + drop_paragraph_identities: bool, } fn remap_body_identities( @@ -7132,7 +7136,10 @@ fn remap_body_identities( ) -> Result { let xml = document.document.to_xml()?; let values = body_identity_values(&xml)?; - let mut remap = BodyIdentityRemap::default(); + let mut remap = BodyIdentityRemap { + drop_paragraph_identities: true, + ..Default::default() + }; for value in values.bookmark_ids { if let std::collections::btree_map::Entry::Vacant(entry) = remap.bookmark_ids.entry(value) { entry.insert(identifiers.reserve_bookmark_id()?.to_string()); @@ -7453,6 +7460,7 @@ fn resolved_element_attribute( } const WP_NS: &str = "http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing"; +const W14_NS: &str = "http://schemas.microsoft.com/office/word/2010/wordml"; const PIC_NS: &str = "http://schemas.openxmlformats.org/drawingml/2006/picture"; const DRAWING_NS: &str = "http://schemas.openxmlformats.org/drawingml/2006/main"; const WPS_NS: &str = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape"; @@ -7646,7 +7654,10 @@ pub(crate) fn freshen_content_fragment_identities( )); } let mut state = BodyIdentityState::from_documents(std::slice::from_ref(document))?; - let mut remap = BodyIdentityRemap::default(); + let mut remap = BodyIdentityRemap { + drop_paragraph_identities: true, + ..Default::default() + }; for value in values.bookmark_ids { if let std::collections::btree_map::Entry::Vacant(entry) = remap.bookmark_ids.entry(value) { entry.insert(document.identifiers.reserve_bookmark_id()?.to_string()); @@ -7807,6 +7818,34 @@ fn collect_body_identity_edits( &remap.non_visual_drawing_ids, edits, )?; + } else if remap.drop_paragraph_identities && namespace.word && matches!(local, b"p" | b"tr") { + for attribute in element.attributes() { + let attribute = attribute + .map_err(|error| Error::Other(format!("invalid XML attribute: {error}")))?; + let (attribute_namespace, attribute_local) = resolver.resolve_attribute(attribute.key); + if !namespace_matches(&attribute_namespace, W14_NS) + || !matches!(attribute_local.as_ref(), b"paraId" | b"textId") + { + continue; + } + let Some((name_start, _, value_end)) = + attribute_source_span(&xml[start..end], attribute.key.as_ref()) + else { + return Err(Error::Other( + "document identity attribute source was not found".to_owned(), + )); + }; + // Remove the whitespace before the name with the attribute. + let removed_start = xml[start..start + name_start] + .iter() + .rposition(|byte| !byte.is_ascii_whitespace()) + .map_or(0, |last| last + 1); + edits.push(FieldSourceEdit { + start: start + removed_start, + end: start + value_end + 1, + replacement: Vec::new(), + }); + } } Ok(()) } @@ -7990,6 +8029,13 @@ fn remap_reference_instruction( } fn attribute_value_span(element: &[u8], attribute_name: &[u8]) -> Option<(usize, usize)> { + attribute_source_span(element, attribute_name) + .map(|(_, value_start, value_end)| (value_start, value_end)) +} + +/// Return where the name of one attribute starts in a start tag, and where +/// its value starts and ends. +fn attribute_source_span(element: &[u8], attribute_name: &[u8]) -> Option<(usize, usize, usize)> { let mut index = 1usize; while index < element.len() && !element[index].is_ascii_whitespace() { index += 1; @@ -8030,7 +8076,7 @@ fn attribute_value_span(element: &[u8], attribute_name: &[u8]) -> Option<(usize, } let value_end = index; if &element[name_start..name_end] == attribute_name { - return Some((value_start, value_end)); + return Some((name_start, value_start, value_end)); } index += 1; } diff --git a/crates/rdocx/src/template.rs b/crates/rdocx/src/template.rs index 3721169cc..36621906a 100644 --- a/crates/rdocx/src/template.rs +++ b/crates/rdocx/src/template.rs @@ -10,7 +10,7 @@ use rdocx_oxml::header_footer::CT_HdrFtr; use rdocx_oxml::namespace::matches_local_name; use rdocx_oxml::placeholder; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_Tc, CellContent}; -use rdocx_oxml::text::{CT_P, RunContent}; +use rdocx_oxml::text::{CT_P, RunContent, drop_w14_paragraph_identities}; use serde_json::Value; use crate::document::Document; @@ -360,11 +360,22 @@ fn render_body( let mut render_item = |content: &mut BodyContent, root: &Value, scopes: &[Scope], - sentinels: &mut SentinelPool| { - render_body_item(content, root, scopes, sentinels, &mut structural_change) + sentinels: &mut SentinelPool, + repeated: bool| { + let count = render_body_item(content, root, scopes, sentinels, &mut structural_change)?; + if repeated { + drop_copied_body_identities(content)?; + } + Ok(count) }; - let (evaluated, count) = - evaluate_blocks(&blocks, data, &mut scopes, sentinels, &mut render_item)?; + let (evaluated, count) = evaluate_blocks( + &blocks, + data, + &mut scopes, + sentinels, + false, + &mut render_item, + )?; document.document.body.content = evaluated .into_iter() .map(|evaluated| evaluated.value) @@ -433,7 +444,8 @@ fn render_table( let mut render_item = |item: &mut TableRowItem, root: &Value, scopes: &[Scope], - sentinels: &mut SentinelPool| { + sentinels: &mut SentinelPool, + repeated: bool| { let mut count = render_nested_tables_in_row(&mut item.row, root, scopes, sentinels, structural_change)?; count += stage_row_scalars(&mut item.row, root, scopes, sentinels)?; @@ -447,6 +459,12 @@ fn render_table( )?; count += stage_control_scalars(control, root, scopes, sentinels)?; } + if repeated { + drop_copied_row_identities(&mut item.row)?; + for (_, control) in &mut item.controls { + drop_copied_control_identities(control)?; + } + } Ok(count) }; let (evaluated, mut count) = evaluate_blocks( @@ -454,6 +472,7 @@ fn render_table( root, &mut local_scopes, sentinels, + false, &mut render_item, )?; @@ -574,6 +593,72 @@ fn render_nested_tables_in_control( Ok(count) } +/// Drop the w14 identities of every paragraph and table row in one loop copy. +/// Copies must not share them, and Word assigns new ones to an element that +/// has none. Text-box paragraphs stay in the raw XML of their drawing. +fn drop_copied_body_identities(content: &mut BodyContent) -> Result<()> { + match content { + BodyContent::Paragraph(paragraph) => drop_copied_paragraph_identities(paragraph), + BodyContent::Table(table) => drop_copied_table_identities(table), + BodyContent::ContentControl(control) => drop_copied_control_identities(control), + BodyContent::RawXml(_) => Ok(()), + } +} + +fn drop_copied_paragraph_identities(paragraph: &mut CT_P) -> Result<()> { + drop_w14_paragraph_identities(&mut paragraph.extra_xml)?; + for (_, _, _, control) in &mut paragraph.content_controls { + drop_copied_control_identities(control)?; + } + Ok(()) +} + +fn drop_copied_table_identities(table: &mut CT_Tbl) -> Result<()> { + for row in &mut table.rows { + drop_copied_row_identities(row)?; + } + for (_, _, control) in &mut table.content_controls { + drop_copied_control_identities(control)?; + } + Ok(()) +} + +fn drop_copied_row_identities(row: &mut CT_Row) -> Result<()> { + drop_w14_paragraph_identities(&mut row.extra_xml)?; + for cell in &mut row.cells { + drop_copied_cell_identities(cell)?; + } + for (_, _, control) in &mut row.content_controls { + drop_copied_control_identities(control)?; + } + Ok(()) +} + +fn drop_copied_cell_identities(cell: &mut CT_Tc) -> Result<()> { + for content in &mut cell.content { + match content { + CellContent::Paragraph(paragraph) => drop_copied_paragraph_identities(paragraph)?, + CellContent::Table(table) => drop_copied_table_identities(table)?, + CellContent::ContentControl(control) => drop_copied_control_identities(control)?, + } + } + Ok(()) +} + +fn drop_copied_control_identities(control: &mut CT_Sdt) -> Result<()> { + for content in &mut control.content { + match content { + SdtContent::Paragraph(paragraph) => drop_copied_paragraph_identities(paragraph)?, + SdtContent::Table(table) => drop_copied_table_identities(table)?, + SdtContent::Row(row) => drop_copied_row_identities(row)?, + SdtContent::Cell(cell) => drop_copied_cell_identities(cell)?, + SdtContent::ContentControl(nested) => drop_copied_control_identities(nested)?, + SdtContent::Run(_) | SdtContent::RawXml(_) => {} + } + } + Ok(()) +} + fn body_marker(content: &BodyContent) -> Result> { match content { BodyContent::Paragraph(paragraph) => marker_from_sources(&[paragraph_text(paragraph)]), @@ -748,15 +833,18 @@ fn blocks_have_controls(blocks: &[Block]) -> bool { .any(|block| !matches!(block, Block::Item { .. })) } +/// Evaluate the blocks of one level. `repeated` is set inside a loop, whose +/// body is written once per item, so every item it renders is a copy. fn evaluate_blocks( blocks: &[Block], root: &Value, scopes: &mut Vec, sentinels: &mut SentinelPool, + repeated: bool, render_item: &mut F, ) -> Result<(Vec>, usize)> where - F: FnMut(&mut T, &Value, &[Scope], &mut SentinelPool) -> Result, + F: FnMut(&mut T, &Value, &[Scope], &mut SentinelPool, bool) -> Result, { let mut output = Vec::new(); let mut count = 0; @@ -767,7 +855,7 @@ where value, } => { let mut value = value.clone(); - count += render_item(&mut value, root, scopes, sentinels)?; + count += render_item(&mut value, root, scopes, sentinels, repeated)?; output.push(Evaluated { source_index: *source_index, value, @@ -791,7 +879,14 @@ where deferred: true, }); let mut validation_sentinels = SentinelPool::new(&[]); - evaluate_blocks(body, root, scopes, &mut validation_sentinels, render_item)?; + evaluate_blocks( + body, + root, + scopes, + &mut validation_sentinels, + true, + render_item, + )?; scopes.pop(); } for value in values { @@ -800,7 +895,8 @@ where value, deferred: false, }); - let evaluated = evaluate_blocks(body, root, scopes, sentinels, render_item)?; + let evaluated = + evaluate_blocks(body, root, scopes, sentinels, true, render_item)?; scopes.pop(); output.extend(evaluated.0); count += evaluated.1; @@ -808,13 +904,21 @@ where } Block::If { path, body } => match resolve_value(root, scopes, path)? { ResolvedValue::Value(value) if is_truthy(value) => { - let evaluated = evaluate_blocks(body, root, scopes, sentinels, render_item)?; + let evaluated = + evaluate_blocks(body, root, scopes, sentinels, repeated, render_item)?; output.extend(evaluated.0); count += evaluated.1; } ResolvedValue::Value(_) | ResolvedValue::Deferred => { let mut validation_sentinels = SentinelPool::new(&[]); - evaluate_blocks(body, root, scopes, &mut validation_sentinels, render_item)?; + evaluate_blocks( + body, + root, + scopes, + &mut validation_sentinels, + repeated, + render_item, + )?; } }, } diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 5e0332961..a5f2112e7 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -14881,7 +14881,7 @@ mod table_row_identity_attribute_regressions { }], ..Default::default() }; - let output = document + let mut output = document .mail_merge_rich(&data, None) .unwrap_or_else(|error| panic!("{error}")) .remove(0); @@ -14895,6 +14895,136 @@ mod table_row_identity_attribute_regressions { assert_eq!(table.row_count(), 2); assert_eq!(table.cell(0, 0).unwrap().text(), "first"); assert_eq!(table.cell(1, 0).unwrap().text(), "second"); + // Every output paragraph and row is a copy of a template element, so + // none keeps its w14 identities. The revision-save identities stay. + let saved = super::document_xml(&mut output); + assert!(!saved.contains("MERGEFIELD"), "{saved}"); + assert_eq!( + saved.matches(r#"w:rsidTr="00A1B2C4""#).count(), + 2, + "{saved}" + ); + assert_eq!(saved.matches(r#"w:rsidR="00B1B2B3""#).count(), 4, "{saved}"); + assert!(!saved.contains("w14:paraId"), "{saved}"); + assert!(!saved.contains("w14:textId"), "{saved}"); + } + + #[test] + fn copied_rows_and_paragraphs_drop_their_w14_identities() { + // Two elements must not share a `w14:paraId`, and Word assigns new + // w14 identities to an element that has none, so a copy drops them. + // The revision-save identities stay, since Word repeats them freely. + let cell_paragraph = + r#""#; + let body_paragraph = + r#""#; + // A row without a cell leaves the table without a valid topology. + let xml = row_document_xml(ALL, "alpha") + .replace(&format!(""), "") + .replacen("", &format!("{cell_paragraph}"), 1) + .replacen("", &format!("{body_paragraph}"), 1); + let mut document = super::document_with_content_controls(&xml); + assert_eq!(document.clone_table_row(0, 0, 2).unwrap(), 2); + let body = super::f254_story(&document, rdocx::StoryKind::Body); + let source = super::f254_item(&document, &body, 0); + document + .clone_content(&source, &rdocx::ContentLocation::end(body)) + .unwrap(); + + let saved = super::document_xml(&mut document); + assert_eq!(saved.matches("alpha").count(), 2, "{saved}"); + assert_eq!( + saved.matches("Body text of section 3.1.").count(), + 2, + "{saved}" + ); + for (attribute, count) in [ + (r#"w14:paraId="1A2B3C4D""#, 3), + (r#"w14:textId="5E6F7A8B""#, 3), + (r#"w:rsidTr="00A1B2C4""#, 4), + (r#"w:rsidDel="00A1B2C6""#, 4), + (r#"w14:paraId="3C4D5E6F""#, 1), + (r#"w14:textId="7A8B9C0D""#, 1), + (r#"w:rsidR="00B1B2B3""#, 2), + (r#"w14:paraId="4D5E6F70""#, 1), + (r#"w14:textId="0A1B2C3D""#, 1), + (r#"w:rsidR="00C1C2C3""#, 2), + ] { + assert_eq!( + saved.matches(attribute).count(), + count, + "{attribute}: {saved}" + ); + } + } + + #[test] + fn template_loop_copies_drop_their_w14_identities() { + // Word writes a distinct `w14:paraId` on every paragraph and row. A + // loop writes its body once per item, so no copy keeps them, while + // the revision-save identities and content outside a loop stay. + let paragraph = |text: &str, id: u8| { + format!( + r#"{text}"# + ) + }; + let row = |text: &str, id: u8| { + format!( + r#"{}"#, + paragraph(text, id + 10) + ) + }; + let body = [ + paragraph("{% for item in items %}", 1), + paragraph("{{ item }}", 2), + paragraph("{% endfor %}", 3), + r#""#.to_owned(), + row("{% for item in items %}", 4), + row("{{ item }}", 5), + row("{% endfor %}", 6), + row("kept row", 7), + "".to_owned(), + paragraph("kept paragraph", 8), + "".to_owned(), + ] + .concat(); + let mut document = + super::document_with_content_controls(&super::wrap_word_body(&body).replacen( + "{item}"); + assert_eq!(saved.matches(&text).count(), 2, "{item}: {saved}"); + } + for (attribute, count) in [ + // The repeated body paragraph, row and cell paragraph. + (r#"w14:paraId="3C4D5E02""#, 0), + (r#"w14:paraId="1A2B3C05""#, 0), + (r#"w14:paraId="3C4D5E15""#, 0), + (r#"w14:textId="7A8B9C02""#, 0), + (r#"w14:textId="5E6F7A05""#, 0), + (r#"w14:textId="7A8B9C15""#, 0), + // The row and paragraphs outside every loop. + (r#"w14:paraId="1A2B3C07""#, 1), + (r#"w14:paraId="3C4D5E17""#, 1), + (r#"w14:paraId="3C4D5E08""#, 1), + (r#"w14:textId="5E6F7A07""#, 1), + (r#"w:rsidTr="00A1B2C4""#, 4), + (r#"w:rsidR="00B1B2B3""#, 8), + ] { + assert_eq!( + saved.matches(attribute).count(), + count, + "{attribute}: {saved}" + ); + } } } diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 97ce19c68..c808503d7 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -813,9 +813,12 @@ each non-final body section properties value to a next-page section-ending paragraph, and retains the final body-level section properties value. A namespace-aware serialized-body pass remaps bookmark, content-control, and drawing identities together with bookmark field and hyperlink references, -including values held in preserved raw XML. The operation does not evaluate -structured template tags, and ordinary field traversal keeps its existing -typed story scope. +including values held in preserved raw XML. The same pass removes `w14:paraId` +and `w14:textId` from every copied paragraph and table row, since a copy must +not share them and Word assigns new ones, and keeps revision-save identities. +Rich merge region copies and imported fragments go through it too. The +operation does not evaluate structured template tags, and ordinary field +traversal keeps its existing typed story scope. The same facade owns additive native rich mail merge over `MailMergeData` and owned text, image, and DOCX fragment values. Whole-paragraph and whole-row @@ -839,11 +842,13 @@ container-aware stack parser. Top-level marker paragraphs clone body entries, including section-ending paragraphs and their section properties. Marker rows clone every row in a multi-row template group inside their owning table. The owning table is retained, and each row and cell is deep-cloned with its merge, -banding, content-control, and ordered raw XML state. Numbered paragraphs in a -loop retain their source `numId` and level, which keeps one continuous list -without allocating definitions. Numbering references are validated before -evaluation. Loop variables form lexical scopes, and dotted lookup searches the -innermost scope before the root value. Structural controls are limited to the +banding, content-control, and ordered raw XML state. Every paragraph and table +row a loop renders drops its `w14:paraId` and `w14:textId`, as other copies do, +and keeps its revision-save identities. Numbered paragraphs in a loop retain +their source `numId` and level, which keeps one continuous list without +allocating definitions. Numbering references are validated before evaluation. +Loop variables form lexical scopes, and dotted lookup searches the innermost +scope before the root value. Structural controls are limited to the main body and its tables. Headers, footers, text boxes, and chart labels retain scalar-only replacement through the existing Word placeholder mapper. A successful render commits the staged document and package together and @@ -1257,7 +1262,8 @@ preserved node. Insert, remove, clone, and move resolve canonical children of the matching kind. `ContentLocation::end` is the distinct boundary after final direct content. It works for empty and self-closing owners and remains before body section properties. Moves stay within one unchanged story -owner. Clones allocate fresh document identities, while relationship-bearing +owner. Clones allocate fresh document identities and drop the `w14:paraId` and +`w14:textId` of their paragraphs and table rows, while relationship-bearing fragments require the unchanged owner scope. Every operation serializes and reopens a staged candidate before publishing it. diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index 18a2bb249..b3bc4b6f0 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -1095,12 +1095,14 @@ immediately adjacent vertical merge ranges, then replace the live table. Existing direct table rows clone and remove through a staged `Document` mutation. A clone retains the complete row, cell, nested-content, relationship, and raw XML model, then freshens bookmark, content-control, and drawing -identities and omits copied comment anchors. Body namespace declaration names -are converted to fragment prefixes before freshening, including the empty -prefix for a root default namespace. Table-level raw XML and content controls -move with their logical row boundary. Removing a vertical-merge restart -promotes a matching continuation below, and a table always retains one direct -row. Invalid indexes, topology, XML, or reopen results discard the candidate. +identities, drops the `w14:paraId` and `w14:textId` of the row and of its +paragraphs, and omits copied comment anchors. Revision-save identities stay. +Body namespace declaration names are converted to fragment prefixes before +freshening, including the empty prefix for a root default namespace. +Table-level raw XML and content controls move with their logical row boundary. +Removing a vertical-merge restart promotes a matching continuation below, and a +table always retains one direct row. Invalid indexes, topology, XML, or reopen +results discard the candidate. Row and cell property readers select modeled elements and attributes by their bound WordprocessingML namespace. Foreign same-local children remain raw in @@ -1343,6 +1345,9 @@ pairs nested controls within one body or table-row container before evaluation. The evaluator clones typed body entries and rows into candidate sequences, so section properties, row properties, and ordered raw-child sidecars travel with their owner. A row loop may clone several adjacent template rows per iteration. +Every paragraph and table row a loop renders is a copy, so it drops the +`w14:paraId` and `w14:textId` of its retained root-attribute record. Text-box +paragraphs stay inside the raw XML of their drawing and keep theirs. The original table and its properties, grid, raw boundaries, content controls, and relationships remain in place. Cloned row and cell property sequences keep grid spans, vertical merge state, and unmodelled children byte for byte. diff --git a/docs/hld/12-testing-strategy.md b/docs/hld/12-testing-strategy.md index e4ef412eb..b5b15bfa9 100644 --- a/docs/hld/12-testing-strategy.md +++ b/docs/hld/12-testing-strategy.md @@ -1850,7 +1850,9 @@ raise no unsupported-content or export diagnostic, and rich merge region markers and whole-paragraph fragment fields still resolve in paragraphs and rows that carry Word identities. A row or paragraph that declares `w14` itself under a root that does not, and a comparison of main and header stories -against a copy whose roots declare `w14`, both write a readable part. +against a copy whose roots declare `w14`, both write a readable part. Rows +and paragraphs copied by `clone_table_row`, `clone_content` and template loops +drop their w14 identities and keep their revision-save identities. The run-level page-break differential gate authors both the break-only paragraph written by python-docx and a break between two pieces of text. Its From 7a8841fbcbc7b903716c6f74bc8ef30951d39fa8 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:40:05 +0200 Subject: [PATCH 7/7] Re-record the archive measurements of three rdocx crates The table-row retention record, the text-box start tag, the w14 root declaration and the helper that drops w14 identities from a copy grow the rdocx-oxml package, the layout cache predicate and its test grow the rdocx-layout package, and the consumer filters, the identity drop for copies and template loops and the new regression tests grow the rdocx package. The README archive rows of rdocx-oxml, rdocx-layout and rdocx and their ARCHIVE_MEASUREMENTS entries are re-measured, with today as their measurement date. GitHub issue #159. --- README.md | 2 +- crates/rdocx-layout/README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 8 ++++---- 4 files changed, 7 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index a0520e66a..48357e2df 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,095,349 compressed bytes, 6,509,769 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | +| Crates.io archive: rdocx | 1,101,657 compressed bytes, 6,540,109 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-layout/README.md b/crates/rdocx-layout/README.md index 8e18a7d23..d9bbc330d 100644 --- a/crates/rdocx-layout/README.md +++ b/crates/rdocx-layout/README.md @@ -18,7 +18,7 @@ sections, and retains source provenance. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-layout | 255,752 compressed bytes, 1,385,701 member bytes, 15 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-layout` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx-layout | 256,067 compressed bytes, 1,386,937 member bytes, 15 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-layout` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 867ca2e35..2f82abc82 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 368,549 compressed bytes, 2,383,545 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | +| Crates.io archive: rdocx-oxml | 372,614 compressed bytes, 2,400,815 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index 1c4a709b9..d3468e31b 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -368,7 +368,7 @@ class ReadmeCase: MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { "rdocx": "2026-09-27", - "rdocx-layout": "2026-09-26", + "rdocx-layout": "2026-09-27", "rdocx-oxml": "2026-09-27", "rpptx": "2026-09-26", } @@ -384,12 +384,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_095_349, 6_509_769, 36), + "rdocx": (1_101_657, 6_540_109, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), - "rdocx-layout": (255_752, 1_385_701, 15), + "rdocx-layout": (256_067, 1_386_937, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (368_549, 2_383_545, 32), + "rdocx-oxml": (372_614, 2_400_815, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),