From c72089c2ada5da7fa3cd7fefb427d879a88fc7b8 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:34:51 +0200 Subject: [PATCH 1/2] Keep every child of a text box that replacement rewrites Replacement in text boxes walks the raw XML of the document part and of each header and footer, parses the w:p children of every w:txbxContent and writes those paragraphs back. Every other child was skipped on read and never written. The document layer stores the rewritten part as soon as one text box in it has a hit, so a single hit deleted the tables, block content controls, bookmarks and empty paragraphs of every text box in that part. try_replace_text, replace_all, replace_regex and render_template all go through this walker, for the DrawingML and the VML copy of a text box alike. The walker now edits each paragraph in place and copies every other child through verbatim, so each text box keeps its content in its order. Only a paragraph that the edit changed is re-serialised. The others are copied as read, so they also keep the start-tag attributes that CT_P does not model, such as w:rsidR and w14:paraId. Two defects of the same loop are fixed with it. The closing tag is the one read from the part instead of a fixed w:txbxContent, which did not match a start tag under another prefix. A part that ends inside a text box is now an error instead of an endless read. GitHub issue #160. --- crates/rdocx-oxml/src/placeholder.rs | 172 +++++++++++++++----------- crates/rdocx/tests/regression_test.rs | 75 +++++++++++ 2 files changed, 174 insertions(+), 73 deletions(-) diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 0bf078dd4..15e863548 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -309,9 +309,13 @@ pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replaceme /// Replace placeholders in text boxes and shapes within a raw XML part. /// -/// Walks the XML, finds `w:txbxContent` elements at any depth, parses their -/// child `w:p` elements using `CT_P::from_xml`, performs replacement, and -/// re-serializes back. Returns the modified XML and replacement count. +/// Walks the XML, finds `w:txbxContent` elements at any depth outside another +/// text box, parses their child `w:p` elements using `CT_P::from_xml`, +/// performs replacement, and re-serializes the paragraphs it changed. Every +/// other child of the text box, such as a table, a content control, a +/// bookmark or a paragraph without a match, is copied through verbatim in its +/// place. A text box nested inside another one is kept as it is, not edited. +/// Returns the modified XML and replacement count. pub fn replace_in_xml_part( xml: &[u8], placeholder: &str, @@ -329,11 +333,11 @@ pub fn replace_many_in_xml_part( xml: &[u8], replacements: &[(&str, &str)], ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { + rewrite_text_boxes(xml, &mut |paragraph| { replacements .iter() .map(|(placeholder, replacement)| { - replace_in_paragraphs(paragraphs, placeholder, replacement) + replace_in_paragraph(paragraph, placeholder, replacement) }) .sum() }) @@ -348,18 +352,22 @@ pub fn replace_regex_in_xml_part( re: ®ex::Regex, replacement: &str, ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { - replace_regex_in_paragraphs(paragraphs, re, replacement) + rewrite_text_boxes(xml, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) }) } -/// Walk `xml`, handing the paragraphs of each `w:txbxContent` element to -/// `edit`, and re-serialise. Returns the rewritten XML and the summed count. +/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit` +/// and re-serialising it in place when `edit` counts a change. Every other +/// paragraph and child of the text box is copied through verbatim. Returns the +/// rewritten XML and the summed count. fn rewrite_text_boxes( xml: &[u8], - edit: &mut dyn FnMut(&mut Vec) -> usize, + edit: &mut dyn FnMut(&mut CT_P) -> usize, ) -> crate::error::Result<(Vec, usize)> { + use crate::error::OxmlError; use crate::namespace::matches_local_name; + use crate::raw_xml::capture_element; use quick_xml::events::Event; use quick_xml::{Reader, Writer}; @@ -374,50 +382,18 @@ fn rewrite_text_boxes( match reader.read_event_into(&mut buf) { Ok(Event::Eof) => break, Ok(Event::Start(ref e)) if matches_local_name(e.name().as_ref(), b"txbxContent") => { - // We found a txbxContent element. Collect its contents as raw XML, - // parse paragraphs, do replacement, and re-serialize. + // We found a txbxContent element. Parse and edit each paragraph + // and copy every other child through verbatim, in document order. writer.write_event(Event::Start(e.clone()))?; - // Read all events inside txbxContent - let mut depth = 1u32; let mut inner_buf = Vec::new(); - // Collect paragraphs from inside txbxContent - let mut paragraphs: Vec = Vec::new(); - loop { match reader.read_event_into(&mut inner_buf) { Ok(Event::Start(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"p") && depth == 1 { + if matches_local_name(ie.name().as_ref(), b"p") { // Parse this paragraph: collect its XML, then parse via CT_P - let mut para_writer = Writer::new(Vec::new()); - // Write the opening tag - para_writer.write_event(Event::Start(ie.clone()))?; - let mut pdepth = 1u32; - let mut pbuf = Vec::new(); - loop { - match reader.read_event_into(&mut pbuf) { - Ok(Event::Start(ref pe)) => { - pdepth += 1; - para_writer.write_event(Event::Start(pe.clone()))?; - } - Ok(Event::End(ref pe)) => { - pdepth -= 1; - para_writer.write_event(Event::End(pe.clone()))?; - if pdepth == 0 { - break; - } - } - Ok(ref ev) => { - para_writer.write_event(ev.clone())?; - } - Err(e) => return Err(e.into()), - } - pbuf.clear(); - } - let para_xml = para_writer.into_inner(); - - // Parse the paragraph + let para_xml = capture_element(&mut reader, ie)?; let mut para_reader = Reader::from_reader(para_xml.as_slice()); para_reader.config_mut().trim_text(true); let mut prbuf = Vec::new(); @@ -434,42 +410,39 @@ fn rewrite_text_boxes( } prbuf.clear(); } - let para = CT_P::from_xml(&mut para_reader)?; - paragraphs.push(para); + let mut para = CT_P::from_xml(&mut para_reader)?; + let count = edit(&mut para); + if count == 0 { + // Nothing changed, so the paragraph keeps its + // bytes, start-tag attributes included. + writer.get_mut().extend_from_slice(¶_xml); + } else { + para.to_xml(&mut writer)?; + } + total_count += count; } else { - depth += 1; - // Non-paragraph element inside txbxContent; skip it - reader.read_to_end_into(ie.name(), &mut Vec::new())?; - depth -= 1; + // A table, a content control or any other element + // the edit does not reach stays as it was. + let raw = capture_element(&mut reader, ie)?; + writer.get_mut().extend_from_slice(&raw); } } - Ok(Event::End(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"txbxContent") && depth == 1 - { - break; - } - depth -= 1; + // Every child element is consumed whole, so the first end + // tag at this level closes the txbxContent element. + Ok(Event::End(ie)) => { + writer.write_event(Event::End(ie))?; + break; } - Ok(_) => { - // Whitespace/text at top level of txbxContent, skip + Ok(Event::Eof) => { + return Err(OxmlError::MissingElement("w:txbxContent end".to_owned())); } + // Empty elements such as `` or a bookmark, whitespace, + // comments and processing instructions. + Ok(ev) => writer.write_event(ev)?, Err(e) => return Err(e.into()), } inner_buf.clear(); } - - // Hand the collected paragraphs to the caller's edit function - total_count += edit(&mut paragraphs); - - // Re-serialize paragraphs into the writer - for p in ¶graphs { - p.to_xml(&mut writer)?; - } - - // Write closing txbxContent tag - writer.write_event(Event::End(quick_xml::events::BytesEnd::new( - "w:txbxContent", - )))?; } Ok(ev) => { writer.write_event(ev)?; @@ -918,6 +891,59 @@ mod tests { assert!(result_str.contains("Company: Acme")); } + /// A text box whose paragraphs sit among a table, a block content + /// control, a bookmark, an empty paragraph, a comment and whitespace. + /// A second text box without a placeholder follows. The walker rewrites + /// every text box of the part, so that one must come back unchanged too. + /// The paragraphs without a placeholder keep their identity attributes. + const TEXT_BOX_WITH_EVERY_KIND_OF_CHILD: &str = r#" +Title +Hello {{name}} +cell +control{{name}} again +other cellNo placeholder here"#; + + #[test] + fn replace_in_textbox_keeps_every_other_child_in_order() { + let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD; + let expected = xml.replace("{{name}}", "Alice"); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + let (result, count) = replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + } + + /// The end tag used to be written as `w:txbxContent` whatever prefix the + /// start tag carried, which left the part ill-formed. + #[test] + fn replace_in_textbox_closes_it_with_its_own_prefix() { + let xml = r#"Hello {{name}}"#; + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 1); + assert_eq!( + String::from_utf8(result).unwrap(), + xml.replace("{{name}}", "Alice") + ); + } + + /// A text box cut short used to keep the walker reading past the end of + /// the part forever. + #[test] + fn replace_in_unterminated_textbox_is_an_error() { + for tail in ["", "Hello {{name}}"] { + let xml = format!( + r#"{tail}"# + ); + assert!(replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").is_err()); + } + } + #[test] fn replace_in_xml_part_no_textbox() { let xml = br#" diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c74..58cf66aa3 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -16180,6 +16180,81 @@ fn zero_width_regex_match_terminates() { assert!(count <= 4, "should not loop indefinitely, got {count}"); } +/// Replacement in a text box parses the paragraphs of its `w:txbxContent` and +/// used to write back nothing else, so a hit there deleted the tables, +/// content controls, bookmarks and empty paragraphs of that text box. +mod text_box_replacement_keeps_every_child { + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + /// One copy of the text box: the paragraph the replacement edits, then a + /// table, a block content control and a bookmark around an empty + /// paragraph. Each copy needs its own bookmark id for the file to open. + fn text_box_content(text: &str, id: u32) -> String { + format!( + r#"{text}cellcontrol"# + ) + } + + /// A text box as Word saves it, the DrawingML shape in `mc:Choice` and its + /// VML copy in `mc:Fallback`, or the bare `wp:anchor` alone. + fn text_box_document(compatibility_block: bool) -> Document { + let content = text_box_content("Dear {{name}}", 7); + let drawing = format!( + r#"00{content}"# + ); + let shape = if compatibility_block { + format!( + r#"{drawing}{}"#, + text_box_content("Dear {{name}}", 8) + ) + } else { + drawing + }; + super::document_with_content_controls(&format!( + r#"Host paragraph{shape}"# + )) + } + + /// Replace in both text box forms, then check that every copy holds the + /// edited paragraph and every other child byte for byte and in order. + fn assert_replacement_keeps_every_child(replace: impl Fn(&mut Document) -> usize) { + for (compatibility_block, ids) in [(true, &[7, 8][..]), (false, &[7][..])] { + let mut document = text_box_document(compatibility_block); + assert_eq!(replace(&mut document), ids.len(), "{compatibility_block}"); + let saved = super::document_xml(&mut document); + assert_eq!(saved.matches("").count(), ids.len()); + for &id in ids { + let expected = text_box_content("Dear Ada", id); + assert!(saved.contains(&expected), "{expected}\n{saved}"); + } + } + } + + #[test] + fn replacing_text_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.try_replace_text("{{name}}", "Ada").unwrap() + }); + } + + #[test] + fn regex_replacement_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.replace_regex(r"\{\{(\w+)\}\}", "Ada").unwrap() + }); + } + + #[test] + fn a_template_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document + .render_template(&serde_json::json!({"name": "Ada"})) + .unwrap() + }); + } +} + /// `9360 / cols` panicked when a caller asked for a zero-column table. #[test] fn zero_column_tables_do_not_panic() { From 242e9984b33a96d32ece1ec3de95cc0c3aed76f1 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 18:31:12 +0200 Subject: [PATCH 2/2] Re-record the archive measurements of rdocx-oxml and rdocx The text box fix grows the rdocx-oxml sources and the rdocx regression tests, and both are packaged, so the crates.io archive rows of the two READMEs and their ARCHIVE_MEASUREMENTS entries are re-measured. Both rows are dated 2026-09-27 through ARCHIVE_REMEASUREMENT_DATES. GitHub issue #160. --- README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 7 ++++--- 3 files changed, 6 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 479b462ef..482ee0252 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,093,637 compressed bytes, 6,503,388 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef2416..791cc1c1d 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 368,405 compressed bytes, 2,382,882 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643be..7decad66e 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,8 +367,9 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", + "rdocx": "2026-09-27", "rdocx-layout": "2026-09-26", + "rdocx-oxml": "2026-09-27", "rpptx": "2026-09-26", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" @@ -383,12 +384,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_093_637, 6_503_388, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (368_405, 2_382_882, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),