diff --git a/README.md b/README.md index 479b462ef..482ee0252 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,093,637 compressed bytes, 6,503,388 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef2416..791cc1c1d 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 368,405 compressed bytes, 2,382,882 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 0bf078dd4..15e863548 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -309,9 +309,13 @@ pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replaceme /// Replace placeholders in text boxes and shapes within a raw XML part. /// -/// Walks the XML, finds `w:txbxContent` elements at any depth, parses their -/// child `w:p` elements using `CT_P::from_xml`, performs replacement, and -/// re-serializes back. Returns the modified XML and replacement count. +/// Walks the XML, finds `w:txbxContent` elements at any depth outside another +/// text box, parses their child `w:p` elements using `CT_P::from_xml`, +/// performs replacement, and re-serializes the paragraphs it changed. Every +/// other child of the text box, such as a table, a content control, a +/// bookmark or a paragraph without a match, is copied through verbatim in its +/// place. A text box nested inside another one is kept as it is, not edited. +/// Returns the modified XML and replacement count. pub fn replace_in_xml_part( xml: &[u8], placeholder: &str, @@ -329,11 +333,11 @@ pub fn replace_many_in_xml_part( xml: &[u8], replacements: &[(&str, &str)], ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { + rewrite_text_boxes(xml, &mut |paragraph| { replacements .iter() .map(|(placeholder, replacement)| { - replace_in_paragraphs(paragraphs, placeholder, replacement) + replace_in_paragraph(paragraph, placeholder, replacement) }) .sum() }) @@ -348,18 +352,22 @@ pub fn replace_regex_in_xml_part( re: ®ex::Regex, replacement: &str, ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { - replace_regex_in_paragraphs(paragraphs, re, replacement) + rewrite_text_boxes(xml, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) }) } -/// Walk `xml`, handing the paragraphs of each `w:txbxContent` element to -/// `edit`, and re-serialise. Returns the rewritten XML and the summed count. +/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit` +/// and re-serialising it in place when `edit` counts a change. Every other +/// paragraph and child of the text box is copied through verbatim. Returns the +/// rewritten XML and the summed count. fn rewrite_text_boxes( xml: &[u8], - edit: &mut dyn FnMut(&mut Vec) -> usize, + edit: &mut dyn FnMut(&mut CT_P) -> usize, ) -> crate::error::Result<(Vec, usize)> { + use crate::error::OxmlError; use crate::namespace::matches_local_name; + use crate::raw_xml::capture_element; use quick_xml::events::Event; use quick_xml::{Reader, Writer}; @@ -374,50 +382,18 @@ fn rewrite_text_boxes( match reader.read_event_into(&mut buf) { Ok(Event::Eof) => break, Ok(Event::Start(ref e)) if matches_local_name(e.name().as_ref(), b"txbxContent") => { - // We found a txbxContent element. Collect its contents as raw XML, - // parse paragraphs, do replacement, and re-serialize. + // We found a txbxContent element. Parse and edit each paragraph + // and copy every other child through verbatim, in document order. writer.write_event(Event::Start(e.clone()))?; - // Read all events inside txbxContent - let mut depth = 1u32; let mut inner_buf = Vec::new(); - // Collect paragraphs from inside txbxContent - let mut paragraphs: Vec = Vec::new(); - loop { match reader.read_event_into(&mut inner_buf) { Ok(Event::Start(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"p") && depth == 1 { + if matches_local_name(ie.name().as_ref(), b"p") { // Parse this paragraph: collect its XML, then parse via CT_P - let mut para_writer = Writer::new(Vec::new()); - // Write the opening tag - para_writer.write_event(Event::Start(ie.clone()))?; - let mut pdepth = 1u32; - let mut pbuf = Vec::new(); - loop { - match reader.read_event_into(&mut pbuf) { - Ok(Event::Start(ref pe)) => { - pdepth += 1; - para_writer.write_event(Event::Start(pe.clone()))?; - } - Ok(Event::End(ref pe)) => { - pdepth -= 1; - para_writer.write_event(Event::End(pe.clone()))?; - if pdepth == 0 { - break; - } - } - Ok(ref ev) => { - para_writer.write_event(ev.clone())?; - } - Err(e) => return Err(e.into()), - } - pbuf.clear(); - } - let para_xml = para_writer.into_inner(); - - // Parse the paragraph + let para_xml = capture_element(&mut reader, ie)?; let mut para_reader = Reader::from_reader(para_xml.as_slice()); para_reader.config_mut().trim_text(true); let mut prbuf = Vec::new(); @@ -434,42 +410,39 @@ fn rewrite_text_boxes( } prbuf.clear(); } - let para = CT_P::from_xml(&mut para_reader)?; - paragraphs.push(para); + let mut para = CT_P::from_xml(&mut para_reader)?; + let count = edit(&mut para); + if count == 0 { + // Nothing changed, so the paragraph keeps its + // bytes, start-tag attributes included. + writer.get_mut().extend_from_slice(¶_xml); + } else { + para.to_xml(&mut writer)?; + } + total_count += count; } else { - depth += 1; - // Non-paragraph element inside txbxContent; skip it - reader.read_to_end_into(ie.name(), &mut Vec::new())?; - depth -= 1; + // A table, a content control or any other element + // the edit does not reach stays as it was. + let raw = capture_element(&mut reader, ie)?; + writer.get_mut().extend_from_slice(&raw); } } - Ok(Event::End(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"txbxContent") && depth == 1 - { - break; - } - depth -= 1; + // Every child element is consumed whole, so the first end + // tag at this level closes the txbxContent element. + Ok(Event::End(ie)) => { + writer.write_event(Event::End(ie))?; + break; } - Ok(_) => { - // Whitespace/text at top level of txbxContent, skip + Ok(Event::Eof) => { + return Err(OxmlError::MissingElement("w:txbxContent end".to_owned())); } + // Empty elements such as `` or a bookmark, whitespace, + // comments and processing instructions. + Ok(ev) => writer.write_event(ev)?, Err(e) => return Err(e.into()), } inner_buf.clear(); } - - // Hand the collected paragraphs to the caller's edit function - total_count += edit(&mut paragraphs); - - // Re-serialize paragraphs into the writer - for p in ¶graphs { - p.to_xml(&mut writer)?; - } - - // Write closing txbxContent tag - writer.write_event(Event::End(quick_xml::events::BytesEnd::new( - "w:txbxContent", - )))?; } Ok(ev) => { writer.write_event(ev)?; @@ -918,6 +891,59 @@ mod tests { assert!(result_str.contains("Company: Acme")); } + /// A text box whose paragraphs sit among a table, a block content + /// control, a bookmark, an empty paragraph, a comment and whitespace. + /// A second text box without a placeholder follows. The walker rewrites + /// every text box of the part, so that one must come back unchanged too. + /// The paragraphs without a placeholder keep their identity attributes. + const TEXT_BOX_WITH_EVERY_KIND_OF_CHILD: &str = r#" +Title +Hello {{name}} +cell +control{{name}} again +other cellNo placeholder here"#; + + #[test] + fn replace_in_textbox_keeps_every_other_child_in_order() { + let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD; + let expected = xml.replace("{{name}}", "Alice"); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + let (result, count) = replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + } + + /// The end tag used to be written as `w:txbxContent` whatever prefix the + /// start tag carried, which left the part ill-formed. + #[test] + fn replace_in_textbox_closes_it_with_its_own_prefix() { + let xml = r#"Hello {{name}}"#; + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 1); + assert_eq!( + String::from_utf8(result).unwrap(), + xml.replace("{{name}}", "Alice") + ); + } + + /// A text box cut short used to keep the walker reading past the end of + /// the part forever. + #[test] + fn replace_in_unterminated_textbox_is_an_error() { + for tail in ["", "Hello {{name}}"] { + let xml = format!( + r#"{tail}"# + ); + assert!(replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").is_err()); + } + } + #[test] fn replace_in_xml_part_no_textbox() { let xml = br#" diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c74..58cf66aa3 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -16180,6 +16180,81 @@ fn zero_width_regex_match_terminates() { assert!(count <= 4, "should not loop indefinitely, got {count}"); } +/// Replacement in a text box parses the paragraphs of its `w:txbxContent` and +/// used to write back nothing else, so a hit there deleted the tables, +/// content controls, bookmarks and empty paragraphs of that text box. +mod text_box_replacement_keeps_every_child { + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + /// One copy of the text box: the paragraph the replacement edits, then a + /// table, a block content control and a bookmark around an empty + /// paragraph. Each copy needs its own bookmark id for the file to open. + fn text_box_content(text: &str, id: u32) -> String { + format!( + r#"{text}cellcontrol"# + ) + } + + /// A text box as Word saves it, the DrawingML shape in `mc:Choice` and its + /// VML copy in `mc:Fallback`, or the bare `wp:anchor` alone. + fn text_box_document(compatibility_block: bool) -> Document { + let content = text_box_content("Dear {{name}}", 7); + let drawing = format!( + r#"00{content}"# + ); + let shape = if compatibility_block { + format!( + r#"{drawing}{}"#, + text_box_content("Dear {{name}}", 8) + ) + } else { + drawing + }; + super::document_with_content_controls(&format!( + r#"Host paragraph{shape}"# + )) + } + + /// Replace in both text box forms, then check that every copy holds the + /// edited paragraph and every other child byte for byte and in order. + fn assert_replacement_keeps_every_child(replace: impl Fn(&mut Document) -> usize) { + for (compatibility_block, ids) in [(true, &[7, 8][..]), (false, &[7][..])] { + let mut document = text_box_document(compatibility_block); + assert_eq!(replace(&mut document), ids.len(), "{compatibility_block}"); + let saved = super::document_xml(&mut document); + assert_eq!(saved.matches("").count(), ids.len()); + for &id in ids { + let expected = text_box_content("Dear Ada", id); + assert!(saved.contains(&expected), "{expected}\n{saved}"); + } + } + } + + #[test] + fn replacing_text_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.try_replace_text("{{name}}", "Ada").unwrap() + }); + } + + #[test] + fn regex_replacement_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.replace_regex(r"\{\{(\w+)\}\}", "Ada").unwrap() + }); + } + + #[test] + fn a_template_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document + .render_template(&serde_json::json!({"name": "Ada"})) + .unwrap() + }); + } +} + /// `9360 / cols` panicked when a caller asked for a zero-column table. #[test] fn zero_column_tables_do_not_panic() { diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643be..7decad66e 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,8 +367,9 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", + "rdocx": "2026-09-27", "rdocx-layout": "2026-09-26", + "rdocx-oxml": "2026-09-27", "rpptx": "2026-09-26", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" @@ -383,12 +384,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_093_637, 6_503_388, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (368_405, 2_382_882, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),