From 267a1524148a0e615231dcaa9ebbc6d4a783f9d2 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:20:30 +0200 Subject: [PATCH 01/10] Ignore the content-control id when comparing documents compare() refused any pair whose content controls differed only by w:sdtPr/w:id, with "comparison cannot revise content-control properties". Word and Google Docs renumber that id when they save, so two versions of a document with a table-of-contents control could not be compared at all, although nothing in the control had changed. The id was part of control_property_signature, the one tuple that drives control alignment, the refusal and the accept and reject postconditions. Leaving it out of that tuple keeps the three consistent. Controls that differ only by id now align as unchanged and the redline keeps the original w:sdtPr bytes, and a change inside such a control is revised like any other. A w:sdtPr that holds none of the compared properties reads like no w:sdtPr, so a control that gains or loses an id-only w:sdtPr compares too. A differing w:tag, alias, control type or data binding still refuses the pair. HLD 03 now says the id is not part of the control shell. GitHub issue #159. --- crates/rdocx/src/comparison.rs | 28 ++++-- crates/rdocx/tests/regression_test.rs | 128 ++++++++++++++++++++++++++ docs/hld/03-architecture.md | 9 +- 3 files changed, 152 insertions(+), 13 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 19c557ed..5553d681 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -29,7 +29,6 @@ thread_local! { type ControlPropertySignature<'a> = Option<( Option<&'a str>, Option<&'a str>, - Option, Option, Option<&'a rdocx_oxml::content_control::CT_DataBinding>, )>; @@ -5478,16 +5477,25 @@ fn control_signature(control: &CT_Sdt) -> String { ) } +/// The content-control properties that alignment, refusal and the accept and +/// reject postconditions compare. +/// +/// `w:id` is left out. Producers renumber it on save and it carries no +/// content, so a pair that differs only by it keeps the original's `w:sdtPr`. +/// A `w:sdtPr` with none of these properties reads like no `w:sdtPr`. fn control_property_signature(control: &CT_Sdt) -> ControlPropertySignature<'_> { - control.properties.as_ref().map(|properties| { - ( - properties.alias.as_deref(), - properties.tag.as_deref(), - properties.id, - properties.control_type, - properties.data_binding.as_ref(), - ) - }) + control + .properties + .as_ref() + .map(|properties| { + ( + properties.alias.as_deref(), + properties.tag.as_deref(), + properties.control_type, + properties.data_binding.as_ref(), + ) + }) + .filter(|signature| !matches!(signature, (None, None, None, None))) } fn modeled_control_content(control: &CT_Sdt) -> Vec<&SdtContent> { diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c7..dbd61d6f 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30379,6 +30379,134 @@ fn comparison_treats_empty_paragraph_properties_as_absent() { assert!(document_xml(&mut empty).contains(" Vec { + document + .revisions() + .iter() + .map(|revision| revision.kind()) + .collect() + } + + /// Check that accepting gives the edited side and rejecting the original, + /// each compared again with no diagnostic and no revision. + fn assert_resolutions(tracked: &[u8], original: &Document, edited: &Document) { + for (resolve, expected) in [ + ( + Document::accept_all as fn(&mut Document) -> rdocx::Result, + edited, + ), + (Document::reject_all, original), + ] { + let mut resolved = Document::from_bytes(tracked).unwrap(); + resolve(&mut resolved).unwrap(); + let diagnostics = resolved + .compare(expected, "postcondition", TIMESTAMP) + .unwrap(); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_eq!(revision_kinds(&resolved), []); + } + } + + /// Compare two bodies and return the revision kinds and the redline. + fn compared_kinds(original_xml: &str, edited_xml: &str) -> (Vec, String) { + let original = document_with_content_controls(original_xml); + let edited = document_with_content_controls(edited_xml); + let mut compared = document_with_content_controls(original_xml); + let diagnostics = compared + .compare(&edited, "R", TIMESTAMP) + .expect("producer noise must not refuse the pair"); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited); + (revision_kinds(&compared), document_xml(&mut compared)) + } + + fn table_of_contents_control(id: Option<&str>, first_entry: &str) -> String { + let id = id.map_or_else(String::new, |value| format!(r#""#)); + wrap_word_body(&format!( + r#"Before the content control.{id}{first_entry} entryBeta entryGamma entryAfter the content control."# + )) + } + + fn google_docs_inline_control(id: &str, word: &str) -> String { + wrap_word_body(&format!( + r#"Before {word} after."# + )) + } + + #[test] + fn a_content_control_identity_is_not_content() { + for (original, edited, kept_id) in [ + (None, Some("-2000000001"), None), + (Some("-2000000001"), None, Some("-2000000001")), + (Some("11"), Some("12"), Some("11")), + ] { + let (kinds, tracked) = compared_kinds( + &table_of_contents_control(original, "Alpha"), + &table_of_contents_control(edited, "Alpha"), + ); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + let ids = [original, edited] + .into_iter() + .flatten() + .filter(|id| tracked.contains(&format!(r#""#))) + .collect::>(); + assert_eq!(ids, kept_id.into_iter().collect::>(), "{tracked}"); + + let (kinds, tracked) = compared_kinds( + &table_of_contents_control(original, "Alpha"), + &table_of_contents_control(edited, "Delta"), + ); + assert_eq!( + kinds, + [RevisionKind::Deletion, RevisionKind::Insertion], + "{original:?} -> {edited:?}" + ); + assert_eq!(tracked.matches("").count(), 1, "{tracked}"); + } + + let (kinds, _) = compared_kinds( + &google_docs_inline_control("-1854911024", "Alpha"), + &google_docs_inline_control("1374263513", "Alpha"), + ); + assert_eq!(kinds, []); + let (kinds, tracked) = compared_kinds( + &google_docs_inline_control("-1854911024", "Alpha"), + &google_docs_inline_control("1374263513", "Delta"), + ); + assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]); + assert!( + tracked.contains(r#""#), + "{tracked}" + ); + + let bare_control = |properties: &str| { + wrap_word_body(&format!( + r#"{properties}Alpha entryAfter the content control."# + )) + }; + let id_only = r#""#; + for (original, edited) in [("", id_only), (id_only, "")] { + let (kinds, tracked) = compared_kinds(&bare_control(original), &bare_control(edited)); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + assert_eq!( + tracked.contains(""), + !original.is_empty(), + "{tracked}" + ); + } + } +} + #[test] fn unmodelled_property_changes_report_a_diagnostic() { for (original, edited, expected_location) in [ diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 97ce19c6..ca07b49e 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -910,9 +910,12 @@ section properties emit property revisions that retain the original property sidecars. Unsupported formatting differences retain the original bytes and produce stable `ComparisonDiagnostic` values at the actual story path. Inputs with existing modeled revisions or differing story and control shells are -rejected unless their story category is ignored. Attributed text alignment -retains owner, formatting, content position, and raw-child boundaries, then -coalesces adjacent equal-owner edits into minimal revision wrappers. +rejected unless their story category is ignored. A content control's `w:id` +is producer identity and not part of its shell, so controls that differ only +by it align, compare, and keep the original `w:sdtPr`. Attributed text +alignment retains owner, formatting, content position, and raw-child +boundaries, then coalesces adjacent equal-owner edits into minimal revision +wrappers. When a main story gains a trailing run of paragraphs, comparison marks the original final paragraph boundary once, marks each intermediate inserted paragraph boundary once, and leaves the final inserted paragraph mark as the From 2e6b3b0cfd211978b8a453d2bbc4ef516a5f406f Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:21:08 +0200 Subject: [PATCH 02/10] Keep xml:space on replaced runs and compare it only at text edges A run rewritten by try_replace_text lost xml:space="preserve" on its w:t, because replace_in_single_run and replace_across_runs recomputed the flag from the new text alone. Google Docs writes the flag on every w:t, so a no-op replacement, common in a scripted pass that normalizes wording, turned an unchanged run into a deletion and an insertion once the edited copy was compared against its source. The three rewrite sites, which regex replacement shares, now keep a flag the producer wrote and still add one when the new text starts or ends with a space. compare() also read the flag as content wherever it appeared, because the run signature formatted CT_Text whole. The flag only changes how text reads when whitespace sits at an edge, so the text signature now counts it only there. The same signature serves alignment and the accept and reject postconditions, so they stay consistent, and a run that matches keeps the original bytes. Word and character granularity and the ignore options compare text units, which the redline writes back as their own w:t. Each unit copied the flag of its text, so a space split out of a flagged text differed from the same space in an unflagged one. A unit with whitespace at an edge now always carries the flag, so on that path whitespace reads as whitespace whatever the source flag was, and the flag is never content. Files from two producers that place the flag differently then compare the same way at every granularity, and a space split out of a text keeps the flag in the redline, where Word used to drop it. On that path, the shortcut that keeps runs whole when every run of a paragraph matches reads the flag the same way. Otherwise a run that differed only by the flag at an edge was rewritten one unit per run, which moved a bookmark end indexed by run and refused the pair. GitHub issue #160. --- crates/rdocx-oxml/src/placeholder.rs | 37 +++++- crates/rdocx/src/comparison.rs | 51 +++++++- crates/rdocx/tests/regression_test.rs | 171 +++++++++++++++++++++++++- 3 files changed, 248 insertions(+), 11 deletions(-) diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 0bf078dd..59780350 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -153,7 +153,9 @@ fn replace_in_single_run( new_text.push_str(replacement); new_text.push_str(&t.text[byte_end..]); t.text = new_text; - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + // Keep the flag the producer wrote. Dropping it rewrites an unchanged + // run, which a later comparison against the source then reports. + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } } @@ -176,7 +178,7 @@ fn replace_across_runs( new_text.push_str(&t.text[..first_byte_offset]); new_text.push_str(replacement); t.text = new_text; - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } // Handle the last run: replace from start to match end within that content item. @@ -189,7 +191,7 @@ fn replace_across_runs( let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0); let byte_end = last_byte_offset + ch_len; t.text = t.text[byte_end..].to_string(); - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } // Clear text content from runs strictly between first and last. @@ -796,6 +798,35 @@ mod tests { assert_eq!(p.runs[1].properties.as_ref().unwrap().italic, Some(true)); } + #[test] + fn replace_keeps_the_producer_space_flag() { + let mut preserved = CT_R::new("WORD"); + if let RunContent::Text(text) = &mut preserved.content[0] { + text.preserve_space = true; + } + let mut p = CT_P::new(); + p.runs.push(preserved.clone()); + p.runs.push(preserved); + p.add_run("tail"); + + assert_eq!(replace_in_paragraph(&mut p, "WORD", "WORD"), 2); + assert_eq!(replace_in_paragraph(&mut p, "DWO", "D-WO"), 1); + assert_eq!(replace_in_paragraph(&mut p, "tail", " end"), 1); + let flags = p + .runs + .iter() + .map(|run| match &run.content[0] { + RunContent::Text(text) => (text.text.as_str(), text.preserve_space), + _ => unreachable!(), + }) + .collect::>(); + assert_eq!( + flags, + [("WORD-WO", true), ("RD", true), (" end", true)], + "a rewritten run keeps a producer flag and gains one at a text edge" + ); + } + #[test] fn replace_multiple_occurrences() { let mut p = make_para(&["{{x}} and {{x}}"]); diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 5553d681..02c33e98 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -2984,8 +2984,16 @@ fn compare_granular_paragraph( ))); } - let original_run_signatures = original.runs.iter().map(run_signature).collect::>(); - let edited_run_signatures = edited.runs.iter().map(run_signature).collect::>(); + let original_run_signatures = original + .runs + .iter() + .map(attributed_run_signature) + .collect::>(); + let edited_run_signatures = edited + .runs + .iter() + .map(attributed_run_signature) + .collect::>(); if original_run_signatures == edited_run_signatures && original.content_controls == edited.content_controls { @@ -3557,12 +3565,30 @@ fn granular_text(text: &CT_Text, options: &ComparisonOptions) -> Vec { fragments .into_iter() .map(|value| CT_Text { + // A unit is written back as its own `w:t`, where edge whitespace + // needs the flag to survive. Whitespace in a unit then reads as + // whitespace whatever the source flag was. + preserve_space: text.preserve_space || has_edge_whitespace(&value), text: value, - preserve_space: text.preserve_space, }) .collect() } +/// The whole-run signature that agrees with the attributed units. +/// +/// It reads the space flag the way `granular_text` writes it on every unit, +/// so a run that differs only by the flag stays whole instead of being +/// rewritten one unit per run, which would move run-indexed bookmark ends. +fn attributed_run_signature(run: &CT_R) -> String { + let mut run = run.clone(); + for content in &mut run.content { + if let RunContent::Text(text) | RunContent::DeletedText(text) = content { + text.preserve_space |= has_edge_whitespace(&text.text); + } + } + run_signature(&run) +} + fn whitespace_fragments(text: &str) -> Vec { let mut output = Vec::new(); let mut current = String::new(); @@ -5410,10 +5436,29 @@ fn run_content_signature(content: &RunContent) -> String { match content { RunContent::Field(_) => "field-owner".to_owned(), RunContent::Drawing(drawing) => format!("Drawing({:?})", drawing_signature(drawing)), + RunContent::Text(text) => format!("Text({:?})", text_signature(text)), + RunContent::DeletedText(text) => format!("DeletedText({:?})", text_signature(text)), content => format!("{content:?}"), } } +/// The text and whether its `xml:space="preserve"` changes how it reads. +/// +/// The flag only protects whitespace at an edge of the text. Producers write +/// it on every `w:t` or only where needed, so elsewhere it is serialization. +fn text_signature(text: &CT_Text) -> (&str, bool) { + ( + &text.text, + text.preserve_space && has_edge_whitespace(&text.text), + ) +} + +/// Whether XML whitespace starts or ends the text. +fn has_edge_whitespace(text: &str) -> bool { + let whitespace = |character: char| matches!(character, ' ' | '\t' | '\n' | '\r'); + text.starts_with(whitespace) || text.ends_with(whitespace) +} + fn table_signature(table: &CT_Tbl) -> String { format!( "{:?}:{:?}:{:?}:{:?}", diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index dbd61d6f..5afa2345 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30385,7 +30385,7 @@ fn comparison_treats_empty_paragraph_properties_as_absent() { /// revision. mod compare_producer_noise { use super::*; - use rdocx::RevisionKind; + use rdocx::{ComparisonGranularity, ComparisonOptions, RevisionKind}; const TIMESTAMP: &str = "2026-09-27T12:00:00Z"; @@ -30399,7 +30399,12 @@ mod compare_producer_noise { /// Check that accepting gives the edited side and rejecting the original, /// each compared again with no diagnostic and no revision. - fn assert_resolutions(tracked: &[u8], original: &Document, edited: &Document) { + fn assert_resolutions( + tracked: &[u8], + original: &Document, + edited: &Document, + options: &ComparisonOptions, + ) { for (resolve, expected) in [ ( Document::accept_all as fn(&mut Document) -> rdocx::Result, @@ -30410,7 +30415,7 @@ mod compare_producer_noise { let mut resolved = Document::from_bytes(tracked).unwrap(); resolve(&mut resolved).unwrap(); let diagnostics = resolved - .compare(expected, "postcondition", TIMESTAMP) + .compare_with_options(expected, "postcondition", TIMESTAMP, options) .unwrap(); assert!(diagnostics.is_empty(), "{diagnostics:?}"); assert_eq!(revision_kinds(&resolved), []); @@ -30419,14 +30424,22 @@ mod compare_producer_noise { /// Compare two bodies and return the revision kinds and the redline. fn compared_kinds(original_xml: &str, edited_xml: &str) -> (Vec, String) { + compared_kinds_with(original_xml, edited_xml, &ComparisonOptions::default()) + } + + fn compared_kinds_with( + original_xml: &str, + edited_xml: &str, + options: &ComparisonOptions, + ) -> (Vec, String) { let original = document_with_content_controls(original_xml); let edited = document_with_content_controls(edited_xml); let mut compared = document_with_content_controls(original_xml); let diagnostics = compared - .compare(&edited, "R", TIMESTAMP) + .compare_with_options(&edited, "R", TIMESTAMP, options) .expect("producer noise must not refuse the pair"); assert!(diagnostics.is_empty(), "{diagnostics:?}"); - assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited); + assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited, options); (revision_kinds(&compared), document_xml(&mut compared)) } @@ -30505,6 +30518,154 @@ mod compare_producer_noise { ); } } + + fn replaced_copy(source_xml: &str, old: &str, new: &str) -> String { + let mut document = document_with_content_controls(source_xml); + assert_eq!(document.try_replace_text(old, new).unwrap(), 1); + let mut edited = Document::from_bytes(&document.to_bytes().unwrap()).unwrap(); + document_xml(&mut edited) + } + + #[test] + fn a_rewritten_run_keeps_its_producer_space_flag() { + let preserved = wrap_word_body( + r#"Paragraph 1.WORD"#, + ); + let edited = replaced_copy(&preserved, "WORD", "WORD"); + assert!( + edited.contains(r#"WORD"#), + "{edited}" + ); + let (kinds, _) = compared_kinds(&preserved, &edited); + assert_eq!(kinds, []); + + let edge_space_removed = replaced_copy( + &wrap_word_body(r#"WORD "#), + "WORD ", + "WORD", + ); + assert!( + edge_space_removed.contains(r#"WORD"#), + "{edge_space_removed}" + ); + } + + #[test] + fn the_space_flag_is_content_only_at_a_text_edge() { + let paragraphs = |first: &str, second: &str| { + wrap_word_body(&format!( + r#"{first}{second}"# + )) + }; + let paragraph = |text: &str| paragraphs("Paragraph 1.", text); + // Bookmark ends are indexed by run, so an unchanged run must stay whole. + let bookmarked = |text: &str| { + wrap_word_body(&format!( + r#"{text}tail"# + )) + }; + let changed = &[RevisionKind::Deletion, RevisionKind::Insertion][..]; + let unchanged = &[][..]; + // Each case gives the kinds of the whole-run path and of the + // attributed path. The attributed path writes every unit with edge + // whitespace with the flag, so the flag is never content there. + let cases = [ + ( + paragraph(r#"WORD"#), + paragraph("WORD"), + unchanged, + unchanged, + ), + ( + paragraph(r#"two words"#), + paragraph("two words"), + unchanged, + unchanged, + ), + ( + paragraph("two words"), + paragraph(r#"two words"#), + unchanged, + unchanged, + ), + ( + paragraph(r#"WORD "#), + paragraph("WORD "), + changed, + unchanged, + ), + ( + bookmarked(r#"two words "#), + bookmarked("two words "), + changed, + unchanged, + ), + ( + paragraph("WORD"), + paragraph("text"), + changed, + changed, + ), + // Google Docs flags every `w:t` and Word only where needed. + ( + paragraphs( + r#"two words"#, + r#"WORD"#, + ), + paragraphs("two words", "text"), + changed, + changed, + ), + ]; + for options in [ + ComparisonOptions::default(), + ComparisonOptions { + ignore_formatting: true, + ..Default::default() + }, + ComparisonOptions { + granularity: ComparisonGranularity::Word, + ..Default::default() + }, + ComparisonOptions { + granularity: ComparisonGranularity::Character, + ..Default::default() + }, + ] { + for (original, edited, whole_run, attributed) in &cases { + let expected = if options == ComparisonOptions::default() { + whole_run + } else { + attributed + }; + for (left, right) in [(original, edited), (edited, original)] { + let (kinds, _) = compared_kinds_with(left, right, &options); + assert_eq!(kinds, *expected, "{options:?}: {left} -> {right}"); + } + } + } + + // A space split out of a text keeps reading as a space in the redline. + for granularity in [ + ComparisonGranularity::Word, + ComparisonGranularity::Character, + ] { + let (kinds, tracked) = compared_kinds_with( + ¶graph("two words"), + ¶graph("two wordy"), + &ComparisonOptions { + granularity, + ..Default::default() + }, + ); + assert_eq!(kinds, changed, "{granularity:?}"); + assert!(!tracked.contains(" "), "{tracked}"); + assert!( + tracked.contains(r#" "#), + "{tracked}" + ); + } + } } #[test] From c8880e276a7ac6c0414af67a5f45355db6a99973 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:21:40 +0200 Subject: [PATCH 03/10] Treat the default portrait orientation as absent in compare Comparing a file whose w:pgSz carries w:orient="portrait" against its rdocx-edited copy reported a section_property_change next to the real edit. The pgSz writer emits w:orient only for landscape, so a modelled edit drops the default value, and section_properties_xml compared the two sections as whole CT_SectPr values, Some(Portrait) against None. Portrait is the schema default, so both modelled sections now read it as absent before the equality test. A paragraph that holds a bookmark, hyperlink, comment range or control compares its whole paragraph properties, section break included, so the same rule applies there. Without it such a section-break paragraph kept a formatting diagnostic, or refused the pair when its text changed. A real change to landscape is still a section property revision. The serializer is left alone, because the typed defaults of every generated document carry Some(Portrait) and writing it would move the document.xml hash baselines. GitHub issue #160. --- crates/rdocx/src/comparison.rs | 24 +++++++++++-- crates/rdocx/tests/regression_test.rs | 49 +++++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 2 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 02c33e98..37d28126 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -11,6 +11,7 @@ use rdocx_oxml::content_control::{CT_Sdt, SdtContent}; use rdocx_oxml::document::{BodyContent, CT_Document}; use rdocx_oxml::namespace::W_NS; use rdocx_oxml::properties::CT_PPr; +use rdocx_oxml::shared::ST_PageOrientation; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_TblPr, CT_Tc, CT_TrPr, CellContent}; use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent}; use sha2::{Digest, Sha256}; @@ -4002,8 +4003,22 @@ fn nonempty_paragraph_properties(mut properties: CT_PPr) -> Option { } fn paragraph_properties_differ(original: Option<&CT_PPr>, edited: Option<&CT_PPr>) -> bool { - original.cloned().and_then(nonempty_paragraph_properties) - != edited.cloned().and_then(nonempty_paragraph_properties) + let modeled = |properties: Option<&CT_PPr>| { + properties.cloned().and_then(|mut properties| { + if let Some(section) = properties.sect_pr.as_mut() { + clear_default_orientation(section); + } + nonempty_paragraph_properties(properties) + }) + }; + modeled(original) != modeled(edited) +} + +/// Portrait is the schema default, which the section writer omits. +fn clear_default_orientation(section: &mut rdocx_oxml::document::CT_SectPr) { + if section.orientation == Some(ST_PageOrientation::Portrait) { + section.orientation = None; + } } fn section_properties_xml( @@ -4033,6 +4048,8 @@ fn section_properties_xml( original_modeled.change = None; let mut edited_modeled = edited.clone(); edited_modeled.change = None; + clear_default_orientation(&mut original_modeled); + clear_default_orientation(&mut edited_modeled); if original_modeled == edited_modeled { return section_property_xml(original); } @@ -5810,6 +5827,9 @@ fn paragraph_formatting(paragraph: &CT_P) -> Option { properties.numbering_revision_position = None; properties.change = None; properties.revision_xml.clear(); + if let Some(section) = properties.sect_pr.as_mut() { + clear_default_orientation(section); + } properties }) } diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 5afa2345..08f17b7a 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30666,6 +30666,55 @@ mod compare_producer_noise { ); } } + + fn page_body(orientation: &str) -> String { + wrap_word_body(&format!( + r#"Paragraph 1, lorem ipsum dolor sit amet.WORDParagraph 3, lorem ipsum dolor sit amet."# + )) + } + + #[test] + fn the_default_page_orientation_is_not_a_section_change() { + let portrait = page_body(r#" w:orient="portrait""#); + let edited = replaced_copy(&portrait, "3, lorem", "3, LOREM"); + assert!(!edited.contains("w:orient"), "{edited}"); + let (kinds, _) = compared_kinds(&portrait, &edited); + assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]); + + let (kinds, _) = compared_kinds(&portrait, &page_body("")); + assert_eq!(kinds, []); + let (kinds, _) = compared_kinds(&page_body(""), &portrait); + assert_eq!(kinds, []); + + let (kinds, tracked) = compared_kinds(&portrait, &page_body(r#" w:orient="landscape""#)); + assert_eq!(kinds, [RevisionKind::SectionPropertyChange]); + assert!(tracked.contains(r#"w:orient="landscape""#), "{tracked}"); + + // A bookmark makes the section-break paragraph take the complex path, + // as Google Docs writes around headings. + let bookmarked_break = |orientation: &str, heading: &str| { + wrap_word_body(&format!( + r#"{heading}Second section."# + )) + }; + let portrait = r#" w:orient="portrait""#; + for (original, edited) in [(portrait, ""), ("", portrait)] { + let (kinds, _) = compared_kinds( + &bookmarked_break(original, "Heading"), + &bookmarked_break(edited, "Heading"), + ); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + let (kinds, _) = compared_kinds( + &bookmarked_break(original, "Heading"), + &bookmarked_break(edited, "Title"), + ); + assert_eq!( + kinds, + [RevisionKind::Deletion, RevisionKind::Insertion], + "{original:?} -> {edited:?}" + ); + } + } } #[test] From a607736f5a7d8b2dc3789ab5893b9ccf2e4da104 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:22:23 +0200 Subject: [PATCH 04/10] Compare a field packed in one run like the same field split into runs compare() failed with "complex field source has no end boundary" on a file against its own copy once update_page_fields had refreshed a field packed in one run, as Google Docs writes footer fields. The refreshed result lands inside that run and changes w:dirty, so compare_field_xml extracts the result on both sides, and complex_field_result looked for the end character only after the end of the run holding separate. In a packed run the end sits inside that same run. complex_field_result now splits the run so that separate ends a run and end starts one before it reads the result. The new runs repeat the original start tag and run properties, and a run with nothing on the far side of the character is left alone, so a field whose separate and end each have a run of their own yields the same bytes as before. The postconditions read a field as its owner only, so the split does not reach them. The same split fixes a quieter loss. update_page_fields writes the result of an uncached one-run-per-part field into the run of end, and the redline then carried an empty insertion without the new page number. It now inserts the result. GitHub issue #160. --- crates/rdocx/src/comparison.rs | 96 ++++++++++++++++++++++++--- crates/rdocx/tests/regression_test.rs | 96 +++++++++++++++++++++++++++ 2 files changed, 182 insertions(+), 10 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 37d28126..7f7ff9d1 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -5141,17 +5141,20 @@ fn deleted_text_xml(xml: &str) -> String { } fn complex_field_result(xml: &str) -> Result<(String, String, String)> { - let separate = xml - .find("fldCharType=\"separate\"") - .or_else(|| xml.find("fldCharType='separate'")) + let separate = field_character(xml, 0, "separate") .ok_or_else(|| Error::Other("complex field source has no separate boundary".to_owned()))?; - let (_, result_start) = containing_run(xml, separate)?; - let end_marker = xml[result_start..] - .find("fldCharType=\"end\"") - .or_else(|| xml[result_start..].find("fldCharType='end'")) - .map(|offset| result_start + offset) + // A producer may pack a whole field in one run, as Google Docs writes page + // fields. Ending a run after `separate` and starting one at `end` reads it + // as the same field written one run per part. + let xml = split_field_run(xml, separate, true)?; + let (_, result_start) = containing_run(&xml, separate)?; + let end_marker = field_character(&xml, result_start, "end") .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?; - let (result_end, _) = containing_run(xml, end_marker)?; + let xml = split_field_run(&xml, end_marker, false)?; + // The split may have moved the `end` character further along. + let end_marker = field_character(&xml, result_start, "end") + .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?; + let (result_end, _) = containing_run(&xml, end_marker)?; Ok(( xml[..result_start].to_owned(), xml[result_start..result_end].to_owned(), @@ -5159,6 +5162,48 @@ fn complex_field_result(xml: &str) -> Result<(String, String, String)> { )) } +fn field_character(xml: &str, from: usize, kind: &str) -> Option { + let tail = &xml[from..]; + tail.find(&format!("fldCharType=\"{kind}\"")) + .or_else(|| tail.find(&format!("fldCharType='{kind}'"))) + .map(|offset| from + offset) +} + +/// Split the run holding the field character at `marker` so that the +/// character ends its run (`after`) or starts it. +/// +/// Both runs repeat the original start tag and run properties. A run with no +/// content on that side of the character is returned unchanged. +fn split_field_run(xml: &str, marker: usize, after: bool) -> Result { + let (run_start, run_end) = containing_run(xml, marker)?; + let run = &xml[run_start..run_end]; + let children = direct_element_spans(run)?; + let character = children + .iter() + .position(|child| child.contains(&(marker - run_start))) + .ok_or_else(|| Error::Other("complex field character is not a run child".to_owned()))?; + let is_properties = |child: &Range| { + let mut reader = Reader::from_reader(run[child.clone()].as_bytes()); + matches!( + reader.read_event(), + Ok(Event::Start(element) | Event::Empty(element)) + if element.local_name().as_ref() == b"rPr" + ) + }; + let first_content = usize::from(children.first().is_some_and(is_properties)); + let split = if after { character + 1 } else { character }; + if split <= first_content || split >= children.len() { + return Ok(xml.to_owned()); + } + let head = &run[..children[first_content].start]; + let close = run + .rfind(" Result<(usize, usize)> { let mut reader = Reader::from_reader(xml.as_bytes()); reader.config_mut().trim_text(false); @@ -6360,7 +6405,8 @@ fn utf8_error(error: impl std::fmt::Display) -> Error { mod tests { use super::{ ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING, - attributed_run_units, comparison_postcondition_error, story_document, word_fragments, + attributed_run_units, comparison_postcondition_error, complex_field_result, story_document, + word_fragments, }; use crate::Document; use rdocx_oxml::document::BodyContent; @@ -6452,6 +6498,36 @@ mod tests { ); } + #[test] + fn a_packed_field_result_is_read_as_one_run_per_part() { + for w in ["w", "q"] { + let shell = format!(r#"<{w}:r {w}:rsidR="00AB12CD"><{w}:rPr><{w}:b/>"#); + let close = format!(""); + let character = |kind: &str| format!(r#"<{w}:fldChar {w}:fldCharType="{kind}"/>"#); + let (begin, separate, end) = + (character("begin"), character("separate"), character("end")); + let code = format!("<{w}:instrText>PAGE"); + let result = format!("<{w}:t>1"); + let expected = ( + format!("{shell}{begin}{code}{separate}{close}"), + format!("{shell}{result}{close}"), + format!("{shell}{end}{close}"), + ); + for field in [ + format!("{shell}{begin}{code}{separate}{result}{end}{close}"), + format!("{shell}{begin}{code}{separate}{close}{shell}{result}{end}{close}"), + format!("{}{}{}", expected.0, expected.1, expected.2), + ] { + assert_eq!(complex_field_result(&field).unwrap(), expected, "{field}"); + } + let uncached = format!("{shell}{begin}{code}{separate}{end}{close}"); + assert_eq!( + complex_field_result(&uncached).unwrap(), + (expected.0, String::new(), expected.2) + ); + } + } + #[test] fn staged_comparison_postcondition_failure_preserves_bytes_and_layout_cache() { let mut original = Document::new(); diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 08f17b7a..db90661d 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30715,6 +30715,102 @@ mod compare_producer_noise { ); } } + + fn document_with_footer(footer_paragraph: &str) -> Document { + let mut seed = Document::new(); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + package.set_part( + "/word/footer1.xml", + format!(r#"{footer_paragraph}"#).into_bytes(), + ); + package.content_types.add_override( + "/word/footer1.xml", + "application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml", + ); + let footer_id = package + .get_or_create_part_rels("/word/document.xml") + .add(oxml_opc::relationship::rel_types::FOOTER, "footer1.xml"); + package.set_part( + "/word/document.xml", + format!( + r#"Lorem ipsum dolor sit amet."# + ) + .into_bytes(), + ); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + fn page_field(packed: bool, cached: bool) -> String { + let parts = [ + r#""#, + r#" PAGE "#, + r#""#, + if cached { "1" } else { "" }, + r#""#, + ]; + let field = if packed { + format!("{}", parts.concat()) + } else { + parts + .iter() + .filter(|part| !part.is_empty()) + .map(|part| format!("{part}")) + .collect() + }; + format!(r#"Page {field}"#) + } + + /// Compare a footer field against its copy refreshed by `update_page_fields`. + fn refreshed_field_comparison(packed: bool, cached: bool) -> String { + let source = document_with_footer(&page_field(packed, cached)) + .to_bytes() + .unwrap(); + let mut refreshed = Document::from_bytes(&source).unwrap(); + refreshed.update_page_fields().unwrap(); + let refreshed = Document::from_bytes(&refreshed.to_bytes().unwrap()).unwrap(); + let mut compared = Document::from_bytes(&source).unwrap(); + let diagnostics = compared + .compare(&refreshed, "R", TIMESTAMP) + .unwrap_or_else(|error| panic!("packed={packed} cached={cached}: {error}")); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_resolutions( + &compared.to_bytes().unwrap(), + &Document::from_bytes(&source).unwrap(), + &refreshed, + &ComparisonOptions::default(), + ); + comparison_part_xml(&mut compared, "/word/footer1.xml") + } + + #[test] + fn a_packed_field_compares_like_the_same_field_split_into_runs() { + let separate = r#""#; + let result_and_end = |footer: &str| footer[footer.find(separate).unwrap()..].to_owned(); + for (cached, deleted) in [(true, "1"), (false, "")] { + let split = refreshed_field_comparison(false, cached); + let packed = refreshed_field_comparison(true, cached); + assert_eq!( + result_and_end(&split), + format!( + r#"{separate}{deleted}1"# + ), + "cached={cached}" + ); + assert_eq!( + result_and_end(&packed), + result_and_end(&split), + "cached={cached}" + ); + assert!( + packed.contains(r#" PAGE "#), + "{packed}" + ); + } + } } #[test] From 83b92bbf69942de8d7c3aacdeee19cc0da59ff26 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:09:51 +0200 Subject: [PATCH 05/10] Re-record the archive measurements of rdocx and rdocx-oxml The comparison fixes, the kept xml:space flag and their tests grow the rdocx and rdocx-oxml packages, so the crates.io archive rows in README.md and crates/rdocx-oxml/README.md and their ARCHIVE_MEASUREMENTS entries are re-measured. GitHub issues #159 and #160. --- README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 4 ++-- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 479b462e..412db0df 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,097,973 compressed bytes, 6,523,992 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef241..dabac199 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 367,851 compressed bytes, 2,381,304 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643b..25c233a4 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -383,12 +383,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_097_973, 6_523_992, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (367_851, 2_381_304, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6), From 3b9d4ccd676a166e09b6e43273446425a19ca6b5 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:34:51 +0200 Subject: [PATCH 06/10] Keep every child of a text box that replacement rewrites Replacement in text boxes walks the raw XML of the document part and of each header and footer, parses the w:p children of every w:txbxContent and writes those paragraphs back. Every other child was skipped on read and never written. The document layer stores the rewritten part as soon as one text box in it has a hit, so a single hit deleted the tables, block content controls, bookmarks and empty paragraphs of every text box in that part. try_replace_text, replace_all, replace_regex and render_template all go through this walker, for the DrawingML and the VML copy of a text box alike. The walker now edits each paragraph in place and copies every other child through verbatim, so each text box keeps its content in its order. Only a paragraph that the edit changed is re-serialised. The others are copied as read, so they also keep the start-tag attributes that CT_P does not model, such as w:rsidR and w14:paraId. Two defects of the same loop are fixed with it. The closing tag is the one read from the part instead of a fixed w:txbxContent, which did not match a start tag under another prefix. A part that ends inside a text box is now an error instead of an endless read. GitHub issue #160. --- crates/rdocx-oxml/src/placeholder.rs | 172 +++++++++++++++----------- crates/rdocx/tests/regression_test.rs | 75 +++++++++++ 2 files changed, 174 insertions(+), 73 deletions(-) diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 59780350..5bc56d33 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -311,9 +311,13 @@ pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replaceme /// Replace placeholders in text boxes and shapes within a raw XML part. /// -/// Walks the XML, finds `w:txbxContent` elements at any depth, parses their -/// child `w:p` elements using `CT_P::from_xml`, performs replacement, and -/// re-serializes back. Returns the modified XML and replacement count. +/// Walks the XML, finds `w:txbxContent` elements at any depth outside another +/// text box, parses their child `w:p` elements using `CT_P::from_xml`, +/// performs replacement, and re-serializes the paragraphs it changed. Every +/// other child of the text box, such as a table, a content control, a +/// bookmark or a paragraph without a match, is copied through verbatim in its +/// place. A text box nested inside another one is kept as it is, not edited. +/// Returns the modified XML and replacement count. pub fn replace_in_xml_part( xml: &[u8], placeholder: &str, @@ -331,11 +335,11 @@ pub fn replace_many_in_xml_part( xml: &[u8], replacements: &[(&str, &str)], ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { + rewrite_text_boxes(xml, &mut |paragraph| { replacements .iter() .map(|(placeholder, replacement)| { - replace_in_paragraphs(paragraphs, placeholder, replacement) + replace_in_paragraph(paragraph, placeholder, replacement) }) .sum() }) @@ -350,18 +354,22 @@ pub fn replace_regex_in_xml_part( re: ®ex::Regex, replacement: &str, ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { - replace_regex_in_paragraphs(paragraphs, re, replacement) + rewrite_text_boxes(xml, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) }) } -/// Walk `xml`, handing the paragraphs of each `w:txbxContent` element to -/// `edit`, and re-serialise. Returns the rewritten XML and the summed count. +/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit` +/// and re-serialising it in place when `edit` counts a change. Every other +/// paragraph and child of the text box is copied through verbatim. Returns the +/// rewritten XML and the summed count. fn rewrite_text_boxes( xml: &[u8], - edit: &mut dyn FnMut(&mut Vec) -> usize, + edit: &mut dyn FnMut(&mut CT_P) -> usize, ) -> crate::error::Result<(Vec, usize)> { + use crate::error::OxmlError; use crate::namespace::matches_local_name; + use crate::raw_xml::capture_element; use quick_xml::events::Event; use quick_xml::{Reader, Writer}; @@ -376,50 +384,18 @@ fn rewrite_text_boxes( match reader.read_event_into(&mut buf) { Ok(Event::Eof) => break, Ok(Event::Start(ref e)) if matches_local_name(e.name().as_ref(), b"txbxContent") => { - // We found a txbxContent element. Collect its contents as raw XML, - // parse paragraphs, do replacement, and re-serialize. + // We found a txbxContent element. Parse and edit each paragraph + // and copy every other child through verbatim, in document order. writer.write_event(Event::Start(e.clone()))?; - // Read all events inside txbxContent - let mut depth = 1u32; let mut inner_buf = Vec::new(); - // Collect paragraphs from inside txbxContent - let mut paragraphs: Vec = Vec::new(); - loop { match reader.read_event_into(&mut inner_buf) { Ok(Event::Start(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"p") && depth == 1 { + if matches_local_name(ie.name().as_ref(), b"p") { // Parse this paragraph: collect its XML, then parse via CT_P - let mut para_writer = Writer::new(Vec::new()); - // Write the opening tag - para_writer.write_event(Event::Start(ie.clone()))?; - let mut pdepth = 1u32; - let mut pbuf = Vec::new(); - loop { - match reader.read_event_into(&mut pbuf) { - Ok(Event::Start(ref pe)) => { - pdepth += 1; - para_writer.write_event(Event::Start(pe.clone()))?; - } - Ok(Event::End(ref pe)) => { - pdepth -= 1; - para_writer.write_event(Event::End(pe.clone()))?; - if pdepth == 0 { - break; - } - } - Ok(ref ev) => { - para_writer.write_event(ev.clone())?; - } - Err(e) => return Err(e.into()), - } - pbuf.clear(); - } - let para_xml = para_writer.into_inner(); - - // Parse the paragraph + let para_xml = capture_element(&mut reader, ie)?; let mut para_reader = Reader::from_reader(para_xml.as_slice()); para_reader.config_mut().trim_text(true); let mut prbuf = Vec::new(); @@ -436,42 +412,39 @@ fn rewrite_text_boxes( } prbuf.clear(); } - let para = CT_P::from_xml(&mut para_reader)?; - paragraphs.push(para); + let mut para = CT_P::from_xml(&mut para_reader)?; + let count = edit(&mut para); + if count == 0 { + // Nothing changed, so the paragraph keeps its + // bytes, start-tag attributes included. + writer.get_mut().extend_from_slice(¶_xml); + } else { + para.to_xml(&mut writer)?; + } + total_count += count; } else { - depth += 1; - // Non-paragraph element inside txbxContent; skip it - reader.read_to_end_into(ie.name(), &mut Vec::new())?; - depth -= 1; + // A table, a content control or any other element + // the edit does not reach stays as it was. + let raw = capture_element(&mut reader, ie)?; + writer.get_mut().extend_from_slice(&raw); } } - Ok(Event::End(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"txbxContent") && depth == 1 - { - break; - } - depth -= 1; + // Every child element is consumed whole, so the first end + // tag at this level closes the txbxContent element. + Ok(Event::End(ie)) => { + writer.write_event(Event::End(ie))?; + break; } - Ok(_) => { - // Whitespace/text at top level of txbxContent, skip + Ok(Event::Eof) => { + return Err(OxmlError::MissingElement("w:txbxContent end".to_owned())); } + // Empty elements such as `` or a bookmark, whitespace, + // comments and processing instructions. + Ok(ev) => writer.write_event(ev)?, Err(e) => return Err(e.into()), } inner_buf.clear(); } - - // Hand the collected paragraphs to the caller's edit function - total_count += edit(&mut paragraphs); - - // Re-serialize paragraphs into the writer - for p in ¶graphs { - p.to_xml(&mut writer)?; - } - - // Write closing txbxContent tag - writer.write_event(Event::End(quick_xml::events::BytesEnd::new( - "w:txbxContent", - )))?; } Ok(ev) => { writer.write_event(ev)?; @@ -949,6 +922,59 @@ mod tests { assert!(result_str.contains("Company: Acme")); } + /// A text box whose paragraphs sit among a table, a block content + /// control, a bookmark, an empty paragraph, a comment and whitespace. + /// A second text box without a placeholder follows. The walker rewrites + /// every text box of the part, so that one must come back unchanged too. + /// The paragraphs without a placeholder keep their identity attributes. + const TEXT_BOX_WITH_EVERY_KIND_OF_CHILD: &str = r#" +Title +Hello {{name}} +cell +control{{name}} again +other cellNo placeholder here"#; + + #[test] + fn replace_in_textbox_keeps_every_other_child_in_order() { + let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD; + let expected = xml.replace("{{name}}", "Alice"); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + let (result, count) = replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + } + + /// The end tag used to be written as `w:txbxContent` whatever prefix the + /// start tag carried, which left the part ill-formed. + #[test] + fn replace_in_textbox_closes_it_with_its_own_prefix() { + let xml = r#"Hello {{name}}"#; + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 1); + assert_eq!( + String::from_utf8(result).unwrap(), + xml.replace("{{name}}", "Alice") + ); + } + + /// A text box cut short used to keep the walker reading past the end of + /// the part forever. + #[test] + fn replace_in_unterminated_textbox_is_an_error() { + for tail in ["", "Hello {{name}}"] { + let xml = format!( + r#"{tail}"# + ); + assert!(replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").is_err()); + } + } + #[test] fn replace_in_xml_part_no_textbox() { let xml = br#" diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index db90661d..a1b129fc 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -16180,6 +16180,81 @@ fn zero_width_regex_match_terminates() { assert!(count <= 4, "should not loop indefinitely, got {count}"); } +/// Replacement in a text box parses the paragraphs of its `w:txbxContent` and +/// used to write back nothing else, so a hit there deleted the tables, +/// content controls, bookmarks and empty paragraphs of that text box. +mod text_box_replacement_keeps_every_child { + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + /// One copy of the text box: the paragraph the replacement edits, then a + /// table, a block content control and a bookmark around an empty + /// paragraph. Each copy needs its own bookmark id for the file to open. + fn text_box_content(text: &str, id: u32) -> String { + format!( + r#"{text}cellcontrol"# + ) + } + + /// A text box as Word saves it, the DrawingML shape in `mc:Choice` and its + /// VML copy in `mc:Fallback`, or the bare `wp:anchor` alone. + fn text_box_document(compatibility_block: bool) -> Document { + let content = text_box_content("Dear {{name}}", 7); + let drawing = format!( + r#"00{content}"# + ); + let shape = if compatibility_block { + format!( + r#"{drawing}{}"#, + text_box_content("Dear {{name}}", 8) + ) + } else { + drawing + }; + super::document_with_content_controls(&format!( + r#"Host paragraph{shape}"# + )) + } + + /// Replace in both text box forms, then check that every copy holds the + /// edited paragraph and every other child byte for byte and in order. + fn assert_replacement_keeps_every_child(replace: impl Fn(&mut Document) -> usize) { + for (compatibility_block, ids) in [(true, &[7, 8][..]), (false, &[7][..])] { + let mut document = text_box_document(compatibility_block); + assert_eq!(replace(&mut document), ids.len(), "{compatibility_block}"); + let saved = super::document_xml(&mut document); + assert_eq!(saved.matches("").count(), ids.len()); + for &id in ids { + let expected = text_box_content("Dear Ada", id); + assert!(saved.contains(&expected), "{expected}\n{saved}"); + } + } + } + + #[test] + fn replacing_text_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.try_replace_text("{{name}}", "Ada").unwrap() + }); + } + + #[test] + fn regex_replacement_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.replace_regex(r"\{\{(\w+)\}\}", "Ada").unwrap() + }); + } + + #[test] + fn a_template_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document + .render_template(&serde_json::json!({"name": "Ada"})) + .unwrap() + }); + } +} + /// `9360 / cols` panicked when a caller asked for a zero-column table. #[test] fn zero_column_tables_do_not_panic() { From fd67fb8320c605a81e23cb04d4feffed6b229e17 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:42:19 +0200 Subject: [PATCH 07/10] Keep paragraph anchors in place when replacement removes a run A replacement that spans runs puts its text in the first run and removes the runs it leaves empty. It removed them with a plain retain over the direct runs and then only clamped the hyperlink spans. The bookmarks, comment ranges, inline content controls, raw children, revisions, equations and ruby spans of a paragraph are kept by run index, so each one after a removed run slid one run later, past the text it preceded, and a hyperlink could lose a run or gain the next one. The retain also removed every other run without typed content in the paragraph: an empty run, and a run holding only raw XML such as a drawing Word writes in mc:AlternateContent, which a single match anywhere in the paragraph deleted. The removal now goes through the remapping that comment removal already used, moved into a shared CT_P::remove_runs, so every anchor stays on the boundary that remains and a hyperlink left without runs is dropped. That remapping left the ruby spans out, so comment removal slid them too. It now moves them, and drops an annotation left without base runs, which the writer skips anyway. Only a run the replacement emptied is removed. A run keeps its start-tag attributes as a raw record, which does not count as content. GitHub issue #160. --- crates/rdocx-oxml/src/placeholder.rs | 120 ++++++++++++++++++++------- crates/rdocx-oxml/src/text.rs | 19 ++++- 2 files changed, 108 insertions(+), 31 deletions(-) diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 5bc56d33..eed2c7f4 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -17,6 +17,7 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st } let mut total = 0; + let mut emptied = Vec::new(); // Byte offset in the concatenated paragraph text at which to look for the // next match. Resuming *after* the text we just inserted is what keeps this @@ -72,14 +73,9 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st last_run, replacement, ); + emptied.extend(first_run + 1..=last_run); } - // Remove runs that became completely empty (no content at all). - para.runs.retain(|r| !r.content.is_empty()); - - // Update hyperlink spans to account for removed runs. - reindex_hyperlinks(para); - // Text before `byte_start` is untouched and the replacement now // occupies `byte_start..byte_start + replacement.len()`, so this stays // on a char boundary of the rebuilt text. @@ -88,9 +84,32 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st total += 1; } + remove_emptied_runs(para, emptied); total } +/// Remove the runs at `candidates` that the replacement left without any +/// content, keeping every anchor of the paragraph on the boundary that +/// remains. A run that was empty before is kept. +fn remove_emptied_runs(para: &mut CT_P, candidates: Vec) { + let mut removed = vec![false; para.runs.len()]; + for index in candidates { + // A run keeps the attributes of its start tag, such as `w:rsidR`, as + // a raw record, which does not make it worth keeping once empty. + let run = ¶.runs[index]; + removed[index] = run.content.is_empty() + && run + .extra_xml_positions + .iter() + .filter(|position| CT_R::raw_child_is_root_attributes(**position)) + .count() + == run.extra_xml.len(); + } + if removed.contains(&true) { + para.remove_runs(&removed); + } +} + /// A mapping from character position in the concatenated text to its source run and content item. #[derive(Debug)] struct CharMapping { @@ -211,27 +230,6 @@ fn replace_across_runs( } } -/// Re-index hyperlink spans after runs may have been removed. -fn reindex_hyperlinks(para: &mut CT_P) { - // After retain, run indices may have shifted. We rebuild by checking - // which runs still exist. Since retain preserves order and only removes - // empty runs, the relative order is maintained. However, hyperlink spans - // referenced by index need adjustment. - // - // For simplicity: we decrement indices for each removed slot. - // But since we already called retain, the runs are already compacted. - // We need to adjust hyperlinks based on the new run count. - // - // The simplest correct approach: hyperlinks that referenced removed runs - // get their range clamped/invalidated. - para.hyperlinks.retain(|hl| hl.run_start < para.runs.len()); - for hl in &mut para.hyperlinks { - if hl.run_end > para.runs.len() { - hl.run_end = para.runs.len(); - } - } -} - /// Replace all occurrences of `placeholder` in all paragraphs of a slice. pub fn replace_in_paragraphs(paras: &mut [CT_P], placeholder: &str, replacement: &str) -> usize { paras @@ -532,6 +530,7 @@ pub fn replace_many_in_chart_xml( /// Returns the number of replacements made. pub fn replace_regex_in_paragraph(para: &mut CT_P, re: ®ex::Regex, replacement: &str) -> usize { let mut total = 0; + let mut emptied = Vec::new(); // See `replace_in_paragraph`: resume after the inserted text so a // replacement that itself matches the pattern cannot loop forever. @@ -598,14 +597,14 @@ pub fn replace_regex_in_paragraph(para: &mut CT_P, re: ®ex::Regex, replacemen last_run, &expanded_replacement, ); + emptied.extend(first_run + 1..=last_run); } - para.runs.retain(|r| !r.content.is_empty()); - reindex_hyperlinks(para); search_from = byte_start + expanded_replacement.len(); total += 1; } + remove_emptied_runs(para, emptied); total } @@ -705,6 +704,7 @@ pub fn replace_regex_in_header_footer( #[cfg(test)] mod tests { use super::*; + use crate::namespace::{MC_NS, W_NS}; use crate::properties::CT_RPr; fn make_para(texts: &[&str]) -> CT_P { @@ -800,6 +800,68 @@ mod tests { ); } + fn paragraph_xml(paragraph: &CT_P) -> String { + let mut writer = quick_xml::Writer::new(Vec::new()); + paragraph.to_xml(&mut writer).unwrap(); + String::from_utf8(writer.into_inner()).unwrap() + } + + /// A match across runs removes the runs it empties. The anchors after + /// them are kept by run index, and used to slide one run later, past the + /// text they preceded. A ruby annotation slid onto the run after its base. + #[test] + fn removing_emptied_runs_keeps_later_anchors_in_place() { + let source = format!( + r#"{{{{name}}}}controlannBASEx"# + ); + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + for regex in [false, true] { + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + let count = if regex { + replace_regex_in_paragraph(&mut p, &re, "Bob") + } else { + replace_in_paragraph(&mut p, "{{name}}", "Bob") + }; + assert_eq!(count, 1); + assert_eq!(p.runs.len(), 3); + let xml = paragraph_xml(&p); + let positions = [ + ">Bob<", + "bookmarkStart", + "commentRangeStart", + "", + "proofErr", + "", + ">BASE<", + "", + ">x<", + "bookmarkEnd", + "commentRangeEnd", + ] + .map(|marker| { + xml.find(marker) + .unwrap_or_else(|| panic!("{marker}: {xml}")) + }); + assert!(positions.is_sorted(), "{xml}"); + } + } + + /// A run whose only child is raw XML, such as a drawing Word writes in + /// `mc:AlternateContent`, has no typed content, and neither has an + /// empty run. A match anywhere in the paragraph used to remove both. + #[test] + fn replace_keeps_the_runs_it_did_not_empty() { + let source = format!( + r#"Hello {{{{name}}}}"# + ); + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + + assert_eq!(replace_in_paragraph(&mut p, "{{name}}", "Ada"), 1); + + assert_eq!(p.runs.len(), 3); + assert!(paragraph_xml(&p).contains("")); + } + #[test] fn replace_multiple_occurrences() { let mut p = make_para(&["{{x}} and {{x}}"]); diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs index cd5501ca..c4f016c4 100644 --- a/crates/rdocx-oxml/src/text.rs +++ b/crates/rdocx-oxml/src/text.rs @@ -4320,10 +4320,19 @@ impl CT_P { if removed.iter().all(|remove| !remove) { return; } + self.remove_runs(&removed); + } + + /// Remove the direct runs flagged in `removed` and move every run-boundary + /// projection onto the boundary that remains. Raw children, comment + /// markers, bookmarks and controls of the boundaries that collapse into + /// one keep their order, and a hyperlink or a ruby annotation left + /// without runs is dropped. + pub(crate) fn remove_runs(&mut self, removed: &[bool]) { let removed_run_addresses = self .runs .iter() - .zip(&removed) + .zip(removed) .filter_map(|(run, remove)| remove.then_some(std::ptr::from_ref(run))) .collect::>(); let removed_projected_indices = accepted_paragraph_runs(self) @@ -4470,6 +4479,12 @@ impl CT_P { *position = boundary_map[old_boundary]; *raw_before = raw_prefixes[old_boundary] + (*raw_before).min(raw_counts[old_boundary]); } + for ruby in &mut self.rubies { + ruby.base_start = boundary_map[ruby.base_start.min(old_run_count)]; + ruby.base_end = boundary_map[ruby.base_end.min(old_run_count)]; + } + // An annotation over no base run is not written, so drop it. + self.rubies.retain(|ruby| ruby.base_start < ruby.base_end); let old_hyperlinks = std::mem::take(&mut self.hyperlinks); let mut hyperlink_map = vec![None; old_hyperlinks.len()]; for (old_index, mut hyperlink) in old_hyperlinks.into_iter().enumerate() { @@ -4517,7 +4532,7 @@ impl CT_P { .runs .drain(..) .zip(removed) - .filter_map(|(run, remove)| (!remove).then_some(run)) + .filter_map(|(run, remove)| (!*remove).then_some(run)) .collect(); let _ = self.refresh_bookmark_projection(); } From 3b6b89c3b587408f4c5b4b11f6e94559c6186baa Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:45:45 +0200 Subject: [PATCH 08/10] Replace text inside content controls at every level try_replace_text, replace_all, the regex replacement, render_template and rdocx replace skipped the text of content controls. The body walkers matched a body-level control to nothing, so a paragraph that a Google Docs export wraps at body level was never read. The paragraph replacement read only the direct runs, so a run wrapped in an inline control was missed at every level, in the body, in table cells, in text boxes and in header and footer paragraphs. Paragraph.text reads those runs since #118, so the read and replace views disagreed on the same run. The text box walker copied its tables and block controls without reading them. The body walkers now go through the visitor that already reaches tables, cells and controls at every level. The paragraph replacement reads its direct runs and the runs of its inline controls, nested ones included, in the order CT_P::runs reads them. A match must lie within one stretch of runs: the direct runs between two controls, or the runs of one control between its nested ones. A match that straddles a control boundary is left as it is, never half replaced, and the search goes on after it, so a shorter match that a pattern such as \d+ finds inside it cannot replace part of it either. The direct runs on both sides of a control no longer read as one text, which matched words a reader does not see. A run the replacement empties inside a control is removed with its content index bookkeeping. The literal and the regex replacement share one table, row, cell and control recursion instead of two copies of it. The text box walker edits its tables and block controls through the typed parsers and writes back only those it changed. The typed writers leave out namespace declarations, so a rewritten element declares those of its start tag again. One that declares a namespace deeper down, other than the part binds it, keeps its bytes and counts nothing, since its rewrite would leave the prefix unbound or bound to another namespace. render_template reads its tags in the same stretches, so a tag inside an inline control, or in a table or control of a text box, is rendered, where it was left in the output before. A tag that straddles a control boundary is reported as an invalid template. GitHub issue #160. --- crates/rdocx-cli/tests/integration.rs | 106 +++ crates/rdocx-oxml/src/content_control.rs | 21 + crates/rdocx-oxml/src/placeholder.rs | 963 ++++++++++++++++------- crates/rdocx/src/document.rs | 56 +- crates/rdocx/src/template.rs | 88 +-- crates/rdocx/tests/regression_test.rs | 405 ++++++++++ 6 files changed, 1246 insertions(+), 393 deletions(-) diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs index e4cdd98f..046adf68 100644 --- a/crates/rdocx-cli/tests/integration.rs +++ b/crates/rdocx-cli/tests/integration.rs @@ -343,6 +343,112 @@ fn cli_replace_reports_namespace_preflight_errors_without_panicking() { assert!(!output_path.exists()); } +/// `rdocx replace --expect 1` found none of the text that Google Docs and +/// Word keep in content controls: a run wrapped inside its paragraph, a +/// paragraph wrapped at body level, a control in a table cell, nested +/// controls and a control in a text box. +#[test] +fn replace_with_expect_counts_the_text_of_content_controls_everywhere() { + let temp = TempWorkspace::new("replace-content-controls"); + let input = temp.path.join("controls.docx"); + write_document(&input, &["seed"]); + + let control = |tag: &str, content: &str| { + format!( + r#"{content}"# + ) + }; + let run = |text: &str| format!("{text}"); + let paragraph = |content: &str| format!("{content}"); + let table = |cell: &str| { + format!( + r#"{cell}"# + ) + }; + let text_box = format!( + r#"{}"#, + control("box", ¶graph(&run("{{box}}"))) + ); + let body = [ + paragraph(&[run("Body "), control("goog_rdk_0", &run("{{inline}}"))].concat()), + control("goog_rdk_1", ¶graph(&run("{{block}}"))), + table(&control("cell", ¶graph(&run("{{cell}}")))), + control("outer", ¶graph(&control("inner", &run("{{nested}}")))), + paragraph(&[run("Host"), text_box].concat()), + ] + .concat(); + let word = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"; + let header = format!( + r#"{}{}"#, + table(¶graph(&run("{{header_table}}"))), + control("header", ¶graph(&run("{{header_control}}"))) + ); + let footer = format!( + r#"{}{}"#, + control("page", ¶graph(&run("{{footer_control}}"))), + paragraph(&run("Confidential")) + ); + + let mut package = + OpcPackage::from_reader(std::io::Cursor::new(fs::read(&input).unwrap())).unwrap(); + let mut references = String::new(); + for (kind, xml, rel_type) in [ + ("header", header, rel_types::HEADER), + ("footer", footer, rel_types::FOOTER), + ] { + let part = format!("/word/{kind}1.xml"); + package.set_part(&part, xml.into_bytes()); + package.content_types.add_override( + &part, + &format!("application/vnd.openxmlformats-officedocument.wordprocessingml.{kind}+xml"), + ); + let id = package + .get_or_create_part_rels("/word/document.xml") + .add(rel_type, &format!("{kind}1.xml")); + references.push_str(&format!( + r#""# + )); + } + package.set_part( + "/word/document.xml", + format!( + r#"{body}{references}"# + ) + .into_bytes(), + ); + package + .write_to(&mut fs::File::create(&input).unwrap()) + .unwrap(); + + for name in ["inline", "block", "cell", "nested", "box"] { + let tag = format!("{{{{{name}}}}}"); + let replaced = temp.path.join(format!("{name}.docx")); + let output = cli(&[ + "replace", + path_text(&input), + "--placeholder", + &tag, + "--value", + "done", + "--expect", + "1", + "--output", + path_text(&replaced), + ]); + assert_success(&output, name); + + let package = OpcPackage::open(&replaced).unwrap(); + let saved = ["document", "header1", "footer1"] + .map(|part| { + let xml = package.get_part(&format!("/word/{part}.xml")).unwrap(); + String::from_utf8(xml.to_vec()).unwrap() + }) + .concat(); + assert!(!saved.contains(&tag), "{name}: {saved}"); + assert_eq!(saved.matches(">done<").count(), 1, "{name}: {saved}"); + } +} + #[test] fn validate_exit_status_is_a_verdict() { let temp = TempWorkspace::new("validate"); diff --git a/crates/rdocx-oxml/src/content_control.rs b/crates/rdocx-oxml/src/content_control.rs index 8ff31cd5..711f3e5e 100644 --- a/crates/rdocx-oxml/src/content_control.rs +++ b/crates/rdocx-oxml/src/content_control.rs @@ -806,6 +806,27 @@ impl CT_Sdt { } } + /// Remove one content child, keeping the revisions and the source bytes + /// of the later runs at their content index. + pub(crate) fn remove_content(&mut self, index: usize) { + if index >= self.content.len() { + return; + } + self.content.remove(index); + for (boundary, _) in &mut self.revisions { + if *boundary > index { + *boundary -= 1; + } + } + self.inline_run_sources + .retain(|source| source.content_index != index); + for source in &mut self.inline_run_sources { + if source.content_index > index { + source.content_index -= 1; + } + } + } + pub(crate) fn word_prefixes(&self) -> &[String] { &self.word_prefixes } diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index eed2c7f4..5052e48a 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -2,22 +2,73 @@ //! //! Handles the cross-run splitting problem: a placeholder like `{{name}}` //! may be split across multiple `` elements in the OOXML source. - +//! +//! A replacement reads the direct runs of a paragraph and the runs of its +//! inline content controls, in document order, and searches their text left +//! to right as one. A match must lie within one stretch of those runs, see +//! [`replaceable_texts`]. A match that straddles a content-control boundary +//! is not replaced, and the search goes on after it, so no part of the text +//! it covers is replaced either. + +use crate::content_control::{CT_Sdt, SdtContent}; use crate::header_footer::CT_HdrFtr; -use crate::table::CT_Tbl; -use crate::text::{CT_P, CT_R, RunContent}; +use crate::namespace::W_NS; +use crate::numbering::{local_namespace_overrides, namespace_bindings, word_prefixes_at}; +use crate::properties::is_word_element; +use crate::table::{CT_Row, CT_Tbl, CT_Tc, CellContent}; +use crate::text::{ + AcceptedRunPath, AcceptedRunPathSegment, CT_P, CT_R, RunContent, raw_with_external_bindings, +}; /// Replace all occurrences of `placeholder` with `replacement` in a paragraph. /// -/// Handles placeholders split across multiple runs. Preserves the formatting -/// of the first matched run. Returns the number of replacements made. +/// Handles placeholders split across multiple runs and reaches the runs of +/// inline content controls. A match that straddles a content-control +/// boundary is not replaced. Preserves the formatting of the first matched +/// run. Returns the number of replacements made. pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &str) -> usize { if placeholder.is_empty() { return 0; } + replace_matches(para, &mut |text, from| { + let start = from + text[from..].find(placeholder)?; + Some((start, start + placeholder.len(), replacement.to_owned())) + }) +} - let mut total = 0; +/// The texts a replacement in `para` matches against, one per stretch of +/// runs. The direct runs between two inline content controls form one +/// stretch, and so do the runs of one control between its nested controls. +/// A match never spans two stretches. +#[doc(hidden)] +pub fn replaceable_texts(para: &CT_P) -> Vec { + let mut texts: Vec = Vec::new(); + let mut previous = None; + for (stretch, path) in text_runs(para) { + if previous != Some(stretch) { + texts.push(String::new()); + previous = Some(stretch); + } + if let (Some(text), Some(run)) = (texts.last_mut(), para.accepted_run(&path)) { + text.extend(run.content.iter().filter_map(|content| match content { + RunContent::Text(t) => Some(t.text.as_str()), + _ => None, + })); + } + } + texts +} + +/// Takes a text and the byte offset to search from, and returns the byte +/// range of the next match and the text that replaces it. +type MatchFinder<'a> = dyn FnMut(&str, usize) -> Option<(usize, usize, String)> + 'a; + +/// Find each match that `next_match` reports in the text of `para` and +/// replace it. +fn replace_matches(para: &mut CT_P, next_match: &mut MatchFinder<'_>) -> usize { + let runs = text_runs(para); let mut emptied = Vec::new(); + let mut total = 0; // Byte offset in the concatenated paragraph text at which to look for the // next match. Resuming *after* the text we just inserted is what keeps this @@ -26,53 +77,54 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st // replacement forever. let mut search_from = 0usize; - // We loop because after one replacement the text layout changes - // and there may be more matches. loop { // 1. Concatenate all run text and build a char map. - let (full_text, char_map) = build_char_map(¶.runs); - - if search_from >= full_text.len() { + let (full_text, char_map) = build_char_map(para, &runs); + if search_from > full_text.len() { break; } - // 2. Find the next match (byte offset in full_text). - let Some(byte_start) = full_text[search_from..] - .find(placeholder) - .map(|i| search_from + i) - else { + // 2. Find the next match (byte offsets in full_text). + let Some((byte_start, byte_end, replacement)) = next_match(&full_text, search_from) else { break; }; + let next_char = full_text[byte_start..] + .chars() + .next() + .map_or(1, char::len_utf8); + + // A zero-width match would neither consume input nor advance the + // cursor, so step past it and the loop always makes progress. + if byte_start == byte_end { + search_from = byte_start + next_char; + continue; + } + // Convert byte offsets to char indices (char_map is indexed by char position). let match_start = full_text[..byte_start].chars().count(); - let match_end = match_start + placeholder.chars().count(); + let match_end = match_start + full_text[byte_start..byte_end].chars().count(); // 3. Determine which runs are affected. - let first_char = &char_map[match_start]; - let last_char = &char_map[match_end - 1]; - let first_run = first_char.run_index; - let last_run = last_char.run_index; + let first_run = char_map[match_start].run_index; + let last_run = char_map[match_end - 1].run_index; + + if runs[first_run].0 != runs[last_run].0 { + // The match straddles a content-control boundary, so it is no + // match. Look again after it, as after a replaced match, so that + // no shorter match inside it, such as `\d+` finds in the digits + // after a boundary, replaces part of the text it covers. + search_from = byte_end; + continue; + } if first_run == last_run { // Single-run match: simple in-place replacement on that run's text content. - replace_in_single_run( - &mut para.runs[first_run], - &char_map, - match_start, - match_end, - replacement, - ); + edit_run(para, &runs[first_run].1, |run| { + replace_in_single_run(run, &char_map, match_start, match_end, &replacement); + }); } else { // Cross-run match: put replacement in first run, clear matched parts from others. - replace_across_runs( - &mut para.runs, - &char_map, - match_start, - match_end, - first_run, - last_run, - replacement, - ); + replace_across_runs(para, &runs, &char_map, match_start, match_end, &replacement); emptied.extend(first_run + 1..=last_run); } @@ -84,36 +136,82 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st total += 1; } - remove_emptied_runs(para, emptied); + remove_emptied_runs(para, &runs, emptied); total } -/// Remove the runs at `candidates` that the replacement left without any -/// content, keeping every anchor of the paragraph on the boundary that -/// remains. A run that was empty before is kept. -fn remove_emptied_runs(para: &mut CT_P, candidates: Vec) { - let mut removed = vec![false; para.runs.len()]; - for index in candidates { - // A run keeps the attributes of its start tag, such as `w:rsidR`, as - // a raw record, which does not make it worth keeping once empty. - let run = ¶.runs[index]; - removed[index] = run.content.is_empty() - && run - .extra_xml_positions - .iter() - .filter(|position| CT_R::raw_child_is_root_attributes(**position)) - .count() - == run.extra_xml.len(); +/// The runs of `para` that a replacement reads, in the order `CT_P::runs` +/// reads them, each with the stretch it belongs to and its address. +fn text_runs(para: &CT_P) -> Vec<(usize, AcceptedRunPath)> { + let mut runs = Vec::new(); + let mut stretch = 0; + for index in 0..=para.runs.len() { + for (control, (_, _, _, sdt)) in para + .content_controls + .iter() + .enumerate() + .filter(|(_, (at, _, _, _))| *at == index) + { + let mut prefix = vec![AcceptedRunPathSegment::ContentControl(control)]; + stretch += 1; + control_text_runs(sdt, &mut prefix, &mut stretch, &mut runs); + stretch += 1; + } + if index < para.runs.len() { + runs.push(( + stretch, + AcceptedRunPath { + segments: vec![AcceptedRunPathSegment::Run(index)], + }, + )); + } } - if removed.contains(&true) { - para.remove_runs(&removed); + runs +} + +fn control_text_runs( + sdt: &CT_Sdt, + prefix: &mut Vec, + stretch: &mut usize, + runs: &mut Vec<(usize, AcceptedRunPath)>, +) { + for (index, content) in sdt.content.iter().enumerate() { + match content { + SdtContent::Run(_) => { + prefix.push(AcceptedRunPathSegment::Run(index)); + runs.push(( + *stretch, + AcceptedRunPath { + segments: prefix.clone(), + }, + )); + prefix.pop(); + } + SdtContent::ContentControl(nested) => { + prefix.push(AcceptedRunPathSegment::ContentControl(index)); + *stretch += 1; + control_text_runs(nested, prefix, stretch, runs); + *stretch += 1; + prefix.pop(); + } + _ => {} + } + } +} + +/// Apply `edit` to the run at `path`. +fn edit_run(para: &mut CT_P, path: &AcceptedRunPath, edit: impl FnOnce(&mut CT_R)) { + if let Some(mut run) = para.accepted_run(path).cloned() { + edit(&mut run); + let replaced = para.replace_accepted_run(path, run); + debug_assert!(matches!(replaced, Ok(true)), "text run path is stale"); } } /// A mapping from character position in the concatenated text to its source run and content item. #[derive(Debug)] struct CharMapping { - /// Index of the run in the paragraph's runs vec. + /// Index of the run in the list of text runs. run_index: usize, /// Index of the RunContent item within the run. content_index: usize, @@ -121,11 +219,14 @@ struct CharMapping { byte_offset: usize, } -fn build_char_map(runs: &[CT_R]) -> (String, Vec) { +fn build_char_map(para: &CT_P, runs: &[(usize, AcceptedRunPath)]) -> (String, Vec) { let mut full_text = String::new(); let mut char_map = Vec::new(); - for (run_idx, run) in runs.iter().enumerate() { + for (run_idx, (_, path)) in runs.iter().enumerate() { + let Some(run) = para.accepted_run(path) else { + continue; + }; for (content_idx, content) in run.content.iter().enumerate() { if let RunContent::Text(t) = content { for (byte_pos, _ch) in t.text.char_indices() { @@ -179,54 +280,107 @@ fn replace_in_single_run( } fn replace_across_runs( - runs: &mut [CT_R], + para: &mut CT_P, + runs: &[(usize, AcceptedRunPath)], char_map: &[CharMapping], match_start: usize, match_end: usize, - first_run: usize, - last_run: usize, replacement: &str, ) { // Handle the first run: replace from match start to end of text in that content item. let first_mapping = &char_map[match_start]; - let first_content_idx = first_mapping.content_index; - let first_byte_offset = first_mapping.byte_offset; - - if let RunContent::Text(t) = &mut runs[first_run].content[first_content_idx] { - let mut new_text = String::new(); - new_text.push_str(&t.text[..first_byte_offset]); - new_text.push_str(replacement); - t.text = new_text; - t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); - } + edit_run(para, &runs[first_mapping.run_index].1, |run| { + if let RunContent::Text(t) = &mut run.content[first_mapping.content_index] { + let mut new_text = String::new(); + new_text.push_str(&t.text[..first_mapping.byte_offset]); + new_text.push_str(replacement); + t.text = new_text; + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); + } + }); // Handle the last run: replace from start to match end within that content item. let last_mapping = &char_map[match_end - 1]; - let last_content_idx = last_mapping.content_index; - let last_byte_offset = last_mapping.byte_offset; - - if let RunContent::Text(t) = &mut runs[last_run].content[last_content_idx] { - let remaining = &t.text[last_byte_offset..]; - let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0); - let byte_end = last_byte_offset + ch_len; - t.text = t.text[byte_end..].to_string(); - t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); - } - - // Clear text content from runs strictly between first and last. - for run in &mut runs[(first_run + 1)..last_run] { - run.content.retain(|c| !matches!(c, RunContent::Text(_))); - } + edit_run(para, &runs[last_mapping.run_index].1, |run| { + if let RunContent::Text(t) = &mut run.content[last_mapping.content_index] { + let remaining = &t.text[last_mapping.byte_offset..]; + let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0); + let byte_end = last_mapping.byte_offset + ch_len; + t.text = t.text[byte_end..].to_string(); + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); + } - // If the last run's text is now empty, remove its text content too. - if last_run != first_run { - runs[last_run].content.retain(|c| { + // If the last run's text is now empty, remove its text content too. + run.content.retain(|c| { if let RunContent::Text(t) = c { !t.text.is_empty() } else { true } }); + }); + + // Clear text content from runs strictly between first and last. + for (_, path) in &runs[first_mapping.run_index + 1..last_mapping.run_index] { + edit_run(para, path, |run| { + run.content.retain(|c| !matches!(c, RunContent::Text(_))); + }); + } +} + +/// Remove the runs at `candidates` (indices into `runs`) that the replacement +/// left without any content, keeping every anchor of the paragraph and of +/// its content controls on the boundary that remains. +fn remove_emptied_runs( + para: &mut CT_P, + runs: &[(usize, AcceptedRunPath)], + mut candidates: Vec, +) { + candidates.sort_unstable(); + candidates.dedup(); + let mut direct = vec![false; para.runs.len()]; + // Later runs first, so that removing one inside a control leaves the + // addresses of the others valid. + for index in candidates.into_iter().rev() { + let path = &runs[index].1; + // A run keeps the attributes of its start tag, such as `w:rsidR`, as + // a raw record, which does not make it worth keeping once empty. + let emptied = para.accepted_run(path).is_some_and(|run| { + run.content.is_empty() + && run + .extra_xml_positions + .iter() + .filter(|position| CT_R::raw_child_is_root_attributes(**position)) + .count() + == run.extra_xml.len() + }); + if !emptied { + continue; + } + match path.segments() { + [AcceptedRunPathSegment::Run(run)] => direct[*run] = true, + [AcceptedRunPathSegment::ContentControl(control), rest @ ..] => { + if let Some((_, _, _, sdt)) = para.content_controls.get_mut(*control) { + remove_control_run(sdt, rest); + } + } + _ => {} + } + } + if direct.contains(&true) { + para.remove_runs(&direct); + } +} + +fn remove_control_run(sdt: &mut CT_Sdt, path: &[AcceptedRunPathSegment]) { + match path { + [AcceptedRunPathSegment::Run(index)] => sdt.remove_content(*index), + [AcceptedRunPathSegment::ContentControl(index), rest @ ..] => { + if let Some(SdtContent::ContentControl(nested)) = sdt.content.get_mut(*index) { + remove_control_run(nested, rest); + } + } + _ => {} } } @@ -238,84 +392,255 @@ pub fn replace_in_paragraphs(paras: &mut [CT_P], placeholder: &str, replacement: .sum() } -/// Replace all occurrences of `placeholder` in a table (recursively handles nested tables). +/// Replace all occurrences of `placeholder` in a table (recursively handles +/// nested tables and content controls). pub fn replace_in_table(table: &mut CT_Tbl, placeholder: &str, replacement: &str) -> usize { + edit_table(table, &mut |paragraph| { + replace_in_paragraph(paragraph, placeholder, replacement) + }) +} + +/// Hand every paragraph of a table to `edit` and sum what it counts. +fn edit_table(table: &mut CT_Tbl, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { let mut count = 0; for (_, _, sdt) in &mut table.content_controls { - count += replace_in_sdt(sdt, placeholder, replacement); + count += edit_control(sdt, edit); } for row in &mut table.rows { - count += replace_in_row(row, placeholder, replacement); + count += edit_row(row, edit); } count } -fn replace_in_sdt( - sdt: &mut crate::content_control::CT_Sdt, - placeholder: &str, - replacement: &str, -) -> usize { - use crate::content_control::SdtContent; - +fn edit_control(sdt: &mut CT_Sdt, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { let mut count = 0; for content in &mut sdt.content { count += match content { - SdtContent::Paragraph(paragraph) => { - replace_in_paragraph(paragraph, placeholder, replacement) - } - SdtContent::Table(table) => replace_in_table(table, placeholder, replacement), - SdtContent::Row(row) => replace_in_row(row, placeholder, replacement), - SdtContent::Cell(cell) => replace_in_cell(cell, placeholder, replacement), - SdtContent::ContentControl(nested) => replace_in_sdt(nested, placeholder, replacement), + SdtContent::Paragraph(paragraph) => edit(paragraph), + SdtContent::Table(table) => edit_table(table, edit), + SdtContent::Row(row) => edit_row(row, edit), + SdtContent::Cell(cell) => edit_cell(cell, edit), + SdtContent::ContentControl(nested) => edit_control(nested, edit), SdtContent::Run(_) | SdtContent::RawXml(_) => 0, }; } count } -fn replace_in_row(row: &mut crate::table::CT_Row, placeholder: &str, replacement: &str) -> usize { - let controls = row - .content_controls - .iter_mut() - .map(|(_, _, sdt)| replace_in_sdt(sdt, placeholder, replacement)) - .sum::(); - controls - + row - .cells - .iter_mut() - .map(|cell| replace_in_cell(cell, placeholder, replacement)) - .sum::() +fn edit_row(row: &mut CT_Row, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { + let mut count = 0; + for (_, _, sdt) in &mut row.content_controls { + count += edit_control(sdt, edit); + } + for cell in &mut row.cells { + count += edit_cell(cell, edit); + } + count } -fn replace_in_cell(cell: &mut crate::table::CT_Tc, placeholder: &str, replacement: &str) -> usize { - use crate::table::CellContent; +fn edit_cell(cell: &mut CT_Tc, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { + let mut count = 0; + for content in &mut cell.content { + count += match content { + CellContent::Paragraph(paragraph) => edit(paragraph), + CellContent::Table(table) => edit_table(table, edit), + CellContent::ContentControl(sdt) => edit_control(sdt, edit), + }; + } + count +} - cell.content - .iter_mut() - .map(|content| match content { - CellContent::Paragraph(paragraph) => { - replace_in_paragraph(paragraph, placeholder, replacement) - } - CellContent::Table(table) => replace_in_table(table, placeholder, replacement), - CellContent::ContentControl(sdt) => replace_in_sdt(sdt, placeholder, replacement), +/// Hand the paragraphs of a table or a block content control kept as raw +/// XML to `edit`, and re-serialise the element in place when the edit +/// counts a change. Any other element, one the typed parsers refuse, or one +/// whose namespaces the rewrite cannot keep, see [`with_source_namespaces`], +/// keeps its bytes and counts nothing. +fn edit_raw_block( + raw: &mut Vec, + word_prefixes: &[String], + edit: &mut dyn FnMut(&mut CT_P) -> usize, +) -> usize { + use quick_xml::events::Event; + use quick_xml::{Reader, Writer}; + + let mut reader = Reader::from_reader(raw.as_slice()); + reader.config_mut().trim_text(false); + let mut buffer = Vec::new(); + let Ok(Event::Start(start)) = reader.read_event_into(&mut buffer) else { + return 0; + }; + let Ok(prefixes) = word_prefixes_at(&start, word_prefixes) else { + return 0; + }; + // The namespaces the start tag declares other than the part does. + let Ok(bindings) = local_namespace_overrides(&start, word_prefixes) else { + return 0; + }; + let mut writer = Writer::new(Vec::new()); + let count = if is_word_element(start.name().as_ref(), b"tbl", &prefixes) { + let Ok(mut table) = + CT_Tbl::from_xml_with_prefixes_and_owner_bindings(&mut reader, &prefixes, &bindings) + else { + return 0; + }; + let count = edit_table(&mut table, edit); + if count == 0 || table.to_xml(&mut writer).is_err() { + return 0; + } + count + } else if is_word_element(start.name().as_ref(), b"sdt", &prefixes) { + let Some(mut control) = CT_Sdt::from_body_raw(raw, word_prefixes) else { + return 0; + }; + let count = edit_control(&mut control, edit); + if count == 0 || control.to_xml(&mut writer).is_err() { + return 0; + } + count + } else { + return 0; + }; + let Some(rewritten) = + with_source_namespaces(raw, &writer.into_inner(), &bindings, word_prefixes) + else { + return 0; + }; + *raw = rewritten; + count +} + +/// Declare the namespaces that the start tag of `source` declares other +/// than the part does, `start_bindings`, again on `rewritten`, the element +/// the typed writers produced from it, since they leave them out. They leave +/// out a declaration below the start tag too, so a prefix of `rewritten` +/// left to the part must not be one that `source` declares with a namespace +/// the part does not bind it to. Otherwise the prefix would be unbound, or +/// name another namespace, and this returns `None`. +fn with_source_namespaces( + source: &[u8], + rewritten: &[u8], + start_bindings: &[(String, String)], + word_prefixes: &[String], +) -> Option> { + let rewritten = raw_with_external_bindings(rewritten, start_bindings).ok()?; + let (_, declared) = namespace_use(source)?; + let (left_to_part, _) = namespace_use(&rewritten)?; + let part = namespace_bindings(word_prefixes); + let part_namespace = |prefix: &str| { + part.iter() + .find(|(candidate, _)| candidate == prefix) + .map(|(_, namespace)| namespace.as_str()) + .or_else(|| word_prefixes.iter().any(|p| p == prefix).then_some(W_NS)) + }; + left_to_part + .iter() + .all(|prefix| { + declared + .iter() + .filter(|(candidate, _)| candidate == prefix) + .all(|(_, namespace)| part_namespace(prefix) == Some(namespace.as_str())) }) - .sum() + .then_some(rewritten) +} + +/// The prefixes an element uses outside every declaration it makes, and +/// every declaration it makes, as prefix and namespace. +type NamespaceUse = (Vec, Vec<(String, String)>); + +/// The [`NamespaceUse`] of `xml`, the empty prefix standing for the default +/// namespace of an unprefixed element. +fn namespace_use(xml: &[u8]) -> Option { + use quick_xml::Reader; + use quick_xml::events::Event; + + let prefix_of = |name: &[u8]| { + let name = std::str::from_utf8(name).ok()?; + Some(name.split_once(':').map(|(prefix, _)| prefix.to_owned())) + }; + let mut reader = Reader::from_reader(xml); + let mut scopes: Vec> = Vec::new(); + let mut unbound: Vec = Vec::new(); + let mut declared = Vec::new(); + let mut buffer = Vec::new(); + loop { + let (element, opens) = match reader.read_event_into(&mut buffer).ok()? { + Event::Start(element) => (element.into_owned(), true), + Event::Empty(element) => (element.into_owned(), false), + Event::End(_) => { + scopes.pop(); + buffer.clear(); + continue; + } + Event::Eof => break, + _ => { + buffer.clear(); + continue; + } + }; + buffer.clear(); + let declarations = local_namespace_overrides(&element, &[]).ok()?; + let mut prefixes = vec![prefix_of(element.name().as_ref())?.unwrap_or_default()]; + for attribute in element.attributes() { + let key = attribute.ok()?.key; + let key = key.as_ref(); + if key != b"xmlns" && !key.starts_with(b"xmlns:") { + prefixes.extend(prefix_of(key)?); + } + } + for prefix in prefixes { + let bound = prefix == "xml" + || declarations + .iter() + .chain(scopes.iter().flatten()) + .any(|(candidate, _)| *candidate == prefix); + if !bound && !unbound.contains(&prefix) { + unbound.push(prefix); + } + } + declared.extend(declarations.iter().cloned()); + if opens { + scopes.push(declarations); + } + } + Some((unbound, declared)) } /// Replace all occurrences of `placeholder` in a header or footer. pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replacement: &str) -> usize { - replace_in_paragraphs(&mut hf.paragraphs, placeholder, replacement) + edit_header_footer(hf, &mut |paragraph| { + replace_in_paragraph(paragraph, placeholder, replacement) + }) +} + +/// The texts a replacement in `hf` matches against, see [`replaceable_texts`]. +#[doc(hidden)] +pub fn header_footer_replaceable_texts(hf: &CT_HdrFtr) -> Vec { + let mut texts = Vec::new(); + edit_header_footer(&mut hf.clone(), &mut |paragraph| { + texts.extend(replaceable_texts(paragraph)); + 0 + }); + texts +} + +fn edit_header_footer(hf: &mut CT_HdrFtr, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { + let mut count = 0; + for paragraph in &mut hf.paragraphs { + count += edit(paragraph); + } + count } /// Replace placeholders in text boxes and shapes within a raw XML part. /// /// Walks the XML, finds `w:txbxContent` elements at any depth outside another -/// text box, parses their child `w:p` elements using `CT_P::from_xml`, -/// performs replacement, and re-serializes the paragraphs it changed. Every -/// other child of the text box, such as a table, a content control, a -/// bookmark or a paragraph without a match, is copied through verbatim in its -/// place. A text box nested inside another one is kept as it is, not edited. -/// Returns the modified XML and replacement count. +/// text box, parses their child `w:p` elements using `CT_P::from_xml`, and +/// their child tables and block content controls with the typed parsers, +/// performs replacement, and re-serializes the children it changed. Every +/// other child of the text box, such as a bookmark or a paragraph without a +/// match, is copied through verbatim in its place. A text box nested inside +/// another one is kept as it is, not edited. Returns the modified XML and +/// replacement count. pub fn replace_in_xml_part( xml: &[u8], placeholder: &str, @@ -357,16 +682,29 @@ pub fn replace_regex_in_xml_part( }) } -/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit` -/// and re-serialising it in place when `edit` counts a change. Every other -/// paragraph and child of the text box is copied through verbatim. Returns the +/// The texts a replacement in the text boxes of a raw XML part matches +/// against, see [`replaceable_texts`]. +#[doc(hidden)] +pub fn xml_part_replaceable_texts(xml: &[u8]) -> crate::error::Result> { + let mut texts = Vec::new(); + rewrite_text_boxes(xml, &mut |paragraph| { + texts.extend(replaceable_texts(paragraph)); + 0 + })?; + Ok(texts) +} + +/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit`, +/// those of its tables and block content controls included, and +/// re-serialising a child in place when `edit` counts a change in it. Every +/// other child of the text box is copied through verbatim. Returns the /// rewritten XML and the summed count. fn rewrite_text_boxes( xml: &[u8], edit: &mut dyn FnMut(&mut CT_P) -> usize, ) -> crate::error::Result<(Vec, usize)> { use crate::error::OxmlError; - use crate::namespace::matches_local_name; + use crate::namespace::{MC_NS, R_NS, matches_local_name}; use crate::raw_xml::capture_element; use quick_xml::events::Event; use quick_xml::{Reader, Writer}; @@ -377,6 +715,12 @@ fn rewrite_text_boxes( let mut writer = Writer::new(Vec::new()); let mut buf = Vec::new(); let mut total_count = 0; + // The bindings `CT_P::from_xml` assumes for a paragraph cut out of the part. + let word_prefixes = [ + "w".to_owned(), + format!("\0r\0{R_NS}"), + format!("\0mc\0{MC_NS}"), + ]; loop { match reader.read_event_into(&mut buf) { @@ -421,9 +765,11 @@ fn rewrite_text_boxes( } total_count += count; } else { - // A table, a content control or any other element - // the edit does not reach stays as it was. - let raw = capture_element(&mut reader, ie)?; + // A table or a content control is edited through + // the typed model, and any other element the edit + // does not reach stays as it was. + let mut raw = capture_element(&mut reader, ie)?; + total_count += edit_raw_block(&mut raw, &word_prefixes, edit); writer.get_mut().extend_from_slice(&raw); } } @@ -526,86 +872,21 @@ pub fn replace_many_in_chart_xml( /// Replace all regex matches in a paragraph with the replacement string. /// /// The `replacement` string supports capture group references: `$1`, `$2`, etc. -/// Uses the same cross-run char map algorithm as literal replacement. +/// Uses the same cross-run char map algorithm as literal replacement, so it +/// reaches the runs of inline content controls and leaves a match that +/// straddles a content-control boundary as it is. /// Returns the number of replacements made. pub fn replace_regex_in_paragraph(para: &mut CT_P, re: ®ex::Regex, replacement: &str) -> usize { - let mut total = 0; - let mut emptied = Vec::new(); - - // See `replace_in_paragraph`: resume after the inserted text so a - // replacement that itself matches the pattern cannot loop forever. // `captures_at` (rather than slicing) keeps anchors and look-around // evaluating against the full paragraph text. - let mut search_from = 0usize; - - loop { - let (full_text, char_map) = build_char_map(¶.runs); - if char_map.is_empty() || search_from > full_text.len() { - break; - } - - // Find the next match - let Some(m) = re.captures_at(&full_text, search_from).and_then(|caps| { - let mat = caps.get(0)?; - // Expand capture groups in replacement - let mut expanded = String::new(); - caps.expand(replacement, &mut expanded); - Some((mat.start(), mat.end(), expanded)) - }) else { - break; - }; - - let (byte_start, byte_end, expanded_replacement) = m; - - // A zero-width match would neither consume input nor advance the - // cursor; step past it so the loop always makes progress. - if byte_start == byte_end { - search_from = full_text[byte_start..] - .chars() - .next() - .map_or(full_text.len() + 1, |c| byte_start + c.len_utf8()); - continue; - } - - // Convert byte offsets to char indices - let match_start = full_text[..byte_start].chars().count(); - let match_end = match_start + full_text[byte_start..byte_end].chars().count(); - - if match_start >= char_map.len() || match_end == 0 || match_end > char_map.len() { - break; - } - - // Determine which runs are affected - let first_run = char_map[match_start].run_index; - let last_run = char_map[match_end - 1].run_index; - - if first_run == last_run { - replace_in_single_run( - &mut para.runs[first_run], - &char_map, - match_start, - match_end, - &expanded_replacement, - ); - } else { - replace_across_runs( - &mut para.runs, - &char_map, - match_start, - match_end, - first_run, - last_run, - &expanded_replacement, - ); - emptied.extend(first_run + 1..=last_run); - } - - search_from = byte_start + expanded_replacement.len(); - total += 1; - } - - remove_emptied_runs(para, emptied); - total + replace_matches(para, &mut |text, from| { + let captures = re.captures_at(text, from)?; + let matched = captures.get(0)?; + // Expand capture groups in replacement + let mut expanded = String::new(); + captures.expand(replacement, &mut expanded); + Some((matched.start(), matched.end(), expanded)) + }) } /// Replace regex matches in all paragraphs. @@ -620,85 +901,24 @@ pub fn replace_regex_in_paragraphs( .sum() } -/// Replace regex matches in a table (recursively handles nested tables). +/// Replace regex matches in a table (recursively handles nested tables and +/// content controls). pub fn replace_regex_in_table(table: &mut CT_Tbl, re: ®ex::Regex, replacement: &str) -> usize { - let mut count = 0; - for (_, _, sdt) in &mut table.content_controls { - count += replace_regex_in_sdt(sdt, re, replacement); - } - for row in &mut table.rows { - count += replace_regex_in_row(row, re, replacement); - } - count -} - -fn replace_regex_in_sdt( - sdt: &mut crate::content_control::CT_Sdt, - re: ®ex::Regex, - replacement: &str, -) -> usize { - use crate::content_control::SdtContent; - - let mut count = 0; - for content in &mut sdt.content { - count += match content { - SdtContent::Paragraph(paragraph) => { - replace_regex_in_paragraph(paragraph, re, replacement) - } - SdtContent::Table(table) => replace_regex_in_table(table, re, replacement), - SdtContent::Row(row) => replace_regex_in_row(row, re, replacement), - SdtContent::Cell(cell) => replace_regex_in_cell(cell, re, replacement), - SdtContent::ContentControl(nested) => replace_regex_in_sdt(nested, re, replacement), - SdtContent::Run(_) | SdtContent::RawXml(_) => 0, - }; - } - count -} - -fn replace_regex_in_row( - row: &mut crate::table::CT_Row, - re: ®ex::Regex, - replacement: &str, -) -> usize { - let controls = row - .content_controls - .iter_mut() - .map(|(_, _, sdt)| replace_regex_in_sdt(sdt, re, replacement)) - .sum::(); - controls - + row - .cells - .iter_mut() - .map(|cell| replace_regex_in_cell(cell, re, replacement)) - .sum::() -} - -fn replace_regex_in_cell( - cell: &mut crate::table::CT_Tc, - re: ®ex::Regex, - replacement: &str, -) -> usize { - use crate::table::CellContent; - - cell.content - .iter_mut() - .map(|content| match content { - CellContent::Paragraph(paragraph) => { - replace_regex_in_paragraph(paragraph, re, replacement) - } - CellContent::Table(table) => replace_regex_in_table(table, re, replacement), - CellContent::ContentControl(sdt) => replace_regex_in_sdt(sdt, re, replacement), - }) - .sum() + edit_table(table, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) + }) } -/// Replace regex matches in a header or footer. +/// Replace regex matches in a header or footer, see +/// [`replace_in_header_footer`]. pub fn replace_regex_in_header_footer( hf: &mut CT_HdrFtr, re: ®ex::Regex, replacement: &str, ) -> usize { - replace_regex_in_paragraphs(&mut hf.paragraphs, re, replacement) + edit_header_footer(hf, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) + }) } #[cfg(test)] @@ -862,6 +1082,61 @@ mod tests { assert!(paragraph_xml(&p).contains("")); } + /// Inside an inline control the replacement removes the runs it empties + /// too. A later run of the control keeps the bytes it was read with. + #[test] + fn a_control_run_after_emptied_ones_keeps_its_source_bytes() { + let tail = r#" tail"#; + let source = format!( + r#"Dear {{{{name}}}}{tail}"# + ); + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + + assert_eq!(replace_in_paragraph(&mut p, "{{name}}", "Bob"), 1); + + assert_eq!(p.text(), "Dear Bob tail"); + assert_eq!(p.content_controls[0].3.content.len(), 2); + let xml = paragraph_xml(&p); + assert!(xml.contains(tail), "{xml}"); + } + + /// Each inline control starts a stretch of its own, and so does each + /// control nested in it. A match never spans two stretches. + #[test] + fn a_match_stays_within_one_stretch_of_runs() { + let source = format!( + r#"alXphabcde"# + ); + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + assert_eq!(replaceable_texts(&p), ["al", "X", "pha", "b", "c", "de"]); + + for straddling in ["alX", "Xpha", "phab", "bc", "cd"] { + assert_eq!(replace_in_paragraph(&mut p, straddling, "-"), 0); + } + let re = regex::Regex::new(r"^al|bcd|e$").unwrap(); + assert_eq!(replace_regex_in_paragraph(&mut p, &re, "-"), 2); + + assert_eq!(p.text(), "-Xphabcd-"); + } + + /// A quantified pattern also matches the digits after a boundary alone. + /// The search goes on after the straddling match, so it no longer + /// replaces part of the number the reader sees. + #[test] + fn no_match_starts_inside_a_straddling_one() { + let source = format!( + r#"1234 end 56"# + ); + for pattern in [r"\d+", r"\d{2,}"] { + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + let re = regex::Regex::new(pattern).unwrap(); + + assert_eq!(replace_regex_in_paragraph(&mut p, &re, "N"), 1, "{pattern}"); + + assert_eq!(p.text(), "1234 end N", "{pattern}"); + } + } + #[test] fn replace_multiple_occurrences() { let mut p = make_para(&["{{x}} and {{x}}"]); @@ -1011,6 +1286,112 @@ mod tests { assert_eq!(String::from_utf8(result).unwrap(), expected); } + /// The tables and block content controls of a text box are edited + /// through the typed model. Those without a match keep their bytes. + #[test] + fn replace_in_textbox_reaches_its_tables_and_controls() { + let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD + .replace(">cell<", ">cell {{name}}<") + .replace(">control<", ">control {{name}}<"); + let other_table = r#"other cell"#; + assert_eq!( + xml_part_replaceable_texts(xml.as_bytes()).unwrap(), + [ + "Title", + "Hello {{name}}", + "cell {{name}}", + "control {{name}}", + "{{name}} again", + "other cell", + "No placeholder here" + ] + ); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + for regex in [false, true] { + let (result, count) = if regex { + replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap() + } else { + replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap() + }; + assert_eq!(count, 4); + let result = String::from_utf8(result).unwrap(); + assert!(!result.contains("{{name}}"), "{result}"); + let positions = [ + "Hello Alice", + "cell Alice", + "", + "control Alice", + "Alice again", + other_table, + ] + .map(|marker| { + result + .find(marker) + .unwrap_or_else(|| panic!("{marker}: {result}")) + }); + assert!(positions.is_sorted(), "{result}"); + } + } + + /// Panic on an element or attribute prefix that no declaration binds. + fn assert_every_prefix_is_bound(xml: &str) { + use quick_xml::events::Event; + use quick_xml::name::ResolveResult; + + let mut reader = quick_xml::NsReader::from_str(xml); + loop { + let (namespace, event) = reader.read_resolved_event().unwrap(); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + match event { + Event::Start(element) | Event::Empty(element) => { + for attribute in element.attributes() { + let key = attribute.unwrap().key; + let (namespace, _) = reader.resolver().resolve_attribute(key); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + } + } + Event::Eof => break, + _ => {} + } + } + } + + /// A table with a paragraph that uses `w14`, declared on the table or on + /// its cell. + fn table_declaring_w14(on_table: bool, text: &str) -> String { + let declaration = r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#; + let (table, cell) = if on_table { + (declaration, "") + } else { + ("", declaration) + }; + format!( + r#"{text}"# + ) + } + + /// The typed writers leave out the namespace declarations of a table, so + /// a paragraph that used a prefix declared on the table came back with + /// it unbound. The table declares it again. A declaration on a cell + /// cannot be put back, so that table keeps its bytes and counts nothing. + #[test] + fn replace_in_textbox_keeps_the_namespaces_its_tables_declare() { + let on_cell = table_declaring_w14(false, "cell {{name}}"); + let xml = format!( + r#"{}{on_cell}"#, + table_declaring_w14(true, "table {{name}}") + ); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Ada").unwrap(); + + assert_eq!(count, 1); + let result = String::from_utf8(result).unwrap(); + assert_every_prefix_is_bound(&result); + assert!(result.contains(">table Ada<"), "{result}"); + assert!(result.contains(&on_cell), "{result}"); + } + /// The end tag used to be written as `w:txbxContent` whatever prefix the /// start tag carried, which left the part ill-formed. #[test] diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 5c4b2b46..b4f59a76 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -21212,8 +21212,10 @@ impl Document { /// Replace all occurrences of `placeholder` with `replacement` throughout the document. /// /// Searches body paragraphs, tables (including nested), headers, footers, - /// text boxes and chart labels. Handles placeholders split across multiple - /// runs. Returns the total number of replacements made. + /// text boxes and chart labels, and the content controls at every level of + /// them, inline controls included. Handles placeholders split across + /// multiple runs. A match that straddles a content-control boundary is not + /// replaced. Returns the total number of replacements made. /// /// A `replacement` that contains `placeholder` is substituted once, not /// repeatedly. @@ -21284,13 +21286,15 @@ impl Document { for (rel_id, is_header) in self.header_footer_rel_ids() { if let Some(header_footer) = self.load_header_footer(&rel_id, is_header) { - sources.extend(crate::template::header_footer_sources(&header_footer)); + sources.extend(rdocx_oxml::placeholder::header_footer_replaceable_texts( + &header_footer, + )); } } for (part_name, _) in self.raw_text_bearing_part_names() { if let Some(xml) = self.package.get_part(&part_name) { - sources.extend(crate::template::text_box_sources(xml)?); + sources.extend(rdocx_oxml::placeholder::xml_part_replaceable_texts(xml)?); } } @@ -21334,23 +21338,14 @@ impl Document { Ok(count) } - /// Run the typed replacement over body paragraphs and tables. + /// Run the typed replacement over every body paragraph, those of tables + /// and content controls at every level included. fn replace_in_body(&mut self, placeholder: &str, replacement: &str) -> usize { - use rdocx_oxml::placeholder; - let mut count = 0; - for content in &mut self.document.body.content { - match content { - BodyContent::Paragraph(p) => { - count += placeholder::replace_in_paragraph(p, placeholder, replacement); - } - BodyContent::Table(t) => { - count += placeholder::replace_in_table(t, placeholder, replacement); - } - BodyContent::ContentControl(_) => {} - BodyContent::RawXml(_) => {} - } - } + visit_body_paragraphs_mut(&mut self.document.body.content, &mut |paragraph| { + count += + rdocx_oxml::placeholder::replace_in_paragraph(paragraph, placeholder, replacement); + }); count } @@ -21448,8 +21443,10 @@ impl Document { /// Replace all regex matches with `replacement` throughout the document. /// /// The `replacement` string supports capture groups: `$1`, `$2`, etc. - /// Searches body paragraphs, tables (including nested), headers, and footers. - /// Returns the total number of replacements made, or an error if the regex is invalid. + /// Searches body paragraphs, tables (including nested), headers, footers + /// and text boxes, and the content controls at every level of them, as + /// [`Self::replace_text`] does. Returns the total number of replacements + /// made, or an error if the regex is invalid. pub fn replace_regex(&mut self, pattern: &str, replacement: &str) -> Result { let re = regex::Regex::new(pattern).map_err(|e| Error::Other(format!("invalid regex: {e}")))?; @@ -21484,19 +21481,10 @@ impl Document { let mut count = 0; - // Replace in body paragraphs and tables - for content in &mut self.document.body.content { - match content { - BodyContent::Paragraph(p) => { - count += placeholder::replace_regex_in_paragraph(p, re, replacement); - } - BodyContent::Table(t) => { - count += placeholder::replace_regex_in_table(t, re, replacement); - } - BodyContent::ContentControl(_) => {} - BodyContent::RawXml(_) => {} - } - } + // Replace in body paragraphs, tables and content controls + visit_body_paragraphs_mut(&mut self.document.body.content, &mut |paragraph| { + count += placeholder::replace_regex_in_paragraph(paragraph, re, replacement); + }); // Replace in headers and footers for (rel_id, is_header) in self.header_footer_rel_ids() { diff --git a/crates/rdocx/src/template.rs b/crates/rdocx/src/template.rs index 3721169c..cc0dcf80 100644 --- a/crates/rdocx/src/template.rs +++ b/crates/rdocx/src/template.rs @@ -6,7 +6,6 @@ use quick_xml::Reader; use quick_xml::events::Event; use rdocx_oxml::content_control::{CT_Sdt, SdtContent}; use rdocx_oxml::document::{BodyContent, CT_Document}; -use rdocx_oxml::header_footer::CT_HdrFtr; use rdocx_oxml::namespace::matches_local_name; use rdocx_oxml::placeholder; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_Tc, CellContent}; @@ -576,7 +575,9 @@ fn render_nested_tables_in_control( fn body_marker(content: &BodyContent) -> Result> { match content { - BodyContent::Paragraph(paragraph) => marker_from_sources(&[paragraph_text(paragraph)]), + BodyContent::Paragraph(paragraph) => { + marker_from_sources(&placeholder::replaceable_texts(paragraph)) + } BodyContent::Table(_) | BodyContent::RawXml(_) => Ok(None), BodyContent::ContentControl(control) => { reject_unsupported_control_sources(&control_sources(control))?; @@ -928,7 +929,8 @@ fn stage_paragraph_scalars( scopes: &[Scope], sentinels: &mut SentinelPool, ) -> Result { - let replacements = resolve_replacements(&[paragraph_text(paragraph)], root, scopes)?; + let replacements = + resolve_replacements(&placeholder::replaceable_texts(paragraph), root, scopes)?; let mut count = 0; for replacement in replacements { let sentinel = sentinels.stage(&replacement.value, replacement.occurrences); @@ -1135,7 +1137,9 @@ pub(crate) fn body_sources(document: &CT_Document) -> Vec { let mut sources = Vec::new(); for content in &document.body.content { match content { - BodyContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + BodyContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } BodyContent::Table(table) => collect_table(table, &mut sources), BodyContent::ContentControl(control) => collect_control(control, &mut sources), BodyContent::RawXml(_) => {} @@ -1144,26 +1148,6 @@ pub(crate) fn body_sources(document: &CT_Document) -> Vec { sources } -pub(crate) fn header_footer_sources(header_footer: &CT_HdrFtr) -> Vec { - header_footer - .paragraphs - .iter() - .map(paragraph_text) - .collect() -} - -fn paragraph_text(paragraph: &CT_P) -> String { - paragraph - .runs - .iter() - .flat_map(|run| &run.content) - .filter_map(|content| match content { - RunContent::Text(text) => Some(text.text.as_str()), - _ => None, - }) - .collect() -} - fn collect_table(table: &CT_Tbl, sources: &mut Vec) { for (_, _, control) in &table.content_controls { collect_control(control, sources); @@ -1182,7 +1166,9 @@ fn control_sources(control: &CT_Sdt) -> Vec { fn collect_control(control: &CT_Sdt, sources: &mut Vec) { for content in &control.content { match content { - SdtContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + SdtContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } SdtContent::Table(table) => collect_table(table, sources), SdtContent::Row(row) => collect_row(row, sources), SdtContent::Cell(cell) => collect_cell(cell, sources), @@ -1213,7 +1199,9 @@ fn collect_row_marker_sources(row: &CT_Row, sources: &mut Vec) { fn collect_cell_marker_sources(cell: &CT_Tc, sources: &mut Vec) { for content in &cell.content { match content { - CellContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + CellContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } CellContent::Table(_) => {} CellContent::ContentControl(control) => { collect_control_marker_sources(control, sources); @@ -1225,7 +1213,9 @@ fn collect_cell_marker_sources(cell: &CT_Tc, sources: &mut Vec) { fn collect_control_marker_sources(control: &CT_Sdt, sources: &mut Vec) { for content in &control.content { match content { - SdtContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + SdtContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } SdtContent::Table(_) => {} SdtContent::Row(row) => collect_row_marker_sources(row, sources), SdtContent::Cell(cell) => collect_cell_marker_sources(cell, sources), @@ -1249,53 +1239,15 @@ fn collect_control_marker_sources(control: &CT_Sdt, sources: &mut Vec) { fn collect_cell(cell: &CT_Tc, sources: &mut Vec) { for content in &cell.content { match content { - CellContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + CellContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } CellContent::Table(table) => collect_table(table, sources), CellContent::ContentControl(control) => collect_control(control, sources), } } } -pub(crate) fn text_box_sources(xml: &[u8]) -> Result> { - let mut reader = Reader::from_reader(xml); - reader.config_mut().trim_text(false); - let mut sources = Vec::new(); - let mut buffer = Vec::new(); - let mut in_text_box = false; - - loop { - match reader.read_event_into(&mut buffer) { - Ok(Event::Eof) => break, - Ok(Event::Start(element)) - if matches_local_name(element.name().as_ref(), b"txbxContent") => - { - in_text_box = true; - } - Ok(Event::Start(element)) - if in_text_box && matches_local_name(element.name().as_ref(), b"p") => - { - let paragraph = CT_P::from_xml(&mut reader)?; - sources.push(paragraph_text(¶graph)); - } - Ok(Event::Start(element)) if in_text_box => { - reader - .read_to_end_into(element.name(), &mut Vec::new()) - .map_err(template_xml_error)?; - } - Ok(Event::End(element)) - if in_text_box && matches_local_name(element.name().as_ref(), b"txbxContent") => - { - in_text_box = false; - } - Ok(_) => {} - Err(error) => return Err(template_xml_error(error)), - } - buffer.clear(); - } - - Ok(sources) -} - pub(crate) fn chart_sources(xml: &[u8]) -> Result> { let mut reader = Reader::from_reader(xml); reader.config_mut().trim_text(false); diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index a1b129fc..ebaa27f9 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -16255,6 +16255,411 @@ mod text_box_replacement_keeps_every_child { } } +/// Replacement skipped the text of content controls. It never entered a +/// body-level control, read the runs of an inline control at no level, in +/// the body, a text box or a footer, and did not reach the tables and +/// controls of a text box either. Each location below holds its own tag, +/// which every walker must replace exactly once. +mod replacement_reaches_content_controls { + use std::collections::HashMap; + + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + const DOCUMENT: &str = "/word/document.xml"; + const HEADER: &str = "/word/header1.xml"; + const FOOTER: &str = "/word/footer1.xml"; + + /// Every location with the part that holds it. The tag of a location is + /// its name in double braces and its replacement the name in brackets. + const LOCATIONS: [(&str, &str); 7] = [ + ("block", DOCUMENT), + ("inline", DOCUMENT), + ("cell", DOCUMENT), + ("nested", DOCUMENT), + ("box", DOCUMENT), + ("box_inline", DOCUMENT), + ("footer_inline", FOOTER), + ]; + + /// The `w:tag` of every control, by part. + const CONTROL_TAGS: [(&str, &[&str]); 3] = [ + ( + DOCUMENT, + &[ + "goog_rdk_1", + "goog_rdk_0", + "cell", + "outer", + "middle", + "inner", + "innermost", + "box", + "box-inline", + ], + ), + (HEADER, &["header"]), + (FOOTER, &["goog_rdk_2"]), + ]; + + fn tag(name: &str) -> String { + format!("{{{{{name}}}}}") + } + + fn value(name: &str) -> String { + format!("[{name}]") + } + + fn run(text: &str) -> String { + format!(r#"{text}"#) + } + + fn paragraph(content: &str) -> String { + format!("{content}") + } + + fn control(tag: &str, content: &str) -> String { + format!( + r#"{content}"# + ) + } + + fn table(cell: &str) -> String { + format!( + r#"{cell}"# + ) + } + + /// A DrawingML text box holding a block control and an inline control. + fn text_box() -> String { + let content = [ + control("box", ¶graph(&run("Boxed {{box}}."))), + paragraph(&[run("Boxed "), control("box-inline", &run("{{box_inline}}"))].concat()), + ] + .concat(); + format!( + r#"00{content}"# + ) + } + + /// The table and the control of the header, which the model keeps as + /// raw XML, before its paragraph. + fn header_blocks() -> [String; 2] { + [ + table(¶graph(&run("Header cell {{header_table}}."))), + control( + "header", + ¶graph(&run("Header control {{header_control}}.")), + ), + ] + } + + /// The page-number control Word writes in a footer, before a paragraph + /// that wraps a run in a Google Docs control. + fn footer_block() -> String { + format!( + r#"{}"#, + paragraph(&run("Footer control {{footer_control}}.")) + ) + } + + fn document() -> Document { + let body = [ + control("goog_rdk_1", ¶graph(&run("Block {{block}}."))), + paragraph( + &[ + run("Body text, "), + control("goog_rdk_0", &run("{{inline}}")), + run(" dolor."), + ] + .concat(), + ), + table(&control("cell", ¶graph(&run("Cell {{cell}}.")))), + control( + "outer", + &control( + "middle", + ¶graph( + &[ + run("Nested "), + control("inner", &control("innermost", &run("{{nested}}"))), + ] + .concat(), + ), + ), + ), + paragraph(&[run("Host "), format!("{}", text_box())].concat()), + ] + .concat(); + let header = format!( + r#"{}{}"#, + header_blocks().concat(), + paragraph(&run("Header paragraph.")) + ); + let footer = format!( + r#"{}{}"#, + footer_block(), + paragraph( + &[ + run("Confidential "), + control("goog_rdk_2", &run("{{footer_inline}}")), + ] + .concat() + ) + ); + + let mut seed = Document::new(); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + let mut references = Vec::new(); + for (part, xml, kind, rel_type) in [ + ( + HEADER, + header, + "header", + oxml_opc::relationship::rel_types::HEADER, + ), + ( + FOOTER, + footer, + "footer", + oxml_opc::relationship::rel_types::FOOTER, + ), + ] { + package.set_part(part, xml.into_bytes()); + package.content_types.add_override( + part, + &format!( + "application/vnd.openxmlformats-officedocument.wordprocessingml.{kind}+xml" + ), + ); + let id = package + .get_or_create_part_rels(DOCUMENT) + .add(rel_type, part.trim_start_matches("/word/")); + references.push(format!( + r#""# + )); + } + package.set_part( + DOCUMENT, + format!( + r#"{body}{}"#, + references.concat() + ) + .into_bytes(), + ); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + fn saved_parts(document: &mut Document) -> HashMap<&'static str, String> { + let package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(document.to_bytes().unwrap())) + .unwrap(); + [DOCUMENT, HEADER, FOOTER] + .into_iter() + .map(|part| { + let xml = String::from_utf8(package.get_part(part).unwrap().to_vec()).unwrap(); + (part, xml) + }) + .collect() + } + + /// Check that the tag of every location in `replaced` gave way to its + /// value once, that every other tag is still there once, that every + /// control kept its `w:tag`, and that the tables and controls of the + /// header and footer kept their bytes where nothing was replaced. + fn assert_replaced(parts: &HashMap<&str, String>, replaced: &[&str]) { + for (name, part) in LOCATIONS { + let xml = &parts[part]; + if replaced.contains(&name) { + assert!(!xml.contains(&tag(name)), "{name}: {xml}"); + assert_eq!(xml.matches(&value(name)).count(), 1, "{name}: {xml}"); + } else { + assert_eq!(xml.matches(&tag(name)).count(), 1, "{name}: {xml}"); + } + } + for (part, tags) in CONTROL_TAGS { + for tag in tags { + let element = format!(r#""#); + assert!(parts[part].contains(&element), "{tag}: {}", parts[part]); + } + } + let raw_blocks = header_blocks() + .into_iter() + .map(|block| (HEADER, block)) + .chain([(FOOTER, footer_block())]); + for (part, block) in raw_blocks { + if !replaced.iter().any(|name| block.contains(&tag(name))) { + assert!(parts[part].contains(&block), "{block}\n{}", parts[part]); + } + } + } + + #[test] + fn every_walker_replaces_the_tag_of_every_location_once() { + for (name, _) in LOCATIONS { + let (tag, value) = (tag(name), value(name)); + for walker in ["try_replace_text", "replace_regex", "replace_all"] { + let mut document = document(); + let count = match walker { + "try_replace_text" => document.try_replace_text(&tag, &value).unwrap(), + "replace_regex" => document + .replace_regex(®ex::escape(&tag), &value) + .unwrap(), + _ => document.replace_all(&HashMap::from([(tag.as_str(), value.as_str())])), + }; + assert_eq!(count, 1, "{walker} at {name}"); + assert_replaced(&saved_parts(&mut document), &[name]); + } + } + } + + #[test] + fn a_template_renders_the_tag_of_every_location() { + let mut document = document(); + let data = LOCATIONS + .iter() + .map(|(name, _)| ((*name).to_owned(), serde_json::Value::from(value(name)))) + .collect::>(); + + let count = document + .render_template(&serde_json::Value::Object(data)) + .unwrap(); + + assert_eq!(count, LOCATIONS.len()); + let names = LOCATIONS.map(|(name, _)| name); + assert_replaced(&saved_parts(&mut document), &names); + } + + /// A match that straddles a control boundary is no match. A reader sees + /// "alpha one" in the first paragraph and "alXpha two" in the second, + /// and neither changes, while the match inside the control of the third + /// is replaced. Direct runs on both sides of a control used to be read as + /// one text, which matched "alpha" in the second paragraph. + #[test] + fn a_match_that_straddles_a_control_boundary_is_not_replaced() { + let body = [ + paragraph(&[run("al"), control("a", &run("pha one"))].concat()), + paragraph(&[run("al"), control("b", &run("X")), run("pha two")].concat()), + paragraph(&[run("al"), control("c", &run("alpha three"))].concat()), + ] + .concat(); + for regex in [false, true] { + let mut document = super::document_with_content_controls(&super::wrap_word_body(&body)); + let count = if regex { + document.replace_regex("alpha", "ALPHA").unwrap() + } else { + document.try_replace_text("alpha", "ALPHA").unwrap() + }; + assert_eq!(count, 1, "regex: {regex}"); + let texts = document + .paragraphs() + .iter() + .map(|paragraph| paragraph.text()) + .collect::>(); + assert_eq!(texts, ["alpha one", "alXpha two", "alALPHA three"]); + let xml = super::document_xml(&mut document); + assert_eq!(xml.matches(">al").count(), 3, "{xml}"); + } + } + + /// A quantified pattern also matches the digits after the boundary + /// alone, and replaced them, which left the number a reader sees half + /// replaced. The search now goes on after the straddling match, and + /// still reaches the number after the control. + #[test] + fn a_quantified_match_that_straddles_a_control_boundary_is_not_replaced() { + let body = paragraph(&[run("Order 12"), control("a", &run("34 end")), run(" 56")].concat()); + for pattern in [r"\d+", r"\d{2,}"] { + let mut document = super::document_with_content_controls(&super::wrap_word_body(&body)); + + assert_eq!( + document.replace_regex(pattern, "N").unwrap(), + 1, + "{pattern}" + ); + + assert_eq!( + document.paragraph(0).unwrap().text(), + "Order 1234 end N", + "{pattern}" + ); + } + } + + /// Word writes its table of contents as a body-level control whose + /// entries repeat the heading text. Replacement reaches them as it does + /// any other control, so a heading word counts once more for its entry, + /// and the entry still reads as its heading until the next update. + #[test] + fn a_table_of_contents_entry_is_replaced_with_its_heading() { + let toc = r#" TOC \o "1-3" \h \z \u Introduction PAGEREF _Toc1 \h 1"#; + let heading = r#"Introduction"#; + let mut document = + super::document_with_content_controls(&super::wrap_word_body(&[toc, heading].concat())); + + assert_eq!( + document + .try_replace_text("Introduction", "Overview") + .unwrap(), + 2 + ); + + let xml = super::document_xml(&mut document); + assert_eq!(xml.matches(">Overview<").count(), 2, "{xml}"); + assert!(xml.contains(r#" Date: Sun, 27 Sep 2026 21:47:06 +0200 Subject: [PATCH 09/10] Replace inside header and footer tables and keep their place The header and footer model types only its paragraphs. Every other child of the part, such as a table or a content control like the page-number control Word writes in a footer, is kept as raw XML without its position, and a rewrite wrote all of them after the last paragraph. Replacement rewrote the part as soon as one paragraph matched, so a footer whose page-number control preceded its text came back with the control moved below it. Replacement never read those raw children either, so text in a header table or in a header or footer control was not replaced. The part was also read with trimmed text events, which a raw child is captured with, so the page-number control holding "Page " came back as "Page" once the part was rewritten. CT_HdrFtr now records how many paragraphs precede each raw child, and the namespace bindings of its root, and writes each child back where it was. A child a caller adds is still written after the last paragraph. It reads the part without trimming, as CT_Document does, and still skips the text between the children of the root. Replacement parses a raw table or block control with the typed parsers under those bindings, as the text box walker does, and writes back only one it changed, so the others keep their bytes. render_template reads the tags of those blocks too. The model stays a list of paragraphs rather than a full typed body, so the published type only gains two fields that are not public. GitHub issue #160. --- crates/rdocx-cli/tests/integration.rs | 14 +++- crates/rdocx-oxml/src/header_footer.rs | 79 +++++++++++++++++-- crates/rdocx-oxml/src/placeholder.rs | 64 ++++++++++++++++ crates/rdocx/tests/regression_test.rs | 100 ++++++++++++++++++++++--- 4 files changed, 240 insertions(+), 17 deletions(-) diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs index 046adf68..28e194b2 100644 --- a/crates/rdocx-cli/tests/integration.rs +++ b/crates/rdocx-cli/tests/integration.rs @@ -346,7 +346,8 @@ fn cli_replace_reports_namespace_preflight_errors_without_panicking() { /// `rdocx replace --expect 1` found none of the text that Google Docs and /// Word keep in content controls: a run wrapped inside its paragraph, a /// paragraph wrapped at body level, a control in a table cell, nested -/// controls and a control in a text box. +/// controls, a control in a text box, and the table and controls of a +/// header or footer. #[test] fn replace_with_expect_counts_the_text_of_content_controls_everywhere() { let temp = TempWorkspace::new("replace-content-controls"); @@ -420,7 +421,16 @@ fn replace_with_expect_counts_the_text_of_content_controls_everywhere() { .write_to(&mut fs::File::create(&input).unwrap()) .unwrap(); - for name in ["inline", "block", "cell", "nested", "box"] { + for name in [ + "inline", + "block", + "cell", + "nested", + "box", + "header_table", + "header_control", + "footer_control", + ] { let tag = format!("{{{{{name}}}}}"); let replaced = temp.path.join(format!("{name}.docx")); let output = cli(&[ diff --git a/crates/rdocx-oxml/src/header_footer.rs b/crates/rdocx-oxml/src/header_footer.rs index 2ec2cb1b..424798b2 100644 --- a/crates/rdocx-oxml/src/header_footer.rs +++ b/crates/rdocx-oxml/src/header_footer.rs @@ -196,6 +196,13 @@ pub struct CT_HdrFtr { pub extra_namespaces: Vec<(String, String)>, /// Unknown child elements captured as raw XML. pub extra_xml: Vec>, + /// How many paragraphs precede each entry of `extra_xml`, so that a + /// rewrite puts a table or a content control back where it was. An entry + /// without a position is written after the last paragraph. + extra_xml_positions: Vec, + /// The namespace bindings of the root element, which a raw child is + /// parsed with when replacement reaches into it. + pub(crate) word_prefixes: Vec, } #[allow(non_snake_case)] @@ -206,6 +213,8 @@ impl CT_HdrFtr { watermarks: Vec::new(), extra_namespaces: Vec::new(), extra_xml: Vec::new(), + extra_xml_positions: Vec::new(), + word_prefixes: vec!["w".to_owned()], } } @@ -227,11 +236,16 @@ impl CT_HdrFtr { pub fn from_xml(xml: &[u8]) -> Result { let watermarks = parse_vml_watermarks(xml); let mut reader = Reader::from_reader(xml); - reader.config_mut().trim_text(true); + // A raw child is captured with the text events of this reader, so + // trimming them would drop the edge spaces of its text, such as the + // one of "Page " before a page number. The text between the children + // of the root is skipped below. + reader.config_mut().trim_text(false); let mut paragraphs = Vec::new(); let mut extra_namespaces = Vec::new(); let mut extra_xml = Vec::new(); + let mut extra_xml_positions = Vec::new(); let mut buf = Vec::new(); let mut word_prefixes = Vec::new(); @@ -267,6 +281,7 @@ impl CT_HdrFtr { } else { // Capture unknown elements as raw XML extra_xml.push(capture_element(&mut reader, e)?); + extra_xml_positions.push(paragraphs.len()); } } Ok(Event::Empty(ref e)) => { @@ -278,6 +293,7 @@ impl CT_HdrFtr { && !matches_local_name(name.as_ref(), b"ftr") { extra_xml.push(capture_empty_element(e)?); + extra_xml_positions.push(paragraphs.len()); } } Ok(Event::Eof) => break, @@ -292,6 +308,8 @@ impl CT_HdrFtr { watermarks, extra_namespaces, extra_xml, + extra_xml_positions, + word_prefixes, }) } @@ -341,12 +359,24 @@ impl CT_HdrFtr { writer.write_event(Event::Start(start))?; - for p in &self.paragraphs { + // Write each captured unknown element before the paragraph it + // preceded, and the rest after the last paragraph. + let mut raw_children = self + .extra_xml + .iter() + .enumerate() + .map(|(index, raw)| { + let position = self.extra_xml_positions.get(index).copied(); + (position.unwrap_or(usize::MAX), raw) + }) + .peekable(); + for (index, p) in self.paragraphs.iter().enumerate() { + while let Some((_, raw)) = raw_children.next_if(|(position, _)| *position <= index) { + writer.get_mut().extend_from_slice(raw); + } p.to_xml(&mut writer)?; } - - // Write captured unknown elements - for raw in &self.extra_xml { + for (_, raw) in raw_children { writer.get_mut().extend_from_slice(raw); } @@ -1219,6 +1249,45 @@ mod tests { assert_eq!(parsed.paragraphs.len(), 0); } + /// A rewrite wrote every table and content control after the last + /// paragraph. A raw child a caller adds still goes after it. + #[test] + fn raw_children_keep_their_place_between_paragraphs() { + let xml = format!( + r#"onetwo"# + ); + let mut parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + parsed + .extra_xml + .push(br#""#.to_vec()); + + let written = String::from_utf8(parsed.to_xml_header().unwrap()).unwrap(); + + let positions = [ + "", + ">one<", + "", + ">two<", + "Page "#; + let xml = format!("\n {control}\n \n"); + + let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + + assert_eq!(parsed.extra_xml, [control.as_bytes()]); + assert_eq!(parsed.paragraphs.len(), 1); + } + #[test] fn aliased_header_paragraph_properties_keep_root_scope() { let xml = format!( diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 5052e48a..14f4a6d0 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -606,6 +606,10 @@ fn namespace_use(xml: &[u8]) -> Option { } /// Replace all occurrences of `placeholder` in a header or footer. +/// +/// Reaches its paragraphs and those of the tables and block content controls +/// it keeps as raw XML. A raw element the replacement changed is written back +/// in its place, and the others keep their bytes. pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replacement: &str) -> usize { edit_header_footer(hf, &mut |paragraph| { replace_in_paragraph(paragraph, placeholder, replacement) @@ -628,6 +632,9 @@ fn edit_header_footer(hf: &mut CT_HdrFtr, edit: &mut dyn FnMut(&mut CT_P) -> usi for paragraph in &mut hf.paragraphs { count += edit(paragraph); } + for raw in &mut hf.extra_xml { + count += edit_raw_block(raw, &hf.word_prefixes, edit); + } count } @@ -1203,6 +1210,63 @@ mod tests { assert_eq!(hf.text(), "Company: Acme Corp"); } + /// The tables and block content controls of a header are raw XML in the + /// model, and replacement used to skip them. One the replacement changes + /// is written back in its place, and one it does not change keeps its + /// bytes, under the prefix the part uses. + #[test] + fn replace_in_header_footer_reaches_its_tables_and_controls() { + let untouched = r#"no tag"#; + let xml = format!( + r#"cell {{{{x}}}}paragraph {{{{x}}}}control {{{{x}}}}{untouched}"# + ); + let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + assert_eq!( + header_footer_replaceable_texts(&parsed), + ["paragraph {{x}}", "cell {{x}}", "control {{x}}", "no tag"] + ); + + let re = regex::Regex::new(r"\{\{x\}\}").unwrap(); + for regex in [false, true] { + let mut hf = parsed.clone(); + let count = if regex { + replace_regex_in_header_footer(&mut hf, &re, "Y") + } else { + replace_in_header_footer(&mut hf, "{{x}}", "Y") + }; + assert_eq!(count, 3); + let written = String::from_utf8(hf.to_xml_header().unwrap()).unwrap(); + assert!(!written.contains("{{x}}"), "{written}"); + let positions = ["cell Y", "paragraph Y", "control Y", untouched].map(|marker| { + written + .find(marker) + .unwrap_or_else(|| panic!("{marker}: {written}")) + }); + assert!(positions.is_sorted(), "{written}"); + } + } + + /// A header rewrites its raw tables as a text box does, see + /// `replace_in_textbox_keeps_the_namespaces_its_tables_declare`. A cell + /// that declares a namespace the root binds the same way is replaced. + #[test] + fn replace_in_header_footer_keeps_the_namespaces_its_tables_declare() { + let on_cell = table_declaring_w14(false, "cell {{x}}"); + let tables = [table_declaring_w14(true, "table {{x}}"), on_cell.clone()].concat(); + let w14 = r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#; + for (root, expected) in [("", 1), (w14, 2)] { + let xml = format!(r#"{tables}"#); + let mut hf = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + + assert_eq!(replace_in_header_footer(&mut hf, "{{x}}", "Y"), expected); + + let written = String::from_utf8(hf.to_xml_header().unwrap()).unwrap(); + assert_every_prefix_is_bound(&written); + assert!(written.contains(">table Y<"), "{written}"); + assert_eq!(written.contains(&on_cell), expected == 1, "{written}"); + } + } + #[test] fn replace_empty_placeholder_noop() { let mut p = make_para(&["Hello"]); diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index ebaa27f9..2f434db3 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -16256,10 +16256,10 @@ mod text_box_replacement_keeps_every_child { } /// Replacement skipped the text of content controls. It never entered a -/// body-level control, read the runs of an inline control at no level, in -/// the body, a text box or a footer, and did not reach the tables and -/// controls of a text box either. Each location below holds its own tag, -/// which every walker must replace exactly once. +/// body-level control, read the runs of an inline control at no level, and +/// reached neither the tables and controls of a header or footer nor those +/// of a text box. Each location below holds its own tag, which every walker +/// must replace exactly once. mod replacement_reaches_content_controls { use std::collections::HashMap; @@ -16272,13 +16272,16 @@ mod replacement_reaches_content_controls { /// Every location with the part that holds it. The tag of a location is /// its name in double braces and its replacement the name in brackets. - const LOCATIONS: [(&str, &str); 7] = [ + const LOCATIONS: [(&str, &str); 10] = [ ("block", DOCUMENT), ("inline", DOCUMENT), ("cell", DOCUMENT), ("nested", DOCUMENT), ("box", DOCUMENT), ("box_inline", DOCUMENT), + ("header_table", HEADER), + ("header_control", HEADER), + ("footer_control", FOOTER), ("footer_inline", FOOTER), ]; @@ -16342,14 +16345,20 @@ mod replacement_reaches_content_controls { ) } + /// The runs of a header or footer block that the model keeps as raw XML. + /// The first one ends in a space, as "Page " does before a page number. + fn raw_block_runs(text: &str, name: &str) -> String { + paragraph(&[run(text), run(&format!("{}.", tag(name)))].concat()) + } + /// The table and the control of the header, which the model keeps as /// raw XML, before its paragraph. fn header_blocks() -> [String; 2] { [ - table(¶graph(&run("Header cell {{header_table}}."))), + table(&raw_block_runs("Header cell ", "header_table")), control( "header", - ¶graph(&run("Header control {{header_control}}.")), + &raw_block_runs("Header control ", "header_control"), ), ] } @@ -16359,7 +16368,7 @@ mod replacement_reaches_content_controls { fn footer_block() -> String { format!( r#"{}"#, - paragraph(&run("Footer control {{footer_control}}.")) + raw_block_runs("Footer control ", "footer_control") ) } @@ -16469,8 +16478,9 @@ mod replacement_reaches_content_controls { /// Check that the tag of every location in `replaced` gave way to its /// value once, that every other tag is still there once, that every - /// control kept its `w:tag`, and that the tables and controls of the - /// header and footer kept their bytes where nothing was replaced. + /// control kept its `w:tag`, and that the header and footer kept their + /// blocks in order, byte for byte where nothing was replaced, and the + /// edge spaces of their text everywhere. fn assert_replaced(parts: &HashMap<&str, String>, replaced: &[&str]) { for (name, part) in LOCATIONS { let xml = &parts[part]; @@ -16487,6 +16497,16 @@ mod replacement_reaches_content_controls { assert!(parts[part].contains(&element), "{tag}: {}", parts[part]); } } + for (part, markers) in [ + ( + HEADER, + ["Header cell", "Header control", "Header paragraph."], + ), + (FOOTER, ["docPartGallery", "Footer control", "Confidential"]), + ] { + let positions = markers.map(|marker| parts[part].find(marker).unwrap()); + assert!(positions.is_sorted(), "{part}: {}", parts[part]); + } let raw_blocks = header_blocks() .into_iter() .map(|block| (HEADER, block)) @@ -16496,6 +16516,13 @@ mod replacement_reaches_content_controls { assert!(parts[part].contains(&block), "{block}\n{}", parts[part]); } } + for (part, text) in [ + (HEADER, ">Header cell "), + (HEADER, ">Header control "), + (FOOTER, ">Footer control "), + ] { + assert!(parts[part].contains(text), "{text}\n{}", parts[part]); + } } #[test] @@ -16613,6 +16640,59 @@ mod replacement_reaches_content_controls { assert!(xml.contains(r#"{text}"# + ) + }; + let on_cell = table("", w14, "cell {{x}}"); + let tables = [table(w14, "", "table {{x}}"), on_cell.clone()].concat(); + let mut header = super::document_with_header_story(&format!( + r#"{tables}"# + )); + let mut text_box = super::document_with_content_controls(&format!( + r#"{tables}"# + )); + + assert_eq!(header.try_replace_text("{{x}}", "Y").unwrap(), 1); + assert_eq!(text_box.try_replace_text("{{x}}", "Y").unwrap(), 1); + + for xml in [ + super::header_story_xml(&mut header), + super::document_xml(&mut text_box), + ] { + // Every element and attribute prefix resolves. + let mut reader = quick_xml::NsReader::from_str(&xml); + loop { + let (namespace, event) = reader.read_resolved_event().unwrap(); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + match event { + XmlEvent::Start(element) | XmlEvent::Empty(element) => { + for attribute in element.attributes() { + let key = attribute.unwrap().key; + let (namespace, _) = reader.resolver().resolve_attribute(key); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + } + } + XmlEvent::Eof => break, + _ => {} + } + } + assert!(xml.contains(">table Y<"), "{xml}"); + assert!(xml.contains(&on_cell), "{xml}"); + } + } + /// A template tag that a control boundary splits cannot be rendered /// whole, so the template is rejected rather than left half rendered. #[test] From 65e31f99ce41aeb34aaada8f2d01cb44f19ebab7 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:49:59 +0200 Subject: [PATCH 10/10] Re-record the archive measurements of rdocx, rdocx-oxml and rdocx-cli The replacement changes, the text box fix this branch is stacked on and their tests grow the rdocx-oxml, rdocx and rdocx-cli packages, so the crates.io archive rows in README.md, crates/rdocx-oxml/README.md and crates/rdocx-cli/README.md and their ARCHIVE_MEASUREMENTS entries are re-measured on top of the stacked tree. GitHub issue #160. --- README.md | 2 +- crates/rdocx-cli/README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 6 +++--- 4 files changed, 6 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index 412db0df..2b4dd87d 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,097,973 compressed bytes, 6,523,992 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,104,100 compressed bytes, 6,548,342 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-cli/README.md b/crates/rdocx-cli/README.md index 25ecc0fe..6b778c8a 100644 --- a/crates/rdocx-cli/README.md +++ b/crates/rdocx-cli/README.md @@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-cli | 35,147 compressed bytes, 149,734 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index dabac199..261b1508 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,851 compressed bytes, 2,381,304 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 375,767 compressed bytes, 2,412,318 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index 25c233a4..0dabe703 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -383,12 +383,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_097_973, 6_523_992, 36), - "rdocx-cli": (33_805, 145_256, 8), + "rdocx": (1_104_100, 6_548_342, 36), + "rdocx-cli": (35_147, 149_734, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_851, 2_381_304, 32), + "rdocx-oxml": (375_767, 2_412_318, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),