diff --git a/README.md b/README.md
index 479b462e..2b4dd87d 100644
--- a/README.md
+++ b/README.md
@@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation.
| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
-| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
+| Crates.io archive: rdocx | 1,104,100 compressed bytes, 6,548,342 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
diff --git a/crates/rdocx-cli/README.md b/crates/rdocx-cli/README.md
index 25ecc0fe..6b778c8a 100644
--- a/crates/rdocx-cli/README.md
+++ b/crates/rdocx-cli/README.md
@@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host.
| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
-| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
+| Crates.io archive: rdocx-cli | 35,147 compressed bytes, 149,734 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
## Use it when
diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs
index e4cdd98f..28e194b2 100644
--- a/crates/rdocx-cli/tests/integration.rs
+++ b/crates/rdocx-cli/tests/integration.rs
@@ -343,6 +343,122 @@ fn cli_replace_reports_namespace_preflight_errors_without_panicking() {
assert!(!output_path.exists());
}
+/// `rdocx replace --expect 1` found none of the text that Google Docs and
+/// Word keep in content controls: a run wrapped inside its paragraph, a
+/// paragraph wrapped at body level, a control in a table cell, nested
+/// controls, a control in a text box, and the table and controls of a
+/// header or footer.
+#[test]
+fn replace_with_expect_counts_the_text_of_content_controls_everywhere() {
+ let temp = TempWorkspace::new("replace-content-controls");
+ let input = temp.path.join("controls.docx");
+ write_document(&input, &["seed"]);
+
+ let control = |tag: &str, content: &str| {
+ format!(
+ r#"{content}"#
+ )
+ };
+ let run = |text: &str| format!("{text}");
+ let paragraph = |content: &str| format!("{content}");
+ let table = |cell: &str| {
+ format!(
+ r#"{cell}"#
+ )
+ };
+ let text_box = format!(
+ r#"{}"#,
+ control("box", ¶graph(&run("{{box}}")))
+ );
+ let body = [
+ paragraph(&[run("Body "), control("goog_rdk_0", &run("{{inline}}"))].concat()),
+ control("goog_rdk_1", ¶graph(&run("{{block}}"))),
+ table(&control("cell", ¶graph(&run("{{cell}}")))),
+ control("outer", ¶graph(&control("inner", &run("{{nested}}")))),
+ paragraph(&[run("Host"), text_box].concat()),
+ ]
+ .concat();
+ let word = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
+ let header = format!(
+ r#"{}{}"#,
+ table(¶graph(&run("{{header_table}}"))),
+ control("header", ¶graph(&run("{{header_control}}")))
+ );
+ let footer = format!(
+ r#"{}{}"#,
+ control("page", ¶graph(&run("{{footer_control}}"))),
+ paragraph(&run("Confidential"))
+ );
+
+ let mut package =
+ OpcPackage::from_reader(std::io::Cursor::new(fs::read(&input).unwrap())).unwrap();
+ let mut references = String::new();
+ for (kind, xml, rel_type) in [
+ ("header", header, rel_types::HEADER),
+ ("footer", footer, rel_types::FOOTER),
+ ] {
+ let part = format!("/word/{kind}1.xml");
+ package.set_part(&part, xml.into_bytes());
+ package.content_types.add_override(
+ &part,
+ &format!("application/vnd.openxmlformats-officedocument.wordprocessingml.{kind}+xml"),
+ );
+ let id = package
+ .get_or_create_part_rels("/word/document.xml")
+ .add(rel_type, &format!("{kind}1.xml"));
+ references.push_str(&format!(
+ r#""#
+ ));
+ }
+ package.set_part(
+ "/word/document.xml",
+ format!(
+ r#"{body}{references}"#
+ )
+ .into_bytes(),
+ );
+ package
+ .write_to(&mut fs::File::create(&input).unwrap())
+ .unwrap();
+
+ for name in [
+ "inline",
+ "block",
+ "cell",
+ "nested",
+ "box",
+ "header_table",
+ "header_control",
+ "footer_control",
+ ] {
+ let tag = format!("{{{{{name}}}}}");
+ let replaced = temp.path.join(format!("{name}.docx"));
+ let output = cli(&[
+ "replace",
+ path_text(&input),
+ "--placeholder",
+ &tag,
+ "--value",
+ "done",
+ "--expect",
+ "1",
+ "--output",
+ path_text(&replaced),
+ ]);
+ assert_success(&output, name);
+
+ let package = OpcPackage::open(&replaced).unwrap();
+ let saved = ["document", "header1", "footer1"]
+ .map(|part| {
+ let xml = package.get_part(&format!("/word/{part}.xml")).unwrap();
+ String::from_utf8(xml.to_vec()).unwrap()
+ })
+ .concat();
+ assert!(!saved.contains(&tag), "{name}: {saved}");
+ assert_eq!(saved.matches(">done<").count(), 1, "{name}: {saved}");
+ }
+}
+
#[test]
fn validate_exit_status_is_a_verdict() {
let temp = TempWorkspace::new("validate");
diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md
index 11fef241..261b1508 100644
--- a/crates/rdocx-oxml/README.md
+++ b/crates/rdocx-oxml/README.md
@@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits.
| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
-| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
+| Crates.io archive: rdocx-oxml | 375,767 compressed bytes, 2,412,318 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
## Use it when
diff --git a/crates/rdocx-oxml/src/content_control.rs b/crates/rdocx-oxml/src/content_control.rs
index 8ff31cd5..711f3e5e 100644
--- a/crates/rdocx-oxml/src/content_control.rs
+++ b/crates/rdocx-oxml/src/content_control.rs
@@ -806,6 +806,27 @@ impl CT_Sdt {
}
}
+ /// Remove one content child, keeping the revisions and the source bytes
+ /// of the later runs at their content index.
+ pub(crate) fn remove_content(&mut self, index: usize) {
+ if index >= self.content.len() {
+ return;
+ }
+ self.content.remove(index);
+ for (boundary, _) in &mut self.revisions {
+ if *boundary > index {
+ *boundary -= 1;
+ }
+ }
+ self.inline_run_sources
+ .retain(|source| source.content_index != index);
+ for source in &mut self.inline_run_sources {
+ if source.content_index > index {
+ source.content_index -= 1;
+ }
+ }
+ }
+
pub(crate) fn word_prefixes(&self) -> &[String] {
&self.word_prefixes
}
diff --git a/crates/rdocx-oxml/src/header_footer.rs b/crates/rdocx-oxml/src/header_footer.rs
index 2ec2cb1b..424798b2 100644
--- a/crates/rdocx-oxml/src/header_footer.rs
+++ b/crates/rdocx-oxml/src/header_footer.rs
@@ -196,6 +196,13 @@ pub struct CT_HdrFtr {
pub extra_namespaces: Vec<(String, String)>,
/// Unknown child elements captured as raw XML.
pub extra_xml: Vec>,
+ /// How many paragraphs precede each entry of `extra_xml`, so that a
+ /// rewrite puts a table or a content control back where it was. An entry
+ /// without a position is written after the last paragraph.
+ extra_xml_positions: Vec,
+ /// The namespace bindings of the root element, which a raw child is
+ /// parsed with when replacement reaches into it.
+ pub(crate) word_prefixes: Vec,
}
#[allow(non_snake_case)]
@@ -206,6 +213,8 @@ impl CT_HdrFtr {
watermarks: Vec::new(),
extra_namespaces: Vec::new(),
extra_xml: Vec::new(),
+ extra_xml_positions: Vec::new(),
+ word_prefixes: vec!["w".to_owned()],
}
}
@@ -227,11 +236,16 @@ impl CT_HdrFtr {
pub fn from_xml(xml: &[u8]) -> Result {
let watermarks = parse_vml_watermarks(xml);
let mut reader = Reader::from_reader(xml);
- reader.config_mut().trim_text(true);
+ // A raw child is captured with the text events of this reader, so
+ // trimming them would drop the edge spaces of its text, such as the
+ // one of "Page " before a page number. The text between the children
+ // of the root is skipped below.
+ reader.config_mut().trim_text(false);
let mut paragraphs = Vec::new();
let mut extra_namespaces = Vec::new();
let mut extra_xml = Vec::new();
+ let mut extra_xml_positions = Vec::new();
let mut buf = Vec::new();
let mut word_prefixes = Vec::new();
@@ -267,6 +281,7 @@ impl CT_HdrFtr {
} else {
// Capture unknown elements as raw XML
extra_xml.push(capture_element(&mut reader, e)?);
+ extra_xml_positions.push(paragraphs.len());
}
}
Ok(Event::Empty(ref e)) => {
@@ -278,6 +293,7 @@ impl CT_HdrFtr {
&& !matches_local_name(name.as_ref(), b"ftr")
{
extra_xml.push(capture_empty_element(e)?);
+ extra_xml_positions.push(paragraphs.len());
}
}
Ok(Event::Eof) => break,
@@ -292,6 +308,8 @@ impl CT_HdrFtr {
watermarks,
extra_namespaces,
extra_xml,
+ extra_xml_positions,
+ word_prefixes,
})
}
@@ -341,12 +359,24 @@ impl CT_HdrFtr {
writer.write_event(Event::Start(start))?;
- for p in &self.paragraphs {
+ // Write each captured unknown element before the paragraph it
+ // preceded, and the rest after the last paragraph.
+ let mut raw_children = self
+ .extra_xml
+ .iter()
+ .enumerate()
+ .map(|(index, raw)| {
+ let position = self.extra_xml_positions.get(index).copied();
+ (position.unwrap_or(usize::MAX), raw)
+ })
+ .peekable();
+ for (index, p) in self.paragraphs.iter().enumerate() {
+ while let Some((_, raw)) = raw_children.next_if(|(position, _)| *position <= index) {
+ writer.get_mut().extend_from_slice(raw);
+ }
p.to_xml(&mut writer)?;
}
-
- // Write captured unknown elements
- for raw in &self.extra_xml {
+ for (_, raw) in raw_children {
writer.get_mut().extend_from_slice(raw);
}
@@ -1219,6 +1249,45 @@ mod tests {
assert_eq!(parsed.paragraphs.len(), 0);
}
+ /// A rewrite wrote every table and content control after the last
+ /// paragraph. A raw child a caller adds still goes after it.
+ #[test]
+ fn raw_children_keep_their_place_between_paragraphs() {
+ let xml = format!(
+ r#"onetwo"#
+ );
+ let mut parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();
+ parsed
+ .extra_xml
+ .push(br#""#.to_vec());
+
+ let written = String::from_utf8(parsed.to_xml_header().unwrap()).unwrap();
+
+ let positions = [
+ "",
+ ">one<",
+ "",
+ ">two<",
+ "Page "#;
+ let xml = format!("\n {control}\n \n");
+
+ let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();
+
+ assert_eq!(parsed.extra_xml, [control.as_bytes()]);
+ assert_eq!(parsed.paragraphs.len(), 1);
+ }
+
#[test]
fn aliased_header_paragraph_properties_keep_root_scope() {
let xml = format!(
diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs
index 0bf078dd..14f4a6d0 100644
--- a/crates/rdocx-oxml/src/placeholder.rs
+++ b/crates/rdocx-oxml/src/placeholder.rs
@@ -2,20 +2,72 @@
//!
//! Handles the cross-run splitting problem: a placeholder like `{{name}}`
//! may be split across multiple `` elements in the OOXML source.
-
+//!
+//! A replacement reads the direct runs of a paragraph and the runs of its
+//! inline content controls, in document order, and searches their text left
+//! to right as one. A match must lie within one stretch of those runs, see
+//! [`replaceable_texts`]. A match that straddles a content-control boundary
+//! is not replaced, and the search goes on after it, so no part of the text
+//! it covers is replaced either.
+
+use crate::content_control::{CT_Sdt, SdtContent};
use crate::header_footer::CT_HdrFtr;
-use crate::table::CT_Tbl;
-use crate::text::{CT_P, CT_R, RunContent};
+use crate::namespace::W_NS;
+use crate::numbering::{local_namespace_overrides, namespace_bindings, word_prefixes_at};
+use crate::properties::is_word_element;
+use crate::table::{CT_Row, CT_Tbl, CT_Tc, CellContent};
+use crate::text::{
+ AcceptedRunPath, AcceptedRunPathSegment, CT_P, CT_R, RunContent, raw_with_external_bindings,
+};
/// Replace all occurrences of `placeholder` with `replacement` in a paragraph.
///
-/// Handles placeholders split across multiple runs. Preserves the formatting
-/// of the first matched run. Returns the number of replacements made.
+/// Handles placeholders split across multiple runs and reaches the runs of
+/// inline content controls. A match that straddles a content-control
+/// boundary is not replaced. Preserves the formatting of the first matched
+/// run. Returns the number of replacements made.
pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &str) -> usize {
if placeholder.is_empty() {
return 0;
}
+ replace_matches(para, &mut |text, from| {
+ let start = from + text[from..].find(placeholder)?;
+ Some((start, start + placeholder.len(), replacement.to_owned()))
+ })
+}
+/// The texts a replacement in `para` matches against, one per stretch of
+/// runs. The direct runs between two inline content controls form one
+/// stretch, and so do the runs of one control between its nested controls.
+/// A match never spans two stretches.
+#[doc(hidden)]
+pub fn replaceable_texts(para: &CT_P) -> Vec {
+ let mut texts: Vec = Vec::new();
+ let mut previous = None;
+ for (stretch, path) in text_runs(para) {
+ if previous != Some(stretch) {
+ texts.push(String::new());
+ previous = Some(stretch);
+ }
+ if let (Some(text), Some(run)) = (texts.last_mut(), para.accepted_run(&path)) {
+ text.extend(run.content.iter().filter_map(|content| match content {
+ RunContent::Text(t) => Some(t.text.as_str()),
+ _ => None,
+ }));
+ }
+ }
+ texts
+}
+
+/// Takes a text and the byte offset to search from, and returns the byte
+/// range of the next match and the text that replaces it.
+type MatchFinder<'a> = dyn FnMut(&str, usize) -> Option<(usize, usize, String)> + 'a;
+
+/// Find each match that `next_match` reports in the text of `para` and
+/// replace it.
+fn replace_matches(para: &mut CT_P, next_match: &mut MatchFinder<'_>) -> usize {
+ let runs = text_runs(para);
+ let mut emptied = Vec::new();
let mut total = 0;
// Byte offset in the concatenated paragraph text at which to look for the
@@ -25,61 +77,57 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st
// replacement forever.
let mut search_from = 0usize;
- // We loop because after one replacement the text layout changes
- // and there may be more matches.
loop {
// 1. Concatenate all run text and build a char map.
- let (full_text, char_map) = build_char_map(¶.runs);
-
- if search_from >= full_text.len() {
+ let (full_text, char_map) = build_char_map(para, &runs);
+ if search_from > full_text.len() {
break;
}
- // 2. Find the next match (byte offset in full_text).
- let Some(byte_start) = full_text[search_from..]
- .find(placeholder)
- .map(|i| search_from + i)
- else {
+ // 2. Find the next match (byte offsets in full_text).
+ let Some((byte_start, byte_end, replacement)) = next_match(&full_text, search_from) else {
break;
};
+ let next_char = full_text[byte_start..]
+ .chars()
+ .next()
+ .map_or(1, char::len_utf8);
+
+ // A zero-width match would neither consume input nor advance the
+ // cursor, so step past it and the loop always makes progress.
+ if byte_start == byte_end {
+ search_from = byte_start + next_char;
+ continue;
+ }
+
// Convert byte offsets to char indices (char_map is indexed by char position).
let match_start = full_text[..byte_start].chars().count();
- let match_end = match_start + placeholder.chars().count();
+ let match_end = match_start + full_text[byte_start..byte_end].chars().count();
// 3. Determine which runs are affected.
- let first_char = &char_map[match_start];
- let last_char = &char_map[match_end - 1];
- let first_run = first_char.run_index;
- let last_run = last_char.run_index;
+ let first_run = char_map[match_start].run_index;
+ let last_run = char_map[match_end - 1].run_index;
+
+ if runs[first_run].0 != runs[last_run].0 {
+ // The match straddles a content-control boundary, so it is no
+ // match. Look again after it, as after a replaced match, so that
+ // no shorter match inside it, such as `\d+` finds in the digits
+ // after a boundary, replaces part of the text it covers.
+ search_from = byte_end;
+ continue;
+ }
if first_run == last_run {
// Single-run match: simple in-place replacement on that run's text content.
- replace_in_single_run(
- &mut para.runs[first_run],
- &char_map,
- match_start,
- match_end,
- replacement,
- );
+ edit_run(para, &runs[first_run].1, |run| {
+ replace_in_single_run(run, &char_map, match_start, match_end, &replacement);
+ });
} else {
// Cross-run match: put replacement in first run, clear matched parts from others.
- replace_across_runs(
- &mut para.runs,
- &char_map,
- match_start,
- match_end,
- first_run,
- last_run,
- replacement,
- );
+ replace_across_runs(para, &runs, &char_map, match_start, match_end, &replacement);
+ emptied.extend(first_run + 1..=last_run);
}
- // Remove runs that became completely empty (no content at all).
- para.runs.retain(|r| !r.content.is_empty());
-
- // Update hyperlink spans to account for removed runs.
- reindex_hyperlinks(para);
-
// Text before `byte_start` is untouched and the replacement now
// occupies `byte_start..byte_start + replacement.len()`, so this stays
// on a char boundary of the rebuilt text.
@@ -88,13 +136,82 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st
total += 1;
}
+ remove_emptied_runs(para, &runs, emptied);
total
}
+/// The runs of `para` that a replacement reads, in the order `CT_P::runs`
+/// reads them, each with the stretch it belongs to and its address.
+fn text_runs(para: &CT_P) -> Vec<(usize, AcceptedRunPath)> {
+ let mut runs = Vec::new();
+ let mut stretch = 0;
+ for index in 0..=para.runs.len() {
+ for (control, (_, _, _, sdt)) in para
+ .content_controls
+ .iter()
+ .enumerate()
+ .filter(|(_, (at, _, _, _))| *at == index)
+ {
+ let mut prefix = vec![AcceptedRunPathSegment::ContentControl(control)];
+ stretch += 1;
+ control_text_runs(sdt, &mut prefix, &mut stretch, &mut runs);
+ stretch += 1;
+ }
+ if index < para.runs.len() {
+ runs.push((
+ stretch,
+ AcceptedRunPath {
+ segments: vec![AcceptedRunPathSegment::Run(index)],
+ },
+ ));
+ }
+ }
+ runs
+}
+
+fn control_text_runs(
+ sdt: &CT_Sdt,
+ prefix: &mut Vec,
+ stretch: &mut usize,
+ runs: &mut Vec<(usize, AcceptedRunPath)>,
+) {
+ for (index, content) in sdt.content.iter().enumerate() {
+ match content {
+ SdtContent::Run(_) => {
+ prefix.push(AcceptedRunPathSegment::Run(index));
+ runs.push((
+ *stretch,
+ AcceptedRunPath {
+ segments: prefix.clone(),
+ },
+ ));
+ prefix.pop();
+ }
+ SdtContent::ContentControl(nested) => {
+ prefix.push(AcceptedRunPathSegment::ContentControl(index));
+ *stretch += 1;
+ control_text_runs(nested, prefix, stretch, runs);
+ *stretch += 1;
+ prefix.pop();
+ }
+ _ => {}
+ }
+ }
+}
+
+/// Apply `edit` to the run at `path`.
+fn edit_run(para: &mut CT_P, path: &AcceptedRunPath, edit: impl FnOnce(&mut CT_R)) {
+ if let Some(mut run) = para.accepted_run(path).cloned() {
+ edit(&mut run);
+ let replaced = para.replace_accepted_run(path, run);
+ debug_assert!(matches!(replaced, Ok(true)), "text run path is stale");
+ }
+}
+
/// A mapping from character position in the concatenated text to its source run and content item.
#[derive(Debug)]
struct CharMapping {
- /// Index of the run in the paragraph's runs vec.
+ /// Index of the run in the list of text runs.
run_index: usize,
/// Index of the RunContent item within the run.
content_index: usize,
@@ -102,11 +219,14 @@ struct CharMapping {
byte_offset: usize,
}
-fn build_char_map(runs: &[CT_R]) -> (String, Vec) {
+fn build_char_map(para: &CT_P, runs: &[(usize, AcceptedRunPath)]) -> (String, Vec) {
let mut full_text = String::new();
let mut char_map = Vec::new();
- for (run_idx, run) in runs.iter().enumerate() {
+ for (run_idx, (_, path)) in runs.iter().enumerate() {
+ let Some(run) = para.accepted_run(path) else {
+ continue;
+ };
for (content_idx, content) in run.content.iter().enumerate() {
if let RunContent::Text(t) = content {
for (byte_pos, _ch) in t.text.char_indices() {
@@ -153,80 +273,114 @@ fn replace_in_single_run(
new_text.push_str(replacement);
new_text.push_str(&t.text[byte_end..]);
t.text = new_text;
- t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' ');
+ // Keep the flag the producer wrote. Dropping it rewrites an unchanged
+ // run, which a later comparison against the source then reports.
+ t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' ');
}
}
fn replace_across_runs(
- runs: &mut [CT_R],
+ para: &mut CT_P,
+ runs: &[(usize, AcceptedRunPath)],
char_map: &[CharMapping],
match_start: usize,
match_end: usize,
- first_run: usize,
- last_run: usize,
replacement: &str,
) {
// Handle the first run: replace from match start to end of text in that content item.
let first_mapping = &char_map[match_start];
- let first_content_idx = first_mapping.content_index;
- let first_byte_offset = first_mapping.byte_offset;
-
- if let RunContent::Text(t) = &mut runs[first_run].content[first_content_idx] {
- let mut new_text = String::new();
- new_text.push_str(&t.text[..first_byte_offset]);
- new_text.push_str(replacement);
- t.text = new_text;
- t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' ');
- }
+ edit_run(para, &runs[first_mapping.run_index].1, |run| {
+ if let RunContent::Text(t) = &mut run.content[first_mapping.content_index] {
+ let mut new_text = String::new();
+ new_text.push_str(&t.text[..first_mapping.byte_offset]);
+ new_text.push_str(replacement);
+ t.text = new_text;
+ t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' ');
+ }
+ });
// Handle the last run: replace from start to match end within that content item.
let last_mapping = &char_map[match_end - 1];
- let last_content_idx = last_mapping.content_index;
- let last_byte_offset = last_mapping.byte_offset;
-
- if let RunContent::Text(t) = &mut runs[last_run].content[last_content_idx] {
- let remaining = &t.text[last_byte_offset..];
- let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0);
- let byte_end = last_byte_offset + ch_len;
- t.text = t.text[byte_end..].to_string();
- t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' ');
- }
-
- // Clear text content from runs strictly between first and last.
- for run in &mut runs[(first_run + 1)..last_run] {
- run.content.retain(|c| !matches!(c, RunContent::Text(_)));
- }
+ edit_run(para, &runs[last_mapping.run_index].1, |run| {
+ if let RunContent::Text(t) = &mut run.content[last_mapping.content_index] {
+ let remaining = &t.text[last_mapping.byte_offset..];
+ let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0);
+ let byte_end = last_mapping.byte_offset + ch_len;
+ t.text = t.text[byte_end..].to_string();
+ t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' ');
+ }
- // If the last run's text is now empty, remove its text content too.
- if last_run != first_run {
- runs[last_run].content.retain(|c| {
+ // If the last run's text is now empty, remove its text content too.
+ run.content.retain(|c| {
if let RunContent::Text(t) = c {
!t.text.is_empty()
} else {
true
}
});
+ });
+
+ // Clear text content from runs strictly between first and last.
+ for (_, path) in &runs[first_mapping.run_index + 1..last_mapping.run_index] {
+ edit_run(para, path, |run| {
+ run.content.retain(|c| !matches!(c, RunContent::Text(_)));
+ });
}
}
-/// Re-index hyperlink spans after runs may have been removed.
-fn reindex_hyperlinks(para: &mut CT_P) {
- // After retain, run indices may have shifted. We rebuild by checking
- // which runs still exist. Since retain preserves order and only removes
- // empty runs, the relative order is maintained. However, hyperlink spans
- // referenced by index need adjustment.
- //
- // For simplicity: we decrement indices for each removed slot.
- // But since we already called retain, the runs are already compacted.
- // We need to adjust hyperlinks based on the new run count.
- //
- // The simplest correct approach: hyperlinks that referenced removed runs
- // get their range clamped/invalidated.
- para.hyperlinks.retain(|hl| hl.run_start < para.runs.len());
- for hl in &mut para.hyperlinks {
- if hl.run_end > para.runs.len() {
- hl.run_end = para.runs.len();
+/// Remove the runs at `candidates` (indices into `runs`) that the replacement
+/// left without any content, keeping every anchor of the paragraph and of
+/// its content controls on the boundary that remains.
+fn remove_emptied_runs(
+ para: &mut CT_P,
+ runs: &[(usize, AcceptedRunPath)],
+ mut candidates: Vec,
+) {
+ candidates.sort_unstable();
+ candidates.dedup();
+ let mut direct = vec![false; para.runs.len()];
+ // Later runs first, so that removing one inside a control leaves the
+ // addresses of the others valid.
+ for index in candidates.into_iter().rev() {
+ let path = &runs[index].1;
+ // A run keeps the attributes of its start tag, such as `w:rsidR`, as
+ // a raw record, which does not make it worth keeping once empty.
+ let emptied = para.accepted_run(path).is_some_and(|run| {
+ run.content.is_empty()
+ && run
+ .extra_xml_positions
+ .iter()
+ .filter(|position| CT_R::raw_child_is_root_attributes(**position))
+ .count()
+ == run.extra_xml.len()
+ });
+ if !emptied {
+ continue;
+ }
+ match path.segments() {
+ [AcceptedRunPathSegment::Run(run)] => direct[*run] = true,
+ [AcceptedRunPathSegment::ContentControl(control), rest @ ..] => {
+ if let Some((_, _, _, sdt)) = para.content_controls.get_mut(*control) {
+ remove_control_run(sdt, rest);
+ }
+ }
+ _ => {}
+ }
+ }
+ if direct.contains(&true) {
+ para.remove_runs(&direct);
+ }
+}
+
+fn remove_control_run(sdt: &mut CT_Sdt, path: &[AcceptedRunPathSegment]) {
+ match path {
+ [AcceptedRunPathSegment::Run(index)] => sdt.remove_content(*index),
+ [AcceptedRunPathSegment::ContentControl(index), rest @ ..] => {
+ if let Some(SdtContent::ContentControl(nested)) = sdt.content.get_mut(*index) {
+ remove_control_run(nested, rest);
+ }
}
+ _ => {}
}
}
@@ -238,80 +392,262 @@ pub fn replace_in_paragraphs(paras: &mut [CT_P], placeholder: &str, replacement:
.sum()
}
-/// Replace all occurrences of `placeholder` in a table (recursively handles nested tables).
+/// Replace all occurrences of `placeholder` in a table (recursively handles
+/// nested tables and content controls).
pub fn replace_in_table(table: &mut CT_Tbl, placeholder: &str, replacement: &str) -> usize {
+ edit_table(table, &mut |paragraph| {
+ replace_in_paragraph(paragraph, placeholder, replacement)
+ })
+}
+
+/// Hand every paragraph of a table to `edit` and sum what it counts.
+fn edit_table(table: &mut CT_Tbl, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize {
let mut count = 0;
for (_, _, sdt) in &mut table.content_controls {
- count += replace_in_sdt(sdt, placeholder, replacement);
+ count += edit_control(sdt, edit);
}
for row in &mut table.rows {
- count += replace_in_row(row, placeholder, replacement);
+ count += edit_row(row, edit);
}
count
}
-fn replace_in_sdt(
- sdt: &mut crate::content_control::CT_Sdt,
- placeholder: &str,
- replacement: &str,
-) -> usize {
- use crate::content_control::SdtContent;
-
+fn edit_control(sdt: &mut CT_Sdt, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize {
let mut count = 0;
for content in &mut sdt.content {
count += match content {
- SdtContent::Paragraph(paragraph) => {
- replace_in_paragraph(paragraph, placeholder, replacement)
- }
- SdtContent::Table(table) => replace_in_table(table, placeholder, replacement),
- SdtContent::Row(row) => replace_in_row(row, placeholder, replacement),
- SdtContent::Cell(cell) => replace_in_cell(cell, placeholder, replacement),
- SdtContent::ContentControl(nested) => replace_in_sdt(nested, placeholder, replacement),
+ SdtContent::Paragraph(paragraph) => edit(paragraph),
+ SdtContent::Table(table) => edit_table(table, edit),
+ SdtContent::Row(row) => edit_row(row, edit),
+ SdtContent::Cell(cell) => edit_cell(cell, edit),
+ SdtContent::ContentControl(nested) => edit_control(nested, edit),
SdtContent::Run(_) | SdtContent::RawXml(_) => 0,
};
}
count
}
-fn replace_in_row(row: &mut crate::table::CT_Row, placeholder: &str, replacement: &str) -> usize {
- let controls = row
- .content_controls
- .iter_mut()
- .map(|(_, _, sdt)| replace_in_sdt(sdt, placeholder, replacement))
- .sum::();
- controls
- + row
- .cells
- .iter_mut()
- .map(|cell| replace_in_cell(cell, placeholder, replacement))
- .sum::()
+fn edit_row(row: &mut CT_Row, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize {
+ let mut count = 0;
+ for (_, _, sdt) in &mut row.content_controls {
+ count += edit_control(sdt, edit);
+ }
+ for cell in &mut row.cells {
+ count += edit_cell(cell, edit);
+ }
+ count
}
-fn replace_in_cell(cell: &mut crate::table::CT_Tc, placeholder: &str, replacement: &str) -> usize {
- use crate::table::CellContent;
+fn edit_cell(cell: &mut CT_Tc, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize {
+ let mut count = 0;
+ for content in &mut cell.content {
+ count += match content {
+ CellContent::Paragraph(paragraph) => edit(paragraph),
+ CellContent::Table(table) => edit_table(table, edit),
+ CellContent::ContentControl(sdt) => edit_control(sdt, edit),
+ };
+ }
+ count
+}
- cell.content
- .iter_mut()
- .map(|content| match content {
- CellContent::Paragraph(paragraph) => {
- replace_in_paragraph(paragraph, placeholder, replacement)
- }
- CellContent::Table(table) => replace_in_table(table, placeholder, replacement),
- CellContent::ContentControl(sdt) => replace_in_sdt(sdt, placeholder, replacement),
+/// Hand the paragraphs of a table or a block content control kept as raw
+/// XML to `edit`, and re-serialise the element in place when the edit
+/// counts a change. Any other element, one the typed parsers refuse, or one
+/// whose namespaces the rewrite cannot keep, see [`with_source_namespaces`],
+/// keeps its bytes and counts nothing.
+fn edit_raw_block(
+ raw: &mut Vec,
+ word_prefixes: &[String],
+ edit: &mut dyn FnMut(&mut CT_P) -> usize,
+) -> usize {
+ use quick_xml::events::Event;
+ use quick_xml::{Reader, Writer};
+
+ let mut reader = Reader::from_reader(raw.as_slice());
+ reader.config_mut().trim_text(false);
+ let mut buffer = Vec::new();
+ let Ok(Event::Start(start)) = reader.read_event_into(&mut buffer) else {
+ return 0;
+ };
+ let Ok(prefixes) = word_prefixes_at(&start, word_prefixes) else {
+ return 0;
+ };
+ // The namespaces the start tag declares other than the part does.
+ let Ok(bindings) = local_namespace_overrides(&start, word_prefixes) else {
+ return 0;
+ };
+ let mut writer = Writer::new(Vec::new());
+ let count = if is_word_element(start.name().as_ref(), b"tbl", &prefixes) {
+ let Ok(mut table) =
+ CT_Tbl::from_xml_with_prefixes_and_owner_bindings(&mut reader, &prefixes, &bindings)
+ else {
+ return 0;
+ };
+ let count = edit_table(&mut table, edit);
+ if count == 0 || table.to_xml(&mut writer).is_err() {
+ return 0;
+ }
+ count
+ } else if is_word_element(start.name().as_ref(), b"sdt", &prefixes) {
+ let Some(mut control) = CT_Sdt::from_body_raw(raw, word_prefixes) else {
+ return 0;
+ };
+ let count = edit_control(&mut control, edit);
+ if count == 0 || control.to_xml(&mut writer).is_err() {
+ return 0;
+ }
+ count
+ } else {
+ return 0;
+ };
+ let Some(rewritten) =
+ with_source_namespaces(raw, &writer.into_inner(), &bindings, word_prefixes)
+ else {
+ return 0;
+ };
+ *raw = rewritten;
+ count
+}
+
+/// Declare the namespaces that the start tag of `source` declares other
+/// than the part does, `start_bindings`, again on `rewritten`, the element
+/// the typed writers produced from it, since they leave them out. They leave
+/// out a declaration below the start tag too, so a prefix of `rewritten`
+/// left to the part must not be one that `source` declares with a namespace
+/// the part does not bind it to. Otherwise the prefix would be unbound, or
+/// name another namespace, and this returns `None`.
+fn with_source_namespaces(
+ source: &[u8],
+ rewritten: &[u8],
+ start_bindings: &[(String, String)],
+ word_prefixes: &[String],
+) -> Option> {
+ let rewritten = raw_with_external_bindings(rewritten, start_bindings).ok()?;
+ let (_, declared) = namespace_use(source)?;
+ let (left_to_part, _) = namespace_use(&rewritten)?;
+ let part = namespace_bindings(word_prefixes);
+ let part_namespace = |prefix: &str| {
+ part.iter()
+ .find(|(candidate, _)| candidate == prefix)
+ .map(|(_, namespace)| namespace.as_str())
+ .or_else(|| word_prefixes.iter().any(|p| p == prefix).then_some(W_NS))
+ };
+ left_to_part
+ .iter()
+ .all(|prefix| {
+ declared
+ .iter()
+ .filter(|(candidate, _)| candidate == prefix)
+ .all(|(_, namespace)| part_namespace(prefix) == Some(namespace.as_str()))
})
- .sum()
+ .then_some(rewritten)
+}
+
+/// The prefixes an element uses outside every declaration it makes, and
+/// every declaration it makes, as prefix and namespace.
+type NamespaceUse = (Vec, Vec<(String, String)>);
+
+/// The [`NamespaceUse`] of `xml`, the empty prefix standing for the default
+/// namespace of an unprefixed element.
+fn namespace_use(xml: &[u8]) -> Option {
+ use quick_xml::Reader;
+ use quick_xml::events::Event;
+
+ let prefix_of = |name: &[u8]| {
+ let name = std::str::from_utf8(name).ok()?;
+ Some(name.split_once(':').map(|(prefix, _)| prefix.to_owned()))
+ };
+ let mut reader = Reader::from_reader(xml);
+ let mut scopes: Vec> = Vec::new();
+ let mut unbound: Vec = Vec::new();
+ let mut declared = Vec::new();
+ let mut buffer = Vec::new();
+ loop {
+ let (element, opens) = match reader.read_event_into(&mut buffer).ok()? {
+ Event::Start(element) => (element.into_owned(), true),
+ Event::Empty(element) => (element.into_owned(), false),
+ Event::End(_) => {
+ scopes.pop();
+ buffer.clear();
+ continue;
+ }
+ Event::Eof => break,
+ _ => {
+ buffer.clear();
+ continue;
+ }
+ };
+ buffer.clear();
+ let declarations = local_namespace_overrides(&element, &[]).ok()?;
+ let mut prefixes = vec![prefix_of(element.name().as_ref())?.unwrap_or_default()];
+ for attribute in element.attributes() {
+ let key = attribute.ok()?.key;
+ let key = key.as_ref();
+ if key != b"xmlns" && !key.starts_with(b"xmlns:") {
+ prefixes.extend(prefix_of(key)?);
+ }
+ }
+ for prefix in prefixes {
+ let bound = prefix == "xml"
+ || declarations
+ .iter()
+ .chain(scopes.iter().flatten())
+ .any(|(candidate, _)| *candidate == prefix);
+ if !bound && !unbound.contains(&prefix) {
+ unbound.push(prefix);
+ }
+ }
+ declared.extend(declarations.iter().cloned());
+ if opens {
+ scopes.push(declarations);
+ }
+ }
+ Some((unbound, declared))
}
/// Replace all occurrences of `placeholder` in a header or footer.
+///
+/// Reaches its paragraphs and those of the tables and block content controls
+/// it keeps as raw XML. A raw element the replacement changed is written back
+/// in its place, and the others keep their bytes.
pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replacement: &str) -> usize {
- replace_in_paragraphs(&mut hf.paragraphs, placeholder, replacement)
+ edit_header_footer(hf, &mut |paragraph| {
+ replace_in_paragraph(paragraph, placeholder, replacement)
+ })
+}
+
+/// The texts a replacement in `hf` matches against, see [`replaceable_texts`].
+#[doc(hidden)]
+pub fn header_footer_replaceable_texts(hf: &CT_HdrFtr) -> Vec {
+ let mut texts = Vec::new();
+ edit_header_footer(&mut hf.clone(), &mut |paragraph| {
+ texts.extend(replaceable_texts(paragraph));
+ 0
+ });
+ texts
+}
+
+fn edit_header_footer(hf: &mut CT_HdrFtr, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize {
+ let mut count = 0;
+ for paragraph in &mut hf.paragraphs {
+ count += edit(paragraph);
+ }
+ for raw in &mut hf.extra_xml {
+ count += edit_raw_block(raw, &hf.word_prefixes, edit);
+ }
+ count
}
/// Replace placeholders in text boxes and shapes within a raw XML part.
///
-/// Walks the XML, finds `w:txbxContent` elements at any depth, parses their
-/// child `w:p` elements using `CT_P::from_xml`, performs replacement, and
-/// re-serializes back. Returns the modified XML and replacement count.
+/// Walks the XML, finds `w:txbxContent` elements at any depth outside another
+/// text box, parses their child `w:p` elements using `CT_P::from_xml`, and
+/// their child tables and block content controls with the typed parsers,
+/// performs replacement, and re-serializes the children it changed. Every
+/// other child of the text box, such as a bookmark or a paragraph without a
+/// match, is copied through verbatim in its place. A text box nested inside
+/// another one is kept as it is, not edited. Returns the modified XML and
+/// replacement count.
pub fn replace_in_xml_part(
xml: &[u8],
placeholder: &str,
@@ -329,11 +665,11 @@ pub fn replace_many_in_xml_part(
xml: &[u8],
replacements: &[(&str, &str)],
) -> crate::error::Result<(Vec, usize)> {
- rewrite_text_boxes(xml, &mut |paragraphs| {
+ rewrite_text_boxes(xml, &mut |paragraph| {
replacements
.iter()
.map(|(placeholder, replacement)| {
- replace_in_paragraphs(paragraphs, placeholder, replacement)
+ replace_in_paragraph(paragraph, placeholder, replacement)
})
.sum()
})
@@ -348,18 +684,35 @@ pub fn replace_regex_in_xml_part(
re: ®ex::Regex,
replacement: &str,
) -> crate::error::Result<(Vec, usize)> {
- rewrite_text_boxes(xml, &mut |paragraphs| {
- replace_regex_in_paragraphs(paragraphs, re, replacement)
+ rewrite_text_boxes(xml, &mut |paragraph| {
+ replace_regex_in_paragraph(paragraph, re, replacement)
})
}
-/// Walk `xml`, handing the paragraphs of each `w:txbxContent` element to
-/// `edit`, and re-serialise. Returns the rewritten XML and the summed count.
+/// The texts a replacement in the text boxes of a raw XML part matches
+/// against, see [`replaceable_texts`].
+#[doc(hidden)]
+pub fn xml_part_replaceable_texts(xml: &[u8]) -> crate::error::Result> {
+ let mut texts = Vec::new();
+ rewrite_text_boxes(xml, &mut |paragraph| {
+ texts.extend(replaceable_texts(paragraph));
+ 0
+ })?;
+ Ok(texts)
+}
+
+/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit`,
+/// those of its tables and block content controls included, and
+/// re-serialising a child in place when `edit` counts a change in it. Every
+/// other child of the text box is copied through verbatim. Returns the
+/// rewritten XML and the summed count.
fn rewrite_text_boxes(
xml: &[u8],
- edit: &mut dyn FnMut(&mut Vec) -> usize,
+ edit: &mut dyn FnMut(&mut CT_P) -> usize,
) -> crate::error::Result<(Vec, usize)> {
- use crate::namespace::matches_local_name;
+ use crate::error::OxmlError;
+ use crate::namespace::{MC_NS, R_NS, matches_local_name};
+ use crate::raw_xml::capture_element;
use quick_xml::events::Event;
use quick_xml::{Reader, Writer};
@@ -369,55 +722,29 @@ fn rewrite_text_boxes(
let mut writer = Writer::new(Vec::new());
let mut buf = Vec::new();
let mut total_count = 0;
+ // The bindings `CT_P::from_xml` assumes for a paragraph cut out of the part.
+ let word_prefixes = [
+ "w".to_owned(),
+ format!("\0r\0{R_NS}"),
+ format!("\0mc\0{MC_NS}"),
+ ];
loop {
match reader.read_event_into(&mut buf) {
Ok(Event::Eof) => break,
Ok(Event::Start(ref e)) if matches_local_name(e.name().as_ref(), b"txbxContent") => {
- // We found a txbxContent element. Collect its contents as raw XML,
- // parse paragraphs, do replacement, and re-serialize.
+ // We found a txbxContent element. Parse and edit each paragraph
+ // and copy every other child through verbatim, in document order.
writer.write_event(Event::Start(e.clone()))?;
- // Read all events inside txbxContent
- let mut depth = 1u32;
let mut inner_buf = Vec::new();
- // Collect paragraphs from inside txbxContent
- let mut paragraphs: Vec = Vec::new();
-
loop {
match reader.read_event_into(&mut inner_buf) {
Ok(Event::Start(ref ie)) => {
- if matches_local_name(ie.name().as_ref(), b"p") && depth == 1 {
+ if matches_local_name(ie.name().as_ref(), b"p") {
// Parse this paragraph: collect its XML, then parse via CT_P
- let mut para_writer = Writer::new(Vec::new());
- // Write the opening tag
- para_writer.write_event(Event::Start(ie.clone()))?;
- let mut pdepth = 1u32;
- let mut pbuf = Vec::new();
- loop {
- match reader.read_event_into(&mut pbuf) {
- Ok(Event::Start(ref pe)) => {
- pdepth += 1;
- para_writer.write_event(Event::Start(pe.clone()))?;
- }
- Ok(Event::End(ref pe)) => {
- pdepth -= 1;
- para_writer.write_event(Event::End(pe.clone()))?;
- if pdepth == 0 {
- break;
- }
- }
- Ok(ref ev) => {
- para_writer.write_event(ev.clone())?;
- }
- Err(e) => return Err(e.into()),
- }
- pbuf.clear();
- }
- let para_xml = para_writer.into_inner();
-
- // Parse the paragraph
+ let para_xml = capture_element(&mut reader, ie)?;
let mut para_reader = Reader::from_reader(para_xml.as_slice());
para_reader.config_mut().trim_text(true);
let mut prbuf = Vec::new();
@@ -434,42 +761,41 @@ fn rewrite_text_boxes(
}
prbuf.clear();
}
- let para = CT_P::from_xml(&mut para_reader)?;
- paragraphs.push(para);
+ let mut para = CT_P::from_xml(&mut para_reader)?;
+ let count = edit(&mut para);
+ if count == 0 {
+ // Nothing changed, so the paragraph keeps its
+ // bytes, start-tag attributes included.
+ writer.get_mut().extend_from_slice(¶_xml);
+ } else {
+ para.to_xml(&mut writer)?;
+ }
+ total_count += count;
} else {
- depth += 1;
- // Non-paragraph element inside txbxContent; skip it
- reader.read_to_end_into(ie.name(), &mut Vec::new())?;
- depth -= 1;
+ // A table or a content control is edited through
+ // the typed model, and any other element the edit
+ // does not reach stays as it was.
+ let mut raw = capture_element(&mut reader, ie)?;
+ total_count += edit_raw_block(&mut raw, &word_prefixes, edit);
+ writer.get_mut().extend_from_slice(&raw);
}
}
- Ok(Event::End(ref ie)) => {
- if matches_local_name(ie.name().as_ref(), b"txbxContent") && depth == 1
- {
- break;
- }
- depth -= 1;
+ // Every child element is consumed whole, so the first end
+ // tag at this level closes the txbxContent element.
+ Ok(Event::End(ie)) => {
+ writer.write_event(Event::End(ie))?;
+ break;
}
- Ok(_) => {
- // Whitespace/text at top level of txbxContent, skip
+ Ok(Event::Eof) => {
+ return Err(OxmlError::MissingElement("w:txbxContent end".to_owned()));
}
+ // Empty elements such as `` or a bookmark, whitespace,
+ // comments and processing instructions.
+ Ok(ev) => writer.write_event(ev)?,
Err(e) => return Err(e.into()),
}
inner_buf.clear();
}
-
- // Hand the collected paragraphs to the caller's edit function
- total_count += edit(&mut paragraphs);
-
- // Re-serialize paragraphs into the writer
- for p in ¶graphs {
- p.to_xml(&mut writer)?;
- }
-
- // Write closing txbxContent tag
- writer.write_event(Event::End(quick_xml::events::BytesEnd::new(
- "w:txbxContent",
- )))?;
}
Ok(ev) => {
writer.write_event(ev)?;
@@ -553,85 +879,21 @@ pub fn replace_many_in_chart_xml(
/// Replace all regex matches in a paragraph with the replacement string.
///
/// The `replacement` string supports capture group references: `$1`, `$2`, etc.
-/// Uses the same cross-run char map algorithm as literal replacement.
+/// Uses the same cross-run char map algorithm as literal replacement, so it
+/// reaches the runs of inline content controls and leaves a match that
+/// straddles a content-control boundary as it is.
/// Returns the number of replacements made.
pub fn replace_regex_in_paragraph(para: &mut CT_P, re: ®ex::Regex, replacement: &str) -> usize {
- let mut total = 0;
-
- // See `replace_in_paragraph`: resume after the inserted text so a
- // replacement that itself matches the pattern cannot loop forever.
// `captures_at` (rather than slicing) keeps anchors and look-around
// evaluating against the full paragraph text.
- let mut search_from = 0usize;
-
- loop {
- let (full_text, char_map) = build_char_map(¶.runs);
- if char_map.is_empty() || search_from > full_text.len() {
- break;
- }
-
- // Find the next match
- let Some(m) = re.captures_at(&full_text, search_from).and_then(|caps| {
- let mat = caps.get(0)?;
- // Expand capture groups in replacement
- let mut expanded = String::new();
- caps.expand(replacement, &mut expanded);
- Some((mat.start(), mat.end(), expanded))
- }) else {
- break;
- };
-
- let (byte_start, byte_end, expanded_replacement) = m;
-
- // A zero-width match would neither consume input nor advance the
- // cursor; step past it so the loop always makes progress.
- if byte_start == byte_end {
- search_from = full_text[byte_start..]
- .chars()
- .next()
- .map_or(full_text.len() + 1, |c| byte_start + c.len_utf8());
- continue;
- }
-
- // Convert byte offsets to char indices
- let match_start = full_text[..byte_start].chars().count();
- let match_end = match_start + full_text[byte_start..byte_end].chars().count();
-
- if match_start >= char_map.len() || match_end == 0 || match_end > char_map.len() {
- break;
- }
-
- // Determine which runs are affected
- let first_run = char_map[match_start].run_index;
- let last_run = char_map[match_end - 1].run_index;
-
- if first_run == last_run {
- replace_in_single_run(
- &mut para.runs[first_run],
- &char_map,
- match_start,
- match_end,
- &expanded_replacement,
- );
- } else {
- replace_across_runs(
- &mut para.runs,
- &char_map,
- match_start,
- match_end,
- first_run,
- last_run,
- &expanded_replacement,
- );
- }
-
- para.runs.retain(|r| !r.content.is_empty());
- reindex_hyperlinks(para);
- search_from = byte_start + expanded_replacement.len();
- total += 1;
- }
-
- total
+ replace_matches(para, &mut |text, from| {
+ let captures = re.captures_at(text, from)?;
+ let matched = captures.get(0)?;
+ // Expand capture groups in replacement
+ let mut expanded = String::new();
+ captures.expand(replacement, &mut expanded);
+ Some((matched.start(), matched.end(), expanded))
+ })
}
/// Replace regex matches in all paragraphs.
@@ -646,90 +908,30 @@ pub fn replace_regex_in_paragraphs(
.sum()
}
-/// Replace regex matches in a table (recursively handles nested tables).
+/// Replace regex matches in a table (recursively handles nested tables and
+/// content controls).
pub fn replace_regex_in_table(table: &mut CT_Tbl, re: ®ex::Regex, replacement: &str) -> usize {
- let mut count = 0;
- for (_, _, sdt) in &mut table.content_controls {
- count += replace_regex_in_sdt(sdt, re, replacement);
- }
- for row in &mut table.rows {
- count += replace_regex_in_row(row, re, replacement);
- }
- count
-}
-
-fn replace_regex_in_sdt(
- sdt: &mut crate::content_control::CT_Sdt,
- re: ®ex::Regex,
- replacement: &str,
-) -> usize {
- use crate::content_control::SdtContent;
-
- let mut count = 0;
- for content in &mut sdt.content {
- count += match content {
- SdtContent::Paragraph(paragraph) => {
- replace_regex_in_paragraph(paragraph, re, replacement)
- }
- SdtContent::Table(table) => replace_regex_in_table(table, re, replacement),
- SdtContent::Row(row) => replace_regex_in_row(row, re, replacement),
- SdtContent::Cell(cell) => replace_regex_in_cell(cell, re, replacement),
- SdtContent::ContentControl(nested) => replace_regex_in_sdt(nested, re, replacement),
- SdtContent::Run(_) | SdtContent::RawXml(_) => 0,
- };
- }
- count
-}
-
-fn replace_regex_in_row(
- row: &mut crate::table::CT_Row,
- re: ®ex::Regex,
- replacement: &str,
-) -> usize {
- let controls = row
- .content_controls
- .iter_mut()
- .map(|(_, _, sdt)| replace_regex_in_sdt(sdt, re, replacement))
- .sum::();
- controls
- + row
- .cells
- .iter_mut()
- .map(|cell| replace_regex_in_cell(cell, re, replacement))
- .sum::()
-}
-
-fn replace_regex_in_cell(
- cell: &mut crate::table::CT_Tc,
- re: ®ex::Regex,
- replacement: &str,
-) -> usize {
- use crate::table::CellContent;
-
- cell.content
- .iter_mut()
- .map(|content| match content {
- CellContent::Paragraph(paragraph) => {
- replace_regex_in_paragraph(paragraph, re, replacement)
- }
- CellContent::Table(table) => replace_regex_in_table(table, re, replacement),
- CellContent::ContentControl(sdt) => replace_regex_in_sdt(sdt, re, replacement),
- })
- .sum()
+ edit_table(table, &mut |paragraph| {
+ replace_regex_in_paragraph(paragraph, re, replacement)
+ })
}
-/// Replace regex matches in a header or footer.
+/// Replace regex matches in a header or footer, see
+/// [`replace_in_header_footer`].
pub fn replace_regex_in_header_footer(
hf: &mut CT_HdrFtr,
re: ®ex::Regex,
replacement: &str,
) -> usize {
- replace_regex_in_paragraphs(&mut hf.paragraphs, re, replacement)
+ edit_header_footer(hf, &mut |paragraph| {
+ replace_regex_in_paragraph(paragraph, re, replacement)
+ })
}
#[cfg(test)]
mod tests {
use super::*;
+ use crate::namespace::{MC_NS, W_NS};
use crate::properties::CT_RPr;
fn make_para(texts: &[&str]) -> CT_P {
@@ -796,6 +998,152 @@ mod tests {
assert_eq!(p.runs[1].properties.as_ref().unwrap().italic, Some(true));
}
+ #[test]
+ fn replace_keeps_the_producer_space_flag() {
+ let mut preserved = CT_R::new("WORD");
+ if let RunContent::Text(text) = &mut preserved.content[0] {
+ text.preserve_space = true;
+ }
+ let mut p = CT_P::new();
+ p.runs.push(preserved.clone());
+ p.runs.push(preserved);
+ p.add_run("tail");
+
+ assert_eq!(replace_in_paragraph(&mut p, "WORD", "WORD"), 2);
+ assert_eq!(replace_in_paragraph(&mut p, "DWO", "D-WO"), 1);
+ assert_eq!(replace_in_paragraph(&mut p, "tail", " end"), 1);
+ let flags = p
+ .runs
+ .iter()
+ .map(|run| match &run.content[0] {
+ RunContent::Text(text) => (text.text.as_str(), text.preserve_space),
+ _ => unreachable!(),
+ })
+ .collect::>();
+ assert_eq!(
+ flags,
+ [("WORD-WO", true), ("RD", true), (" end", true)],
+ "a rewritten run keeps a producer flag and gains one at a text edge"
+ );
+ }
+
+ fn paragraph_xml(paragraph: &CT_P) -> String {
+ let mut writer = quick_xml::Writer::new(Vec::new());
+ paragraph.to_xml(&mut writer).unwrap();
+ String::from_utf8(writer.into_inner()).unwrap()
+ }
+
+ /// A match across runs removes the runs it empties. The anchors after
+ /// them are kept by run index, and used to slide one run later, past the
+ /// text they preceded. A ruby annotation slid onto the run after its base.
+ #[test]
+ fn removing_emptied_runs_keeps_later_anchors_in_place() {
+ let source = format!(
+ r#"{{{{name}}}}controlannBASEx"#
+ );
+ let re = regex::Regex::new(r"\{\{name\}\}").unwrap();
+ for regex in [false, true] {
+ let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap();
+ let count = if regex {
+ replace_regex_in_paragraph(&mut p, &re, "Bob")
+ } else {
+ replace_in_paragraph(&mut p, "{{name}}", "Bob")
+ };
+ assert_eq!(count, 1);
+ assert_eq!(p.runs.len(), 3);
+ let xml = paragraph_xml(&p);
+ let positions = [
+ ">Bob<",
+ "bookmarkStart",
+ "commentRangeStart",
+ "",
+ "proofErr",
+ "",
+ ">BASE<",
+ "",
+ ">x<",
+ "bookmarkEnd",
+ "commentRangeEnd",
+ ]
+ .map(|marker| {
+ xml.find(marker)
+ .unwrap_or_else(|| panic!("{marker}: {xml}"))
+ });
+ assert!(positions.is_sorted(), "{xml}");
+ }
+ }
+
+ /// A run whose only child is raw XML, such as a drawing Word writes in
+ /// `mc:AlternateContent`, has no typed content, and neither has an
+ /// empty run. A match anywhere in the paragraph used to remove both.
+ #[test]
+ fn replace_keeps_the_runs_it_did_not_empty() {
+ let source = format!(
+ r#"Hello {{{{name}}}}"#
+ );
+ let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap();
+
+ assert_eq!(replace_in_paragraph(&mut p, "{{name}}", "Ada"), 1);
+
+ assert_eq!(p.runs.len(), 3);
+ assert!(paragraph_xml(&p).contains(""));
+ }
+
+ /// Inside an inline control the replacement removes the runs it empties
+ /// too. A later run of the control keeps the bytes it was read with.
+ #[test]
+ fn a_control_run_after_emptied_ones_keeps_its_source_bytes() {
+ let tail = r#" tail"#;
+ let source = format!(
+ r#"Dear {{{{name}}}}{tail}"#
+ );
+ let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap();
+
+ assert_eq!(replace_in_paragraph(&mut p, "{{name}}", "Bob"), 1);
+
+ assert_eq!(p.text(), "Dear Bob tail");
+ assert_eq!(p.content_controls[0].3.content.len(), 2);
+ let xml = paragraph_xml(&p);
+ assert!(xml.contains(tail), "{xml}");
+ }
+
+ /// Each inline control starts a stretch of its own, and so does each
+ /// control nested in it. A match never spans two stretches.
+ #[test]
+ fn a_match_stays_within_one_stretch_of_runs() {
+ let source = format!(
+ r#"alXphabcde"#
+ );
+ let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap();
+ assert_eq!(replaceable_texts(&p), ["al", "X", "pha", "b", "c", "de"]);
+
+ for straddling in ["alX", "Xpha", "phab", "bc", "cd"] {
+ assert_eq!(replace_in_paragraph(&mut p, straddling, "-"), 0);
+ }
+ let re = regex::Regex::new(r"^al|bcd|e$").unwrap();
+ assert_eq!(replace_regex_in_paragraph(&mut p, &re, "-"), 2);
+
+ assert_eq!(p.text(), "-Xphabcd-");
+ }
+
+ /// A quantified pattern also matches the digits after a boundary alone.
+ /// The search goes on after the straddling match, so it no longer
+ /// replaces part of the number the reader sees.
+ #[test]
+ fn no_match_starts_inside_a_straddling_one() {
+ let source = format!(
+ r#"1234 end 56"#
+ );
+ for pattern in [r"\d+", r"\d{2,}"] {
+ let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap();
+ let re = regex::Regex::new(pattern).unwrap();
+
+ assert_eq!(replace_regex_in_paragraph(&mut p, &re, "N"), 1, "{pattern}");
+
+ assert_eq!(p.text(), "1234 end N", "{pattern}");
+ }
+ }
+
#[test]
fn replace_multiple_occurrences() {
let mut p = make_para(&["{{x}} and {{x}}"]);
@@ -862,6 +1210,63 @@ mod tests {
assert_eq!(hf.text(), "Company: Acme Corp");
}
+ /// The tables and block content controls of a header are raw XML in the
+ /// model, and replacement used to skip them. One the replacement changes
+ /// is written back in its place, and one it does not change keeps its
+ /// bytes, under the prefix the part uses.
+ #[test]
+ fn replace_in_header_footer_reaches_its_tables_and_controls() {
+ let untouched = r#"no tag"#;
+ let xml = format!(
+ r#"cell {{{{x}}}}paragraph {{{{x}}}}control {{{{x}}}}{untouched}"#
+ );
+ let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();
+ assert_eq!(
+ header_footer_replaceable_texts(&parsed),
+ ["paragraph {{x}}", "cell {{x}}", "control {{x}}", "no tag"]
+ );
+
+ let re = regex::Regex::new(r"\{\{x\}\}").unwrap();
+ for regex in [false, true] {
+ let mut hf = parsed.clone();
+ let count = if regex {
+ replace_regex_in_header_footer(&mut hf, &re, "Y")
+ } else {
+ replace_in_header_footer(&mut hf, "{{x}}", "Y")
+ };
+ assert_eq!(count, 3);
+ let written = String::from_utf8(hf.to_xml_header().unwrap()).unwrap();
+ assert!(!written.contains("{{x}}"), "{written}");
+ let positions = ["cell Y", "paragraph Y", "control Y", untouched].map(|marker| {
+ written
+ .find(marker)
+ .unwrap_or_else(|| panic!("{marker}: {written}"))
+ });
+ assert!(positions.is_sorted(), "{written}");
+ }
+ }
+
+ /// A header rewrites its raw tables as a text box does, see
+ /// `replace_in_textbox_keeps_the_namespaces_its_tables_declare`. A cell
+ /// that declares a namespace the root binds the same way is replaced.
+ #[test]
+ fn replace_in_header_footer_keeps_the_namespaces_its_tables_declare() {
+ let on_cell = table_declaring_w14(false, "cell {{x}}");
+ let tables = [table_declaring_w14(true, "table {{x}}"), on_cell.clone()].concat();
+ let w14 = r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#;
+ for (root, expected) in [("", 1), (w14, 2)] {
+ let xml = format!(r#"{tables}"#);
+ let mut hf = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();
+
+ assert_eq!(replace_in_header_footer(&mut hf, "{{x}}", "Y"), expected);
+
+ let written = String::from_utf8(hf.to_xml_header().unwrap()).unwrap();
+ assert_every_prefix_is_bound(&written);
+ assert!(written.contains(">table Y<"), "{written}");
+ assert_eq!(written.contains(&on_cell), expected == 1, "{written}");
+ }
+ }
+
#[test]
fn replace_empty_placeholder_noop() {
let mut p = make_para(&["Hello"]);
@@ -918,6 +1323,165 @@ mod tests {
assert!(result_str.contains("Company: Acme"));
}
+ /// A text box whose paragraphs sit among a table, a block content
+ /// control, a bookmark, an empty paragraph, a comment and whitespace.
+ /// A second text box without a placeholder follows. The walker rewrites
+ /// every text box of the part, so that one must come back unchanged too.
+ /// The paragraphs without a placeholder keep their identity attributes.
+ const TEXT_BOX_WITH_EVERY_KIND_OF_CHILD: &str = r#"
+Title
+Hello {{name}}
+cell
+control{{name}} again
+other cellNo placeholder here"#;
+
+ #[test]
+ fn replace_in_textbox_keeps_every_other_child_in_order() {
+ let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD;
+ let expected = xml.replace("{{name}}", "Alice");
+
+ let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap();
+ assert_eq!(count, 2);
+ assert_eq!(String::from_utf8(result).unwrap(), expected);
+
+ let re = regex::Regex::new(r"\{\{name\}\}").unwrap();
+ let (result, count) = replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap();
+ assert_eq!(count, 2);
+ assert_eq!(String::from_utf8(result).unwrap(), expected);
+ }
+
+ /// The tables and block content controls of a text box are edited
+ /// through the typed model. Those without a match keep their bytes.
+ #[test]
+ fn replace_in_textbox_reaches_its_tables_and_controls() {
+ let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD
+ .replace(">cell<", ">cell {{name}}<")
+ .replace(">control<", ">control {{name}}<");
+ let other_table = r#"other cell"#;
+ assert_eq!(
+ xml_part_replaceable_texts(xml.as_bytes()).unwrap(),
+ [
+ "Title",
+ "Hello {{name}}",
+ "cell {{name}}",
+ "control {{name}}",
+ "{{name}} again",
+ "other cell",
+ "No placeholder here"
+ ]
+ );
+
+ let re = regex::Regex::new(r"\{\{name\}\}").unwrap();
+ for regex in [false, true] {
+ let (result, count) = if regex {
+ replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap()
+ } else {
+ replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap()
+ };
+ assert_eq!(count, 4);
+ let result = String::from_utf8(result).unwrap();
+ assert!(!result.contains("{{name}}"), "{result}");
+ let positions = [
+ "Hello Alice",
+ "cell Alice",
+ "",
+ "control Alice",
+ "Alice again",
+ other_table,
+ ]
+ .map(|marker| {
+ result
+ .find(marker)
+ .unwrap_or_else(|| panic!("{marker}: {result}"))
+ });
+ assert!(positions.is_sorted(), "{result}");
+ }
+ }
+
+ /// Panic on an element or attribute prefix that no declaration binds.
+ fn assert_every_prefix_is_bound(xml: &str) {
+ use quick_xml::events::Event;
+ use quick_xml::name::ResolveResult;
+
+ let mut reader = quick_xml::NsReader::from_str(xml);
+ loop {
+ let (namespace, event) = reader.read_resolved_event().unwrap();
+ assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}");
+ match event {
+ Event::Start(element) | Event::Empty(element) => {
+ for attribute in element.attributes() {
+ let key = attribute.unwrap().key;
+ let (namespace, _) = reader.resolver().resolve_attribute(key);
+ assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}");
+ }
+ }
+ Event::Eof => break,
+ _ => {}
+ }
+ }
+ }
+
+ /// A table with a paragraph that uses `w14`, declared on the table or on
+ /// its cell.
+ fn table_declaring_w14(on_table: bool, text: &str) -> String {
+ let declaration = r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#;
+ let (table, cell) = if on_table {
+ (declaration, "")
+ } else {
+ ("", declaration)
+ };
+ format!(
+ r#"{text}"#
+ )
+ }
+
+ /// The typed writers leave out the namespace declarations of a table, so
+ /// a paragraph that used a prefix declared on the table came back with
+ /// it unbound. The table declares it again. A declaration on a cell
+ /// cannot be put back, so that table keeps its bytes and counts nothing.
+ #[test]
+ fn replace_in_textbox_keeps_the_namespaces_its_tables_declare() {
+ let on_cell = table_declaring_w14(false, "cell {{name}}");
+ let xml = format!(
+ r#"{}{on_cell}"#,
+ table_declaring_w14(true, "table {{name}}")
+ );
+
+ let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Ada").unwrap();
+
+ assert_eq!(count, 1);
+ let result = String::from_utf8(result).unwrap();
+ assert_every_prefix_is_bound(&result);
+ assert!(result.contains(">table Ada<"), "{result}");
+ assert!(result.contains(&on_cell), "{result}");
+ }
+
+ /// The end tag used to be written as `w:txbxContent` whatever prefix the
+ /// start tag carried, which left the part ill-formed.
+ #[test]
+ fn replace_in_textbox_closes_it_with_its_own_prefix() {
+ let xml = r#"Hello {{name}}"#;
+
+ let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap();
+ assert_eq!(count, 1);
+ assert_eq!(
+ String::from_utf8(result).unwrap(),
+ xml.replace("{{name}}", "Alice")
+ );
+ }
+
+ /// A text box cut short used to keep the walker reading past the end of
+ /// the part forever.
+ #[test]
+ fn replace_in_unterminated_textbox_is_an_error() {
+ for tail in ["", "Hello {{name}}"] {
+ let xml = format!(
+ r#"{tail}"#
+ );
+ assert!(replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").is_err());
+ }
+ }
+
#[test]
fn replace_in_xml_part_no_textbox() {
let xml = br#"
diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs
index cd5501ca..c4f016c4 100644
--- a/crates/rdocx-oxml/src/text.rs
+++ b/crates/rdocx-oxml/src/text.rs
@@ -4320,10 +4320,19 @@ impl CT_P {
if removed.iter().all(|remove| !remove) {
return;
}
+ self.remove_runs(&removed);
+ }
+
+ /// Remove the direct runs flagged in `removed` and move every run-boundary
+ /// projection onto the boundary that remains. Raw children, comment
+ /// markers, bookmarks and controls of the boundaries that collapse into
+ /// one keep their order, and a hyperlink or a ruby annotation left
+ /// without runs is dropped.
+ pub(crate) fn remove_runs(&mut self, removed: &[bool]) {
let removed_run_addresses = self
.runs
.iter()
- .zip(&removed)
+ .zip(removed)
.filter_map(|(run, remove)| remove.then_some(std::ptr::from_ref(run)))
.collect::>();
let removed_projected_indices = accepted_paragraph_runs(self)
@@ -4470,6 +4479,12 @@ impl CT_P {
*position = boundary_map[old_boundary];
*raw_before = raw_prefixes[old_boundary] + (*raw_before).min(raw_counts[old_boundary]);
}
+ for ruby in &mut self.rubies {
+ ruby.base_start = boundary_map[ruby.base_start.min(old_run_count)];
+ ruby.base_end = boundary_map[ruby.base_end.min(old_run_count)];
+ }
+ // An annotation over no base run is not written, so drop it.
+ self.rubies.retain(|ruby| ruby.base_start < ruby.base_end);
let old_hyperlinks = std::mem::take(&mut self.hyperlinks);
let mut hyperlink_map = vec![None; old_hyperlinks.len()];
for (old_index, mut hyperlink) in old_hyperlinks.into_iter().enumerate() {
@@ -4517,7 +4532,7 @@ impl CT_P {
.runs
.drain(..)
.zip(removed)
- .filter_map(|(run, remove)| (!remove).then_some(run))
+ .filter_map(|(run, remove)| (!*remove).then_some(run))
.collect();
let _ = self.refresh_bookmark_projection();
}
diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs
index 19c557ed..7f7ff9d1 100644
--- a/crates/rdocx/src/comparison.rs
+++ b/crates/rdocx/src/comparison.rs
@@ -11,6 +11,7 @@ use rdocx_oxml::content_control::{CT_Sdt, SdtContent};
use rdocx_oxml::document::{BodyContent, CT_Document};
use rdocx_oxml::namespace::W_NS;
use rdocx_oxml::properties::CT_PPr;
+use rdocx_oxml::shared::ST_PageOrientation;
use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_TblPr, CT_Tc, CT_TrPr, CellContent};
use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent};
use sha2::{Digest, Sha256};
@@ -29,7 +30,6 @@ thread_local! {
type ControlPropertySignature<'a> = Option<(
Option<&'a str>,
Option<&'a str>,
- Option,
Option,
Option<&'a rdocx_oxml::content_control::CT_DataBinding>,
)>;
@@ -2985,8 +2985,16 @@ fn compare_granular_paragraph(
)));
}
- let original_run_signatures = original.runs.iter().map(run_signature).collect::>();
- let edited_run_signatures = edited.runs.iter().map(run_signature).collect::>();
+ let original_run_signatures = original
+ .runs
+ .iter()
+ .map(attributed_run_signature)
+ .collect::>();
+ let edited_run_signatures = edited
+ .runs
+ .iter()
+ .map(attributed_run_signature)
+ .collect::>();
if original_run_signatures == edited_run_signatures
&& original.content_controls == edited.content_controls
{
@@ -3558,12 +3566,30 @@ fn granular_text(text: &CT_Text, options: &ComparisonOptions) -> Vec {
fragments
.into_iter()
.map(|value| CT_Text {
+ // A unit is written back as its own `w:t`, where edge whitespace
+ // needs the flag to survive. Whitespace in a unit then reads as
+ // whitespace whatever the source flag was.
+ preserve_space: text.preserve_space || has_edge_whitespace(&value),
text: value,
- preserve_space: text.preserve_space,
})
.collect()
}
+/// The whole-run signature that agrees with the attributed units.
+///
+/// It reads the space flag the way `granular_text` writes it on every unit,
+/// so a run that differs only by the flag stays whole instead of being
+/// rewritten one unit per run, which would move run-indexed bookmark ends.
+fn attributed_run_signature(run: &CT_R) -> String {
+ let mut run = run.clone();
+ for content in &mut run.content {
+ if let RunContent::Text(text) | RunContent::DeletedText(text) = content {
+ text.preserve_space |= has_edge_whitespace(&text.text);
+ }
+ }
+ run_signature(&run)
+}
+
fn whitespace_fragments(text: &str) -> Vec {
let mut output = Vec::new();
let mut current = String::new();
@@ -3977,8 +4003,22 @@ fn nonempty_paragraph_properties(mut properties: CT_PPr) -> Option {
}
fn paragraph_properties_differ(original: Option<&CT_PPr>, edited: Option<&CT_PPr>) -> bool {
- original.cloned().and_then(nonempty_paragraph_properties)
- != edited.cloned().and_then(nonempty_paragraph_properties)
+ let modeled = |properties: Option<&CT_PPr>| {
+ properties.cloned().and_then(|mut properties| {
+ if let Some(section) = properties.sect_pr.as_mut() {
+ clear_default_orientation(section);
+ }
+ nonempty_paragraph_properties(properties)
+ })
+ };
+ modeled(original) != modeled(edited)
+}
+
+/// Portrait is the schema default, which the section writer omits.
+fn clear_default_orientation(section: &mut rdocx_oxml::document::CT_SectPr) {
+ if section.orientation == Some(ST_PageOrientation::Portrait) {
+ section.orientation = None;
+ }
}
fn section_properties_xml(
@@ -4008,6 +4048,8 @@ fn section_properties_xml(
original_modeled.change = None;
let mut edited_modeled = edited.clone();
edited_modeled.change = None;
+ clear_default_orientation(&mut original_modeled);
+ clear_default_orientation(&mut edited_modeled);
if original_modeled == edited_modeled {
return section_property_xml(original);
}
@@ -5099,17 +5141,20 @@ fn deleted_text_xml(xml: &str) -> String {
}
fn complex_field_result(xml: &str) -> Result<(String, String, String)> {
- let separate = xml
- .find("fldCharType=\"separate\"")
- .or_else(|| xml.find("fldCharType='separate'"))
+ let separate = field_character(xml, 0, "separate")
.ok_or_else(|| Error::Other("complex field source has no separate boundary".to_owned()))?;
- let (_, result_start) = containing_run(xml, separate)?;
- let end_marker = xml[result_start..]
- .find("fldCharType=\"end\"")
- .or_else(|| xml[result_start..].find("fldCharType='end'"))
- .map(|offset| result_start + offset)
+ // A producer may pack a whole field in one run, as Google Docs writes page
+ // fields. Ending a run after `separate` and starting one at `end` reads it
+ // as the same field written one run per part.
+ let xml = split_field_run(xml, separate, true)?;
+ let (_, result_start) = containing_run(&xml, separate)?;
+ let end_marker = field_character(&xml, result_start, "end")
+ .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?;
+ let xml = split_field_run(&xml, end_marker, false)?;
+ // The split may have moved the `end` character further along.
+ let end_marker = field_character(&xml, result_start, "end")
.ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?;
- let (result_end, _) = containing_run(xml, end_marker)?;
+ let (result_end, _) = containing_run(&xml, end_marker)?;
Ok((
xml[..result_start].to_owned(),
xml[result_start..result_end].to_owned(),
@@ -5117,6 +5162,48 @@ fn complex_field_result(xml: &str) -> Result<(String, String, String)> {
))
}
+fn field_character(xml: &str, from: usize, kind: &str) -> Option {
+ let tail = &xml[from..];
+ tail.find(&format!("fldCharType=\"{kind}\""))
+ .or_else(|| tail.find(&format!("fldCharType='{kind}'")))
+ .map(|offset| from + offset)
+}
+
+/// Split the run holding the field character at `marker` so that the
+/// character ends its run (`after`) or starts it.
+///
+/// Both runs repeat the original start tag and run properties. A run with no
+/// content on that side of the character is returned unchanged.
+fn split_field_run(xml: &str, marker: usize, after: bool) -> Result {
+ let (run_start, run_end) = containing_run(xml, marker)?;
+ let run = &xml[run_start..run_end];
+ let children = direct_element_spans(run)?;
+ let character = children
+ .iter()
+ .position(|child| child.contains(&(marker - run_start)))
+ .ok_or_else(|| Error::Other("complex field character is not a run child".to_owned()))?;
+ let is_properties = |child: &Range| {
+ let mut reader = Reader::from_reader(run[child.clone()].as_bytes());
+ matches!(
+ reader.read_event(),
+ Ok(Event::Start(element) | Event::Empty(element))
+ if element.local_name().as_ref() == b"rPr"
+ )
+ };
+ let first_content = usize::from(children.first().is_some_and(is_properties));
+ let split = if after { character + 1 } else { character };
+ if split <= first_content || split >= children.len() {
+ return Ok(xml.to_owned());
+ }
+ let head = &run[..children[first_content].start];
+ let close = run
+ .rfind("")
+ .map(|at| &run[at..])
+ .ok_or_else(|| Error::Other("complex field run has no end tag".to_owned()))?;
+ let at = run_start + children[split].start;
+ Ok(format!("{}{close}{head}{}", &xml[..at], &xml[at..]))
+}
+
fn containing_run(xml: &str, at: usize) -> Result<(usize, usize)> {
let mut reader = Reader::from_reader(xml.as_bytes());
reader.config_mut().trim_text(false);
@@ -5411,10 +5498,29 @@ fn run_content_signature(content: &RunContent) -> String {
match content {
RunContent::Field(_) => "field-owner".to_owned(),
RunContent::Drawing(drawing) => format!("Drawing({:?})", drawing_signature(drawing)),
+ RunContent::Text(text) => format!("Text({:?})", text_signature(text)),
+ RunContent::DeletedText(text) => format!("DeletedText({:?})", text_signature(text)),
content => format!("{content:?}"),
}
}
+/// The text and whether its `xml:space="preserve"` changes how it reads.
+///
+/// The flag only protects whitespace at an edge of the text. Producers write
+/// it on every `w:t` or only where needed, so elsewhere it is serialization.
+fn text_signature(text: &CT_Text) -> (&str, bool) {
+ (
+ &text.text,
+ text.preserve_space && has_edge_whitespace(&text.text),
+ )
+}
+
+/// Whether XML whitespace starts or ends the text.
+fn has_edge_whitespace(text: &str) -> bool {
+ let whitespace = |character: char| matches!(character, ' ' | '\t' | '\n' | '\r');
+ text.starts_with(whitespace) || text.ends_with(whitespace)
+}
+
fn table_signature(table: &CT_Tbl) -> String {
format!(
"{:?}:{:?}:{:?}:{:?}",
@@ -5478,16 +5584,25 @@ fn control_signature(control: &CT_Sdt) -> String {
)
}
+/// The content-control properties that alignment, refusal and the accept and
+/// reject postconditions compare.
+///
+/// `w:id` is left out. Producers renumber it on save and it carries no
+/// content, so a pair that differs only by it keeps the original's `w:sdtPr`.
+/// A `w:sdtPr` with none of these properties reads like no `w:sdtPr`.
fn control_property_signature(control: &CT_Sdt) -> ControlPropertySignature<'_> {
- control.properties.as_ref().map(|properties| {
- (
- properties.alias.as_deref(),
- properties.tag.as_deref(),
- properties.id,
- properties.control_type,
- properties.data_binding.as_ref(),
- )
- })
+ control
+ .properties
+ .as_ref()
+ .map(|properties| {
+ (
+ properties.alias.as_deref(),
+ properties.tag.as_deref(),
+ properties.control_type,
+ properties.data_binding.as_ref(),
+ )
+ })
+ .filter(|signature| !matches!(signature, (None, None, None, None)))
}
fn modeled_control_content(control: &CT_Sdt) -> Vec<&SdtContent> {
@@ -5757,6 +5872,9 @@ fn paragraph_formatting(paragraph: &CT_P) -> Option {
properties.numbering_revision_position = None;
properties.change = None;
properties.revision_xml.clear();
+ if let Some(section) = properties.sect_pr.as_mut() {
+ clear_default_orientation(section);
+ }
properties
})
}
@@ -6287,7 +6405,8 @@ fn utf8_error(error: impl std::fmt::Display) -> Error {
mod tests {
use super::{
ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING,
- attributed_run_units, comparison_postcondition_error, story_document, word_fragments,
+ attributed_run_units, comparison_postcondition_error, complex_field_result, story_document,
+ word_fragments,
};
use crate::Document;
use rdocx_oxml::document::BodyContent;
@@ -6379,6 +6498,36 @@ mod tests {
);
}
+ #[test]
+ fn a_packed_field_result_is_read_as_one_run_per_part() {
+ for w in ["w", "q"] {
+ let shell = format!(r#"<{w}:r {w}:rsidR="00AB12CD"><{w}:rPr><{w}:b/>{w}:rPr>"#);
+ let close = format!("{w}:r>");
+ let character = |kind: &str| format!(r#"<{w}:fldChar {w}:fldCharType="{kind}"/>"#);
+ let (begin, separate, end) =
+ (character("begin"), character("separate"), character("end"));
+ let code = format!("<{w}:instrText>PAGE{w}:instrText>");
+ let result = format!("<{w}:t>1{w}:t>");
+ let expected = (
+ format!("{shell}{begin}{code}{separate}{close}"),
+ format!("{shell}{result}{close}"),
+ format!("{shell}{end}{close}"),
+ );
+ for field in [
+ format!("{shell}{begin}{code}{separate}{result}{end}{close}"),
+ format!("{shell}{begin}{code}{separate}{close}{shell}{result}{end}{close}"),
+ format!("{}{}{}", expected.0, expected.1, expected.2),
+ ] {
+ assert_eq!(complex_field_result(&field).unwrap(), expected, "{field}");
+ }
+ let uncached = format!("{shell}{begin}{code}{separate}{end}{close}");
+ assert_eq!(
+ complex_field_result(&uncached).unwrap(),
+ (expected.0, String::new(), expected.2)
+ );
+ }
+ }
+
#[test]
fn staged_comparison_postcondition_failure_preserves_bytes_and_layout_cache() {
let mut original = Document::new();
diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs
index 5c4b2b46..b4f59a76 100644
--- a/crates/rdocx/src/document.rs
+++ b/crates/rdocx/src/document.rs
@@ -21212,8 +21212,10 @@ impl Document {
/// Replace all occurrences of `placeholder` with `replacement` throughout the document.
///
/// Searches body paragraphs, tables (including nested), headers, footers,
- /// text boxes and chart labels. Handles placeholders split across multiple
- /// runs. Returns the total number of replacements made.
+ /// text boxes and chart labels, and the content controls at every level of
+ /// them, inline controls included. Handles placeholders split across
+ /// multiple runs. A match that straddles a content-control boundary is not
+ /// replaced. Returns the total number of replacements made.
///
/// A `replacement` that contains `placeholder` is substituted once, not
/// repeatedly.
@@ -21284,13 +21286,15 @@ impl Document {
for (rel_id, is_header) in self.header_footer_rel_ids() {
if let Some(header_footer) = self.load_header_footer(&rel_id, is_header) {
- sources.extend(crate::template::header_footer_sources(&header_footer));
+ sources.extend(rdocx_oxml::placeholder::header_footer_replaceable_texts(
+ &header_footer,
+ ));
}
}
for (part_name, _) in self.raw_text_bearing_part_names() {
if let Some(xml) = self.package.get_part(&part_name) {
- sources.extend(crate::template::text_box_sources(xml)?);
+ sources.extend(rdocx_oxml::placeholder::xml_part_replaceable_texts(xml)?);
}
}
@@ -21334,23 +21338,14 @@ impl Document {
Ok(count)
}
- /// Run the typed replacement over body paragraphs and tables.
+ /// Run the typed replacement over every body paragraph, those of tables
+ /// and content controls at every level included.
fn replace_in_body(&mut self, placeholder: &str, replacement: &str) -> usize {
- use rdocx_oxml::placeholder;
-
let mut count = 0;
- for content in &mut self.document.body.content {
- match content {
- BodyContent::Paragraph(p) => {
- count += placeholder::replace_in_paragraph(p, placeholder, replacement);
- }
- BodyContent::Table(t) => {
- count += placeholder::replace_in_table(t, placeholder, replacement);
- }
- BodyContent::ContentControl(_) => {}
- BodyContent::RawXml(_) => {}
- }
- }
+ visit_body_paragraphs_mut(&mut self.document.body.content, &mut |paragraph| {
+ count +=
+ rdocx_oxml::placeholder::replace_in_paragraph(paragraph, placeholder, replacement);
+ });
count
}
@@ -21448,8 +21443,10 @@ impl Document {
/// Replace all regex matches with `replacement` throughout the document.
///
/// The `replacement` string supports capture groups: `$1`, `$2`, etc.
- /// Searches body paragraphs, tables (including nested), headers, and footers.
- /// Returns the total number of replacements made, or an error if the regex is invalid.
+ /// Searches body paragraphs, tables (including nested), headers, footers
+ /// and text boxes, and the content controls at every level of them, as
+ /// [`Self::replace_text`] does. Returns the total number of replacements
+ /// made, or an error if the regex is invalid.
pub fn replace_regex(&mut self, pattern: &str, replacement: &str) -> Result {
let re =
regex::Regex::new(pattern).map_err(|e| Error::Other(format!("invalid regex: {e}")))?;
@@ -21484,19 +21481,10 @@ impl Document {
let mut count = 0;
- // Replace in body paragraphs and tables
- for content in &mut self.document.body.content {
- match content {
- BodyContent::Paragraph(p) => {
- count += placeholder::replace_regex_in_paragraph(p, re, replacement);
- }
- BodyContent::Table(t) => {
- count += placeholder::replace_regex_in_table(t, re, replacement);
- }
- BodyContent::ContentControl(_) => {}
- BodyContent::RawXml(_) => {}
- }
- }
+ // Replace in body paragraphs, tables and content controls
+ visit_body_paragraphs_mut(&mut self.document.body.content, &mut |paragraph| {
+ count += placeholder::replace_regex_in_paragraph(paragraph, re, replacement);
+ });
// Replace in headers and footers
for (rel_id, is_header) in self.header_footer_rel_ids() {
diff --git a/crates/rdocx/src/template.rs b/crates/rdocx/src/template.rs
index 3721169c..cc0dcf80 100644
--- a/crates/rdocx/src/template.rs
+++ b/crates/rdocx/src/template.rs
@@ -6,7 +6,6 @@ use quick_xml::Reader;
use quick_xml::events::Event;
use rdocx_oxml::content_control::{CT_Sdt, SdtContent};
use rdocx_oxml::document::{BodyContent, CT_Document};
-use rdocx_oxml::header_footer::CT_HdrFtr;
use rdocx_oxml::namespace::matches_local_name;
use rdocx_oxml::placeholder;
use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_Tc, CellContent};
@@ -576,7 +575,9 @@ fn render_nested_tables_in_control(
fn body_marker(content: &BodyContent) -> Result"),
+ (HEADER, ">Header control "),
+ (FOOTER, ">Footer control "),
+ ] {
+ assert!(parts[part].contains(text), "{text}\n{}", parts[part]);
+ }
+ }
+
+ #[test]
+ fn every_walker_replaces_the_tag_of_every_location_once() {
+ for (name, _) in LOCATIONS {
+ let (tag, value) = (tag(name), value(name));
+ for walker in ["try_replace_text", "replace_regex", "replace_all"] {
+ let mut document = document();
+ let count = match walker {
+ "try_replace_text" => document.try_replace_text(&tag, &value).unwrap(),
+ "replace_regex" => document
+ .replace_regex(®ex::escape(&tag), &value)
+ .unwrap(),
+ _ => document.replace_all(&HashMap::from([(tag.as_str(), value.as_str())])),
+ };
+ assert_eq!(count, 1, "{walker} at {name}");
+ assert_replaced(&saved_parts(&mut document), &[name]);
+ }
+ }
+ }
+
+ #[test]
+ fn a_template_renders_the_tag_of_every_location() {
+ let mut document = document();
+ let data = LOCATIONS
+ .iter()
+ .map(|(name, _)| ((*name).to_owned(), serde_json::Value::from(value(name))))
+ .collect::>();
+
+ let count = document
+ .render_template(&serde_json::Value::Object(data))
+ .unwrap();
+
+ assert_eq!(count, LOCATIONS.len());
+ let names = LOCATIONS.map(|(name, _)| name);
+ assert_replaced(&saved_parts(&mut document), &names);
+ }
+
+ /// A match that straddles a control boundary is no match. A reader sees
+ /// "alpha one" in the first paragraph and "alXpha two" in the second,
+ /// and neither changes, while the match inside the control of the third
+ /// is replaced. Direct runs on both sides of a control used to be read as
+ /// one text, which matched "alpha" in the second paragraph.
+ #[test]
+ fn a_match_that_straddles_a_control_boundary_is_not_replaced() {
+ let body = [
+ paragraph(&[run("al"), control("a", &run("pha one"))].concat()),
+ paragraph(&[run("al"), control("b", &run("X")), run("pha two")].concat()),
+ paragraph(&[run("al"), control("c", &run("alpha three"))].concat()),
+ ]
+ .concat();
+ for regex in [false, true] {
+ let mut document = super::document_with_content_controls(&super::wrap_word_body(&body));
+ let count = if regex {
+ document.replace_regex("alpha", "ALPHA").unwrap()
+ } else {
+ document.try_replace_text("alpha", "ALPHA").unwrap()
+ };
+ assert_eq!(count, 1, "regex: {regex}");
+ let texts = document
+ .paragraphs()
+ .iter()
+ .map(|paragraph| paragraph.text())
+ .collect::>();
+ assert_eq!(texts, ["alpha one", "alXpha two", "alALPHA three"]);
+ let xml = super::document_xml(&mut document);
+ assert_eq!(xml.matches(">al").count(), 3, "{xml}");
+ }
+ }
+
+ /// A quantified pattern also matches the digits after the boundary
+ /// alone, and replaced them, which left the number a reader sees half
+ /// replaced. The search now goes on after the straddling match, and
+ /// still reaches the number after the control.
+ #[test]
+ fn a_quantified_match_that_straddles_a_control_boundary_is_not_replaced() {
+ let body = paragraph(&[run("Order 12"), control("a", &run("34 end")), run(" 56")].concat());
+ for pattern in [r"\d+", r"\d{2,}"] {
+ let mut document = super::document_with_content_controls(&super::wrap_word_body(&body));
+
+ assert_eq!(
+ document.replace_regex(pattern, "N").unwrap(),
+ 1,
+ "{pattern}"
+ );
+
+ assert_eq!(
+ document.paragraph(0).unwrap().text(),
+ "Order 1234 end N",
+ "{pattern}"
+ );
+ }
+ }
+
+ /// Word writes its table of contents as a body-level control whose
+ /// entries repeat the heading text. Replacement reaches them as it does
+ /// any other control, so a heading word counts once more for its entry,
+ /// and the entry still reads as its heading until the next update.
+ #[test]
+ fn a_table_of_contents_entry_is_replaced_with_its_heading() {
+ let toc = r#" TOC \o "1-3" \h \z \u Introduction PAGEREF _Toc1 \h 1"#;
+ let heading = r#"Introduction"#;
+ let mut document =
+ super::document_with_content_controls(&super::wrap_word_body(&[toc, heading].concat()));
+
+ assert_eq!(
+ document
+ .try_replace_text("Introduction", "Overview")
+ .unwrap(),
+ 2
+ );
+
+ let xml = super::document_xml(&mut document);
+ assert_eq!(xml.matches(">Overview<").count(), 2, "{xml}");
+ assert!(xml.contains(r#"{text}"#
+ )
+ };
+ let on_cell = table("", w14, "cell {{x}}");
+ let tables = [table(w14, "", "table {{x}}"), on_cell.clone()].concat();
+ let mut header = super::document_with_header_story(&format!(
+ r#"{tables}"#
+ ));
+ let mut text_box = super::document_with_content_controls(&format!(
+ r#"{tables}"#
+ ));
+
+ assert_eq!(header.try_replace_text("{{x}}", "Y").unwrap(), 1);
+ assert_eq!(text_box.try_replace_text("{{x}}", "Y").unwrap(), 1);
+
+ for xml in [
+ super::header_story_xml(&mut header),
+ super::document_xml(&mut text_box),
+ ] {
+ // Every element and attribute prefix resolves.
+ let mut reader = quick_xml::NsReader::from_str(&xml);
+ loop {
+ let (namespace, event) = reader.read_resolved_event().unwrap();
+ assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}");
+ match event {
+ XmlEvent::Start(element) | XmlEvent::Empty(element) => {
+ for attribute in element.attributes() {
+ let key = attribute.unwrap().key;
+ let (namespace, _) = reader.resolver().resolve_attribute(key);
+ assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}");
+ }
+ }
+ XmlEvent::Eof => break,
+ _ => {}
+ }
+ }
+ assert!(xml.contains(">table Y<"), "{xml}");
+ assert!(xml.contains(&on_cell), "{xml}");
+ }
+ }
+
+ /// A template tag that a control boundary splits cannot be rendered
+ /// whole, so the template is rejected rather than left half rendered.
+ #[test]
+ fn a_template_tag_that_straddles_a_control_boundary_is_an_error() {
+ let body = paragraph(&[run("{{na"), control("a", &run("me}}"))].concat());
+ let mut document = super::document_with_content_controls(&super::wrap_word_body(&body));
+ let before = document.to_bytes().unwrap();
+
+ let error = document
+ .render_template(&serde_json::json!({"name": "Ada"}))
+ .unwrap_err();
+
+ assert!(error.to_string().contains("invalid template"), "{error}");
+ assert_eq!(document.to_bytes().unwrap(), before);
+ }
+
+ /// The two shapes of the report: a Google Docs export wraps a run in a
+ /// control inside its paragraph, or a whole paragraph at body level.
+ /// Both replaced nothing. The text the paragraph reads is the text the
+ /// replacement changed.
+ #[test]
+ fn the_reported_google_docs_controls_are_replaced() {
+ let text = run("Body text, lorem alpha dolor.");
+ for body in [
+ paragraph(&text),
+ paragraph(&control("goog_rdk_0", &text)),
+ control("goog_rdk_1", ¶graph(&text)),
+ ] {
+ let mut document = super::document_with_content_controls(&super::wrap_word_body(&body));
+ assert_eq!(
+ document.paragraph(0).unwrap().text(),
+ "Body text, lorem alpha dolor."
+ );
+
+ assert_eq!(document.try_replace_text("alpha", "ALPHA").unwrap(), 1);
+
+ assert_eq!(
+ document.paragraph(0).unwrap().text(),
+ "Body text, lorem ALPHA dolor."
+ );
+ let xml = super::document_xml(&mut document);
+ assert!(xml.contains("Body text, lorem ALPHA dolor."), "{xml}");
+ assert_eq!(xml.contains("goog_rdk"), body.contains("goog_rdk"), "{xml}");
+ }
+ }
+}
+
/// `9360 / cols` panicked when a caller asked for a zero-column table.
#[test]
fn zero_column_tables_do_not_panic() {
@@ -30379,6 +30939,440 @@ fn comparison_treats_empty_paragraph_properties_as_absent() {
assert!(document_xml(&mut empty).contains(" Vec {
+ document
+ .revisions()
+ .iter()
+ .map(|revision| revision.kind())
+ .collect()
+ }
+
+ /// Check that accepting gives the edited side and rejecting the original,
+ /// each compared again with no diagnostic and no revision.
+ fn assert_resolutions(
+ tracked: &[u8],
+ original: &Document,
+ edited: &Document,
+ options: &ComparisonOptions,
+ ) {
+ for (resolve, expected) in [
+ (
+ Document::accept_all as fn(&mut Document) -> rdocx::Result,
+ edited,
+ ),
+ (Document::reject_all, original),
+ ] {
+ let mut resolved = Document::from_bytes(tracked).unwrap();
+ resolve(&mut resolved).unwrap();
+ let diagnostics = resolved
+ .compare_with_options(expected, "postcondition", TIMESTAMP, options)
+ .unwrap();
+ assert!(diagnostics.is_empty(), "{diagnostics:?}");
+ assert_eq!(revision_kinds(&resolved), []);
+ }
+ }
+
+ /// Compare two bodies and return the revision kinds and the redline.
+ fn compared_kinds(original_xml: &str, edited_xml: &str) -> (Vec, String) {
+ compared_kinds_with(original_xml, edited_xml, &ComparisonOptions::default())
+ }
+
+ fn compared_kinds_with(
+ original_xml: &str,
+ edited_xml: &str,
+ options: &ComparisonOptions,
+ ) -> (Vec, String) {
+ let original = document_with_content_controls(original_xml);
+ let edited = document_with_content_controls(edited_xml);
+ let mut compared = document_with_content_controls(original_xml);
+ let diagnostics = compared
+ .compare_with_options(&edited, "R", TIMESTAMP, options)
+ .expect("producer noise must not refuse the pair");
+ assert!(diagnostics.is_empty(), "{diagnostics:?}");
+ assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited, options);
+ (revision_kinds(&compared), document_xml(&mut compared))
+ }
+
+ fn table_of_contents_control(id: Option<&str>, first_entry: &str) -> String {
+ let id = id.map_or_else(String::new, |value| format!(r#""#));
+ wrap_word_body(&format!(
+ r#"Before the content control.{id}{first_entry} entryBeta entryGamma entryAfter the content control."#
+ ))
+ }
+
+ fn google_docs_inline_control(id: &str, word: &str) -> String {
+ wrap_word_body(&format!(
+ r#"Before {word} after."#
+ ))
+ }
+
+ #[test]
+ fn a_content_control_identity_is_not_content() {
+ for (original, edited, kept_id) in [
+ (None, Some("-2000000001"), None),
+ (Some("-2000000001"), None, Some("-2000000001")),
+ (Some("11"), Some("12"), Some("11")),
+ ] {
+ let (kinds, tracked) = compared_kinds(
+ &table_of_contents_control(original, "Alpha"),
+ &table_of_contents_control(edited, "Alpha"),
+ );
+ assert_eq!(kinds, [], "{original:?} -> {edited:?}");
+ let ids = [original, edited]
+ .into_iter()
+ .flatten()
+ .filter(|id| tracked.contains(&format!(r#""#)))
+ .collect::>();
+ assert_eq!(ids, kept_id.into_iter().collect::>(), "{tracked}");
+
+ let (kinds, tracked) = compared_kinds(
+ &table_of_contents_control(original, "Alpha"),
+ &table_of_contents_control(edited, "Delta"),
+ );
+ assert_eq!(
+ kinds,
+ [RevisionKind::Deletion, RevisionKind::Insertion],
+ "{original:?} -> {edited:?}"
+ );
+ assert_eq!(tracked.matches("").count(), 1, "{tracked}");
+ }
+
+ let (kinds, _) = compared_kinds(
+ &google_docs_inline_control("-1854911024", "Alpha"),
+ &google_docs_inline_control("1374263513", "Alpha"),
+ );
+ assert_eq!(kinds, []);
+ let (kinds, tracked) = compared_kinds(
+ &google_docs_inline_control("-1854911024", "Alpha"),
+ &google_docs_inline_control("1374263513", "Delta"),
+ );
+ assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]);
+ assert!(
+ tracked.contains(r#""#),
+ "{tracked}"
+ );
+
+ let bare_control = |properties: &str| {
+ wrap_word_body(&format!(
+ r#"{properties}Alpha entryAfter the content control."#
+ ))
+ };
+ let id_only = r#""#;
+ for (original, edited) in [("", id_only), (id_only, "")] {
+ let (kinds, tracked) = compared_kinds(&bare_control(original), &bare_control(edited));
+ assert_eq!(kinds, [], "{original:?} -> {edited:?}");
+ assert_eq!(
+ tracked.contains(""),
+ !original.is_empty(),
+ "{tracked}"
+ );
+ }
+ }
+
+ fn replaced_copy(source_xml: &str, old: &str, new: &str) -> String {
+ let mut document = document_with_content_controls(source_xml);
+ assert_eq!(document.try_replace_text(old, new).unwrap(), 1);
+ let mut edited = Document::from_bytes(&document.to_bytes().unwrap()).unwrap();
+ document_xml(&mut edited)
+ }
+
+ #[test]
+ fn a_rewritten_run_keeps_its_producer_space_flag() {
+ let preserved = wrap_word_body(
+ r#"Paragraph 1.WORD"#,
+ );
+ let edited = replaced_copy(&preserved, "WORD", "WORD");
+ assert!(
+ edited.contains(r#"WORD"#),
+ "{edited}"
+ );
+ let (kinds, _) = compared_kinds(&preserved, &edited);
+ assert_eq!(kinds, []);
+
+ let edge_space_removed = replaced_copy(
+ &wrap_word_body(r#"WORD "#),
+ "WORD ",
+ "WORD",
+ );
+ assert!(
+ edge_space_removed.contains(r#"WORD"#),
+ "{edge_space_removed}"
+ );
+ }
+
+ #[test]
+ fn the_space_flag_is_content_only_at_a_text_edge() {
+ let paragraphs = |first: &str, second: &str| {
+ wrap_word_body(&format!(
+ r#"{first}{second}"#
+ ))
+ };
+ let paragraph = |text: &str| paragraphs("Paragraph 1.", text);
+ // Bookmark ends are indexed by run, so an unchanged run must stay whole.
+ let bookmarked = |text: &str| {
+ wrap_word_body(&format!(
+ r#"{text}tail"#
+ ))
+ };
+ let changed = &[RevisionKind::Deletion, RevisionKind::Insertion][..];
+ let unchanged = &[][..];
+ // Each case gives the kinds of the whole-run path and of the
+ // attributed path. The attributed path writes every unit with edge
+ // whitespace with the flag, so the flag is never content there.
+ let cases = [
+ (
+ paragraph(r#"WORD"#),
+ paragraph("WORD"),
+ unchanged,
+ unchanged,
+ ),
+ (
+ paragraph(r#"two words"#),
+ paragraph("two words"),
+ unchanged,
+ unchanged,
+ ),
+ (
+ paragraph("two words"),
+ paragraph(r#"two words"#),
+ unchanged,
+ unchanged,
+ ),
+ (
+ paragraph(r#"WORD "#),
+ paragraph("WORD "),
+ changed,
+ unchanged,
+ ),
+ (
+ bookmarked(r#"two words "#),
+ bookmarked("two words "),
+ changed,
+ unchanged,
+ ),
+ (
+ paragraph("WORD"),
+ paragraph("text"),
+ changed,
+ changed,
+ ),
+ // Google Docs flags every `w:t` and Word only where needed.
+ (
+ paragraphs(
+ r#"two words"#,
+ r#"WORD"#,
+ ),
+ paragraphs("two words", "text"),
+ changed,
+ changed,
+ ),
+ ];
+ for options in [
+ ComparisonOptions::default(),
+ ComparisonOptions {
+ ignore_formatting: true,
+ ..Default::default()
+ },
+ ComparisonOptions {
+ granularity: ComparisonGranularity::Word,
+ ..Default::default()
+ },
+ ComparisonOptions {
+ granularity: ComparisonGranularity::Character,
+ ..Default::default()
+ },
+ ] {
+ for (original, edited, whole_run, attributed) in &cases {
+ let expected = if options == ComparisonOptions::default() {
+ whole_run
+ } else {
+ attributed
+ };
+ for (left, right) in [(original, edited), (edited, original)] {
+ let (kinds, _) = compared_kinds_with(left, right, &options);
+ assert_eq!(kinds, *expected, "{options:?}: {left} -> {right}");
+ }
+ }
+ }
+
+ // A space split out of a text keeps reading as a space in the redline.
+ for granularity in [
+ ComparisonGranularity::Word,
+ ComparisonGranularity::Character,
+ ] {
+ let (kinds, tracked) = compared_kinds_with(
+ ¶graph("two words"),
+ ¶graph("two wordy"),
+ &ComparisonOptions {
+ granularity,
+ ..Default::default()
+ },
+ );
+ assert_eq!(kinds, changed, "{granularity:?}");
+ assert!(!tracked.contains(""), "{tracked}");
+ assert!(
+ tracked.contains(r#""#),
+ "{tracked}"
+ );
+ }
+ }
+
+ fn page_body(orientation: &str) -> String {
+ wrap_word_body(&format!(
+ r#"Paragraph 1, lorem ipsum dolor sit amet.WORDParagraph 3, lorem ipsum dolor sit amet."#
+ ))
+ }
+
+ #[test]
+ fn the_default_page_orientation_is_not_a_section_change() {
+ let portrait = page_body(r#" w:orient="portrait""#);
+ let edited = replaced_copy(&portrait, "3, lorem", "3, LOREM");
+ assert!(!edited.contains("w:orient"), "{edited}");
+ let (kinds, _) = compared_kinds(&portrait, &edited);
+ assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]);
+
+ let (kinds, _) = compared_kinds(&portrait, &page_body(""));
+ assert_eq!(kinds, []);
+ let (kinds, _) = compared_kinds(&page_body(""), &portrait);
+ assert_eq!(kinds, []);
+
+ let (kinds, tracked) = compared_kinds(&portrait, &page_body(r#" w:orient="landscape""#));
+ assert_eq!(kinds, [RevisionKind::SectionPropertyChange]);
+ assert!(tracked.contains(r#"w:orient="landscape""#), "{tracked}");
+
+ // A bookmark makes the section-break paragraph take the complex path,
+ // as Google Docs writes around headings.
+ let bookmarked_break = |orientation: &str, heading: &str| {
+ wrap_word_body(&format!(
+ r#"{heading}Second section."#
+ ))
+ };
+ let portrait = r#" w:orient="portrait""#;
+ for (original, edited) in [(portrait, ""), ("", portrait)] {
+ let (kinds, _) = compared_kinds(
+ &bookmarked_break(original, "Heading"),
+ &bookmarked_break(edited, "Heading"),
+ );
+ assert_eq!(kinds, [], "{original:?} -> {edited:?}");
+ let (kinds, _) = compared_kinds(
+ &bookmarked_break(original, "Heading"),
+ &bookmarked_break(edited, "Title"),
+ );
+ assert_eq!(
+ kinds,
+ [RevisionKind::Deletion, RevisionKind::Insertion],
+ "{original:?} -> {edited:?}"
+ );
+ }
+ }
+
+ fn document_with_footer(footer_paragraph: &str) -> Document {
+ let mut seed = Document::new();
+ let mut package =
+ oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap()))
+ .unwrap();
+ package.set_part(
+ "/word/footer1.xml",
+ format!(r#"{footer_paragraph}"#).into_bytes(),
+ );
+ package.content_types.add_override(
+ "/word/footer1.xml",
+ "application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml",
+ );
+ let footer_id = package
+ .get_or_create_part_rels("/word/document.xml")
+ .add(oxml_opc::relationship::rel_types::FOOTER, "footer1.xml");
+ package.set_part(
+ "/word/document.xml",
+ format!(
+ r#"Lorem ipsum dolor sit amet."#
+ )
+ .into_bytes(),
+ );
+ let mut bytes = std::io::Cursor::new(Vec::new());
+ package.write_to(&mut bytes).unwrap();
+ Document::from_bytes(bytes.get_ref()).unwrap()
+ }
+
+ fn page_field(packed: bool, cached: bool) -> String {
+ let parts = [
+ r#""#,
+ r#" PAGE "#,
+ r#""#,
+ if cached { "1" } else { "" },
+ r#""#,
+ ];
+ let field = if packed {
+ format!("{}", parts.concat())
+ } else {
+ parts
+ .iter()
+ .filter(|part| !part.is_empty())
+ .map(|part| format!("{part}"))
+ .collect()
+ };
+ format!(r#"Page {field}"#)
+ }
+
+ /// Compare a footer field against its copy refreshed by `update_page_fields`.
+ fn refreshed_field_comparison(packed: bool, cached: bool) -> String {
+ let source = document_with_footer(&page_field(packed, cached))
+ .to_bytes()
+ .unwrap();
+ let mut refreshed = Document::from_bytes(&source).unwrap();
+ refreshed.update_page_fields().unwrap();
+ let refreshed = Document::from_bytes(&refreshed.to_bytes().unwrap()).unwrap();
+ let mut compared = Document::from_bytes(&source).unwrap();
+ let diagnostics = compared
+ .compare(&refreshed, "R", TIMESTAMP)
+ .unwrap_or_else(|error| panic!("packed={packed} cached={cached}: {error}"));
+ assert!(diagnostics.is_empty(), "{diagnostics:?}");
+ assert_resolutions(
+ &compared.to_bytes().unwrap(),
+ &Document::from_bytes(&source).unwrap(),
+ &refreshed,
+ &ComparisonOptions::default(),
+ );
+ comparison_part_xml(&mut compared, "/word/footer1.xml")
+ }
+
+ #[test]
+ fn a_packed_field_compares_like_the_same_field_split_into_runs() {
+ let separate = r#""#;
+ let result_and_end = |footer: &str| footer[footer.find(separate).unwrap()..].to_owned();
+ for (cached, deleted) in [(true, "1"), (false, "")] {
+ let split = refreshed_field_comparison(false, cached);
+ let packed = refreshed_field_comparison(true, cached);
+ assert_eq!(
+ result_and_end(&split),
+ format!(
+ r#"{separate}{deleted}1"#
+ ),
+ "cached={cached}"
+ );
+ assert_eq!(
+ result_and_end(&packed),
+ result_and_end(&split),
+ "cached={cached}"
+ );
+ assert!(
+ packed.contains(r#" PAGE "#),
+ "{packed}"
+ );
+ }
+ }
+}
+
#[test]
fn unmodelled_property_changes_report_a_diagnostic() {
for (original, edited, expected_location) in [
diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md
index 97ce19c6..ca07b49e 100644
--- a/docs/hld/03-architecture.md
+++ b/docs/hld/03-architecture.md
@@ -910,9 +910,12 @@ section properties emit property revisions that retain the original property
sidecars. Unsupported formatting differences retain the original bytes and
produce stable `ComparisonDiagnostic` values at the actual story path. Inputs
with existing modeled revisions or differing story and control shells are
-rejected unless their story category is ignored. Attributed text alignment
-retains owner, formatting, content position, and raw-child boundaries, then
-coalesces adjacent equal-owner edits into minimal revision wrappers.
+rejected unless their story category is ignored. A content control's `w:id`
+is producer identity and not part of its shell, so controls that differ only
+by it align, compare, and keep the original `w:sdtPr`. Attributed text
+alignment retains owner, formatting, content position, and raw-child
+boundaries, then coalesces adjacent equal-owner edits into minimal revision
+wrappers.
When a main story gains a trailing run of paragraphs, comparison marks the
original final paragraph boundary once, marks each intermediate inserted
paragraph boundary once, and leaves the final inserted paragraph mark as the
diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py
index e24c643b..0dabe703 100644
--- a/scripts/readme_doctests.py
+++ b/scripts/readme_doctests.py
@@ -383,12 +383,12 @@ class ReadmeCase:
"oxml-opc": (92_122, 355_510, 12),
"oxml-pdf": (66_015, 304_432, 14),
"oxml-sml": (12_511, 49_803, 6),
- "rdocx": (1_092_256, 6_498_484, 36),
- "rdocx-cli": (33_805, 145_256, 8),
+ "rdocx": (1_104_100, 6_548_342, 36),
+ "rdocx-cli": (35_147, 149_734, 8),
"rdocx-html": (15_486, 63_894, 11),
"rdocx-layout": (255_752, 1_385_701, 15),
"rdocx-opc": (3_655, 9_668, 6),
- "rdocx-oxml": (367_500, 2_380_047, 32),
+ "rdocx-oxml": (375_767, 2_412_318, 32),
"rdocx-pdf": (8_111, 26_758, 6),
"rpptx": (407_658, 2_122_094, 16),
"rpptx-chart": (6_648, 21_136, 6),