diff --git a/README.md b/README.md index 479b462e..2b4dd87d 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,104,100 compressed bytes, 6,548,342 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-cli/README.md b/crates/rdocx-cli/README.md index 25ecc0fe..6b778c8a 100644 --- a/crates/rdocx-cli/README.md +++ b/crates/rdocx-cli/README.md @@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-cli | 35,147 compressed bytes, 149,734 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs index e4cdd98f..28e194b2 100644 --- a/crates/rdocx-cli/tests/integration.rs +++ b/crates/rdocx-cli/tests/integration.rs @@ -343,6 +343,122 @@ fn cli_replace_reports_namespace_preflight_errors_without_panicking() { assert!(!output_path.exists()); } +/// `rdocx replace --expect 1` found none of the text that Google Docs and +/// Word keep in content controls: a run wrapped inside its paragraph, a +/// paragraph wrapped at body level, a control in a table cell, nested +/// controls, a control in a text box, and the table and controls of a +/// header or footer. +#[test] +fn replace_with_expect_counts_the_text_of_content_controls_everywhere() { + let temp = TempWorkspace::new("replace-content-controls"); + let input = temp.path.join("controls.docx"); + write_document(&input, &["seed"]); + + let control = |tag: &str, content: &str| { + format!( + r#"{content}"# + ) + }; + let run = |text: &str| format!("{text}"); + let paragraph = |content: &str| format!("{content}"); + let table = |cell: &str| { + format!( + r#"{cell}"# + ) + }; + let text_box = format!( + r#"{}"#, + control("box", ¶graph(&run("{{box}}"))) + ); + let body = [ + paragraph(&[run("Body "), control("goog_rdk_0", &run("{{inline}}"))].concat()), + control("goog_rdk_1", ¶graph(&run("{{block}}"))), + table(&control("cell", ¶graph(&run("{{cell}}")))), + control("outer", ¶graph(&control("inner", &run("{{nested}}")))), + paragraph(&[run("Host"), text_box].concat()), + ] + .concat(); + let word = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"; + let header = format!( + r#"{}{}"#, + table(¶graph(&run("{{header_table}}"))), + control("header", ¶graph(&run("{{header_control}}"))) + ); + let footer = format!( + r#"{}{}"#, + control("page", ¶graph(&run("{{footer_control}}"))), + paragraph(&run("Confidential")) + ); + + let mut package = + OpcPackage::from_reader(std::io::Cursor::new(fs::read(&input).unwrap())).unwrap(); + let mut references = String::new(); + for (kind, xml, rel_type) in [ + ("header", header, rel_types::HEADER), + ("footer", footer, rel_types::FOOTER), + ] { + let part = format!("/word/{kind}1.xml"); + package.set_part(&part, xml.into_bytes()); + package.content_types.add_override( + &part, + &format!("application/vnd.openxmlformats-officedocument.wordprocessingml.{kind}+xml"), + ); + let id = package + .get_or_create_part_rels("/word/document.xml") + .add(rel_type, &format!("{kind}1.xml")); + references.push_str(&format!( + r#""# + )); + } + package.set_part( + "/word/document.xml", + format!( + r#"{body}{references}"# + ) + .into_bytes(), + ); + package + .write_to(&mut fs::File::create(&input).unwrap()) + .unwrap(); + + for name in [ + "inline", + "block", + "cell", + "nested", + "box", + "header_table", + "header_control", + "footer_control", + ] { + let tag = format!("{{{{{name}}}}}"); + let replaced = temp.path.join(format!("{name}.docx")); + let output = cli(&[ + "replace", + path_text(&input), + "--placeholder", + &tag, + "--value", + "done", + "--expect", + "1", + "--output", + path_text(&replaced), + ]); + assert_success(&output, name); + + let package = OpcPackage::open(&replaced).unwrap(); + let saved = ["document", "header1", "footer1"] + .map(|part| { + let xml = package.get_part(&format!("/word/{part}.xml")).unwrap(); + String::from_utf8(xml.to_vec()).unwrap() + }) + .concat(); + assert!(!saved.contains(&tag), "{name}: {saved}"); + assert_eq!(saved.matches(">done<").count(), 1, "{name}: {saved}"); + } +} + #[test] fn validate_exit_status_is_a_verdict() { let temp = TempWorkspace::new("validate"); diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef241..261b1508 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 375,767 compressed bytes, 2,412,318 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/crates/rdocx-oxml/src/content_control.rs b/crates/rdocx-oxml/src/content_control.rs index 8ff31cd5..711f3e5e 100644 --- a/crates/rdocx-oxml/src/content_control.rs +++ b/crates/rdocx-oxml/src/content_control.rs @@ -806,6 +806,27 @@ impl CT_Sdt { } } + /// Remove one content child, keeping the revisions and the source bytes + /// of the later runs at their content index. + pub(crate) fn remove_content(&mut self, index: usize) { + if index >= self.content.len() { + return; + } + self.content.remove(index); + for (boundary, _) in &mut self.revisions { + if *boundary > index { + *boundary -= 1; + } + } + self.inline_run_sources + .retain(|source| source.content_index != index); + for source in &mut self.inline_run_sources { + if source.content_index > index { + source.content_index -= 1; + } + } + } + pub(crate) fn word_prefixes(&self) -> &[String] { &self.word_prefixes } diff --git a/crates/rdocx-oxml/src/header_footer.rs b/crates/rdocx-oxml/src/header_footer.rs index 2ec2cb1b..424798b2 100644 --- a/crates/rdocx-oxml/src/header_footer.rs +++ b/crates/rdocx-oxml/src/header_footer.rs @@ -196,6 +196,13 @@ pub struct CT_HdrFtr { pub extra_namespaces: Vec<(String, String)>, /// Unknown child elements captured as raw XML. pub extra_xml: Vec>, + /// How many paragraphs precede each entry of `extra_xml`, so that a + /// rewrite puts a table or a content control back where it was. An entry + /// without a position is written after the last paragraph. + extra_xml_positions: Vec, + /// The namespace bindings of the root element, which a raw child is + /// parsed with when replacement reaches into it. + pub(crate) word_prefixes: Vec, } #[allow(non_snake_case)] @@ -206,6 +213,8 @@ impl CT_HdrFtr { watermarks: Vec::new(), extra_namespaces: Vec::new(), extra_xml: Vec::new(), + extra_xml_positions: Vec::new(), + word_prefixes: vec!["w".to_owned()], } } @@ -227,11 +236,16 @@ impl CT_HdrFtr { pub fn from_xml(xml: &[u8]) -> Result { let watermarks = parse_vml_watermarks(xml); let mut reader = Reader::from_reader(xml); - reader.config_mut().trim_text(true); + // A raw child is captured with the text events of this reader, so + // trimming them would drop the edge spaces of its text, such as the + // one of "Page " before a page number. The text between the children + // of the root is skipped below. + reader.config_mut().trim_text(false); let mut paragraphs = Vec::new(); let mut extra_namespaces = Vec::new(); let mut extra_xml = Vec::new(); + let mut extra_xml_positions = Vec::new(); let mut buf = Vec::new(); let mut word_prefixes = Vec::new(); @@ -267,6 +281,7 @@ impl CT_HdrFtr { } else { // Capture unknown elements as raw XML extra_xml.push(capture_element(&mut reader, e)?); + extra_xml_positions.push(paragraphs.len()); } } Ok(Event::Empty(ref e)) => { @@ -278,6 +293,7 @@ impl CT_HdrFtr { && !matches_local_name(name.as_ref(), b"ftr") { extra_xml.push(capture_empty_element(e)?); + extra_xml_positions.push(paragraphs.len()); } } Ok(Event::Eof) => break, @@ -292,6 +308,8 @@ impl CT_HdrFtr { watermarks, extra_namespaces, extra_xml, + extra_xml_positions, + word_prefixes, }) } @@ -341,12 +359,24 @@ impl CT_HdrFtr { writer.write_event(Event::Start(start))?; - for p in &self.paragraphs { + // Write each captured unknown element before the paragraph it + // preceded, and the rest after the last paragraph. + let mut raw_children = self + .extra_xml + .iter() + .enumerate() + .map(|(index, raw)| { + let position = self.extra_xml_positions.get(index).copied(); + (position.unwrap_or(usize::MAX), raw) + }) + .peekable(); + for (index, p) in self.paragraphs.iter().enumerate() { + while let Some((_, raw)) = raw_children.next_if(|(position, _)| *position <= index) { + writer.get_mut().extend_from_slice(raw); + } p.to_xml(&mut writer)?; } - - // Write captured unknown elements - for raw in &self.extra_xml { + for (_, raw) in raw_children { writer.get_mut().extend_from_slice(raw); } @@ -1219,6 +1249,45 @@ mod tests { assert_eq!(parsed.paragraphs.len(), 0); } + /// A rewrite wrote every table and content control after the last + /// paragraph. A raw child a caller adds still goes after it. + #[test] + fn raw_children_keep_their_place_between_paragraphs() { + let xml = format!( + r#"onetwo"# + ); + let mut parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + parsed + .extra_xml + .push(br#""#.to_vec()); + + let written = String::from_utf8(parsed.to_xml_header().unwrap()).unwrap(); + + let positions = [ + "", + ">one<", + "", + ">two<", + "Page "#; + let xml = format!("\n {control}\n \n"); + + let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + + assert_eq!(parsed.extra_xml, [control.as_bytes()]); + assert_eq!(parsed.paragraphs.len(), 1); + } + #[test] fn aliased_header_paragraph_properties_keep_root_scope() { let xml = format!( diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 0bf078dd..14f4a6d0 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -2,20 +2,72 @@ //! //! Handles the cross-run splitting problem: a placeholder like `{{name}}` //! may be split across multiple `` elements in the OOXML source. - +//! +//! A replacement reads the direct runs of a paragraph and the runs of its +//! inline content controls, in document order, and searches their text left +//! to right as one. A match must lie within one stretch of those runs, see +//! [`replaceable_texts`]. A match that straddles a content-control boundary +//! is not replaced, and the search goes on after it, so no part of the text +//! it covers is replaced either. + +use crate::content_control::{CT_Sdt, SdtContent}; use crate::header_footer::CT_HdrFtr; -use crate::table::CT_Tbl; -use crate::text::{CT_P, CT_R, RunContent}; +use crate::namespace::W_NS; +use crate::numbering::{local_namespace_overrides, namespace_bindings, word_prefixes_at}; +use crate::properties::is_word_element; +use crate::table::{CT_Row, CT_Tbl, CT_Tc, CellContent}; +use crate::text::{ + AcceptedRunPath, AcceptedRunPathSegment, CT_P, CT_R, RunContent, raw_with_external_bindings, +}; /// Replace all occurrences of `placeholder` with `replacement` in a paragraph. /// -/// Handles placeholders split across multiple runs. Preserves the formatting -/// of the first matched run. Returns the number of replacements made. +/// Handles placeholders split across multiple runs and reaches the runs of +/// inline content controls. A match that straddles a content-control +/// boundary is not replaced. Preserves the formatting of the first matched +/// run. Returns the number of replacements made. pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &str) -> usize { if placeholder.is_empty() { return 0; } + replace_matches(para, &mut |text, from| { + let start = from + text[from..].find(placeholder)?; + Some((start, start + placeholder.len(), replacement.to_owned())) + }) +} +/// The texts a replacement in `para` matches against, one per stretch of +/// runs. The direct runs between two inline content controls form one +/// stretch, and so do the runs of one control between its nested controls. +/// A match never spans two stretches. +#[doc(hidden)] +pub fn replaceable_texts(para: &CT_P) -> Vec { + let mut texts: Vec = Vec::new(); + let mut previous = None; + for (stretch, path) in text_runs(para) { + if previous != Some(stretch) { + texts.push(String::new()); + previous = Some(stretch); + } + if let (Some(text), Some(run)) = (texts.last_mut(), para.accepted_run(&path)) { + text.extend(run.content.iter().filter_map(|content| match content { + RunContent::Text(t) => Some(t.text.as_str()), + _ => None, + })); + } + } + texts +} + +/// Takes a text and the byte offset to search from, and returns the byte +/// range of the next match and the text that replaces it. +type MatchFinder<'a> = dyn FnMut(&str, usize) -> Option<(usize, usize, String)> + 'a; + +/// Find each match that `next_match` reports in the text of `para` and +/// replace it. +fn replace_matches(para: &mut CT_P, next_match: &mut MatchFinder<'_>) -> usize { + let runs = text_runs(para); + let mut emptied = Vec::new(); let mut total = 0; // Byte offset in the concatenated paragraph text at which to look for the @@ -25,61 +77,57 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st // replacement forever. let mut search_from = 0usize; - // We loop because after one replacement the text layout changes - // and there may be more matches. loop { // 1. Concatenate all run text and build a char map. - let (full_text, char_map) = build_char_map(¶.runs); - - if search_from >= full_text.len() { + let (full_text, char_map) = build_char_map(para, &runs); + if search_from > full_text.len() { break; } - // 2. Find the next match (byte offset in full_text). - let Some(byte_start) = full_text[search_from..] - .find(placeholder) - .map(|i| search_from + i) - else { + // 2. Find the next match (byte offsets in full_text). + let Some((byte_start, byte_end, replacement)) = next_match(&full_text, search_from) else { break; }; + let next_char = full_text[byte_start..] + .chars() + .next() + .map_or(1, char::len_utf8); + + // A zero-width match would neither consume input nor advance the + // cursor, so step past it and the loop always makes progress. + if byte_start == byte_end { + search_from = byte_start + next_char; + continue; + } + // Convert byte offsets to char indices (char_map is indexed by char position). let match_start = full_text[..byte_start].chars().count(); - let match_end = match_start + placeholder.chars().count(); + let match_end = match_start + full_text[byte_start..byte_end].chars().count(); // 3. Determine which runs are affected. - let first_char = &char_map[match_start]; - let last_char = &char_map[match_end - 1]; - let first_run = first_char.run_index; - let last_run = last_char.run_index; + let first_run = char_map[match_start].run_index; + let last_run = char_map[match_end - 1].run_index; + + if runs[first_run].0 != runs[last_run].0 { + // The match straddles a content-control boundary, so it is no + // match. Look again after it, as after a replaced match, so that + // no shorter match inside it, such as `\d+` finds in the digits + // after a boundary, replaces part of the text it covers. + search_from = byte_end; + continue; + } if first_run == last_run { // Single-run match: simple in-place replacement on that run's text content. - replace_in_single_run( - &mut para.runs[first_run], - &char_map, - match_start, - match_end, - replacement, - ); + edit_run(para, &runs[first_run].1, |run| { + replace_in_single_run(run, &char_map, match_start, match_end, &replacement); + }); } else { // Cross-run match: put replacement in first run, clear matched parts from others. - replace_across_runs( - &mut para.runs, - &char_map, - match_start, - match_end, - first_run, - last_run, - replacement, - ); + replace_across_runs(para, &runs, &char_map, match_start, match_end, &replacement); + emptied.extend(first_run + 1..=last_run); } - // Remove runs that became completely empty (no content at all). - para.runs.retain(|r| !r.content.is_empty()); - - // Update hyperlink spans to account for removed runs. - reindex_hyperlinks(para); - // Text before `byte_start` is untouched and the replacement now // occupies `byte_start..byte_start + replacement.len()`, so this stays // on a char boundary of the rebuilt text. @@ -88,13 +136,82 @@ pub fn replace_in_paragraph(para: &mut CT_P, placeholder: &str, replacement: &st total += 1; } + remove_emptied_runs(para, &runs, emptied); total } +/// The runs of `para` that a replacement reads, in the order `CT_P::runs` +/// reads them, each with the stretch it belongs to and its address. +fn text_runs(para: &CT_P) -> Vec<(usize, AcceptedRunPath)> { + let mut runs = Vec::new(); + let mut stretch = 0; + for index in 0..=para.runs.len() { + for (control, (_, _, _, sdt)) in para + .content_controls + .iter() + .enumerate() + .filter(|(_, (at, _, _, _))| *at == index) + { + let mut prefix = vec![AcceptedRunPathSegment::ContentControl(control)]; + stretch += 1; + control_text_runs(sdt, &mut prefix, &mut stretch, &mut runs); + stretch += 1; + } + if index < para.runs.len() { + runs.push(( + stretch, + AcceptedRunPath { + segments: vec![AcceptedRunPathSegment::Run(index)], + }, + )); + } + } + runs +} + +fn control_text_runs( + sdt: &CT_Sdt, + prefix: &mut Vec, + stretch: &mut usize, + runs: &mut Vec<(usize, AcceptedRunPath)>, +) { + for (index, content) in sdt.content.iter().enumerate() { + match content { + SdtContent::Run(_) => { + prefix.push(AcceptedRunPathSegment::Run(index)); + runs.push(( + *stretch, + AcceptedRunPath { + segments: prefix.clone(), + }, + )); + prefix.pop(); + } + SdtContent::ContentControl(nested) => { + prefix.push(AcceptedRunPathSegment::ContentControl(index)); + *stretch += 1; + control_text_runs(nested, prefix, stretch, runs); + *stretch += 1; + prefix.pop(); + } + _ => {} + } + } +} + +/// Apply `edit` to the run at `path`. +fn edit_run(para: &mut CT_P, path: &AcceptedRunPath, edit: impl FnOnce(&mut CT_R)) { + if let Some(mut run) = para.accepted_run(path).cloned() { + edit(&mut run); + let replaced = para.replace_accepted_run(path, run); + debug_assert!(matches!(replaced, Ok(true)), "text run path is stale"); + } +} + /// A mapping from character position in the concatenated text to its source run and content item. #[derive(Debug)] struct CharMapping { - /// Index of the run in the paragraph's runs vec. + /// Index of the run in the list of text runs. run_index: usize, /// Index of the RunContent item within the run. content_index: usize, @@ -102,11 +219,14 @@ struct CharMapping { byte_offset: usize, } -fn build_char_map(runs: &[CT_R]) -> (String, Vec) { +fn build_char_map(para: &CT_P, runs: &[(usize, AcceptedRunPath)]) -> (String, Vec) { let mut full_text = String::new(); let mut char_map = Vec::new(); - for (run_idx, run) in runs.iter().enumerate() { + for (run_idx, (_, path)) in runs.iter().enumerate() { + let Some(run) = para.accepted_run(path) else { + continue; + }; for (content_idx, content) in run.content.iter().enumerate() { if let RunContent::Text(t) = content { for (byte_pos, _ch) in t.text.char_indices() { @@ -153,80 +273,114 @@ fn replace_in_single_run( new_text.push_str(replacement); new_text.push_str(&t.text[byte_end..]); t.text = new_text; - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + // Keep the flag the producer wrote. Dropping it rewrites an unchanged + // run, which a later comparison against the source then reports. + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } } fn replace_across_runs( - runs: &mut [CT_R], + para: &mut CT_P, + runs: &[(usize, AcceptedRunPath)], char_map: &[CharMapping], match_start: usize, match_end: usize, - first_run: usize, - last_run: usize, replacement: &str, ) { // Handle the first run: replace from match start to end of text in that content item. let first_mapping = &char_map[match_start]; - let first_content_idx = first_mapping.content_index; - let first_byte_offset = first_mapping.byte_offset; - - if let RunContent::Text(t) = &mut runs[first_run].content[first_content_idx] { - let mut new_text = String::new(); - new_text.push_str(&t.text[..first_byte_offset]); - new_text.push_str(replacement); - t.text = new_text; - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); - } + edit_run(para, &runs[first_mapping.run_index].1, |run| { + if let RunContent::Text(t) = &mut run.content[first_mapping.content_index] { + let mut new_text = String::new(); + new_text.push_str(&t.text[..first_mapping.byte_offset]); + new_text.push_str(replacement); + t.text = new_text; + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); + } + }); // Handle the last run: replace from start to match end within that content item. let last_mapping = &char_map[match_end - 1]; - let last_content_idx = last_mapping.content_index; - let last_byte_offset = last_mapping.byte_offset; - - if let RunContent::Text(t) = &mut runs[last_run].content[last_content_idx] { - let remaining = &t.text[last_byte_offset..]; - let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0); - let byte_end = last_byte_offset + ch_len; - t.text = t.text[byte_end..].to_string(); - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); - } - - // Clear text content from runs strictly between first and last. - for run in &mut runs[(first_run + 1)..last_run] { - run.content.retain(|c| !matches!(c, RunContent::Text(_))); - } + edit_run(para, &runs[last_mapping.run_index].1, |run| { + if let RunContent::Text(t) = &mut run.content[last_mapping.content_index] { + let remaining = &t.text[last_mapping.byte_offset..]; + let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0); + let byte_end = last_mapping.byte_offset + ch_len; + t.text = t.text[byte_end..].to_string(); + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); + } - // If the last run's text is now empty, remove its text content too. - if last_run != first_run { - runs[last_run].content.retain(|c| { + // If the last run's text is now empty, remove its text content too. + run.content.retain(|c| { if let RunContent::Text(t) = c { !t.text.is_empty() } else { true } }); + }); + + // Clear text content from runs strictly between first and last. + for (_, path) in &runs[first_mapping.run_index + 1..last_mapping.run_index] { + edit_run(para, path, |run| { + run.content.retain(|c| !matches!(c, RunContent::Text(_))); + }); } } -/// Re-index hyperlink spans after runs may have been removed. -fn reindex_hyperlinks(para: &mut CT_P) { - // After retain, run indices may have shifted. We rebuild by checking - // which runs still exist. Since retain preserves order and only removes - // empty runs, the relative order is maintained. However, hyperlink spans - // referenced by index need adjustment. - // - // For simplicity: we decrement indices for each removed slot. - // But since we already called retain, the runs are already compacted. - // We need to adjust hyperlinks based on the new run count. - // - // The simplest correct approach: hyperlinks that referenced removed runs - // get their range clamped/invalidated. - para.hyperlinks.retain(|hl| hl.run_start < para.runs.len()); - for hl in &mut para.hyperlinks { - if hl.run_end > para.runs.len() { - hl.run_end = para.runs.len(); +/// Remove the runs at `candidates` (indices into `runs`) that the replacement +/// left without any content, keeping every anchor of the paragraph and of +/// its content controls on the boundary that remains. +fn remove_emptied_runs( + para: &mut CT_P, + runs: &[(usize, AcceptedRunPath)], + mut candidates: Vec, +) { + candidates.sort_unstable(); + candidates.dedup(); + let mut direct = vec![false; para.runs.len()]; + // Later runs first, so that removing one inside a control leaves the + // addresses of the others valid. + for index in candidates.into_iter().rev() { + let path = &runs[index].1; + // A run keeps the attributes of its start tag, such as `w:rsidR`, as + // a raw record, which does not make it worth keeping once empty. + let emptied = para.accepted_run(path).is_some_and(|run| { + run.content.is_empty() + && run + .extra_xml_positions + .iter() + .filter(|position| CT_R::raw_child_is_root_attributes(**position)) + .count() + == run.extra_xml.len() + }); + if !emptied { + continue; + } + match path.segments() { + [AcceptedRunPathSegment::Run(run)] => direct[*run] = true, + [AcceptedRunPathSegment::ContentControl(control), rest @ ..] => { + if let Some((_, _, _, sdt)) = para.content_controls.get_mut(*control) { + remove_control_run(sdt, rest); + } + } + _ => {} + } + } + if direct.contains(&true) { + para.remove_runs(&direct); + } +} + +fn remove_control_run(sdt: &mut CT_Sdt, path: &[AcceptedRunPathSegment]) { + match path { + [AcceptedRunPathSegment::Run(index)] => sdt.remove_content(*index), + [AcceptedRunPathSegment::ContentControl(index), rest @ ..] => { + if let Some(SdtContent::ContentControl(nested)) = sdt.content.get_mut(*index) { + remove_control_run(nested, rest); + } } + _ => {} } } @@ -238,80 +392,262 @@ pub fn replace_in_paragraphs(paras: &mut [CT_P], placeholder: &str, replacement: .sum() } -/// Replace all occurrences of `placeholder` in a table (recursively handles nested tables). +/// Replace all occurrences of `placeholder` in a table (recursively handles +/// nested tables and content controls). pub fn replace_in_table(table: &mut CT_Tbl, placeholder: &str, replacement: &str) -> usize { + edit_table(table, &mut |paragraph| { + replace_in_paragraph(paragraph, placeholder, replacement) + }) +} + +/// Hand every paragraph of a table to `edit` and sum what it counts. +fn edit_table(table: &mut CT_Tbl, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { let mut count = 0; for (_, _, sdt) in &mut table.content_controls { - count += replace_in_sdt(sdt, placeholder, replacement); + count += edit_control(sdt, edit); } for row in &mut table.rows { - count += replace_in_row(row, placeholder, replacement); + count += edit_row(row, edit); } count } -fn replace_in_sdt( - sdt: &mut crate::content_control::CT_Sdt, - placeholder: &str, - replacement: &str, -) -> usize { - use crate::content_control::SdtContent; - +fn edit_control(sdt: &mut CT_Sdt, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { let mut count = 0; for content in &mut sdt.content { count += match content { - SdtContent::Paragraph(paragraph) => { - replace_in_paragraph(paragraph, placeholder, replacement) - } - SdtContent::Table(table) => replace_in_table(table, placeholder, replacement), - SdtContent::Row(row) => replace_in_row(row, placeholder, replacement), - SdtContent::Cell(cell) => replace_in_cell(cell, placeholder, replacement), - SdtContent::ContentControl(nested) => replace_in_sdt(nested, placeholder, replacement), + SdtContent::Paragraph(paragraph) => edit(paragraph), + SdtContent::Table(table) => edit_table(table, edit), + SdtContent::Row(row) => edit_row(row, edit), + SdtContent::Cell(cell) => edit_cell(cell, edit), + SdtContent::ContentControl(nested) => edit_control(nested, edit), SdtContent::Run(_) | SdtContent::RawXml(_) => 0, }; } count } -fn replace_in_row(row: &mut crate::table::CT_Row, placeholder: &str, replacement: &str) -> usize { - let controls = row - .content_controls - .iter_mut() - .map(|(_, _, sdt)| replace_in_sdt(sdt, placeholder, replacement)) - .sum::(); - controls - + row - .cells - .iter_mut() - .map(|cell| replace_in_cell(cell, placeholder, replacement)) - .sum::() +fn edit_row(row: &mut CT_Row, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { + let mut count = 0; + for (_, _, sdt) in &mut row.content_controls { + count += edit_control(sdt, edit); + } + for cell in &mut row.cells { + count += edit_cell(cell, edit); + } + count } -fn replace_in_cell(cell: &mut crate::table::CT_Tc, placeholder: &str, replacement: &str) -> usize { - use crate::table::CellContent; +fn edit_cell(cell: &mut CT_Tc, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { + let mut count = 0; + for content in &mut cell.content { + count += match content { + CellContent::Paragraph(paragraph) => edit(paragraph), + CellContent::Table(table) => edit_table(table, edit), + CellContent::ContentControl(sdt) => edit_control(sdt, edit), + }; + } + count +} - cell.content - .iter_mut() - .map(|content| match content { - CellContent::Paragraph(paragraph) => { - replace_in_paragraph(paragraph, placeholder, replacement) - } - CellContent::Table(table) => replace_in_table(table, placeholder, replacement), - CellContent::ContentControl(sdt) => replace_in_sdt(sdt, placeholder, replacement), +/// Hand the paragraphs of a table or a block content control kept as raw +/// XML to `edit`, and re-serialise the element in place when the edit +/// counts a change. Any other element, one the typed parsers refuse, or one +/// whose namespaces the rewrite cannot keep, see [`with_source_namespaces`], +/// keeps its bytes and counts nothing. +fn edit_raw_block( + raw: &mut Vec, + word_prefixes: &[String], + edit: &mut dyn FnMut(&mut CT_P) -> usize, +) -> usize { + use quick_xml::events::Event; + use quick_xml::{Reader, Writer}; + + let mut reader = Reader::from_reader(raw.as_slice()); + reader.config_mut().trim_text(false); + let mut buffer = Vec::new(); + let Ok(Event::Start(start)) = reader.read_event_into(&mut buffer) else { + return 0; + }; + let Ok(prefixes) = word_prefixes_at(&start, word_prefixes) else { + return 0; + }; + // The namespaces the start tag declares other than the part does. + let Ok(bindings) = local_namespace_overrides(&start, word_prefixes) else { + return 0; + }; + let mut writer = Writer::new(Vec::new()); + let count = if is_word_element(start.name().as_ref(), b"tbl", &prefixes) { + let Ok(mut table) = + CT_Tbl::from_xml_with_prefixes_and_owner_bindings(&mut reader, &prefixes, &bindings) + else { + return 0; + }; + let count = edit_table(&mut table, edit); + if count == 0 || table.to_xml(&mut writer).is_err() { + return 0; + } + count + } else if is_word_element(start.name().as_ref(), b"sdt", &prefixes) { + let Some(mut control) = CT_Sdt::from_body_raw(raw, word_prefixes) else { + return 0; + }; + let count = edit_control(&mut control, edit); + if count == 0 || control.to_xml(&mut writer).is_err() { + return 0; + } + count + } else { + return 0; + }; + let Some(rewritten) = + with_source_namespaces(raw, &writer.into_inner(), &bindings, word_prefixes) + else { + return 0; + }; + *raw = rewritten; + count +} + +/// Declare the namespaces that the start tag of `source` declares other +/// than the part does, `start_bindings`, again on `rewritten`, the element +/// the typed writers produced from it, since they leave them out. They leave +/// out a declaration below the start tag too, so a prefix of `rewritten` +/// left to the part must not be one that `source` declares with a namespace +/// the part does not bind it to. Otherwise the prefix would be unbound, or +/// name another namespace, and this returns `None`. +fn with_source_namespaces( + source: &[u8], + rewritten: &[u8], + start_bindings: &[(String, String)], + word_prefixes: &[String], +) -> Option> { + let rewritten = raw_with_external_bindings(rewritten, start_bindings).ok()?; + let (_, declared) = namespace_use(source)?; + let (left_to_part, _) = namespace_use(&rewritten)?; + let part = namespace_bindings(word_prefixes); + let part_namespace = |prefix: &str| { + part.iter() + .find(|(candidate, _)| candidate == prefix) + .map(|(_, namespace)| namespace.as_str()) + .or_else(|| word_prefixes.iter().any(|p| p == prefix).then_some(W_NS)) + }; + left_to_part + .iter() + .all(|prefix| { + declared + .iter() + .filter(|(candidate, _)| candidate == prefix) + .all(|(_, namespace)| part_namespace(prefix) == Some(namespace.as_str())) }) - .sum() + .then_some(rewritten) +} + +/// The prefixes an element uses outside every declaration it makes, and +/// every declaration it makes, as prefix and namespace. +type NamespaceUse = (Vec, Vec<(String, String)>); + +/// The [`NamespaceUse`] of `xml`, the empty prefix standing for the default +/// namespace of an unprefixed element. +fn namespace_use(xml: &[u8]) -> Option { + use quick_xml::Reader; + use quick_xml::events::Event; + + let prefix_of = |name: &[u8]| { + let name = std::str::from_utf8(name).ok()?; + Some(name.split_once(':').map(|(prefix, _)| prefix.to_owned())) + }; + let mut reader = Reader::from_reader(xml); + let mut scopes: Vec> = Vec::new(); + let mut unbound: Vec = Vec::new(); + let mut declared = Vec::new(); + let mut buffer = Vec::new(); + loop { + let (element, opens) = match reader.read_event_into(&mut buffer).ok()? { + Event::Start(element) => (element.into_owned(), true), + Event::Empty(element) => (element.into_owned(), false), + Event::End(_) => { + scopes.pop(); + buffer.clear(); + continue; + } + Event::Eof => break, + _ => { + buffer.clear(); + continue; + } + }; + buffer.clear(); + let declarations = local_namespace_overrides(&element, &[]).ok()?; + let mut prefixes = vec![prefix_of(element.name().as_ref())?.unwrap_or_default()]; + for attribute in element.attributes() { + let key = attribute.ok()?.key; + let key = key.as_ref(); + if key != b"xmlns" && !key.starts_with(b"xmlns:") { + prefixes.extend(prefix_of(key)?); + } + } + for prefix in prefixes { + let bound = prefix == "xml" + || declarations + .iter() + .chain(scopes.iter().flatten()) + .any(|(candidate, _)| *candidate == prefix); + if !bound && !unbound.contains(&prefix) { + unbound.push(prefix); + } + } + declared.extend(declarations.iter().cloned()); + if opens { + scopes.push(declarations); + } + } + Some((unbound, declared)) } /// Replace all occurrences of `placeholder` in a header or footer. +/// +/// Reaches its paragraphs and those of the tables and block content controls +/// it keeps as raw XML. A raw element the replacement changed is written back +/// in its place, and the others keep their bytes. pub fn replace_in_header_footer(hf: &mut CT_HdrFtr, placeholder: &str, replacement: &str) -> usize { - replace_in_paragraphs(&mut hf.paragraphs, placeholder, replacement) + edit_header_footer(hf, &mut |paragraph| { + replace_in_paragraph(paragraph, placeholder, replacement) + }) +} + +/// The texts a replacement in `hf` matches against, see [`replaceable_texts`]. +#[doc(hidden)] +pub fn header_footer_replaceable_texts(hf: &CT_HdrFtr) -> Vec { + let mut texts = Vec::new(); + edit_header_footer(&mut hf.clone(), &mut |paragraph| { + texts.extend(replaceable_texts(paragraph)); + 0 + }); + texts +} + +fn edit_header_footer(hf: &mut CT_HdrFtr, edit: &mut dyn FnMut(&mut CT_P) -> usize) -> usize { + let mut count = 0; + for paragraph in &mut hf.paragraphs { + count += edit(paragraph); + } + for raw in &mut hf.extra_xml { + count += edit_raw_block(raw, &hf.word_prefixes, edit); + } + count } /// Replace placeholders in text boxes and shapes within a raw XML part. /// -/// Walks the XML, finds `w:txbxContent` elements at any depth, parses their -/// child `w:p` elements using `CT_P::from_xml`, performs replacement, and -/// re-serializes back. Returns the modified XML and replacement count. +/// Walks the XML, finds `w:txbxContent` elements at any depth outside another +/// text box, parses their child `w:p` elements using `CT_P::from_xml`, and +/// their child tables and block content controls with the typed parsers, +/// performs replacement, and re-serializes the children it changed. Every +/// other child of the text box, such as a bookmark or a paragraph without a +/// match, is copied through verbatim in its place. A text box nested inside +/// another one is kept as it is, not edited. Returns the modified XML and +/// replacement count. pub fn replace_in_xml_part( xml: &[u8], placeholder: &str, @@ -329,11 +665,11 @@ pub fn replace_many_in_xml_part( xml: &[u8], replacements: &[(&str, &str)], ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { + rewrite_text_boxes(xml, &mut |paragraph| { replacements .iter() .map(|(placeholder, replacement)| { - replace_in_paragraphs(paragraphs, placeholder, replacement) + replace_in_paragraph(paragraph, placeholder, replacement) }) .sum() }) @@ -348,18 +684,35 @@ pub fn replace_regex_in_xml_part( re: ®ex::Regex, replacement: &str, ) -> crate::error::Result<(Vec, usize)> { - rewrite_text_boxes(xml, &mut |paragraphs| { - replace_regex_in_paragraphs(paragraphs, re, replacement) + rewrite_text_boxes(xml, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) }) } -/// Walk `xml`, handing the paragraphs of each `w:txbxContent` element to -/// `edit`, and re-serialise. Returns the rewritten XML and the summed count. +/// The texts a replacement in the text boxes of a raw XML part matches +/// against, see [`replaceable_texts`]. +#[doc(hidden)] +pub fn xml_part_replaceable_texts(xml: &[u8]) -> crate::error::Result> { + let mut texts = Vec::new(); + rewrite_text_boxes(xml, &mut |paragraph| { + texts.extend(replaceable_texts(paragraph)); + 0 + })?; + Ok(texts) +} + +/// Walk `xml`, handing each paragraph of a `w:txbxContent` element to `edit`, +/// those of its tables and block content controls included, and +/// re-serialising a child in place when `edit` counts a change in it. Every +/// other child of the text box is copied through verbatim. Returns the +/// rewritten XML and the summed count. fn rewrite_text_boxes( xml: &[u8], - edit: &mut dyn FnMut(&mut Vec) -> usize, + edit: &mut dyn FnMut(&mut CT_P) -> usize, ) -> crate::error::Result<(Vec, usize)> { - use crate::namespace::matches_local_name; + use crate::error::OxmlError; + use crate::namespace::{MC_NS, R_NS, matches_local_name}; + use crate::raw_xml::capture_element; use quick_xml::events::Event; use quick_xml::{Reader, Writer}; @@ -369,55 +722,29 @@ fn rewrite_text_boxes( let mut writer = Writer::new(Vec::new()); let mut buf = Vec::new(); let mut total_count = 0; + // The bindings `CT_P::from_xml` assumes for a paragraph cut out of the part. + let word_prefixes = [ + "w".to_owned(), + format!("\0r\0{R_NS}"), + format!("\0mc\0{MC_NS}"), + ]; loop { match reader.read_event_into(&mut buf) { Ok(Event::Eof) => break, Ok(Event::Start(ref e)) if matches_local_name(e.name().as_ref(), b"txbxContent") => { - // We found a txbxContent element. Collect its contents as raw XML, - // parse paragraphs, do replacement, and re-serialize. + // We found a txbxContent element. Parse and edit each paragraph + // and copy every other child through verbatim, in document order. writer.write_event(Event::Start(e.clone()))?; - // Read all events inside txbxContent - let mut depth = 1u32; let mut inner_buf = Vec::new(); - // Collect paragraphs from inside txbxContent - let mut paragraphs: Vec = Vec::new(); - loop { match reader.read_event_into(&mut inner_buf) { Ok(Event::Start(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"p") && depth == 1 { + if matches_local_name(ie.name().as_ref(), b"p") { // Parse this paragraph: collect its XML, then parse via CT_P - let mut para_writer = Writer::new(Vec::new()); - // Write the opening tag - para_writer.write_event(Event::Start(ie.clone()))?; - let mut pdepth = 1u32; - let mut pbuf = Vec::new(); - loop { - match reader.read_event_into(&mut pbuf) { - Ok(Event::Start(ref pe)) => { - pdepth += 1; - para_writer.write_event(Event::Start(pe.clone()))?; - } - Ok(Event::End(ref pe)) => { - pdepth -= 1; - para_writer.write_event(Event::End(pe.clone()))?; - if pdepth == 0 { - break; - } - } - Ok(ref ev) => { - para_writer.write_event(ev.clone())?; - } - Err(e) => return Err(e.into()), - } - pbuf.clear(); - } - let para_xml = para_writer.into_inner(); - - // Parse the paragraph + let para_xml = capture_element(&mut reader, ie)?; let mut para_reader = Reader::from_reader(para_xml.as_slice()); para_reader.config_mut().trim_text(true); let mut prbuf = Vec::new(); @@ -434,42 +761,41 @@ fn rewrite_text_boxes( } prbuf.clear(); } - let para = CT_P::from_xml(&mut para_reader)?; - paragraphs.push(para); + let mut para = CT_P::from_xml(&mut para_reader)?; + let count = edit(&mut para); + if count == 0 { + // Nothing changed, so the paragraph keeps its + // bytes, start-tag attributes included. + writer.get_mut().extend_from_slice(¶_xml); + } else { + para.to_xml(&mut writer)?; + } + total_count += count; } else { - depth += 1; - // Non-paragraph element inside txbxContent; skip it - reader.read_to_end_into(ie.name(), &mut Vec::new())?; - depth -= 1; + // A table or a content control is edited through + // the typed model, and any other element the edit + // does not reach stays as it was. + let mut raw = capture_element(&mut reader, ie)?; + total_count += edit_raw_block(&mut raw, &word_prefixes, edit); + writer.get_mut().extend_from_slice(&raw); } } - Ok(Event::End(ref ie)) => { - if matches_local_name(ie.name().as_ref(), b"txbxContent") && depth == 1 - { - break; - } - depth -= 1; + // Every child element is consumed whole, so the first end + // tag at this level closes the txbxContent element. + Ok(Event::End(ie)) => { + writer.write_event(Event::End(ie))?; + break; } - Ok(_) => { - // Whitespace/text at top level of txbxContent, skip + Ok(Event::Eof) => { + return Err(OxmlError::MissingElement("w:txbxContent end".to_owned())); } + // Empty elements such as `` or a bookmark, whitespace, + // comments and processing instructions. + Ok(ev) => writer.write_event(ev)?, Err(e) => return Err(e.into()), } inner_buf.clear(); } - - // Hand the collected paragraphs to the caller's edit function - total_count += edit(&mut paragraphs); - - // Re-serialize paragraphs into the writer - for p in ¶graphs { - p.to_xml(&mut writer)?; - } - - // Write closing txbxContent tag - writer.write_event(Event::End(quick_xml::events::BytesEnd::new( - "w:txbxContent", - )))?; } Ok(ev) => { writer.write_event(ev)?; @@ -553,85 +879,21 @@ pub fn replace_many_in_chart_xml( /// Replace all regex matches in a paragraph with the replacement string. /// /// The `replacement` string supports capture group references: `$1`, `$2`, etc. -/// Uses the same cross-run char map algorithm as literal replacement. +/// Uses the same cross-run char map algorithm as literal replacement, so it +/// reaches the runs of inline content controls and leaves a match that +/// straddles a content-control boundary as it is. /// Returns the number of replacements made. pub fn replace_regex_in_paragraph(para: &mut CT_P, re: ®ex::Regex, replacement: &str) -> usize { - let mut total = 0; - - // See `replace_in_paragraph`: resume after the inserted text so a - // replacement that itself matches the pattern cannot loop forever. // `captures_at` (rather than slicing) keeps anchors and look-around // evaluating against the full paragraph text. - let mut search_from = 0usize; - - loop { - let (full_text, char_map) = build_char_map(¶.runs); - if char_map.is_empty() || search_from > full_text.len() { - break; - } - - // Find the next match - let Some(m) = re.captures_at(&full_text, search_from).and_then(|caps| { - let mat = caps.get(0)?; - // Expand capture groups in replacement - let mut expanded = String::new(); - caps.expand(replacement, &mut expanded); - Some((mat.start(), mat.end(), expanded)) - }) else { - break; - }; - - let (byte_start, byte_end, expanded_replacement) = m; - - // A zero-width match would neither consume input nor advance the - // cursor; step past it so the loop always makes progress. - if byte_start == byte_end { - search_from = full_text[byte_start..] - .chars() - .next() - .map_or(full_text.len() + 1, |c| byte_start + c.len_utf8()); - continue; - } - - // Convert byte offsets to char indices - let match_start = full_text[..byte_start].chars().count(); - let match_end = match_start + full_text[byte_start..byte_end].chars().count(); - - if match_start >= char_map.len() || match_end == 0 || match_end > char_map.len() { - break; - } - - // Determine which runs are affected - let first_run = char_map[match_start].run_index; - let last_run = char_map[match_end - 1].run_index; - - if first_run == last_run { - replace_in_single_run( - &mut para.runs[first_run], - &char_map, - match_start, - match_end, - &expanded_replacement, - ); - } else { - replace_across_runs( - &mut para.runs, - &char_map, - match_start, - match_end, - first_run, - last_run, - &expanded_replacement, - ); - } - - para.runs.retain(|r| !r.content.is_empty()); - reindex_hyperlinks(para); - search_from = byte_start + expanded_replacement.len(); - total += 1; - } - - total + replace_matches(para, &mut |text, from| { + let captures = re.captures_at(text, from)?; + let matched = captures.get(0)?; + // Expand capture groups in replacement + let mut expanded = String::new(); + captures.expand(replacement, &mut expanded); + Some((matched.start(), matched.end(), expanded)) + }) } /// Replace regex matches in all paragraphs. @@ -646,90 +908,30 @@ pub fn replace_regex_in_paragraphs( .sum() } -/// Replace regex matches in a table (recursively handles nested tables). +/// Replace regex matches in a table (recursively handles nested tables and +/// content controls). pub fn replace_regex_in_table(table: &mut CT_Tbl, re: ®ex::Regex, replacement: &str) -> usize { - let mut count = 0; - for (_, _, sdt) in &mut table.content_controls { - count += replace_regex_in_sdt(sdt, re, replacement); - } - for row in &mut table.rows { - count += replace_regex_in_row(row, re, replacement); - } - count -} - -fn replace_regex_in_sdt( - sdt: &mut crate::content_control::CT_Sdt, - re: ®ex::Regex, - replacement: &str, -) -> usize { - use crate::content_control::SdtContent; - - let mut count = 0; - for content in &mut sdt.content { - count += match content { - SdtContent::Paragraph(paragraph) => { - replace_regex_in_paragraph(paragraph, re, replacement) - } - SdtContent::Table(table) => replace_regex_in_table(table, re, replacement), - SdtContent::Row(row) => replace_regex_in_row(row, re, replacement), - SdtContent::Cell(cell) => replace_regex_in_cell(cell, re, replacement), - SdtContent::ContentControl(nested) => replace_regex_in_sdt(nested, re, replacement), - SdtContent::Run(_) | SdtContent::RawXml(_) => 0, - }; - } - count -} - -fn replace_regex_in_row( - row: &mut crate::table::CT_Row, - re: ®ex::Regex, - replacement: &str, -) -> usize { - let controls = row - .content_controls - .iter_mut() - .map(|(_, _, sdt)| replace_regex_in_sdt(sdt, re, replacement)) - .sum::(); - controls - + row - .cells - .iter_mut() - .map(|cell| replace_regex_in_cell(cell, re, replacement)) - .sum::() -} - -fn replace_regex_in_cell( - cell: &mut crate::table::CT_Tc, - re: ®ex::Regex, - replacement: &str, -) -> usize { - use crate::table::CellContent; - - cell.content - .iter_mut() - .map(|content| match content { - CellContent::Paragraph(paragraph) => { - replace_regex_in_paragraph(paragraph, re, replacement) - } - CellContent::Table(table) => replace_regex_in_table(table, re, replacement), - CellContent::ContentControl(sdt) => replace_regex_in_sdt(sdt, re, replacement), - }) - .sum() + edit_table(table, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) + }) } -/// Replace regex matches in a header or footer. +/// Replace regex matches in a header or footer, see +/// [`replace_in_header_footer`]. pub fn replace_regex_in_header_footer( hf: &mut CT_HdrFtr, re: ®ex::Regex, replacement: &str, ) -> usize { - replace_regex_in_paragraphs(&mut hf.paragraphs, re, replacement) + edit_header_footer(hf, &mut |paragraph| { + replace_regex_in_paragraph(paragraph, re, replacement) + }) } #[cfg(test)] mod tests { use super::*; + use crate::namespace::{MC_NS, W_NS}; use crate::properties::CT_RPr; fn make_para(texts: &[&str]) -> CT_P { @@ -796,6 +998,152 @@ mod tests { assert_eq!(p.runs[1].properties.as_ref().unwrap().italic, Some(true)); } + #[test] + fn replace_keeps_the_producer_space_flag() { + let mut preserved = CT_R::new("WORD"); + if let RunContent::Text(text) = &mut preserved.content[0] { + text.preserve_space = true; + } + let mut p = CT_P::new(); + p.runs.push(preserved.clone()); + p.runs.push(preserved); + p.add_run("tail"); + + assert_eq!(replace_in_paragraph(&mut p, "WORD", "WORD"), 2); + assert_eq!(replace_in_paragraph(&mut p, "DWO", "D-WO"), 1); + assert_eq!(replace_in_paragraph(&mut p, "tail", " end"), 1); + let flags = p + .runs + .iter() + .map(|run| match &run.content[0] { + RunContent::Text(text) => (text.text.as_str(), text.preserve_space), + _ => unreachable!(), + }) + .collect::>(); + assert_eq!( + flags, + [("WORD-WO", true), ("RD", true), (" end", true)], + "a rewritten run keeps a producer flag and gains one at a text edge" + ); + } + + fn paragraph_xml(paragraph: &CT_P) -> String { + let mut writer = quick_xml::Writer::new(Vec::new()); + paragraph.to_xml(&mut writer).unwrap(); + String::from_utf8(writer.into_inner()).unwrap() + } + + /// A match across runs removes the runs it empties. The anchors after + /// them are kept by run index, and used to slide one run later, past the + /// text they preceded. A ruby annotation slid onto the run after its base. + #[test] + fn removing_emptied_runs_keeps_later_anchors_in_place() { + let source = format!( + r#"{{{{name}}}}controlannBASEx"# + ); + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + for regex in [false, true] { + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + let count = if regex { + replace_regex_in_paragraph(&mut p, &re, "Bob") + } else { + replace_in_paragraph(&mut p, "{{name}}", "Bob") + }; + assert_eq!(count, 1); + assert_eq!(p.runs.len(), 3); + let xml = paragraph_xml(&p); + let positions = [ + ">Bob<", + "bookmarkStart", + "commentRangeStart", + "", + "proofErr", + "", + ">BASE<", + "", + ">x<", + "bookmarkEnd", + "commentRangeEnd", + ] + .map(|marker| { + xml.find(marker) + .unwrap_or_else(|| panic!("{marker}: {xml}")) + }); + assert!(positions.is_sorted(), "{xml}"); + } + } + + /// A run whose only child is raw XML, such as a drawing Word writes in + /// `mc:AlternateContent`, has no typed content, and neither has an + /// empty run. A match anywhere in the paragraph used to remove both. + #[test] + fn replace_keeps_the_runs_it_did_not_empty() { + let source = format!( + r#"Hello {{{{name}}}}"# + ); + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + + assert_eq!(replace_in_paragraph(&mut p, "{{name}}", "Ada"), 1); + + assert_eq!(p.runs.len(), 3); + assert!(paragraph_xml(&p).contains("")); + } + + /// Inside an inline control the replacement removes the runs it empties + /// too. A later run of the control keeps the bytes it was read with. + #[test] + fn a_control_run_after_emptied_ones_keeps_its_source_bytes() { + let tail = r#" tail"#; + let source = format!( + r#"Dear {{{{name}}}}{tail}"# + ); + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + + assert_eq!(replace_in_paragraph(&mut p, "{{name}}", "Bob"), 1); + + assert_eq!(p.text(), "Dear Bob tail"); + assert_eq!(p.content_controls[0].3.content.len(), 2); + let xml = paragraph_xml(&p); + assert!(xml.contains(tail), "{xml}"); + } + + /// Each inline control starts a stretch of its own, and so does each + /// control nested in it. A match never spans two stretches. + #[test] + fn a_match_stays_within_one_stretch_of_runs() { + let source = format!( + r#"alXphabcde"# + ); + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + assert_eq!(replaceable_texts(&p), ["al", "X", "pha", "b", "c", "de"]); + + for straddling in ["alX", "Xpha", "phab", "bc", "cd"] { + assert_eq!(replace_in_paragraph(&mut p, straddling, "-"), 0); + } + let re = regex::Regex::new(r"^al|bcd|e$").unwrap(); + assert_eq!(replace_regex_in_paragraph(&mut p, &re, "-"), 2); + + assert_eq!(p.text(), "-Xphabcd-"); + } + + /// A quantified pattern also matches the digits after a boundary alone. + /// The search goes on after the straddling match, so it no longer + /// replaces part of the number the reader sees. + #[test] + fn no_match_starts_inside_a_straddling_one() { + let source = format!( + r#"1234 end 56"# + ); + for pattern in [r"\d+", r"\d{2,}"] { + let mut p = CT_P::from_xml_fragment(source.as_bytes()).unwrap(); + let re = regex::Regex::new(pattern).unwrap(); + + assert_eq!(replace_regex_in_paragraph(&mut p, &re, "N"), 1, "{pattern}"); + + assert_eq!(p.text(), "1234 end N", "{pattern}"); + } + } + #[test] fn replace_multiple_occurrences() { let mut p = make_para(&["{{x}} and {{x}}"]); @@ -862,6 +1210,63 @@ mod tests { assert_eq!(hf.text(), "Company: Acme Corp"); } + /// The tables and block content controls of a header are raw XML in the + /// model, and replacement used to skip them. One the replacement changes + /// is written back in its place, and one it does not change keeps its + /// bytes, under the prefix the part uses. + #[test] + fn replace_in_header_footer_reaches_its_tables_and_controls() { + let untouched = r#"no tag"#; + let xml = format!( + r#"cell {{{{x}}}}paragraph {{{{x}}}}control {{{{x}}}}{untouched}"# + ); + let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + assert_eq!( + header_footer_replaceable_texts(&parsed), + ["paragraph {{x}}", "cell {{x}}", "control {{x}}", "no tag"] + ); + + let re = regex::Regex::new(r"\{\{x\}\}").unwrap(); + for regex in [false, true] { + let mut hf = parsed.clone(); + let count = if regex { + replace_regex_in_header_footer(&mut hf, &re, "Y") + } else { + replace_in_header_footer(&mut hf, "{{x}}", "Y") + }; + assert_eq!(count, 3); + let written = String::from_utf8(hf.to_xml_header().unwrap()).unwrap(); + assert!(!written.contains("{{x}}"), "{written}"); + let positions = ["cell Y", "paragraph Y", "control Y", untouched].map(|marker| { + written + .find(marker) + .unwrap_or_else(|| panic!("{marker}: {written}")) + }); + assert!(positions.is_sorted(), "{written}"); + } + } + + /// A header rewrites its raw tables as a text box does, see + /// `replace_in_textbox_keeps_the_namespaces_its_tables_declare`. A cell + /// that declares a namespace the root binds the same way is replaced. + #[test] + fn replace_in_header_footer_keeps_the_namespaces_its_tables_declare() { + let on_cell = table_declaring_w14(false, "cell {{x}}"); + let tables = [table_declaring_w14(true, "table {{x}}"), on_cell.clone()].concat(); + let w14 = r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#; + for (root, expected) in [("", 1), (w14, 2)] { + let xml = format!(r#"{tables}"#); + let mut hf = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + + assert_eq!(replace_in_header_footer(&mut hf, "{{x}}", "Y"), expected); + + let written = String::from_utf8(hf.to_xml_header().unwrap()).unwrap(); + assert_every_prefix_is_bound(&written); + assert!(written.contains(">table Y<"), "{written}"); + assert_eq!(written.contains(&on_cell), expected == 1, "{written}"); + } + } + #[test] fn replace_empty_placeholder_noop() { let mut p = make_para(&["Hello"]); @@ -918,6 +1323,165 @@ mod tests { assert!(result_str.contains("Company: Acme")); } + /// A text box whose paragraphs sit among a table, a block content + /// control, a bookmark, an empty paragraph, a comment and whitespace. + /// A second text box without a placeholder follows. The walker rewrites + /// every text box of the part, so that one must come back unchanged too. + /// The paragraphs without a placeholder keep their identity attributes. + const TEXT_BOX_WITH_EVERY_KIND_OF_CHILD: &str = r#" +Title +Hello {{name}} +cell +control{{name}} again +other cellNo placeholder here"#; + + #[test] + fn replace_in_textbox_keeps_every_other_child_in_order() { + let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD; + let expected = xml.replace("{{name}}", "Alice"); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + let (result, count) = replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap(); + assert_eq!(count, 2); + assert_eq!(String::from_utf8(result).unwrap(), expected); + } + + /// The tables and block content controls of a text box are edited + /// through the typed model. Those without a match keep their bytes. + #[test] + fn replace_in_textbox_reaches_its_tables_and_controls() { + let xml = TEXT_BOX_WITH_EVERY_KIND_OF_CHILD + .replace(">cell<", ">cell {{name}}<") + .replace(">control<", ">control {{name}}<"); + let other_table = r#"other cell"#; + assert_eq!( + xml_part_replaceable_texts(xml.as_bytes()).unwrap(), + [ + "Title", + "Hello {{name}}", + "cell {{name}}", + "control {{name}}", + "{{name}} again", + "other cell", + "No placeholder here" + ] + ); + + let re = regex::Regex::new(r"\{\{name\}\}").unwrap(); + for regex in [false, true] { + let (result, count) = if regex { + replace_regex_in_xml_part(xml.as_bytes(), &re, "Alice").unwrap() + } else { + replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap() + }; + assert_eq!(count, 4); + let result = String::from_utf8(result).unwrap(); + assert!(!result.contains("{{name}}"), "{result}"); + let positions = [ + "Hello Alice", + "cell Alice", + "", + "control Alice", + "Alice again", + other_table, + ] + .map(|marker| { + result + .find(marker) + .unwrap_or_else(|| panic!("{marker}: {result}")) + }); + assert!(positions.is_sorted(), "{result}"); + } + } + + /// Panic on an element or attribute prefix that no declaration binds. + fn assert_every_prefix_is_bound(xml: &str) { + use quick_xml::events::Event; + use quick_xml::name::ResolveResult; + + let mut reader = quick_xml::NsReader::from_str(xml); + loop { + let (namespace, event) = reader.read_resolved_event().unwrap(); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + match event { + Event::Start(element) | Event::Empty(element) => { + for attribute in element.attributes() { + let key = attribute.unwrap().key; + let (namespace, _) = reader.resolver().resolve_attribute(key); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + } + } + Event::Eof => break, + _ => {} + } + } + } + + /// A table with a paragraph that uses `w14`, declared on the table or on + /// its cell. + fn table_declaring_w14(on_table: bool, text: &str) -> String { + let declaration = r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml""#; + let (table, cell) = if on_table { + (declaration, "") + } else { + ("", declaration) + }; + format!( + r#"{text}"# + ) + } + + /// The typed writers leave out the namespace declarations of a table, so + /// a paragraph that used a prefix declared on the table came back with + /// it unbound. The table declares it again. A declaration on a cell + /// cannot be put back, so that table keeps its bytes and counts nothing. + #[test] + fn replace_in_textbox_keeps_the_namespaces_its_tables_declare() { + let on_cell = table_declaring_w14(false, "cell {{name}}"); + let xml = format!( + r#"{}{on_cell}"#, + table_declaring_w14(true, "table {{name}}") + ); + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Ada").unwrap(); + + assert_eq!(count, 1); + let result = String::from_utf8(result).unwrap(); + assert_every_prefix_is_bound(&result); + assert!(result.contains(">table Ada<"), "{result}"); + assert!(result.contains(&on_cell), "{result}"); + } + + /// The end tag used to be written as `w:txbxContent` whatever prefix the + /// start tag carried, which left the part ill-formed. + #[test] + fn replace_in_textbox_closes_it_with_its_own_prefix() { + let xml = r#"Hello {{name}}"#; + + let (result, count) = replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").unwrap(); + assert_eq!(count, 1); + assert_eq!( + String::from_utf8(result).unwrap(), + xml.replace("{{name}}", "Alice") + ); + } + + /// A text box cut short used to keep the walker reading past the end of + /// the part forever. + #[test] + fn replace_in_unterminated_textbox_is_an_error() { + for tail in ["", "Hello {{name}}"] { + let xml = format!( + r#"{tail}"# + ); + assert!(replace_in_xml_part(xml.as_bytes(), "{{name}}", "Alice").is_err()); + } + } + #[test] fn replace_in_xml_part_no_textbox() { let xml = br#" diff --git a/crates/rdocx-oxml/src/text.rs b/crates/rdocx-oxml/src/text.rs index cd5501ca..c4f016c4 100644 --- a/crates/rdocx-oxml/src/text.rs +++ b/crates/rdocx-oxml/src/text.rs @@ -4320,10 +4320,19 @@ impl CT_P { if removed.iter().all(|remove| !remove) { return; } + self.remove_runs(&removed); + } + + /// Remove the direct runs flagged in `removed` and move every run-boundary + /// projection onto the boundary that remains. Raw children, comment + /// markers, bookmarks and controls of the boundaries that collapse into + /// one keep their order, and a hyperlink or a ruby annotation left + /// without runs is dropped. + pub(crate) fn remove_runs(&mut self, removed: &[bool]) { let removed_run_addresses = self .runs .iter() - .zip(&removed) + .zip(removed) .filter_map(|(run, remove)| remove.then_some(std::ptr::from_ref(run))) .collect::>(); let removed_projected_indices = accepted_paragraph_runs(self) @@ -4470,6 +4479,12 @@ impl CT_P { *position = boundary_map[old_boundary]; *raw_before = raw_prefixes[old_boundary] + (*raw_before).min(raw_counts[old_boundary]); } + for ruby in &mut self.rubies { + ruby.base_start = boundary_map[ruby.base_start.min(old_run_count)]; + ruby.base_end = boundary_map[ruby.base_end.min(old_run_count)]; + } + // An annotation over no base run is not written, so drop it. + self.rubies.retain(|ruby| ruby.base_start < ruby.base_end); let old_hyperlinks = std::mem::take(&mut self.hyperlinks); let mut hyperlink_map = vec![None; old_hyperlinks.len()]; for (old_index, mut hyperlink) in old_hyperlinks.into_iter().enumerate() { @@ -4517,7 +4532,7 @@ impl CT_P { .runs .drain(..) .zip(removed) - .filter_map(|(run, remove)| (!remove).then_some(run)) + .filter_map(|(run, remove)| (!*remove).then_some(run)) .collect(); let _ = self.refresh_bookmark_projection(); } diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 19c557ed..7f7ff9d1 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -11,6 +11,7 @@ use rdocx_oxml::content_control::{CT_Sdt, SdtContent}; use rdocx_oxml::document::{BodyContent, CT_Document}; use rdocx_oxml::namespace::W_NS; use rdocx_oxml::properties::CT_PPr; +use rdocx_oxml::shared::ST_PageOrientation; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_TblPr, CT_Tc, CT_TrPr, CellContent}; use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent}; use sha2::{Digest, Sha256}; @@ -29,7 +30,6 @@ thread_local! { type ControlPropertySignature<'a> = Option<( Option<&'a str>, Option<&'a str>, - Option, Option, Option<&'a rdocx_oxml::content_control::CT_DataBinding>, )>; @@ -2985,8 +2985,16 @@ fn compare_granular_paragraph( ))); } - let original_run_signatures = original.runs.iter().map(run_signature).collect::>(); - let edited_run_signatures = edited.runs.iter().map(run_signature).collect::>(); + let original_run_signatures = original + .runs + .iter() + .map(attributed_run_signature) + .collect::>(); + let edited_run_signatures = edited + .runs + .iter() + .map(attributed_run_signature) + .collect::>(); if original_run_signatures == edited_run_signatures && original.content_controls == edited.content_controls { @@ -3558,12 +3566,30 @@ fn granular_text(text: &CT_Text, options: &ComparisonOptions) -> Vec { fragments .into_iter() .map(|value| CT_Text { + // A unit is written back as its own `w:t`, where edge whitespace + // needs the flag to survive. Whitespace in a unit then reads as + // whitespace whatever the source flag was. + preserve_space: text.preserve_space || has_edge_whitespace(&value), text: value, - preserve_space: text.preserve_space, }) .collect() } +/// The whole-run signature that agrees with the attributed units. +/// +/// It reads the space flag the way `granular_text` writes it on every unit, +/// so a run that differs only by the flag stays whole instead of being +/// rewritten one unit per run, which would move run-indexed bookmark ends. +fn attributed_run_signature(run: &CT_R) -> String { + let mut run = run.clone(); + for content in &mut run.content { + if let RunContent::Text(text) | RunContent::DeletedText(text) = content { + text.preserve_space |= has_edge_whitespace(&text.text); + } + } + run_signature(&run) +} + fn whitespace_fragments(text: &str) -> Vec { let mut output = Vec::new(); let mut current = String::new(); @@ -3977,8 +4003,22 @@ fn nonempty_paragraph_properties(mut properties: CT_PPr) -> Option { } fn paragraph_properties_differ(original: Option<&CT_PPr>, edited: Option<&CT_PPr>) -> bool { - original.cloned().and_then(nonempty_paragraph_properties) - != edited.cloned().and_then(nonempty_paragraph_properties) + let modeled = |properties: Option<&CT_PPr>| { + properties.cloned().and_then(|mut properties| { + if let Some(section) = properties.sect_pr.as_mut() { + clear_default_orientation(section); + } + nonempty_paragraph_properties(properties) + }) + }; + modeled(original) != modeled(edited) +} + +/// Portrait is the schema default, which the section writer omits. +fn clear_default_orientation(section: &mut rdocx_oxml::document::CT_SectPr) { + if section.orientation == Some(ST_PageOrientation::Portrait) { + section.orientation = None; + } } fn section_properties_xml( @@ -4008,6 +4048,8 @@ fn section_properties_xml( original_modeled.change = None; let mut edited_modeled = edited.clone(); edited_modeled.change = None; + clear_default_orientation(&mut original_modeled); + clear_default_orientation(&mut edited_modeled); if original_modeled == edited_modeled { return section_property_xml(original); } @@ -5099,17 +5141,20 @@ fn deleted_text_xml(xml: &str) -> String { } fn complex_field_result(xml: &str) -> Result<(String, String, String)> { - let separate = xml - .find("fldCharType=\"separate\"") - .or_else(|| xml.find("fldCharType='separate'")) + let separate = field_character(xml, 0, "separate") .ok_or_else(|| Error::Other("complex field source has no separate boundary".to_owned()))?; - let (_, result_start) = containing_run(xml, separate)?; - let end_marker = xml[result_start..] - .find("fldCharType=\"end\"") - .or_else(|| xml[result_start..].find("fldCharType='end'")) - .map(|offset| result_start + offset) + // A producer may pack a whole field in one run, as Google Docs writes page + // fields. Ending a run after `separate` and starting one at `end` reads it + // as the same field written one run per part. + let xml = split_field_run(xml, separate, true)?; + let (_, result_start) = containing_run(&xml, separate)?; + let end_marker = field_character(&xml, result_start, "end") + .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?; + let xml = split_field_run(&xml, end_marker, false)?; + // The split may have moved the `end` character further along. + let end_marker = field_character(&xml, result_start, "end") .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?; - let (result_end, _) = containing_run(xml, end_marker)?; + let (result_end, _) = containing_run(&xml, end_marker)?; Ok(( xml[..result_start].to_owned(), xml[result_start..result_end].to_owned(), @@ -5117,6 +5162,48 @@ fn complex_field_result(xml: &str) -> Result<(String, String, String)> { )) } +fn field_character(xml: &str, from: usize, kind: &str) -> Option { + let tail = &xml[from..]; + tail.find(&format!("fldCharType=\"{kind}\"")) + .or_else(|| tail.find(&format!("fldCharType='{kind}'"))) + .map(|offset| from + offset) +} + +/// Split the run holding the field character at `marker` so that the +/// character ends its run (`after`) or starts it. +/// +/// Both runs repeat the original start tag and run properties. A run with no +/// content on that side of the character is returned unchanged. +fn split_field_run(xml: &str, marker: usize, after: bool) -> Result { + let (run_start, run_end) = containing_run(xml, marker)?; + let run = &xml[run_start..run_end]; + let children = direct_element_spans(run)?; + let character = children + .iter() + .position(|child| child.contains(&(marker - run_start))) + .ok_or_else(|| Error::Other("complex field character is not a run child".to_owned()))?; + let is_properties = |child: &Range| { + let mut reader = Reader::from_reader(run[child.clone()].as_bytes()); + matches!( + reader.read_event(), + Ok(Event::Start(element) | Event::Empty(element)) + if element.local_name().as_ref() == b"rPr" + ) + }; + let first_content = usize::from(children.first().is_some_and(is_properties)); + let split = if after { character + 1 } else { character }; + if split <= first_content || split >= children.len() { + return Ok(xml.to_owned()); + } + let head = &run[..children[first_content].start]; + let close = run + .rfind(" Result<(usize, usize)> { let mut reader = Reader::from_reader(xml.as_bytes()); reader.config_mut().trim_text(false); @@ -5411,10 +5498,29 @@ fn run_content_signature(content: &RunContent) -> String { match content { RunContent::Field(_) => "field-owner".to_owned(), RunContent::Drawing(drawing) => format!("Drawing({:?})", drawing_signature(drawing)), + RunContent::Text(text) => format!("Text({:?})", text_signature(text)), + RunContent::DeletedText(text) => format!("DeletedText({:?})", text_signature(text)), content => format!("{content:?}"), } } +/// The text and whether its `xml:space="preserve"` changes how it reads. +/// +/// The flag only protects whitespace at an edge of the text. Producers write +/// it on every `w:t` or only where needed, so elsewhere it is serialization. +fn text_signature(text: &CT_Text) -> (&str, bool) { + ( + &text.text, + text.preserve_space && has_edge_whitespace(&text.text), + ) +} + +/// Whether XML whitespace starts or ends the text. +fn has_edge_whitespace(text: &str) -> bool { + let whitespace = |character: char| matches!(character, ' ' | '\t' | '\n' | '\r'); + text.starts_with(whitespace) || text.ends_with(whitespace) +} + fn table_signature(table: &CT_Tbl) -> String { format!( "{:?}:{:?}:{:?}:{:?}", @@ -5478,16 +5584,25 @@ fn control_signature(control: &CT_Sdt) -> String { ) } +/// The content-control properties that alignment, refusal and the accept and +/// reject postconditions compare. +/// +/// `w:id` is left out. Producers renumber it on save and it carries no +/// content, so a pair that differs only by it keeps the original's `w:sdtPr`. +/// A `w:sdtPr` with none of these properties reads like no `w:sdtPr`. fn control_property_signature(control: &CT_Sdt) -> ControlPropertySignature<'_> { - control.properties.as_ref().map(|properties| { - ( - properties.alias.as_deref(), - properties.tag.as_deref(), - properties.id, - properties.control_type, - properties.data_binding.as_ref(), - ) - }) + control + .properties + .as_ref() + .map(|properties| { + ( + properties.alias.as_deref(), + properties.tag.as_deref(), + properties.control_type, + properties.data_binding.as_ref(), + ) + }) + .filter(|signature| !matches!(signature, (None, None, None, None))) } fn modeled_control_content(control: &CT_Sdt) -> Vec<&SdtContent> { @@ -5757,6 +5872,9 @@ fn paragraph_formatting(paragraph: &CT_P) -> Option { properties.numbering_revision_position = None; properties.change = None; properties.revision_xml.clear(); + if let Some(section) = properties.sect_pr.as_mut() { + clear_default_orientation(section); + } properties }) } @@ -6287,7 +6405,8 @@ fn utf8_error(error: impl std::fmt::Display) -> Error { mod tests { use super::{ ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING, - attributed_run_units, comparison_postcondition_error, story_document, word_fragments, + attributed_run_units, comparison_postcondition_error, complex_field_result, story_document, + word_fragments, }; use crate::Document; use rdocx_oxml::document::BodyContent; @@ -6379,6 +6498,36 @@ mod tests { ); } + #[test] + fn a_packed_field_result_is_read_as_one_run_per_part() { + for w in ["w", "q"] { + let shell = format!(r#"<{w}:r {w}:rsidR="00AB12CD"><{w}:rPr><{w}:b/>"#); + let close = format!(""); + let character = |kind: &str| format!(r#"<{w}:fldChar {w}:fldCharType="{kind}"/>"#); + let (begin, separate, end) = + (character("begin"), character("separate"), character("end")); + let code = format!("<{w}:instrText>PAGE"); + let result = format!("<{w}:t>1"); + let expected = ( + format!("{shell}{begin}{code}{separate}{close}"), + format!("{shell}{result}{close}"), + format!("{shell}{end}{close}"), + ); + for field in [ + format!("{shell}{begin}{code}{separate}{result}{end}{close}"), + format!("{shell}{begin}{code}{separate}{close}{shell}{result}{end}{close}"), + format!("{}{}{}", expected.0, expected.1, expected.2), + ] { + assert_eq!(complex_field_result(&field).unwrap(), expected, "{field}"); + } + let uncached = format!("{shell}{begin}{code}{separate}{end}{close}"); + assert_eq!( + complex_field_result(&uncached).unwrap(), + (expected.0, String::new(), expected.2) + ); + } + } + #[test] fn staged_comparison_postcondition_failure_preserves_bytes_and_layout_cache() { let mut original = Document::new(); diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 5c4b2b46..b4f59a76 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -21212,8 +21212,10 @@ impl Document { /// Replace all occurrences of `placeholder` with `replacement` throughout the document. /// /// Searches body paragraphs, tables (including nested), headers, footers, - /// text boxes and chart labels. Handles placeholders split across multiple - /// runs. Returns the total number of replacements made. + /// text boxes and chart labels, and the content controls at every level of + /// them, inline controls included. Handles placeholders split across + /// multiple runs. A match that straddles a content-control boundary is not + /// replaced. Returns the total number of replacements made. /// /// A `replacement` that contains `placeholder` is substituted once, not /// repeatedly. @@ -21284,13 +21286,15 @@ impl Document { for (rel_id, is_header) in self.header_footer_rel_ids() { if let Some(header_footer) = self.load_header_footer(&rel_id, is_header) { - sources.extend(crate::template::header_footer_sources(&header_footer)); + sources.extend(rdocx_oxml::placeholder::header_footer_replaceable_texts( + &header_footer, + )); } } for (part_name, _) in self.raw_text_bearing_part_names() { if let Some(xml) = self.package.get_part(&part_name) { - sources.extend(crate::template::text_box_sources(xml)?); + sources.extend(rdocx_oxml::placeholder::xml_part_replaceable_texts(xml)?); } } @@ -21334,23 +21338,14 @@ impl Document { Ok(count) } - /// Run the typed replacement over body paragraphs and tables. + /// Run the typed replacement over every body paragraph, those of tables + /// and content controls at every level included. fn replace_in_body(&mut self, placeholder: &str, replacement: &str) -> usize { - use rdocx_oxml::placeholder; - let mut count = 0; - for content in &mut self.document.body.content { - match content { - BodyContent::Paragraph(p) => { - count += placeholder::replace_in_paragraph(p, placeholder, replacement); - } - BodyContent::Table(t) => { - count += placeholder::replace_in_table(t, placeholder, replacement); - } - BodyContent::ContentControl(_) => {} - BodyContent::RawXml(_) => {} - } - } + visit_body_paragraphs_mut(&mut self.document.body.content, &mut |paragraph| { + count += + rdocx_oxml::placeholder::replace_in_paragraph(paragraph, placeholder, replacement); + }); count } @@ -21448,8 +21443,10 @@ impl Document { /// Replace all regex matches with `replacement` throughout the document. /// /// The `replacement` string supports capture groups: `$1`, `$2`, etc. - /// Searches body paragraphs, tables (including nested), headers, and footers. - /// Returns the total number of replacements made, or an error if the regex is invalid. + /// Searches body paragraphs, tables (including nested), headers, footers + /// and text boxes, and the content controls at every level of them, as + /// [`Self::replace_text`] does. Returns the total number of replacements + /// made, or an error if the regex is invalid. pub fn replace_regex(&mut self, pattern: &str, replacement: &str) -> Result { let re = regex::Regex::new(pattern).map_err(|e| Error::Other(format!("invalid regex: {e}")))?; @@ -21484,19 +21481,10 @@ impl Document { let mut count = 0; - // Replace in body paragraphs and tables - for content in &mut self.document.body.content { - match content { - BodyContent::Paragraph(p) => { - count += placeholder::replace_regex_in_paragraph(p, re, replacement); - } - BodyContent::Table(t) => { - count += placeholder::replace_regex_in_table(t, re, replacement); - } - BodyContent::ContentControl(_) => {} - BodyContent::RawXml(_) => {} - } - } + // Replace in body paragraphs, tables and content controls + visit_body_paragraphs_mut(&mut self.document.body.content, &mut |paragraph| { + count += placeholder::replace_regex_in_paragraph(paragraph, re, replacement); + }); // Replace in headers and footers for (rel_id, is_header) in self.header_footer_rel_ids() { diff --git a/crates/rdocx/src/template.rs b/crates/rdocx/src/template.rs index 3721169c..cc0dcf80 100644 --- a/crates/rdocx/src/template.rs +++ b/crates/rdocx/src/template.rs @@ -6,7 +6,6 @@ use quick_xml::Reader; use quick_xml::events::Event; use rdocx_oxml::content_control::{CT_Sdt, SdtContent}; use rdocx_oxml::document::{BodyContent, CT_Document}; -use rdocx_oxml::header_footer::CT_HdrFtr; use rdocx_oxml::namespace::matches_local_name; use rdocx_oxml::placeholder; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_Tc, CellContent}; @@ -576,7 +575,9 @@ fn render_nested_tables_in_control( fn body_marker(content: &BodyContent) -> Result> { match content { - BodyContent::Paragraph(paragraph) => marker_from_sources(&[paragraph_text(paragraph)]), + BodyContent::Paragraph(paragraph) => { + marker_from_sources(&placeholder::replaceable_texts(paragraph)) + } BodyContent::Table(_) | BodyContent::RawXml(_) => Ok(None), BodyContent::ContentControl(control) => { reject_unsupported_control_sources(&control_sources(control))?; @@ -928,7 +929,8 @@ fn stage_paragraph_scalars( scopes: &[Scope], sentinels: &mut SentinelPool, ) -> Result { - let replacements = resolve_replacements(&[paragraph_text(paragraph)], root, scopes)?; + let replacements = + resolve_replacements(&placeholder::replaceable_texts(paragraph), root, scopes)?; let mut count = 0; for replacement in replacements { let sentinel = sentinels.stage(&replacement.value, replacement.occurrences); @@ -1135,7 +1137,9 @@ pub(crate) fn body_sources(document: &CT_Document) -> Vec { let mut sources = Vec::new(); for content in &document.body.content { match content { - BodyContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + BodyContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } BodyContent::Table(table) => collect_table(table, &mut sources), BodyContent::ContentControl(control) => collect_control(control, &mut sources), BodyContent::RawXml(_) => {} @@ -1144,26 +1148,6 @@ pub(crate) fn body_sources(document: &CT_Document) -> Vec { sources } -pub(crate) fn header_footer_sources(header_footer: &CT_HdrFtr) -> Vec { - header_footer - .paragraphs - .iter() - .map(paragraph_text) - .collect() -} - -fn paragraph_text(paragraph: &CT_P) -> String { - paragraph - .runs - .iter() - .flat_map(|run| &run.content) - .filter_map(|content| match content { - RunContent::Text(text) => Some(text.text.as_str()), - _ => None, - }) - .collect() -} - fn collect_table(table: &CT_Tbl, sources: &mut Vec) { for (_, _, control) in &table.content_controls { collect_control(control, sources); @@ -1182,7 +1166,9 @@ fn control_sources(control: &CT_Sdt) -> Vec { fn collect_control(control: &CT_Sdt, sources: &mut Vec) { for content in &control.content { match content { - SdtContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + SdtContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } SdtContent::Table(table) => collect_table(table, sources), SdtContent::Row(row) => collect_row(row, sources), SdtContent::Cell(cell) => collect_cell(cell, sources), @@ -1213,7 +1199,9 @@ fn collect_row_marker_sources(row: &CT_Row, sources: &mut Vec) { fn collect_cell_marker_sources(cell: &CT_Tc, sources: &mut Vec) { for content in &cell.content { match content { - CellContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + CellContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } CellContent::Table(_) => {} CellContent::ContentControl(control) => { collect_control_marker_sources(control, sources); @@ -1225,7 +1213,9 @@ fn collect_cell_marker_sources(cell: &CT_Tc, sources: &mut Vec) { fn collect_control_marker_sources(control: &CT_Sdt, sources: &mut Vec) { for content in &control.content { match content { - SdtContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + SdtContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } SdtContent::Table(_) => {} SdtContent::Row(row) => collect_row_marker_sources(row, sources), SdtContent::Cell(cell) => collect_cell_marker_sources(cell, sources), @@ -1249,53 +1239,15 @@ fn collect_control_marker_sources(control: &CT_Sdt, sources: &mut Vec) { fn collect_cell(cell: &CT_Tc, sources: &mut Vec) { for content in &cell.content { match content { - CellContent::Paragraph(paragraph) => sources.push(paragraph_text(paragraph)), + CellContent::Paragraph(paragraph) => { + sources.extend(placeholder::replaceable_texts(paragraph)); + } CellContent::Table(table) => collect_table(table, sources), CellContent::ContentControl(control) => collect_control(control, sources), } } } -pub(crate) fn text_box_sources(xml: &[u8]) -> Result> { - let mut reader = Reader::from_reader(xml); - reader.config_mut().trim_text(false); - let mut sources = Vec::new(); - let mut buffer = Vec::new(); - let mut in_text_box = false; - - loop { - match reader.read_event_into(&mut buffer) { - Ok(Event::Eof) => break, - Ok(Event::Start(element)) - if matches_local_name(element.name().as_ref(), b"txbxContent") => - { - in_text_box = true; - } - Ok(Event::Start(element)) - if in_text_box && matches_local_name(element.name().as_ref(), b"p") => - { - let paragraph = CT_P::from_xml(&mut reader)?; - sources.push(paragraph_text(¶graph)); - } - Ok(Event::Start(element)) if in_text_box => { - reader - .read_to_end_into(element.name(), &mut Vec::new()) - .map_err(template_xml_error)?; - } - Ok(Event::End(element)) - if in_text_box && matches_local_name(element.name().as_ref(), b"txbxContent") => - { - in_text_box = false; - } - Ok(_) => {} - Err(error) => return Err(template_xml_error(error)), - } - buffer.clear(); - } - - Ok(sources) -} - pub(crate) fn chart_sources(xml: &[u8]) -> Result> { let mut reader = Reader::from_reader(xml); reader.config_mut().trim_text(false); diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c7..2f434db3 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -16180,6 +16180,566 @@ fn zero_width_regex_match_terminates() { assert!(count <= 4, "should not loop indefinitely, got {count}"); } +/// Replacement in a text box parses the paragraphs of its `w:txbxContent` and +/// used to write back nothing else, so a hit there deleted the tables, +/// content controls, bookmarks and empty paragraphs of that text box. +mod text_box_replacement_keeps_every_child { + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + /// One copy of the text box: the paragraph the replacement edits, then a + /// table, a block content control and a bookmark around an empty + /// paragraph. Each copy needs its own bookmark id for the file to open. + fn text_box_content(text: &str, id: u32) -> String { + format!( + r#"{text}cellcontrol"# + ) + } + + /// A text box as Word saves it, the DrawingML shape in `mc:Choice` and its + /// VML copy in `mc:Fallback`, or the bare `wp:anchor` alone. + fn text_box_document(compatibility_block: bool) -> Document { + let content = text_box_content("Dear {{name}}", 7); + let drawing = format!( + r#"00{content}"# + ); + let shape = if compatibility_block { + format!( + r#"{drawing}{}"#, + text_box_content("Dear {{name}}", 8) + ) + } else { + drawing + }; + super::document_with_content_controls(&format!( + r#"Host paragraph{shape}"# + )) + } + + /// Replace in both text box forms, then check that every copy holds the + /// edited paragraph and every other child byte for byte and in order. + fn assert_replacement_keeps_every_child(replace: impl Fn(&mut Document) -> usize) { + for (compatibility_block, ids) in [(true, &[7, 8][..]), (false, &[7][..])] { + let mut document = text_box_document(compatibility_block); + assert_eq!(replace(&mut document), ids.len(), "{compatibility_block}"); + let saved = super::document_xml(&mut document); + assert_eq!(saved.matches("").count(), ids.len()); + for &id in ids { + let expected = text_box_content("Dear Ada", id); + assert!(saved.contains(&expected), "{expected}\n{saved}"); + } + } + } + + #[test] + fn replacing_text_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.try_replace_text("{{name}}", "Ada").unwrap() + }); + } + + #[test] + fn regex_replacement_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document.replace_regex(r"\{\{(\w+)\}\}", "Ada").unwrap() + }); + } + + #[test] + fn a_template_keeps_the_tables_controls_and_bookmarks_of_a_text_box() { + assert_replacement_keeps_every_child(|document| { + document + .render_template(&serde_json::json!({"name": "Ada"})) + .unwrap() + }); + } +} + +/// Replacement skipped the text of content controls. It never entered a +/// body-level control, read the runs of an inline control at no level, and +/// reached neither the tables and controls of a header or footer nor those +/// of a text box. Each location below holds its own tag, which every walker +/// must replace exactly once. +mod replacement_reaches_content_controls { + use std::collections::HashMap; + + use rdocx::Document; + use rdocx_oxml::namespace::W_NS; + + const DOCUMENT: &str = "/word/document.xml"; + const HEADER: &str = "/word/header1.xml"; + const FOOTER: &str = "/word/footer1.xml"; + + /// Every location with the part that holds it. The tag of a location is + /// its name in double braces and its replacement the name in brackets. + const LOCATIONS: [(&str, &str); 10] = [ + ("block", DOCUMENT), + ("inline", DOCUMENT), + ("cell", DOCUMENT), + ("nested", DOCUMENT), + ("box", DOCUMENT), + ("box_inline", DOCUMENT), + ("header_table", HEADER), + ("header_control", HEADER), + ("footer_control", FOOTER), + ("footer_inline", FOOTER), + ]; + + /// The `w:tag` of every control, by part. + const CONTROL_TAGS: [(&str, &[&str]); 3] = [ + ( + DOCUMENT, + &[ + "goog_rdk_1", + "goog_rdk_0", + "cell", + "outer", + "middle", + "inner", + "innermost", + "box", + "box-inline", + ], + ), + (HEADER, &["header"]), + (FOOTER, &["goog_rdk_2"]), + ]; + + fn tag(name: &str) -> String { + format!("{{{{{name}}}}}") + } + + fn value(name: &str) -> String { + format!("[{name}]") + } + + fn run(text: &str) -> String { + format!(r#"{text}"#) + } + + fn paragraph(content: &str) -> String { + format!("{content}") + } + + fn control(tag: &str, content: &str) -> String { + format!( + r#"{content}"# + ) + } + + fn table(cell: &str) -> String { + format!( + r#"{cell}"# + ) + } + + /// A DrawingML text box holding a block control and an inline control. + fn text_box() -> String { + let content = [ + control("box", ¶graph(&run("Boxed {{box}}."))), + paragraph(&[run("Boxed "), control("box-inline", &run("{{box_inline}}"))].concat()), + ] + .concat(); + format!( + r#"00{content}"# + ) + } + + /// The runs of a header or footer block that the model keeps as raw XML. + /// The first one ends in a space, as "Page " does before a page number. + fn raw_block_runs(text: &str, name: &str) -> String { + paragraph(&[run(text), run(&format!("{}.", tag(name)))].concat()) + } + + /// The table and the control of the header, which the model keeps as + /// raw XML, before its paragraph. + fn header_blocks() -> [String; 2] { + [ + table(&raw_block_runs("Header cell ", "header_table")), + control( + "header", + &raw_block_runs("Header control ", "header_control"), + ), + ] + } + + /// The page-number control Word writes in a footer, before a paragraph + /// that wraps a run in a Google Docs control. + fn footer_block() -> String { + format!( + r#"{}"#, + raw_block_runs("Footer control ", "footer_control") + ) + } + + fn document() -> Document { + let body = [ + control("goog_rdk_1", ¶graph(&run("Block {{block}}."))), + paragraph( + &[ + run("Body text, "), + control("goog_rdk_0", &run("{{inline}}")), + run(" dolor."), + ] + .concat(), + ), + table(&control("cell", ¶graph(&run("Cell {{cell}}.")))), + control( + "outer", + &control( + "middle", + ¶graph( + &[ + run("Nested "), + control("inner", &control("innermost", &run("{{nested}}"))), + ] + .concat(), + ), + ), + ), + paragraph(&[run("Host "), format!("{}", text_box())].concat()), + ] + .concat(); + let header = format!( + r#"{}{}"#, + header_blocks().concat(), + paragraph(&run("Header paragraph.")) + ); + let footer = format!( + r#"{}{}"#, + footer_block(), + paragraph( + &[ + run("Confidential "), + control("goog_rdk_2", &run("{{footer_inline}}")), + ] + .concat() + ) + ); + + let mut seed = Document::new(); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + let mut references = Vec::new(); + for (part, xml, kind, rel_type) in [ + ( + HEADER, + header, + "header", + oxml_opc::relationship::rel_types::HEADER, + ), + ( + FOOTER, + footer, + "footer", + oxml_opc::relationship::rel_types::FOOTER, + ), + ] { + package.set_part(part, xml.into_bytes()); + package.content_types.add_override( + part, + &format!( + "application/vnd.openxmlformats-officedocument.wordprocessingml.{kind}+xml" + ), + ); + let id = package + .get_or_create_part_rels(DOCUMENT) + .add(rel_type, part.trim_start_matches("/word/")); + references.push(format!( + r#""# + )); + } + package.set_part( + DOCUMENT, + format!( + r#"{body}{}"#, + references.concat() + ) + .into_bytes(), + ); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + fn saved_parts(document: &mut Document) -> HashMap<&'static str, String> { + let package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(document.to_bytes().unwrap())) + .unwrap(); + [DOCUMENT, HEADER, FOOTER] + .into_iter() + .map(|part| { + let xml = String::from_utf8(package.get_part(part).unwrap().to_vec()).unwrap(); + (part, xml) + }) + .collect() + } + + /// Check that the tag of every location in `replaced` gave way to its + /// value once, that every other tag is still there once, that every + /// control kept its `w:tag`, and that the header and footer kept their + /// blocks in order, byte for byte where nothing was replaced, and the + /// edge spaces of their text everywhere. + fn assert_replaced(parts: &HashMap<&str, String>, replaced: &[&str]) { + for (name, part) in LOCATIONS { + let xml = &parts[part]; + if replaced.contains(&name) { + assert!(!xml.contains(&tag(name)), "{name}: {xml}"); + assert_eq!(xml.matches(&value(name)).count(), 1, "{name}: {xml}"); + } else { + assert_eq!(xml.matches(&tag(name)).count(), 1, "{name}: {xml}"); + } + } + for (part, tags) in CONTROL_TAGS { + for tag in tags { + let element = format!(r#""#); + assert!(parts[part].contains(&element), "{tag}: {}", parts[part]); + } + } + for (part, markers) in [ + ( + HEADER, + ["Header cell", "Header control", "Header paragraph."], + ), + (FOOTER, ["docPartGallery", "Footer control", "Confidential"]), + ] { + let positions = markers.map(|marker| parts[part].find(marker).unwrap()); + assert!(positions.is_sorted(), "{part}: {}", parts[part]); + } + let raw_blocks = header_blocks() + .into_iter() + .map(|block| (HEADER, block)) + .chain([(FOOTER, footer_block())]); + for (part, block) in raw_blocks { + if !replaced.iter().any(|name| block.contains(&tag(name))) { + assert!(parts[part].contains(&block), "{block}\n{}", parts[part]); + } + } + for (part, text) in [ + (HEADER, ">Header cell "), + (HEADER, ">Header control "), + (FOOTER, ">Footer control "), + ] { + assert!(parts[part].contains(text), "{text}\n{}", parts[part]); + } + } + + #[test] + fn every_walker_replaces_the_tag_of_every_location_once() { + for (name, _) in LOCATIONS { + let (tag, value) = (tag(name), value(name)); + for walker in ["try_replace_text", "replace_regex", "replace_all"] { + let mut document = document(); + let count = match walker { + "try_replace_text" => document.try_replace_text(&tag, &value).unwrap(), + "replace_regex" => document + .replace_regex(®ex::escape(&tag), &value) + .unwrap(), + _ => document.replace_all(&HashMap::from([(tag.as_str(), value.as_str())])), + }; + assert_eq!(count, 1, "{walker} at {name}"); + assert_replaced(&saved_parts(&mut document), &[name]); + } + } + } + + #[test] + fn a_template_renders_the_tag_of_every_location() { + let mut document = document(); + let data = LOCATIONS + .iter() + .map(|(name, _)| ((*name).to_owned(), serde_json::Value::from(value(name)))) + .collect::>(); + + let count = document + .render_template(&serde_json::Value::Object(data)) + .unwrap(); + + assert_eq!(count, LOCATIONS.len()); + let names = LOCATIONS.map(|(name, _)| name); + assert_replaced(&saved_parts(&mut document), &names); + } + + /// A match that straddles a control boundary is no match. A reader sees + /// "alpha one" in the first paragraph and "alXpha two" in the second, + /// and neither changes, while the match inside the control of the third + /// is replaced. Direct runs on both sides of a control used to be read as + /// one text, which matched "alpha" in the second paragraph. + #[test] + fn a_match_that_straddles_a_control_boundary_is_not_replaced() { + let body = [ + paragraph(&[run("al"), control("a", &run("pha one"))].concat()), + paragraph(&[run("al"), control("b", &run("X")), run("pha two")].concat()), + paragraph(&[run("al"), control("c", &run("alpha three"))].concat()), + ] + .concat(); + for regex in [false, true] { + let mut document = super::document_with_content_controls(&super::wrap_word_body(&body)); + let count = if regex { + document.replace_regex("alpha", "ALPHA").unwrap() + } else { + document.try_replace_text("alpha", "ALPHA").unwrap() + }; + assert_eq!(count, 1, "regex: {regex}"); + let texts = document + .paragraphs() + .iter() + .map(|paragraph| paragraph.text()) + .collect::>(); + assert_eq!(texts, ["alpha one", "alXpha two", "alALPHA three"]); + let xml = super::document_xml(&mut document); + assert_eq!(xml.matches(">al").count(), 3, "{xml}"); + } + } + + /// A quantified pattern also matches the digits after the boundary + /// alone, and replaced them, which left the number a reader sees half + /// replaced. The search now goes on after the straddling match, and + /// still reaches the number after the control. + #[test] + fn a_quantified_match_that_straddles_a_control_boundary_is_not_replaced() { + let body = paragraph(&[run("Order 12"), control("a", &run("34 end")), run(" 56")].concat()); + for pattern in [r"\d+", r"\d{2,}"] { + let mut document = super::document_with_content_controls(&super::wrap_word_body(&body)); + + assert_eq!( + document.replace_regex(pattern, "N").unwrap(), + 1, + "{pattern}" + ); + + assert_eq!( + document.paragraph(0).unwrap().text(), + "Order 1234 end N", + "{pattern}" + ); + } + } + + /// Word writes its table of contents as a body-level control whose + /// entries repeat the heading text. Replacement reaches them as it does + /// any other control, so a heading word counts once more for its entry, + /// and the entry still reads as its heading until the next update. + #[test] + fn a_table_of_contents_entry_is_replaced_with_its_heading() { + let toc = r#" TOC \o "1-3" \h \z \u Introduction PAGEREF _Toc1 \h 1"#; + let heading = r#"Introduction"#; + let mut document = + super::document_with_content_controls(&super::wrap_word_body(&[toc, heading].concat())); + + assert_eq!( + document + .try_replace_text("Introduction", "Overview") + .unwrap(), + 2 + ); + + let xml = super::document_xml(&mut document); + assert_eq!(xml.matches(">Overview<").count(), 2, "{xml}"); + assert!(xml.contains(r#"{text}"# + ) + }; + let on_cell = table("", w14, "cell {{x}}"); + let tables = [table(w14, "", "table {{x}}"), on_cell.clone()].concat(); + let mut header = super::document_with_header_story(&format!( + r#"{tables}"# + )); + let mut text_box = super::document_with_content_controls(&format!( + r#"{tables}"# + )); + + assert_eq!(header.try_replace_text("{{x}}", "Y").unwrap(), 1); + assert_eq!(text_box.try_replace_text("{{x}}", "Y").unwrap(), 1); + + for xml in [ + super::header_story_xml(&mut header), + super::document_xml(&mut text_box), + ] { + // Every element and attribute prefix resolves. + let mut reader = quick_xml::NsReader::from_str(&xml); + loop { + let (namespace, event) = reader.read_resolved_event().unwrap(); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + match event { + XmlEvent::Start(element) | XmlEvent::Empty(element) => { + for attribute in element.attributes() { + let key = attribute.unwrap().key; + let (namespace, _) = reader.resolver().resolve_attribute(key); + assert!(!matches!(namespace, ResolveResult::Unknown(_)), "{xml}"); + } + } + XmlEvent::Eof => break, + _ => {} + } + } + assert!(xml.contains(">table Y<"), "{xml}"); + assert!(xml.contains(&on_cell), "{xml}"); + } + } + + /// A template tag that a control boundary splits cannot be rendered + /// whole, so the template is rejected rather than left half rendered. + #[test] + fn a_template_tag_that_straddles_a_control_boundary_is_an_error() { + let body = paragraph(&[run("{{na"), control("a", &run("me}}"))].concat()); + let mut document = super::document_with_content_controls(&super::wrap_word_body(&body)); + let before = document.to_bytes().unwrap(); + + let error = document + .render_template(&serde_json::json!({"name": "Ada"})) + .unwrap_err(); + + assert!(error.to_string().contains("invalid template"), "{error}"); + assert_eq!(document.to_bytes().unwrap(), before); + } + + /// The two shapes of the report: a Google Docs export wraps a run in a + /// control inside its paragraph, or a whole paragraph at body level. + /// Both replaced nothing. The text the paragraph reads is the text the + /// replacement changed. + #[test] + fn the_reported_google_docs_controls_are_replaced() { + let text = run("Body text, lorem alpha dolor."); + for body in [ + paragraph(&text), + paragraph(&control("goog_rdk_0", &text)), + control("goog_rdk_1", ¶graph(&text)), + ] { + let mut document = super::document_with_content_controls(&super::wrap_word_body(&body)); + assert_eq!( + document.paragraph(0).unwrap().text(), + "Body text, lorem alpha dolor." + ); + + assert_eq!(document.try_replace_text("alpha", "ALPHA").unwrap(), 1); + + assert_eq!( + document.paragraph(0).unwrap().text(), + "Body text, lorem ALPHA dolor." + ); + let xml = super::document_xml(&mut document); + assert!(xml.contains("Body text, lorem ALPHA dolor."), "{xml}"); + assert_eq!(xml.contains("goog_rdk"), body.contains("goog_rdk"), "{xml}"); + } + } +} + /// `9360 / cols` panicked when a caller asked for a zero-column table. #[test] fn zero_column_tables_do_not_panic() { @@ -30379,6 +30939,440 @@ fn comparison_treats_empty_paragraph_properties_as_absent() { assert!(document_xml(&mut empty).contains(" Vec { + document + .revisions() + .iter() + .map(|revision| revision.kind()) + .collect() + } + + /// Check that accepting gives the edited side and rejecting the original, + /// each compared again with no diagnostic and no revision. + fn assert_resolutions( + tracked: &[u8], + original: &Document, + edited: &Document, + options: &ComparisonOptions, + ) { + for (resolve, expected) in [ + ( + Document::accept_all as fn(&mut Document) -> rdocx::Result, + edited, + ), + (Document::reject_all, original), + ] { + let mut resolved = Document::from_bytes(tracked).unwrap(); + resolve(&mut resolved).unwrap(); + let diagnostics = resolved + .compare_with_options(expected, "postcondition", TIMESTAMP, options) + .unwrap(); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_eq!(revision_kinds(&resolved), []); + } + } + + /// Compare two bodies and return the revision kinds and the redline. + fn compared_kinds(original_xml: &str, edited_xml: &str) -> (Vec, String) { + compared_kinds_with(original_xml, edited_xml, &ComparisonOptions::default()) + } + + fn compared_kinds_with( + original_xml: &str, + edited_xml: &str, + options: &ComparisonOptions, + ) -> (Vec, String) { + let original = document_with_content_controls(original_xml); + let edited = document_with_content_controls(edited_xml); + let mut compared = document_with_content_controls(original_xml); + let diagnostics = compared + .compare_with_options(&edited, "R", TIMESTAMP, options) + .expect("producer noise must not refuse the pair"); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited, options); + (revision_kinds(&compared), document_xml(&mut compared)) + } + + fn table_of_contents_control(id: Option<&str>, first_entry: &str) -> String { + let id = id.map_or_else(String::new, |value| format!(r#""#)); + wrap_word_body(&format!( + r#"Before the content control.{id}{first_entry} entryBeta entryGamma entryAfter the content control."# + )) + } + + fn google_docs_inline_control(id: &str, word: &str) -> String { + wrap_word_body(&format!( + r#"Before {word} after."# + )) + } + + #[test] + fn a_content_control_identity_is_not_content() { + for (original, edited, kept_id) in [ + (None, Some("-2000000001"), None), + (Some("-2000000001"), None, Some("-2000000001")), + (Some("11"), Some("12"), Some("11")), + ] { + let (kinds, tracked) = compared_kinds( + &table_of_contents_control(original, "Alpha"), + &table_of_contents_control(edited, "Alpha"), + ); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + let ids = [original, edited] + .into_iter() + .flatten() + .filter(|id| tracked.contains(&format!(r#""#))) + .collect::>(); + assert_eq!(ids, kept_id.into_iter().collect::>(), "{tracked}"); + + let (kinds, tracked) = compared_kinds( + &table_of_contents_control(original, "Alpha"), + &table_of_contents_control(edited, "Delta"), + ); + assert_eq!( + kinds, + [RevisionKind::Deletion, RevisionKind::Insertion], + "{original:?} -> {edited:?}" + ); + assert_eq!(tracked.matches("").count(), 1, "{tracked}"); + } + + let (kinds, _) = compared_kinds( + &google_docs_inline_control("-1854911024", "Alpha"), + &google_docs_inline_control("1374263513", "Alpha"), + ); + assert_eq!(kinds, []); + let (kinds, tracked) = compared_kinds( + &google_docs_inline_control("-1854911024", "Alpha"), + &google_docs_inline_control("1374263513", "Delta"), + ); + assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]); + assert!( + tracked.contains(r#""#), + "{tracked}" + ); + + let bare_control = |properties: &str| { + wrap_word_body(&format!( + r#"{properties}Alpha entryAfter the content control."# + )) + }; + let id_only = r#""#; + for (original, edited) in [("", id_only), (id_only, "")] { + let (kinds, tracked) = compared_kinds(&bare_control(original), &bare_control(edited)); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + assert_eq!( + tracked.contains(""), + !original.is_empty(), + "{tracked}" + ); + } + } + + fn replaced_copy(source_xml: &str, old: &str, new: &str) -> String { + let mut document = document_with_content_controls(source_xml); + assert_eq!(document.try_replace_text(old, new).unwrap(), 1); + let mut edited = Document::from_bytes(&document.to_bytes().unwrap()).unwrap(); + document_xml(&mut edited) + } + + #[test] + fn a_rewritten_run_keeps_its_producer_space_flag() { + let preserved = wrap_word_body( + r#"Paragraph 1.WORD"#, + ); + let edited = replaced_copy(&preserved, "WORD", "WORD"); + assert!( + edited.contains(r#"WORD"#), + "{edited}" + ); + let (kinds, _) = compared_kinds(&preserved, &edited); + assert_eq!(kinds, []); + + let edge_space_removed = replaced_copy( + &wrap_word_body(r#"WORD "#), + "WORD ", + "WORD", + ); + assert!( + edge_space_removed.contains(r#"WORD"#), + "{edge_space_removed}" + ); + } + + #[test] + fn the_space_flag_is_content_only_at_a_text_edge() { + let paragraphs = |first: &str, second: &str| { + wrap_word_body(&format!( + r#"{first}{second}"# + )) + }; + let paragraph = |text: &str| paragraphs("Paragraph 1.", text); + // Bookmark ends are indexed by run, so an unchanged run must stay whole. + let bookmarked = |text: &str| { + wrap_word_body(&format!( + r#"{text}tail"# + )) + }; + let changed = &[RevisionKind::Deletion, RevisionKind::Insertion][..]; + let unchanged = &[][..]; + // Each case gives the kinds of the whole-run path and of the + // attributed path. The attributed path writes every unit with edge + // whitespace with the flag, so the flag is never content there. + let cases = [ + ( + paragraph(r#"WORD"#), + paragraph("WORD"), + unchanged, + unchanged, + ), + ( + paragraph(r#"two words"#), + paragraph("two words"), + unchanged, + unchanged, + ), + ( + paragraph("two words"), + paragraph(r#"two words"#), + unchanged, + unchanged, + ), + ( + paragraph(r#"WORD "#), + paragraph("WORD "), + changed, + unchanged, + ), + ( + bookmarked(r#"two words "#), + bookmarked("two words "), + changed, + unchanged, + ), + ( + paragraph("WORD"), + paragraph("text"), + changed, + changed, + ), + // Google Docs flags every `w:t` and Word only where needed. + ( + paragraphs( + r#"two words"#, + r#"WORD"#, + ), + paragraphs("two words", "text"), + changed, + changed, + ), + ]; + for options in [ + ComparisonOptions::default(), + ComparisonOptions { + ignore_formatting: true, + ..Default::default() + }, + ComparisonOptions { + granularity: ComparisonGranularity::Word, + ..Default::default() + }, + ComparisonOptions { + granularity: ComparisonGranularity::Character, + ..Default::default() + }, + ] { + for (original, edited, whole_run, attributed) in &cases { + let expected = if options == ComparisonOptions::default() { + whole_run + } else { + attributed + }; + for (left, right) in [(original, edited), (edited, original)] { + let (kinds, _) = compared_kinds_with(left, right, &options); + assert_eq!(kinds, *expected, "{options:?}: {left} -> {right}"); + } + } + } + + // A space split out of a text keeps reading as a space in the redline. + for granularity in [ + ComparisonGranularity::Word, + ComparisonGranularity::Character, + ] { + let (kinds, tracked) = compared_kinds_with( + ¶graph("two words"), + ¶graph("two wordy"), + &ComparisonOptions { + granularity, + ..Default::default() + }, + ); + assert_eq!(kinds, changed, "{granularity:?}"); + assert!(!tracked.contains(" "), "{tracked}"); + assert!( + tracked.contains(r#" "#), + "{tracked}" + ); + } + } + + fn page_body(orientation: &str) -> String { + wrap_word_body(&format!( + r#"Paragraph 1, lorem ipsum dolor sit amet.WORDParagraph 3, lorem ipsum dolor sit amet."# + )) + } + + #[test] + fn the_default_page_orientation_is_not_a_section_change() { + let portrait = page_body(r#" w:orient="portrait""#); + let edited = replaced_copy(&portrait, "3, lorem", "3, LOREM"); + assert!(!edited.contains("w:orient"), "{edited}"); + let (kinds, _) = compared_kinds(&portrait, &edited); + assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]); + + let (kinds, _) = compared_kinds(&portrait, &page_body("")); + assert_eq!(kinds, []); + let (kinds, _) = compared_kinds(&page_body(""), &portrait); + assert_eq!(kinds, []); + + let (kinds, tracked) = compared_kinds(&portrait, &page_body(r#" w:orient="landscape""#)); + assert_eq!(kinds, [RevisionKind::SectionPropertyChange]); + assert!(tracked.contains(r#"w:orient="landscape""#), "{tracked}"); + + // A bookmark makes the section-break paragraph take the complex path, + // as Google Docs writes around headings. + let bookmarked_break = |orientation: &str, heading: &str| { + wrap_word_body(&format!( + r#"{heading}Second section."# + )) + }; + let portrait = r#" w:orient="portrait""#; + for (original, edited) in [(portrait, ""), ("", portrait)] { + let (kinds, _) = compared_kinds( + &bookmarked_break(original, "Heading"), + &bookmarked_break(edited, "Heading"), + ); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + let (kinds, _) = compared_kinds( + &bookmarked_break(original, "Heading"), + &bookmarked_break(edited, "Title"), + ); + assert_eq!( + kinds, + [RevisionKind::Deletion, RevisionKind::Insertion], + "{original:?} -> {edited:?}" + ); + } + } + + fn document_with_footer(footer_paragraph: &str) -> Document { + let mut seed = Document::new(); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + package.set_part( + "/word/footer1.xml", + format!(r#"{footer_paragraph}"#).into_bytes(), + ); + package.content_types.add_override( + "/word/footer1.xml", + "application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml", + ); + let footer_id = package + .get_or_create_part_rels("/word/document.xml") + .add(oxml_opc::relationship::rel_types::FOOTER, "footer1.xml"); + package.set_part( + "/word/document.xml", + format!( + r#"Lorem ipsum dolor sit amet."# + ) + .into_bytes(), + ); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + fn page_field(packed: bool, cached: bool) -> String { + let parts = [ + r#""#, + r#" PAGE "#, + r#""#, + if cached { "1" } else { "" }, + r#""#, + ]; + let field = if packed { + format!("{}", parts.concat()) + } else { + parts + .iter() + .filter(|part| !part.is_empty()) + .map(|part| format!("{part}")) + .collect() + }; + format!(r#"Page {field}"#) + } + + /// Compare a footer field against its copy refreshed by `update_page_fields`. + fn refreshed_field_comparison(packed: bool, cached: bool) -> String { + let source = document_with_footer(&page_field(packed, cached)) + .to_bytes() + .unwrap(); + let mut refreshed = Document::from_bytes(&source).unwrap(); + refreshed.update_page_fields().unwrap(); + let refreshed = Document::from_bytes(&refreshed.to_bytes().unwrap()).unwrap(); + let mut compared = Document::from_bytes(&source).unwrap(); + let diagnostics = compared + .compare(&refreshed, "R", TIMESTAMP) + .unwrap_or_else(|error| panic!("packed={packed} cached={cached}: {error}")); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_resolutions( + &compared.to_bytes().unwrap(), + &Document::from_bytes(&source).unwrap(), + &refreshed, + &ComparisonOptions::default(), + ); + comparison_part_xml(&mut compared, "/word/footer1.xml") + } + + #[test] + fn a_packed_field_compares_like_the_same_field_split_into_runs() { + let separate = r#""#; + let result_and_end = |footer: &str| footer[footer.find(separate).unwrap()..].to_owned(); + for (cached, deleted) in [(true, "1"), (false, "")] { + let split = refreshed_field_comparison(false, cached); + let packed = refreshed_field_comparison(true, cached); + assert_eq!( + result_and_end(&split), + format!( + r#"{separate}{deleted}1"# + ), + "cached={cached}" + ); + assert_eq!( + result_and_end(&packed), + result_and_end(&split), + "cached={cached}" + ); + assert!( + packed.contains(r#" PAGE "#), + "{packed}" + ); + } + } +} + #[test] fn unmodelled_property_changes_report_a_diagnostic() { for (original, edited, expected_location) in [ diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 97ce19c6..ca07b49e 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -910,9 +910,12 @@ section properties emit property revisions that retain the original property sidecars. Unsupported formatting differences retain the original bytes and produce stable `ComparisonDiagnostic` values at the actual story path. Inputs with existing modeled revisions or differing story and control shells are -rejected unless their story category is ignored. Attributed text alignment -retains owner, formatting, content position, and raw-child boundaries, then -coalesces adjacent equal-owner edits into minimal revision wrappers. +rejected unless their story category is ignored. A content control's `w:id` +is producer identity and not part of its shell, so controls that differ only +by it align, compare, and keep the original `w:sdtPr`. Attributed text +alignment retains owner, formatting, content position, and raw-child +boundaries, then coalesces adjacent equal-owner edits into minimal revision +wrappers. When a main story gains a trailing run of paragraphs, comparison marks the original final paragraph boundary once, marks each intermediate inserted paragraph boundary once, and leaves the final inserted paragraph mark as the diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643b..0dabe703 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -383,12 +383,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), - "rdocx-cli": (33_805, 145_256, 8), + "rdocx": (1_104_100, 6_548_342, 36), + "rdocx-cli": (35_147, 149_734, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (375_767, 2_412_318, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),