From 267a1524148a0e615231dcaa9ebbc6d4a783f9d2 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:20:30 +0200 Subject: [PATCH 01/10] Ignore the content-control id when comparing documents compare() refused any pair whose content controls differed only by w:sdtPr/w:id, with "comparison cannot revise content-control properties". Word and Google Docs renumber that id when they save, so two versions of a document with a table-of-contents control could not be compared at all, although nothing in the control had changed. The id was part of control_property_signature, the one tuple that drives control alignment, the refusal and the accept and reject postconditions. Leaving it out of that tuple keeps the three consistent. Controls that differ only by id now align as unchanged and the redline keeps the original w:sdtPr bytes, and a change inside such a control is revised like any other. A w:sdtPr that holds none of the compared properties reads like no w:sdtPr, so a control that gains or loses an id-only w:sdtPr compares too. A differing w:tag, alias, control type or data binding still refuses the pair. HLD 03 now says the id is not part of the control shell. GitHub issue #159. --- crates/rdocx/src/comparison.rs | 28 ++++-- crates/rdocx/tests/regression_test.rs | 128 ++++++++++++++++++++++++++ docs/hld/03-architecture.md | 9 +- 3 files changed, 152 insertions(+), 13 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 19c557ed0..5553d681f 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -29,7 +29,6 @@ thread_local! { type ControlPropertySignature<'a> = Option<( Option<&'a str>, Option<&'a str>, - Option, Option, Option<&'a rdocx_oxml::content_control::CT_DataBinding>, )>; @@ -5478,16 +5477,25 @@ fn control_signature(control: &CT_Sdt) -> String { ) } +/// The content-control properties that alignment, refusal and the accept and +/// reject postconditions compare. +/// +/// `w:id` is left out. Producers renumber it on save and it carries no +/// content, so a pair that differs only by it keeps the original's `w:sdtPr`. +/// A `w:sdtPr` with none of these properties reads like no `w:sdtPr`. fn control_property_signature(control: &CT_Sdt) -> ControlPropertySignature<'_> { - control.properties.as_ref().map(|properties| { - ( - properties.alias.as_deref(), - properties.tag.as_deref(), - properties.id, - properties.control_type, - properties.data_binding.as_ref(), - ) - }) + control + .properties + .as_ref() + .map(|properties| { + ( + properties.alias.as_deref(), + properties.tag.as_deref(), + properties.control_type, + properties.data_binding.as_ref(), + ) + }) + .filter(|signature| !matches!(signature, (None, None, None, None))) } fn modeled_control_content(control: &CT_Sdt) -> Vec<&SdtContent> { diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c74..dbd61d6fd 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30379,6 +30379,134 @@ fn comparison_treats_empty_paragraph_properties_as_absent() { assert!(document_xml(&mut empty).contains(" Vec { + document + .revisions() + .iter() + .map(|revision| revision.kind()) + .collect() + } + + /// Check that accepting gives the edited side and rejecting the original, + /// each compared again with no diagnostic and no revision. + fn assert_resolutions(tracked: &[u8], original: &Document, edited: &Document) { + for (resolve, expected) in [ + ( + Document::accept_all as fn(&mut Document) -> rdocx::Result, + edited, + ), + (Document::reject_all, original), + ] { + let mut resolved = Document::from_bytes(tracked).unwrap(); + resolve(&mut resolved).unwrap(); + let diagnostics = resolved + .compare(expected, "postcondition", TIMESTAMP) + .unwrap(); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_eq!(revision_kinds(&resolved), []); + } + } + + /// Compare two bodies and return the revision kinds and the redline. + fn compared_kinds(original_xml: &str, edited_xml: &str) -> (Vec, String) { + let original = document_with_content_controls(original_xml); + let edited = document_with_content_controls(edited_xml); + let mut compared = document_with_content_controls(original_xml); + let diagnostics = compared + .compare(&edited, "R", TIMESTAMP) + .expect("producer noise must not refuse the pair"); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited); + (revision_kinds(&compared), document_xml(&mut compared)) + } + + fn table_of_contents_control(id: Option<&str>, first_entry: &str) -> String { + let id = id.map_or_else(String::new, |value| format!(r#""#)); + wrap_word_body(&format!( + r#"Before the content control.{id}{first_entry} entryBeta entryGamma entryAfter the content control."# + )) + } + + fn google_docs_inline_control(id: &str, word: &str) -> String { + wrap_word_body(&format!( + r#"Before {word} after."# + )) + } + + #[test] + fn a_content_control_identity_is_not_content() { + for (original, edited, kept_id) in [ + (None, Some("-2000000001"), None), + (Some("-2000000001"), None, Some("-2000000001")), + (Some("11"), Some("12"), Some("11")), + ] { + let (kinds, tracked) = compared_kinds( + &table_of_contents_control(original, "Alpha"), + &table_of_contents_control(edited, "Alpha"), + ); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + let ids = [original, edited] + .into_iter() + .flatten() + .filter(|id| tracked.contains(&format!(r#""#))) + .collect::>(); + assert_eq!(ids, kept_id.into_iter().collect::>(), "{tracked}"); + + let (kinds, tracked) = compared_kinds( + &table_of_contents_control(original, "Alpha"), + &table_of_contents_control(edited, "Delta"), + ); + assert_eq!( + kinds, + [RevisionKind::Deletion, RevisionKind::Insertion], + "{original:?} -> {edited:?}" + ); + assert_eq!(tracked.matches("").count(), 1, "{tracked}"); + } + + let (kinds, _) = compared_kinds( + &google_docs_inline_control("-1854911024", "Alpha"), + &google_docs_inline_control("1374263513", "Alpha"), + ); + assert_eq!(kinds, []); + let (kinds, tracked) = compared_kinds( + &google_docs_inline_control("-1854911024", "Alpha"), + &google_docs_inline_control("1374263513", "Delta"), + ); + assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]); + assert!( + tracked.contains(r#""#), + "{tracked}" + ); + + let bare_control = |properties: &str| { + wrap_word_body(&format!( + r#"{properties}Alpha entryAfter the content control."# + )) + }; + let id_only = r#""#; + for (original, edited) in [("", id_only), (id_only, "")] { + let (kinds, tracked) = compared_kinds(&bare_control(original), &bare_control(edited)); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + assert_eq!( + tracked.contains(""), + !original.is_empty(), + "{tracked}" + ); + } + } +} + #[test] fn unmodelled_property_changes_report_a_diagnostic() { for (original, edited, expected_location) in [ diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 97ce19c68..ca07b49e6 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -910,9 +910,12 @@ section properties emit property revisions that retain the original property sidecars. Unsupported formatting differences retain the original bytes and produce stable `ComparisonDiagnostic` values at the actual story path. Inputs with existing modeled revisions or differing story and control shells are -rejected unless their story category is ignored. Attributed text alignment -retains owner, formatting, content position, and raw-child boundaries, then -coalesces adjacent equal-owner edits into minimal revision wrappers. +rejected unless their story category is ignored. A content control's `w:id` +is producer identity and not part of its shell, so controls that differ only +by it align, compare, and keep the original `w:sdtPr`. Attributed text +alignment retains owner, formatting, content position, and raw-child +boundaries, then coalesces adjacent equal-owner edits into minimal revision +wrappers. When a main story gains a trailing run of paragraphs, comparison marks the original final paragraph boundary once, marks each intermediate inserted paragraph boundary once, and leaves the final inserted paragraph mark as the From 2e6b3b0cfd211978b8a453d2bbc4ef516a5f406f Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:21:08 +0200 Subject: [PATCH 02/10] Keep xml:space on replaced runs and compare it only at text edges A run rewritten by try_replace_text lost xml:space="preserve" on its w:t, because replace_in_single_run and replace_across_runs recomputed the flag from the new text alone. Google Docs writes the flag on every w:t, so a no-op replacement, common in a scripted pass that normalizes wording, turned an unchanged run into a deletion and an insertion once the edited copy was compared against its source. The three rewrite sites, which regex replacement shares, now keep a flag the producer wrote and still add one when the new text starts or ends with a space. compare() also read the flag as content wherever it appeared, because the run signature formatted CT_Text whole. The flag only changes how text reads when whitespace sits at an edge, so the text signature now counts it only there. The same signature serves alignment and the accept and reject postconditions, so they stay consistent, and a run that matches keeps the original bytes. Word and character granularity and the ignore options compare text units, which the redline writes back as their own w:t. Each unit copied the flag of its text, so a space split out of a flagged text differed from the same space in an unflagged one. A unit with whitespace at an edge now always carries the flag, so on that path whitespace reads as whitespace whatever the source flag was, and the flag is never content. Files from two producers that place the flag differently then compare the same way at every granularity, and a space split out of a text keeps the flag in the redline, where Word used to drop it. On that path, the shortcut that keeps runs whole when every run of a paragraph matches reads the flag the same way. Otherwise a run that differed only by the flag at an edge was rewritten one unit per run, which moved a bookmark end indexed by run and refused the pair. GitHub issue #160. --- crates/rdocx-oxml/src/placeholder.rs | 37 +++++- crates/rdocx/src/comparison.rs | 51 +++++++- crates/rdocx/tests/regression_test.rs | 171 +++++++++++++++++++++++++- 3 files changed, 248 insertions(+), 11 deletions(-) diff --git a/crates/rdocx-oxml/src/placeholder.rs b/crates/rdocx-oxml/src/placeholder.rs index 0bf078dd4..597803505 100644 --- a/crates/rdocx-oxml/src/placeholder.rs +++ b/crates/rdocx-oxml/src/placeholder.rs @@ -153,7 +153,9 @@ fn replace_in_single_run( new_text.push_str(replacement); new_text.push_str(&t.text[byte_end..]); t.text = new_text; - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + // Keep the flag the producer wrote. Dropping it rewrites an unchanged + // run, which a later comparison against the source then reports. + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } } @@ -176,7 +178,7 @@ fn replace_across_runs( new_text.push_str(&t.text[..first_byte_offset]); new_text.push_str(replacement); t.text = new_text; - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } // Handle the last run: replace from start to match end within that content item. @@ -189,7 +191,7 @@ fn replace_across_runs( let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0); let byte_end = last_byte_offset + ch_len; t.text = t.text[byte_end..].to_string(); - t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' '); + t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' '); } // Clear text content from runs strictly between first and last. @@ -796,6 +798,35 @@ mod tests { assert_eq!(p.runs[1].properties.as_ref().unwrap().italic, Some(true)); } + #[test] + fn replace_keeps_the_producer_space_flag() { + let mut preserved = CT_R::new("WORD"); + if let RunContent::Text(text) = &mut preserved.content[0] { + text.preserve_space = true; + } + let mut p = CT_P::new(); + p.runs.push(preserved.clone()); + p.runs.push(preserved); + p.add_run("tail"); + + assert_eq!(replace_in_paragraph(&mut p, "WORD", "WORD"), 2); + assert_eq!(replace_in_paragraph(&mut p, "DWO", "D-WO"), 1); + assert_eq!(replace_in_paragraph(&mut p, "tail", " end"), 1); + let flags = p + .runs + .iter() + .map(|run| match &run.content[0] { + RunContent::Text(text) => (text.text.as_str(), text.preserve_space), + _ => unreachable!(), + }) + .collect::>(); + assert_eq!( + flags, + [("WORD-WO", true), ("RD", true), (" end", true)], + "a rewritten run keeps a producer flag and gains one at a text edge" + ); + } + #[test] fn replace_multiple_occurrences() { let mut p = make_para(&["{{x}} and {{x}}"]); diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 5553d681f..02c33e980 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -2984,8 +2984,16 @@ fn compare_granular_paragraph( ))); } - let original_run_signatures = original.runs.iter().map(run_signature).collect::>(); - let edited_run_signatures = edited.runs.iter().map(run_signature).collect::>(); + let original_run_signatures = original + .runs + .iter() + .map(attributed_run_signature) + .collect::>(); + let edited_run_signatures = edited + .runs + .iter() + .map(attributed_run_signature) + .collect::>(); if original_run_signatures == edited_run_signatures && original.content_controls == edited.content_controls { @@ -3557,12 +3565,30 @@ fn granular_text(text: &CT_Text, options: &ComparisonOptions) -> Vec { fragments .into_iter() .map(|value| CT_Text { + // A unit is written back as its own `w:t`, where edge whitespace + // needs the flag to survive. Whitespace in a unit then reads as + // whitespace whatever the source flag was. + preserve_space: text.preserve_space || has_edge_whitespace(&value), text: value, - preserve_space: text.preserve_space, }) .collect() } +/// The whole-run signature that agrees with the attributed units. +/// +/// It reads the space flag the way `granular_text` writes it on every unit, +/// so a run that differs only by the flag stays whole instead of being +/// rewritten one unit per run, which would move run-indexed bookmark ends. +fn attributed_run_signature(run: &CT_R) -> String { + let mut run = run.clone(); + for content in &mut run.content { + if let RunContent::Text(text) | RunContent::DeletedText(text) = content { + text.preserve_space |= has_edge_whitespace(&text.text); + } + } + run_signature(&run) +} + fn whitespace_fragments(text: &str) -> Vec { let mut output = Vec::new(); let mut current = String::new(); @@ -5410,10 +5436,29 @@ fn run_content_signature(content: &RunContent) -> String { match content { RunContent::Field(_) => "field-owner".to_owned(), RunContent::Drawing(drawing) => format!("Drawing({:?})", drawing_signature(drawing)), + RunContent::Text(text) => format!("Text({:?})", text_signature(text)), + RunContent::DeletedText(text) => format!("DeletedText({:?})", text_signature(text)), content => format!("{content:?}"), } } +/// The text and whether its `xml:space="preserve"` changes how it reads. +/// +/// The flag only protects whitespace at an edge of the text. Producers write +/// it on every `w:t` or only where needed, so elsewhere it is serialization. +fn text_signature(text: &CT_Text) -> (&str, bool) { + ( + &text.text, + text.preserve_space && has_edge_whitespace(&text.text), + ) +} + +/// Whether XML whitespace starts or ends the text. +fn has_edge_whitespace(text: &str) -> bool { + let whitespace = |character: char| matches!(character, ' ' | '\t' | '\n' | '\r'); + text.starts_with(whitespace) || text.ends_with(whitespace) +} + fn table_signature(table: &CT_Tbl) -> String { format!( "{:?}:{:?}:{:?}:{:?}", diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index dbd61d6fd..5afa2345b 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30385,7 +30385,7 @@ fn comparison_treats_empty_paragraph_properties_as_absent() { /// revision. mod compare_producer_noise { use super::*; - use rdocx::RevisionKind; + use rdocx::{ComparisonGranularity, ComparisonOptions, RevisionKind}; const TIMESTAMP: &str = "2026-09-27T12:00:00Z"; @@ -30399,7 +30399,12 @@ mod compare_producer_noise { /// Check that accepting gives the edited side and rejecting the original, /// each compared again with no diagnostic and no revision. - fn assert_resolutions(tracked: &[u8], original: &Document, edited: &Document) { + fn assert_resolutions( + tracked: &[u8], + original: &Document, + edited: &Document, + options: &ComparisonOptions, + ) { for (resolve, expected) in [ ( Document::accept_all as fn(&mut Document) -> rdocx::Result, @@ -30410,7 +30415,7 @@ mod compare_producer_noise { let mut resolved = Document::from_bytes(tracked).unwrap(); resolve(&mut resolved).unwrap(); let diagnostics = resolved - .compare(expected, "postcondition", TIMESTAMP) + .compare_with_options(expected, "postcondition", TIMESTAMP, options) .unwrap(); assert!(diagnostics.is_empty(), "{diagnostics:?}"); assert_eq!(revision_kinds(&resolved), []); @@ -30419,14 +30424,22 @@ mod compare_producer_noise { /// Compare two bodies and return the revision kinds and the redline. fn compared_kinds(original_xml: &str, edited_xml: &str) -> (Vec, String) { + compared_kinds_with(original_xml, edited_xml, &ComparisonOptions::default()) + } + + fn compared_kinds_with( + original_xml: &str, + edited_xml: &str, + options: &ComparisonOptions, + ) -> (Vec, String) { let original = document_with_content_controls(original_xml); let edited = document_with_content_controls(edited_xml); let mut compared = document_with_content_controls(original_xml); let diagnostics = compared - .compare(&edited, "R", TIMESTAMP) + .compare_with_options(&edited, "R", TIMESTAMP, options) .expect("producer noise must not refuse the pair"); assert!(diagnostics.is_empty(), "{diagnostics:?}"); - assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited); + assert_resolutions(&compared.to_bytes().unwrap(), &original, &edited, options); (revision_kinds(&compared), document_xml(&mut compared)) } @@ -30505,6 +30518,154 @@ mod compare_producer_noise { ); } } + + fn replaced_copy(source_xml: &str, old: &str, new: &str) -> String { + let mut document = document_with_content_controls(source_xml); + assert_eq!(document.try_replace_text(old, new).unwrap(), 1); + let mut edited = Document::from_bytes(&document.to_bytes().unwrap()).unwrap(); + document_xml(&mut edited) + } + + #[test] + fn a_rewritten_run_keeps_its_producer_space_flag() { + let preserved = wrap_word_body( + r#"Paragraph 1.WORD"#, + ); + let edited = replaced_copy(&preserved, "WORD", "WORD"); + assert!( + edited.contains(r#"WORD"#), + "{edited}" + ); + let (kinds, _) = compared_kinds(&preserved, &edited); + assert_eq!(kinds, []); + + let edge_space_removed = replaced_copy( + &wrap_word_body(r#"WORD "#), + "WORD ", + "WORD", + ); + assert!( + edge_space_removed.contains(r#"WORD"#), + "{edge_space_removed}" + ); + } + + #[test] + fn the_space_flag_is_content_only_at_a_text_edge() { + let paragraphs = |first: &str, second: &str| { + wrap_word_body(&format!( + r#"{first}{second}"# + )) + }; + let paragraph = |text: &str| paragraphs("Paragraph 1.", text); + // Bookmark ends are indexed by run, so an unchanged run must stay whole. + let bookmarked = |text: &str| { + wrap_word_body(&format!( + r#"{text}tail"# + )) + }; + let changed = &[RevisionKind::Deletion, RevisionKind::Insertion][..]; + let unchanged = &[][..]; + // Each case gives the kinds of the whole-run path and of the + // attributed path. The attributed path writes every unit with edge + // whitespace with the flag, so the flag is never content there. + let cases = [ + ( + paragraph(r#"WORD"#), + paragraph("WORD"), + unchanged, + unchanged, + ), + ( + paragraph(r#"two words"#), + paragraph("two words"), + unchanged, + unchanged, + ), + ( + paragraph("two words"), + paragraph(r#"two words"#), + unchanged, + unchanged, + ), + ( + paragraph(r#"WORD "#), + paragraph("WORD "), + changed, + unchanged, + ), + ( + bookmarked(r#"two words "#), + bookmarked("two words "), + changed, + unchanged, + ), + ( + paragraph("WORD"), + paragraph("text"), + changed, + changed, + ), + // Google Docs flags every `w:t` and Word only where needed. + ( + paragraphs( + r#"two words"#, + r#"WORD"#, + ), + paragraphs("two words", "text"), + changed, + changed, + ), + ]; + for options in [ + ComparisonOptions::default(), + ComparisonOptions { + ignore_formatting: true, + ..Default::default() + }, + ComparisonOptions { + granularity: ComparisonGranularity::Word, + ..Default::default() + }, + ComparisonOptions { + granularity: ComparisonGranularity::Character, + ..Default::default() + }, + ] { + for (original, edited, whole_run, attributed) in &cases { + let expected = if options == ComparisonOptions::default() { + whole_run + } else { + attributed + }; + for (left, right) in [(original, edited), (edited, original)] { + let (kinds, _) = compared_kinds_with(left, right, &options); + assert_eq!(kinds, *expected, "{options:?}: {left} -> {right}"); + } + } + } + + // A space split out of a text keeps reading as a space in the redline. + for granularity in [ + ComparisonGranularity::Word, + ComparisonGranularity::Character, + ] { + let (kinds, tracked) = compared_kinds_with( + ¶graph("two words"), + ¶graph("two wordy"), + &ComparisonOptions { + granularity, + ..Default::default() + }, + ); + assert_eq!(kinds, changed, "{granularity:?}"); + assert!(!tracked.contains(" "), "{tracked}"); + assert!( + tracked.contains(r#" "#), + "{tracked}" + ); + } + } } #[test] From c8880e276a7ac6c0414af67a5f45355db6a99973 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:21:40 +0200 Subject: [PATCH 03/10] Treat the default portrait orientation as absent in compare Comparing a file whose w:pgSz carries w:orient="portrait" against its rdocx-edited copy reported a section_property_change next to the real edit. The pgSz writer emits w:orient only for landscape, so a modelled edit drops the default value, and section_properties_xml compared the two sections as whole CT_SectPr values, Some(Portrait) against None. Portrait is the schema default, so both modelled sections now read it as absent before the equality test. A paragraph that holds a bookmark, hyperlink, comment range or control compares its whole paragraph properties, section break included, so the same rule applies there. Without it such a section-break paragraph kept a formatting diagnostic, or refused the pair when its text changed. A real change to landscape is still a section property revision. The serializer is left alone, because the typed defaults of every generated document carry Some(Portrait) and writing it would move the document.xml hash baselines. GitHub issue #160. --- crates/rdocx/src/comparison.rs | 24 +++++++++++-- crates/rdocx/tests/regression_test.rs | 49 +++++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 2 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 02c33e980..37d281268 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -11,6 +11,7 @@ use rdocx_oxml::content_control::{CT_Sdt, SdtContent}; use rdocx_oxml::document::{BodyContent, CT_Document}; use rdocx_oxml::namespace::W_NS; use rdocx_oxml::properties::CT_PPr; +use rdocx_oxml::shared::ST_PageOrientation; use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_TblPr, CT_Tc, CT_TrPr, CellContent}; use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent}; use sha2::{Digest, Sha256}; @@ -4002,8 +4003,22 @@ fn nonempty_paragraph_properties(mut properties: CT_PPr) -> Option { } fn paragraph_properties_differ(original: Option<&CT_PPr>, edited: Option<&CT_PPr>) -> bool { - original.cloned().and_then(nonempty_paragraph_properties) - != edited.cloned().and_then(nonempty_paragraph_properties) + let modeled = |properties: Option<&CT_PPr>| { + properties.cloned().and_then(|mut properties| { + if let Some(section) = properties.sect_pr.as_mut() { + clear_default_orientation(section); + } + nonempty_paragraph_properties(properties) + }) + }; + modeled(original) != modeled(edited) +} + +/// Portrait is the schema default, which the section writer omits. +fn clear_default_orientation(section: &mut rdocx_oxml::document::CT_SectPr) { + if section.orientation == Some(ST_PageOrientation::Portrait) { + section.orientation = None; + } } fn section_properties_xml( @@ -4033,6 +4048,8 @@ fn section_properties_xml( original_modeled.change = None; let mut edited_modeled = edited.clone(); edited_modeled.change = None; + clear_default_orientation(&mut original_modeled); + clear_default_orientation(&mut edited_modeled); if original_modeled == edited_modeled { return section_property_xml(original); } @@ -5810,6 +5827,9 @@ fn paragraph_formatting(paragraph: &CT_P) -> Option { properties.numbering_revision_position = None; properties.change = None; properties.revision_xml.clear(); + if let Some(section) = properties.sect_pr.as_mut() { + clear_default_orientation(section); + } properties }) } diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 5afa2345b..08f17b7a9 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30666,6 +30666,55 @@ mod compare_producer_noise { ); } } + + fn page_body(orientation: &str) -> String { + wrap_word_body(&format!( + r#"Paragraph 1, lorem ipsum dolor sit amet.WORDParagraph 3, lorem ipsum dolor sit amet."# + )) + } + + #[test] + fn the_default_page_orientation_is_not_a_section_change() { + let portrait = page_body(r#" w:orient="portrait""#); + let edited = replaced_copy(&portrait, "3, lorem", "3, LOREM"); + assert!(!edited.contains("w:orient"), "{edited}"); + let (kinds, _) = compared_kinds(&portrait, &edited); + assert_eq!(kinds, [RevisionKind::Deletion, RevisionKind::Insertion]); + + let (kinds, _) = compared_kinds(&portrait, &page_body("")); + assert_eq!(kinds, []); + let (kinds, _) = compared_kinds(&page_body(""), &portrait); + assert_eq!(kinds, []); + + let (kinds, tracked) = compared_kinds(&portrait, &page_body(r#" w:orient="landscape""#)); + assert_eq!(kinds, [RevisionKind::SectionPropertyChange]); + assert!(tracked.contains(r#"w:orient="landscape""#), "{tracked}"); + + // A bookmark makes the section-break paragraph take the complex path, + // as Google Docs writes around headings. + let bookmarked_break = |orientation: &str, heading: &str| { + wrap_word_body(&format!( + r#"{heading}Second section."# + )) + }; + let portrait = r#" w:orient="portrait""#; + for (original, edited) in [(portrait, ""), ("", portrait)] { + let (kinds, _) = compared_kinds( + &bookmarked_break(original, "Heading"), + &bookmarked_break(edited, "Heading"), + ); + assert_eq!(kinds, [], "{original:?} -> {edited:?}"); + let (kinds, _) = compared_kinds( + &bookmarked_break(original, "Heading"), + &bookmarked_break(edited, "Title"), + ); + assert_eq!( + kinds, + [RevisionKind::Deletion, RevisionKind::Insertion], + "{original:?} -> {edited:?}" + ); + } + } } #[test] From a607736f5a7d8b2dc3789ab5893b9ccf2e4da104 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:22:23 +0200 Subject: [PATCH 04/10] Compare a field packed in one run like the same field split into runs compare() failed with "complex field source has no end boundary" on a file against its own copy once update_page_fields had refreshed a field packed in one run, as Google Docs writes footer fields. The refreshed result lands inside that run and changes w:dirty, so compare_field_xml extracts the result on both sides, and complex_field_result looked for the end character only after the end of the run holding separate. In a packed run the end sits inside that same run. complex_field_result now splits the run so that separate ends a run and end starts one before it reads the result. The new runs repeat the original start tag and run properties, and a run with nothing on the far side of the character is left alone, so a field whose separate and end each have a run of their own yields the same bytes as before. The postconditions read a field as its owner only, so the split does not reach them. The same split fixes a quieter loss. update_page_fields writes the result of an uncached one-run-per-part field into the run of end, and the redline then carried an empty insertion without the new page number. It now inserts the result. GitHub issue #160. --- crates/rdocx/src/comparison.rs | 96 ++++++++++++++++++++++++--- crates/rdocx/tests/regression_test.rs | 96 +++++++++++++++++++++++++++ 2 files changed, 182 insertions(+), 10 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 37d281268..7f7ff9d1a 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -5141,17 +5141,20 @@ fn deleted_text_xml(xml: &str) -> String { } fn complex_field_result(xml: &str) -> Result<(String, String, String)> { - let separate = xml - .find("fldCharType=\"separate\"") - .or_else(|| xml.find("fldCharType='separate'")) + let separate = field_character(xml, 0, "separate") .ok_or_else(|| Error::Other("complex field source has no separate boundary".to_owned()))?; - let (_, result_start) = containing_run(xml, separate)?; - let end_marker = xml[result_start..] - .find("fldCharType=\"end\"") - .or_else(|| xml[result_start..].find("fldCharType='end'")) - .map(|offset| result_start + offset) + // A producer may pack a whole field in one run, as Google Docs writes page + // fields. Ending a run after `separate` and starting one at `end` reads it + // as the same field written one run per part. + let xml = split_field_run(xml, separate, true)?; + let (_, result_start) = containing_run(&xml, separate)?; + let end_marker = field_character(&xml, result_start, "end") .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?; - let (result_end, _) = containing_run(xml, end_marker)?; + let xml = split_field_run(&xml, end_marker, false)?; + // The split may have moved the `end` character further along. + let end_marker = field_character(&xml, result_start, "end") + .ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?; + let (result_end, _) = containing_run(&xml, end_marker)?; Ok(( xml[..result_start].to_owned(), xml[result_start..result_end].to_owned(), @@ -5159,6 +5162,48 @@ fn complex_field_result(xml: &str) -> Result<(String, String, String)> { )) } +fn field_character(xml: &str, from: usize, kind: &str) -> Option { + let tail = &xml[from..]; + tail.find(&format!("fldCharType=\"{kind}\"")) + .or_else(|| tail.find(&format!("fldCharType='{kind}'"))) + .map(|offset| from + offset) +} + +/// Split the run holding the field character at `marker` so that the +/// character ends its run (`after`) or starts it. +/// +/// Both runs repeat the original start tag and run properties. A run with no +/// content on that side of the character is returned unchanged. +fn split_field_run(xml: &str, marker: usize, after: bool) -> Result { + let (run_start, run_end) = containing_run(xml, marker)?; + let run = &xml[run_start..run_end]; + let children = direct_element_spans(run)?; + let character = children + .iter() + .position(|child| child.contains(&(marker - run_start))) + .ok_or_else(|| Error::Other("complex field character is not a run child".to_owned()))?; + let is_properties = |child: &Range| { + let mut reader = Reader::from_reader(run[child.clone()].as_bytes()); + matches!( + reader.read_event(), + Ok(Event::Start(element) | Event::Empty(element)) + if element.local_name().as_ref() == b"rPr" + ) + }; + let first_content = usize::from(children.first().is_some_and(is_properties)); + let split = if after { character + 1 } else { character }; + if split <= first_content || split >= children.len() { + return Ok(xml.to_owned()); + } + let head = &run[..children[first_content].start]; + let close = run + .rfind(" Result<(usize, usize)> { let mut reader = Reader::from_reader(xml.as_bytes()); reader.config_mut().trim_text(false); @@ -6360,7 +6405,8 @@ fn utf8_error(error: impl std::fmt::Display) -> Error { mod tests { use super::{ ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING, - attributed_run_units, comparison_postcondition_error, story_document, word_fragments, + attributed_run_units, comparison_postcondition_error, complex_field_result, story_document, + word_fragments, }; use crate::Document; use rdocx_oxml::document::BodyContent; @@ -6452,6 +6498,36 @@ mod tests { ); } + #[test] + fn a_packed_field_result_is_read_as_one_run_per_part() { + for w in ["w", "q"] { + let shell = format!(r#"<{w}:r {w}:rsidR="00AB12CD"><{w}:rPr><{w}:b/>"#); + let close = format!(""); + let character = |kind: &str| format!(r#"<{w}:fldChar {w}:fldCharType="{kind}"/>"#); + let (begin, separate, end) = + (character("begin"), character("separate"), character("end")); + let code = format!("<{w}:instrText>PAGE"); + let result = format!("<{w}:t>1"); + let expected = ( + format!("{shell}{begin}{code}{separate}{close}"), + format!("{shell}{result}{close}"), + format!("{shell}{end}{close}"), + ); + for field in [ + format!("{shell}{begin}{code}{separate}{result}{end}{close}"), + format!("{shell}{begin}{code}{separate}{close}{shell}{result}{end}{close}"), + format!("{}{}{}", expected.0, expected.1, expected.2), + ] { + assert_eq!(complex_field_result(&field).unwrap(), expected, "{field}"); + } + let uncached = format!("{shell}{begin}{code}{separate}{end}{close}"); + assert_eq!( + complex_field_result(&uncached).unwrap(), + (expected.0, String::new(), expected.2) + ); + } + } + #[test] fn staged_comparison_postcondition_failure_preserves_bytes_and_layout_cache() { let mut original = Document::new(); diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 08f17b7a9..db90661d1 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -30715,6 +30715,102 @@ mod compare_producer_noise { ); } } + + fn document_with_footer(footer_paragraph: &str) -> Document { + let mut seed = Document::new(); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + package.set_part( + "/word/footer1.xml", + format!(r#"{footer_paragraph}"#).into_bytes(), + ); + package.content_types.add_override( + "/word/footer1.xml", + "application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml", + ); + let footer_id = package + .get_or_create_part_rels("/word/document.xml") + .add(oxml_opc::relationship::rel_types::FOOTER, "footer1.xml"); + package.set_part( + "/word/document.xml", + format!( + r#"Lorem ipsum dolor sit amet."# + ) + .into_bytes(), + ); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + fn page_field(packed: bool, cached: bool) -> String { + let parts = [ + r#""#, + r#" PAGE "#, + r#""#, + if cached { "1" } else { "" }, + r#""#, + ]; + let field = if packed { + format!("{}", parts.concat()) + } else { + parts + .iter() + .filter(|part| !part.is_empty()) + .map(|part| format!("{part}")) + .collect() + }; + format!(r#"Page {field}"#) + } + + /// Compare a footer field against its copy refreshed by `update_page_fields`. + fn refreshed_field_comparison(packed: bool, cached: bool) -> String { + let source = document_with_footer(&page_field(packed, cached)) + .to_bytes() + .unwrap(); + let mut refreshed = Document::from_bytes(&source).unwrap(); + refreshed.update_page_fields().unwrap(); + let refreshed = Document::from_bytes(&refreshed.to_bytes().unwrap()).unwrap(); + let mut compared = Document::from_bytes(&source).unwrap(); + let diagnostics = compared + .compare(&refreshed, "R", TIMESTAMP) + .unwrap_or_else(|error| panic!("packed={packed} cached={cached}: {error}")); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_resolutions( + &compared.to_bytes().unwrap(), + &Document::from_bytes(&source).unwrap(), + &refreshed, + &ComparisonOptions::default(), + ); + comparison_part_xml(&mut compared, "/word/footer1.xml") + } + + #[test] + fn a_packed_field_compares_like_the_same_field_split_into_runs() { + let separate = r#""#; + let result_and_end = |footer: &str| footer[footer.find(separate).unwrap()..].to_owned(); + for (cached, deleted) in [(true, "1"), (false, "")] { + let split = refreshed_field_comparison(false, cached); + let packed = refreshed_field_comparison(true, cached); + assert_eq!( + result_and_end(&split), + format!( + r#"{separate}{deleted}1"# + ), + "cached={cached}" + ); + assert_eq!( + result_and_end(&packed), + result_and_end(&split), + "cached={cached}" + ); + assert!( + packed.contains(r#" PAGE "#), + "{packed}" + ); + } + } } #[test] From 83b92bbf69942de8d7c3aacdeee19cc0da59ff26 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 19:09:51 +0200 Subject: [PATCH 05/10] Re-record the archive measurements of rdocx and rdocx-oxml The comparison fixes, the kept xml:space flag and their tests grow the rdocx and rdocx-oxml packages, so the crates.io archive rows in README.md and crates/rdocx-oxml/README.md and their ARCHIVE_MEASUREMENTS entries are re-measured. GitHub issues #159 and #160. --- README.md | 2 +- crates/rdocx-oxml/README.md | 2 +- scripts/readme_doctests.py | 4 ++-- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 479b462ef..412db0df8 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,097,973 compressed bytes, 6,523,992 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef2416..dabac1996 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 367,851 compressed bytes, 2,381,304 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643be..25c233a49 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -383,12 +383,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), + "rdocx": (1_097_973, 6_523_992, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (367_851, 2_381_304, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6), From 2fe59125e52199a496a56ddec95c21c32b925d38 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:18:26 +0200 Subject: [PATCH 06/10] Follow hyperlink and control boundaries through word alignment compare() at word or character granularity refused a pair whose only difference was a word inserted or deleted beside a hyperlink or an inline control, with "comparison cannot revise paragraph boundary structures". The attributed path compared each hyperlink by the absolute number of units before its start and end, so any edit earlier in the paragraph moved both. An inline control was compared by its run index, so a word that Word writes in a run of its own before the control refused the same way, and a redline whose runs were split into units could not be compared again against the edited side. Hyperlinks and inline controls are now compared without their place in the paragraph. The units are aligned separately between consecutive shell boundaries, so the space before a control never matches the space after it, and the redline copies the original bytes before the first run of each segment, shell tags included, before anything written for that segment. Words inserted or deleted on either side of a shell, and text inserted at the start of a hyperlink, land where the edited side has them. A word moved into a hyperlink becomes a deletion outside and an insertion inside. Text inserted between two shells with no original run between them, such as before a hyperlink that opens its paragraph, still refuses the pair. The whole-run path is unchanged. GitHub issue #161. --- crates/rdocx/src/comparison.rs | 158 +++++++++++++++++-- crates/rdocx/tests/regression_test.rs | 208 ++++++++++++++++++++++++++ docs/hld/03-architecture.md | 7 +- 3 files changed, 359 insertions(+), 14 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 7f7ff9d1a..5e70e6083 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -2975,15 +2975,27 @@ fn compare_granular_paragraph( metadata: &mut Metadata<'_>, diagnostics: &mut Vec, ) -> Result { + let control_slots = |paragraph: &CT_P| { + paragraph + .content_controls + .iter() + .map(|(_, raw_before, markers_before, _)| (*raw_before, *markers_before)) + .collect::>() + }; + let boundary_error = || { + Error::Other(format!( + "comparison cannot revise paragraph boundary structures at {location}" + )) + }; if hyperlink_shells(original, metadata.options) != hyperlink_shells(edited, metadata.options) || (!metadata.options.ignore_comments && original.comment_ranges != edited.comment_ranges) || original.bookmark_markers != edited.bookmark_markers - || paragraph_control_boundaries(original) != paragraph_control_boundaries(edited) + || control_slots(original) != control_slots(edited) { - return Err(Error::Other(format!( - "comparison cannot revise paragraph boundary structures at {location}" - ))); + return Err(boundary_error()); } + let original_boundaries = shell_run_boundaries(original); + let edited_boundaries = shell_run_boundaries(edited); let original_run_signatures = original .runs @@ -2997,6 +3009,7 @@ fn compare_granular_paragraph( .collect::>(); if original_run_signatures == edited_run_signatures && original.content_controls == edited.content_controls + && original_boundaries == edited_boundaries { let properties = paragraph_properties_xml(original, edited, location, metadata, diagnostics)?; @@ -3050,7 +3063,23 @@ fn compare_granular_paragraph( .collect::>(); let properties = paragraph_properties_xml(original, edited, location, metadata, diagnostics)?; - let aligned = align(&original_signatures, &edited_signatures); + let Some(cuts) = shell_unit_cuts( + &original_units, + &edited_units, + original_boundaries.into_iter().zip(edited_boundaries), + ) else { + return Err(boundary_error()); + }; + let (aligned, segments) = align_between_shells(&original_signatures, &edited_signatures, &cuts); + // The original run each shell segment starts at, before which the + // redline copies the original bytes, shell tags included. + let segment_runs = std::iter::once(0) + .chain(cuts.iter().map(|&(cut, _)| { + original_units + .get(cut) + .map_or(original.runs.len(), |unit| unit.owner) + })) + .collect::>(); if !metadata.options.ignore_fields { validate_field_alignment( &aligned, @@ -3075,10 +3104,12 @@ fn compare_granular_paragraph( &edited_signatures, ); let mut grouped_alignment = Vec::with_capacity(grouped.len()); + let mut grouped_segment_runs = Vec::with_capacity(grouped.len()); let mut replacements = Vec::with_capacity(grouped.len()); for (action, members) in grouped { let first = aligned[members.start]; grouped_alignment.push(first); + grouped_segment_runs.push(segment_runs[segments[members.start]]); let left_indices = members .clone() .filter_map(|index| aligned[index].0) @@ -3139,6 +3170,7 @@ fn compare_granular_paragraph( &original_units, &edited_units, &grouped_alignment, + &grouped_segment_runs, &replacements, &properties, location, @@ -3147,13 +3179,15 @@ fn compare_granular_paragraph( ) } +/// The hyperlink owners without their place in the paragraph, which +/// [`shell_unit_cuts`] follows through the unit alignment. fn hyperlink_shells(paragraph: &CT_P, options: &ComparisonOptions) -> Vec { paragraph .hyperlinks .iter() .map(|link| { format!( - "{:?}:{:?}:{:?}:{:?}:{:?}:{:?}:{:?}:{}:{}", + "{:?}:{:?}:{:?}:{:?}:{:?}:{:?}:{:?}", link.rel_id, link.anchor, link.tooltip, @@ -3161,13 +3195,95 @@ fn hyperlink_shells(paragraph: &CT_P, options: &ComparisonOptions) -> Vec Vec { + paragraph + .hyperlinks + .iter() + .flat_map(|link| [link.run_start, link.run_end]) + .chain(paragraph.content_controls.iter().map(|(at, ..)| *at)) + .collect() +} + +/// The `(original, edited)` unit indices where the shell boundaries fall, in +/// document order, for `(original, edited)` run boundaries. +/// +/// The units between two consecutive boundaries form a segment, and the +/// redline copies the original bytes before the first run of a segment, +/// shell tags included, before anything of that segment, so each boundary +/// moves with the words inserted or deleted around it. `None` when the +/// shells are not in the same order on both sides, or when the edited side +/// writes a unit between two boundaries that fall between the same two +/// original runs, because the original bytes there are not split. Ignorable +/// and empty units are left out, as in the accept and reject postconditions. +fn shell_unit_cuts( + original_units: &[AttributedRunUnit], + edited_units: &[AttributedRunUnit], + boundaries: impl IntoIterator, +) -> Option> { + let mut boundaries = boundaries.into_iter().collect::>(); + boundaries.sort_unstable(); + if boundaries.windows(2).any(|pair| pair[0].1 > pair[1].1) { + return None; + } + let cuts = boundaries + .into_iter() + .map(|(original, edited)| { + ( + original_units.partition_point(|unit| unit.owner < original), + edited_units.partition_point(|unit| unit.owner < edited), + ) + }) + .collect::>(); + let mut start = (0, 0); + cuts.iter() + .all(|&end| { + let writable = start.0 < end.0 + || edited_units[start.1..end.1] + .iter() + .all(|unit| unit_is_ignorable(unit) || unit_is_empty(unit)); + start = end; + writable + }) + .then_some(cuts) +} + +/// The unit alignment made segment by segment between the shell boundaries +/// at `cuts`, so no unit is matched across a hyperlink or an inline control, +/// and the segment of each pair. +#[allow(clippy::type_complexity)] +fn align_between_shells( + original: &[String], + edited: &[String], + cuts: &[(usize, usize)], +) -> (Vec<(Option, Option)>, Vec) { + let mut aligned = Vec::with_capacity(original.len().max(edited.len())); + let mut segments = Vec::with_capacity(aligned.capacity()); + let mut start = (0, 0); + for (segment, &end) in cuts + .iter() + .chain(std::iter::once(&(original.len(), edited.len()))) + .enumerate() + { + for (left, right) in align(&original[start.0..end.0], &edited[start.1..end.1]) { + aligned.push(( + left.map(|index| index + start.0), + right.map(|index| index + start.1), + )); + segments.push(segment); + } + start = end; + } + (aligned, segments) +} + fn hyperlink_raw_boundaries( paragraph: &CT_P, link: &rdocx_oxml::text::HyperlinkSpan, @@ -3347,6 +3463,7 @@ fn interleave_granular_paragraph( original_units: &[AttributedRunUnit], edited_units: &[AttributedRunUnit], aligned: &[(Option, Option)], + segment_runs: &[usize], replacements: &[String], properties: &str, location: &str, @@ -3370,14 +3487,20 @@ fn interleave_granular_paragraph( (None, true) => {} } let spans = modeled_paragraph_run_spans(original, &source)?; - if spans.len() != original.runs.len() || aligned.len() != replacements.len() { + if spans.len() != original.runs.len() + || aligned.len() != replacements.len() + || aligned.len() != segment_runs.len() + { return Err(Error::Other(format!( "comparison could not correlate granular run owners at {location}" ))); } - let insertion_boundary = spans - .first() - .map_or_else(|| paragraph_close_start(&source), |span| Ok(span.start))?; + let run_start = |run: usize| { + spans + .get(run) + .map_or_else(|| paragraph_close_start(&source), |span| Ok(span.start)) + }; + let insertion_boundary = run_start(0)?; let mut output = source[..insertion_boundary].to_owned(); let mut cursor = insertion_boundary; let mut consumed_owner = None; @@ -3389,7 +3512,16 @@ fn interleave_granular_paragraph( for unit in edited_units { edited_owner_units[unit.owner] += 1; } - for ((left, right), replacement) in aligned.iter().zip(replacements) { + for (((left, right), replacement), &segment_run) in + aligned.iter().zip(replacements).zip(segment_runs) + { + // Text inserted at the start of a shell segment comes after the + // shell tags that open it. + let segment_start = run_start(segment_run)?; + if segment_start > cursor { + output.push_str(&source[cursor..segment_start]); + cursor = segment_start; + } let mut exact_run = None; if let Some(unit) = left.map(|index| &original_units[index]) && consumed_owner != Some(unit.owner) diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index db90661d1..f46fac5f2 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -23816,6 +23816,214 @@ fn granular_hyperlink_edits_preserve_the_owner_shell() { ); } +/// Words inserted or deleted outside a hyperlink or an inline control move +/// its boundaries, which the word and character paths follow (#161). +#[test] +fn granular_edits_beside_a_shell_move_its_boundaries() { + let run = |text: &str| format!(r#"{text}"#); + let link = |text: &str| { + format!(r#"{text}"#) + }; + let control = |text: &str| { + format!( + r#"{text}"# + ) + }; + let paragraph = |parts: &[String]| wrap_word_body(&format!("{}", parts.concat())); + let compare = |original_xml: &str, edited_xml: &str, options: &rdocx::ComparisonOptions| { + let mut tracked = document_with_content_controls(original_xml); + tracked + .compare_with_options( + &document_with_content_controls(edited_xml), + "Ada", + "2026-09-04T09:00:00Z", + options, + ) + .map(|diagnostics| { + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + tracked + }) + }; + // The redline, after checking that accepting and rejecting it give the + // edited and the original side with no revision left. + let redline = |original_xml: &str, edited_xml: &str, options: &rdocx::ComparisonOptions| { + let mut tracked = compare(original_xml, edited_xml, options) + .unwrap_or_else(|error| panic!("{:?}: {edited_xml}: {error}", options.granularity)); + let bytes = tracked.to_bytes().unwrap(); + for (resolve, expected_xml) in [ + ( + Document::accept_all as fn(&mut Document) -> rdocx::Result, + edited_xml, + ), + (Document::reject_all, original_xml), + ] { + let mut resolved = Document::from_bytes(&bytes).unwrap(); + resolve(&mut resolved).unwrap(); + let diagnostics = resolved + .compare_with_options( + &document_with_content_controls(expected_xml), + "postcondition", + "2026-09-04T09:01:00Z", + options, + ) + .unwrap(); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert!(resolved.revisions().is_empty(), "{expected_xml}"); + } + document_xml(&mut tracked) + }; + let cases = [ + // Word writes an inserted word in a run of its own, other producers + // in the run it extends. + ( + vec![run("see "), link("site")], + vec![run("please see "), link("site")], + ), + ( + vec![run("see "), link("site")], + vec![run("please "), run("see "), link("site")], + ), + ( + vec![run("please see "), link("site")], + vec![run("see "), link("site")], + ), + // At character granularity the "s" of "see" must not match the one + // of "site". + (vec![run("see "), link("site")], vec![link("site")]), + ( + vec![run("see "), link("site")], + vec![run("see "), link("site"), run(" now")], + ), + ( + vec![run("see "), link("site"), run(", now")], + vec![run("see "), link("site"), run(" today, now")], + ), + ( + vec![run("see "), link("site"), run("now")], + vec![run("see "), link("site"), run("right now")], + ), + ( + vec![link("one"), run(" and "), link("two")], + vec![link("one"), run(" and also "), link("two")], + ), + ( + vec![run("see "), control("field"), run(" now")], + vec![run("please see "), control("field"), run(" now")], + ), + ( + vec![run("see "), control("field"), run(" now")], + vec![run("please "), run("see "), control("field"), run(" now")], + ), + // The spaces on both sides of a control must not match each other. + ( + vec![run("Enter "), control("field"), run(" today")], + vec![run("Enter it "), control("field"), run(" today")], + ), + ( + vec![run("Your name "), control("field"), run(" here")], + vec![run("Your name is "), control("field"), run(" here")], + ), + ( + vec![run("see "), control("field"), run(" now")], + vec![run("see the "), control("field"), run(" now")], + ), + ( + vec![run("see the "), control("field"), run(" now")], + vec![run("see "), control("field"), run(" now")], + ), + ( + vec![run("see "), control("field"), run("now")], + vec![run("see "), control("field"), run("right now")], + ), + ( + vec![run("see "), control("field"), run(" now")], + vec![run("see "), control("field")], + ), + ]; + for granularity in [ + rdocx::ComparisonGranularity::Word, + rdocx::ComparisonGranularity::Character, + ] { + let options = rdocx::ComparisonOptions { + granularity, + ..Default::default() + }; + for (original, edited) in &cases { + let xml = redline(¶graph(original), ¶graph(edited), &options); + for (open, close) in [(""), ("", "")] { + assert_eq!( + xml.matches(open).count(), + original.concat().matches(open).count(), + "{xml}" + ); + for (start, _) in xml.match_indices(open) { + let shell = &xml[start..start + xml[start..].find(close).unwrap()]; + assert!( + !shell.contains("").unwrap(), + ); + let (before, shell, after) = (&xml[..open], &xml[open..close], &xml[close..]); + assert!( + shell.contains(" Date: Sun, 27 Sep 2026 21:27:20 +0200 Subject: [PATCH 07/10] Report content-control metadata changes instead of refusing the pair compare() refused any pair whose content controls differed by w:tag or w:alias, with "comparison cannot revise content-control properties". Google Docs numbers its goog_rdk_N tags per export, so two exports of one document could not be compared at all. A change to w:lock, w:placeholder or the docPartGallery was the opposite problem: it sat outside the compared properties and was dropped without a word. These five properties name, protect or file a control without changing what it holds. They leave the signature that drives alignment, refusal and the accept and reject postconditions, the redline keeps the original w:sdtPr, and each difference becomes a ComparisonDiagnostic whose message starts with "content-control differs". The control type and its data binding decide what a control holds, so a difference there still refuses the pair. ComparisonDiagnostic keeps its two fields, so no struct literal breaks, and its doc comment now lists the stable message prefixes. On the word and character paths a paragraph whose runs all match now keeps them whole and compares only the controls that differ. Before, a control that differed sent the paragraph through the unit rewrite, which split unchanged runs and moved the run-indexed bookmarks Google Docs writes around headings, so the acceptance check failed. The pinned test that expected a changed tag to refuse the pair now expects the diagnostic, and HLD 03 and 10 describe the three classes. GitHub issue #159. --- crates/rdocx/src/comparison.rs | 176 ++++++++++++++++--- crates/rdocx/tests/regression_test.rs | 238 +++++++++++++++++++++++++- docs/hld/03-architecture.md | 28 +-- docs/hld/10-bindings-spec.md | 16 +- 4 files changed, 414 insertions(+), 44 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 5e70e6083..7b697ff60 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -28,13 +28,21 @@ thread_local! { } type ControlPropertySignature<'a> = Option<( - Option<&'a str>, - Option<&'a str>, Option, Option<&'a rdocx_oxml::content_control::CT_DataBinding>, )>; +/// The content-control properties that name, protect or file a control +/// without changing what it holds, by their `w:sdtPr` element names. +const CONTROL_METADATA: [&str; 5] = ["tag", "alias", "lock", "placeholder", "docPartGallery"]; + /// A comparison difference that cannot be represented as a content revision. +/// +/// The redline keeps the original for every diagnostic. The message starts +/// with a stable prefix naming the difference: `formatting differs` for +/// formatting that cannot be revised, and `content-control differs` +/// for a control's `tag`, `alias`, `lock`, `placeholder` or +/// `docPartGallery`. #[derive(Debug, Clone, PartialEq, Eq)] pub struct ComparisonDiagnostic { pub location: String, @@ -3007,9 +3015,9 @@ fn compare_granular_paragraph( .iter() .map(attributed_run_signature) .collect::>(); - if original_run_signatures == edited_run_signatures - && original.content_controls == edited.content_controls - && original_boundaries == edited_boundaries + // Runs that all match stay whole even when a control differs, so a + // control's metadata or content never moves run-indexed markers. + if original_run_signatures == edited_run_signatures && original_boundaries == edited_boundaries { let properties = paragraph_properties_xml(original, edited, location, metadata, diagnostics)?; @@ -3043,11 +3051,22 @@ fn compare_granular_paragraph( } }) .collect::>>()?; - return replace_paragraph_properties_and_runs( + let output = replace_paragraph_properties_and_runs( original, original_source, &properties, &replacements, + )?; + if original.content_controls == edited.content_controls { + return Ok(output); + } + return compare_granular_controls( + original, + edited, + &output, + location, + metadata, + diagnostics, ); } @@ -3541,8 +3560,20 @@ fn interleave_granular_paragraph( output.push_str(exact_run.unwrap_or(replacement)); } output.push_str(&source[cursor..]); + compare_granular_controls(original, edited, &output, location, metadata, diagnostics) +} - let control_spans = direct_word_element_spans(&output, "sdt")?; +/// Compare the inline controls of a paragraph whose revised runs are in +/// `output`, which still holds the original controls. +fn compare_granular_controls( + original: &CT_P, + edited: &CT_P, + output: &str, + location: &str, + metadata: &mut Metadata<'_>, + diagnostics: &mut Vec, +) -> Result { + let control_spans = direct_word_element_spans(output, "sdt")?; if control_spans.len() != original.content_controls.len() { return Err(Error::Other(format!( "comparison could not correlate granular content controls at {location}" @@ -3564,7 +3595,7 @@ fn interleave_granular_paragraph( &output[control_spans[index].clone()], )?); } - replace_direct_word_elements(&output, "sdt", &control_replacements) + replace_direct_word_elements(output, "sdt", &control_replacements) } fn attributed_run_units(runs: &[CT_R], options: &ComparisonOptions) -> Vec { @@ -4691,6 +4722,7 @@ fn compare_control_from_xml( "comparison cannot revise content-control properties at {location}" ))); } + control_metadata_diagnostics(original, edited, location, diagnostics)?; let original_content = modeled_control_content(original); let edited_content = modeled_control_content(edited); let direct_run_or_raw = @@ -5716,25 +5748,125 @@ fn control_signature(control: &CT_Sdt) -> String { ) } -/// The content-control properties that alignment, refusal and the accept and -/// reject postconditions compare. +/// The content-control properties that decide what a control holds, its type +/// and data binding, which alignment, refusal and the accept and reject +/// postconditions compare. /// -/// `w:id` is left out. Producers renumber it on save and it carries no -/// content, so a pair that differs only by it keeps the original's `w:sdtPr`. -/// A `w:sdtPr` with none of these properties reads like no `w:sdtPr`. +/// `w:id` is left out because producers renumber it on save, and the +/// [`CONTROL_METADATA`] because a difference there is reported as a +/// diagnostic. A pair that differs only by those keeps the original's +/// `w:sdtPr`. A `w:sdtPr` with neither property reads like no `w:sdtPr`. fn control_property_signature(control: &CT_Sdt) -> ControlPropertySignature<'_> { control .properties .as_ref() - .map(|properties| { - ( - properties.alias.as_deref(), - properties.tag.as_deref(), - properties.control_type, - properties.data_binding.as_ref(), - ) - }) - .filter(|signature| !matches!(signature, (None, None, None, None))) + .map(|properties| (properties.control_type, properties.data_binding.as_ref())) + .filter(|signature| !matches!(signature, (None, None))) +} + +/// Report each [`CONTROL_METADATA`] property that differs between the two +/// controls. The redline keeps the original `w:sdtPr`. +fn control_metadata_diagnostics( + original: &CT_Sdt, + edited: &CT_Sdt, + location: &str, + diagnostics: &mut Vec, +) -> Result<()> { + if original.properties == edited.properties { + return Ok(()); + } + let edited_values = control_metadata(edited)?; + for ((name, original_value), edited_value) in CONTROL_METADATA + .iter() + .zip(control_metadata(original)?) + .zip(edited_values) + { + if original_value != edited_value { + diagnostics.push(ComparisonDiagnostic { + location: location.to_owned(), + message: format!( + "content-control {name} differs and the original {name} was retained" + ), + }); + } + } + Ok(()) +} + +/// The [`CONTROL_METADATA`] values of a control, in that order. +/// +/// The lock, placeholder and gallery are kept as raw `w:sdtPr` children, so +/// they are read from the serialized properties by local name. +fn control_metadata(control: &CT_Sdt) -> Result<[Option; 5]> { + let Some(properties) = &control.properties else { + return Ok(Default::default()); + }; + let mut values = [ + properties.tag.clone(), + properties.alias.clone(), + None, + None, + None, + ]; + let mut shell = control.clone(); + shell.content.clear(); + let xml = control_xml(&shell)?; + let mut reader = Reader::from_str(&xml); + let mut path = Vec::>::new(); + let mut buffer = Vec::new(); + loop { + let event = reader.read_event_into(&mut buffer).map_err(|error| { + Error::Other(format!("comparison content-control scan failed: {error}")) + })?; + let (element, empty) = match event { + Event::Start(element) => (element, false), + Event::Empty(element) => (element, true), + Event::End(_) => { + path.pop(); + buffer.clear(); + continue; + } + Event::Eof => break, + _ => { + buffer.clear(); + continue; + } + }; + let local = element.local_name().as_ref().to_vec(); + let parents = path.iter().map(Vec::as_slice).collect::>(); + let slot = match (parents.as_slice(), local.as_slice()) { + ([b"sdt", b"sdtPr"], b"lock") => Some(2), + ([b"sdt", b"sdtPr", b"placeholder"], b"docPart") => Some(3), + ([b"sdt", b"sdtPr", b"docPartObj" | b"docPartList"], b"docPartGallery") => Some(4), + _ => None, + }; + if let Some(slot) = slot { + let mut value = String::new(); + for attribute in element.attributes() { + let attribute = attribute.map_err(|error| { + Error::Other(format!( + "comparison content-control attribute failed: {error}" + )) + })?; + if attribute.key.local_name().as_ref() == b"val" { + value = attribute + .decoded_and_normalized_value(XmlVersion::Implicit1_0, element.decoder()) + .map_err(|error| { + Error::Other(format!( + "comparison content-control value failed: {error}" + )) + })? + .into_owned(); + } + } + values[slot] = Some(value); + } + if !empty { + path.push(local); + } + buffer.clear(); + } + Ok(values) } fn modeled_control_content(control: &CT_Sdt) -> Vec<&SdtContent> { diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index f46fac5f2..997aa66dd 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -17683,16 +17683,27 @@ fn comparison_revises_nested_control_content_without_replacing_its_shell() { .unwrap(); assert!(diagnostics.is_empty(), "{diagnostics:?}"); + // A tag is metadata: the original shell stays and the change is reported. let changed_shell_xml = body("old", "changed-shell"); let changed_shell = document_with_content_controls(&changed_shell_xml); let mut unchanged = document_with_content_controls(&original_xml); - let before = unchanged.to_bytes().unwrap(); + let diagnostics = unchanged + .compare(&changed_shell, "Ada", "2026-08-21T09:30:00Z") + .unwrap(); + assert_eq!( + diagnostics, + vec![rdocx::ComparisonDiagnostic { + location: "body/paragraph[0]/content-control[0]".to_owned(), + message: "content-control tag differs and the original tag was retained".to_owned(), + }] + ); + assert!(unchanged.revisions().is_empty()); + let kept_xml = document_xml(&mut unchanged); assert!( - unchanged - .compare(&changed_shell, "Ada", "2026-08-21T09:30:00Z") - .is_err() + kept_xml.contains(r#"w:val="paragraph-control""#), + "{kept_xml}" ); - assert_eq!(unchanged.to_bytes().unwrap(), before); + assert!(!kept_xml.contains("changed-shell"), "{kept_xml}"); } #[test] @@ -30727,6 +30738,223 @@ mod compare_producer_noise { } } + fn metadata_diagnostic(location: &str, property: &str) -> rdocx::ComparisonDiagnostic { + rdocx::ComparisonDiagnostic { + location: location.to_owned(), + message: format!( + "content-control {property} differs and the original {property} was retained" + ), + } + } + + /// Compare two bodies whose controls differ by metadata and return the + /// revision kinds, the diagnostics and the redline. + /// + /// Accepting keeps the original metadata, so it reads like the edited + /// side with the same diagnostics, and rejecting gives the original back. + fn compared_metadata( + original_xml: &str, + edited_xml: &str, + options: &ComparisonOptions, + ) -> (Vec, Vec, String) { + let original = document_with_content_controls(original_xml); + let edited = document_with_content_controls(edited_xml); + let mut compared = document_with_content_controls(original_xml); + let diagnostics = compared + .compare_with_options(&edited, "R", TIMESTAMP, options) + .expect("content-control metadata must not refuse the pair"); + let tracked = compared.to_bytes().unwrap(); + let mut accepted = Document::from_bytes(&tracked).unwrap(); + accepted.accept_all().unwrap(); + assert_eq!( + accepted + .compare_with_options(&edited, "postcondition", TIMESTAMP, options) + .unwrap(), + diagnostics + ); + assert_eq!(revision_kinds(&accepted), []); + let mut rejected = Document::from_bytes(&tracked).unwrap(); + rejected.reject_all().unwrap(); + assert_eq!( + rejected + .compare_with_options(&original, "postcondition", TIMESTAMP, options) + .unwrap(), + [] + ); + assert_eq!(revision_kinds(&rejected), []); + ( + revision_kinds(&compared), + diagnostics, + document_xml(&mut compared), + ) + } + + /// A control's metadata names, protects or files it without changing + /// what it holds, and Google Docs renumbers its `goog_rdk_N` tags between + /// exports (#159 section 2). A difference is reported and the original + /// `w:sdtPr` is kept. + #[test] + fn content_control_metadata_is_reported_and_the_original_kept() { + let block = |properties: &str, first_entry: &str| { + wrap_word_body(&format!( + r#"Before the content control.{properties}{first_entry} entryBeta entryAfter the content control."# + )) + }; + let gallery = |value: &str| { + format!( + r#""# + ) + }; + let placeholder = + |value: &str| format!(r#""#); + let changed = &[RevisionKind::Deletion, RevisionKind::Insertion][..]; + for (property, original, edited) in [ + ( + "tag", + format!( + r#"{}"#, + gallery("Table of Contents") + ), + format!( + r#"{}"#, + gallery("Table of Contents") + ), + ), + ( + "alias", + r#""#.to_owned(), + r#""#.to_owned(), + ), + ( + "lock", + String::new(), + r#""#.to_owned(), + ), + ( + "placeholder", + placeholder("DefaultPlaceholder_1"), + placeholder("DefaultPlaceholder_2"), + ), + ( + "docPartGallery", + gallery("Table of Contents"), + gallery("Custom Table of Contents"), + ), + ] { + for (first_entry, expected) in [("Alpha", &[][..]), ("Delta", changed)] { + let (kinds, diagnostics, tracked) = compared_metadata( + &block(&original, "Alpha"), + &block(&edited, first_entry), + &ComparisonOptions::default(), + ); + assert_eq!(kinds, expected, "{property}"); + assert_eq!( + diagnostics, + [metadata_diagnostic("body/content-control[1]", property)] + ); + assert!( + tracked.contains(&format!("{original}")), + "{tracked}" + ); + } + } + + // A renamed tag and a new lock are two diagnostics on one control. + let (_, diagnostics, _) = compared_metadata( + &block(r#""#, "Alpha"), + &block( + r#""#, + "Alpha", + ), + &ComparisonOptions::default(), + ); + assert_eq!( + diagnostics, + [ + metadata_diagnostic("body/content-control[1]", "tag"), + metadata_diagnostic("body/content-control[1]", "lock"), + ] + ); + + // Google Docs writes a bookmark around a heading, and bookmarks are + // indexed by run, so the runs around the control must stay whole. + let inline = |tag: &str, word: &str| { + wrap_word_body(&format!( + r#"Before {word} after."# + )) + }; + for granularity in [ + ComparisonGranularity::Run, + ComparisonGranularity::Word, + ComparisonGranularity::Character, + ] { + let options = ComparisonOptions { + granularity, + ..Default::default() + }; + // "Omens" has the length of "Alpha" and none of its characters, + // so every granularity gives one deletion and one insertion. + for (word, expected) in [("Alpha", &[][..]), ("Omens", changed)] { + let (kinds, diagnostics, tracked) = compared_metadata( + &inline("goog_rdk_0", "Alpha"), + &inline("goog_rdk_5", word), + &options, + ); + assert_eq!(kinds, expected, "{granularity:?}"); + assert_eq!( + diagnostics, + [metadata_diagnostic( + "body/paragraph[0]/content-control[0]", + "tag" + )], + "{granularity:?}" + ); + assert!( + tracked.contains(r#""#), + "{tracked}" + ); + assert!(!tracked.contains("goog_rdk_5"), "{tracked}"); + } + } + } + + /// The control type and its data binding decide what a control holds, so + /// a difference in either still refuses the pair. + #[test] + fn a_content_control_type_or_binding_change_still_refuses() { + let control = |properties: &str| { + wrap_word_body(&format!( + r#"{properties}Alpha entryAfter the content control."# + )) + }; + let binding = |xpath: &str| { + format!( + r#""# + ) + }; + for (original, edited) in [ + ("".to_owned(), "".to_owned()), + (binding("/root/first"), binding("/root/second")), + ] { + let mut compared = document_with_content_controls(&control(&original)); + let before = compared.to_bytes().unwrap(); + let error = compared + .compare( + &document_with_content_controls(&control(&edited)), + "R", + TIMESTAMP, + ) + .unwrap_err(); + assert!( + error.to_string().contains( + "cannot revise content-control properties at body/content-control[0]" + ), + "{error}" + ); + assert_eq!(compared.to_bytes().unwrap(), before); + } + } + fn replaced_copy(source_xml: &str, old: &str, new: &str) -> String { let mut document = document_with_content_controls(source_xml); assert_eq!(document.try_replace_text(old, new).unwrap(), 1); diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 09e31ed56..273b8af13 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -909,18 +909,22 @@ form changes replace that complete owner. Supported run, paragraph, table, and section properties emit property revisions that retain the original property sidecars. Unsupported formatting differences retain the original bytes and produce stable `ComparisonDiagnostic` values at the actual story path. Inputs -with existing modeled revisions or differing story and control shells are -rejected unless their story category is ignored. A content control's `w:id` -is producer identity and not part of its shell, so controls that differ only -by it align, compare, and keep the original `w:sdtPr`. Attributed text -alignment retains owner, formatting, content position, and raw-child -boundaries, then coalesces adjacent equal-owner edits into minimal revision -wrappers. That alignment runs separately between consecutive hyperlink and -inline-control boundaries, so no text matches across a shell and words -inserted or deleted beside a shell move it. Text inserted between two -boundaries with no original run between them, such as before a hyperlink -that opens its paragraph, has no original bytes to go between and refuses -the pair. +with existing modeled revisions or differing story shells are rejected unless +their story category is ignored. A content control's shell is its type and +data binding, and a difference there is rejected too. Its `w:id` is producer +identity and ignored. Its tag, alias, lock, placeholder, and document-part +gallery are metadata, so controls that differ only by those align, compare, +keep the original `w:sdtPr`, and report one `content-control differs` +diagnostic per property. Attributed text alignment retains owner, +formatting, content position, and raw-child boundaries, then coalesces +adjacent equal-owner edits into minimal revision wrappers. That alignment +runs separately between consecutive hyperlink and inline-control boundaries, +so no text matches across a shell and words inserted or deleted beside a +shell move it. Text inserted between two boundaries with no original run +between them, such as before a hyperlink that opens its paragraph, has no +original bytes to go between and refuses the pair. When every run of a +paragraph matches, the runs stay whole and only the differing inline +controls are compared. When a main story gains a trailing run of paragraphs, comparison marks the original final paragraph boundary once, marks each intermediate inserted paragraph boundary once, and leaves the final inserted paragraph mark as the diff --git a/docs/hld/10-bindings-spec.md b/docs/hld/10-bindings-spec.md index a6a75eb7c..02305f5e6 100644 --- a/docs/hld/10-bindings-spec.md +++ b/docs/hld/10-bindings-spec.md @@ -1271,11 +1271,17 @@ continue to preserve the resulting document when they save it. Native callers generate tracked changes with `Document::compare`, supplying an edited document, author, and RFC 3339 timestamp. The additive -`ComparisonDiagnostic` value reports stable formatting-only locations and -messages without turning those differences into revisions. Comparison rejects -existing modeled revisions and unsupported structural shell differences, and -it commits only after accepting and rejecting staged copies reproduce their -respective package-wide modeled baselines. `Document::compare` keeps its +`ComparisonDiagnostic` value reports stable locations and messages for +differences that stay out of the revisions, and the redline keeps the +original for each. A message starts with a stable prefix, +`formatting differs` for unsupported formatting and +`content-control differs` for a content control's metadata, where +`` is `tag`, `alias`, `lock`, `placeholder`, or `docPartGallery`. +Comparison rejects existing modeled +revisions and unsupported structural shell differences, a content control's +type or data binding included, and it commits only after accepting and +rejecting staged copies reproduce their respective package-wide modeled +baselines. `Document::compare` keeps its source-compatible whole-run default and delegates to the additive `compare_with_options` method. The concrete `ComparisonOptions` value selects `Run`, `Word`, or `Character` granularity and left-biased ignores for From 9023bc40cdc28c68bbfb096927fb8d446d1692c6 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:33:50 +0200 Subject: [PATCH 08/10] Compare comment and note story shells as XML, not as bytes compare() refused a file against its own rdocx save whenever the save wrote its comments part again, with "comments story root shell changed". The part root, with each comment or note replaced by a placeholder, and each owner start tag were compared as bytes. The XML declaration, the order of attributes and namespace declarations, a dropped declaration and an empty root written as a start and an end tag all refused the pair. A Google Docs export carries an empty self-closed comments part, and rdocx writes the w:comment attributes in another order than Word does. The root, the owner start tags and the footnote and endnote separators are now read as trees with namespace-resolved names and a sorted set of attributes. The declaration, comments, processing instructions, whitespace-only text, namespace declarations and Markup Compatibility attributes are left out, since they say how the part is written, and the redline keeps the original bytes as before. A root child or an owner attribute whose value differs still refuses the pair, so a re-dated comment keeps waiting for the redline to carry the edited comment threads. Headers and footers had no such check. Their root is not compared and their content is compared by model, which the new test pins next to the comments and notes cases. GitHub issue #160. --- crates/rdocx/src/comparison.rs | 157 ++++++++++++++++++++++++-- crates/rdocx/tests/regression_test.rs | 151 +++++++++++++++++++++++++ docs/hld/03-architecture.md | 7 +- 3 files changed, 302 insertions(+), 13 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index 7b697ff60..e749b77ef 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -912,8 +912,15 @@ fn compare_owned_story( story.part_name ))); } - let original_skeleton = story_skeleton(original, &original_spans); - let edited_skeleton = story_skeleton(edited, &edited_spans); + let (original_skeleton, original_owners) = canonical_owned_story(original, owner_local)?; + let (edited_skeleton, edited_owners) = canonical_owned_story(edited, owner_local)?; + if original_owners.len() != original_spans.len() || edited_owners.len() != edited_spans.len() { + return Err(Error::Other(format!( + "comparison could not correlate {} story owners in {}", + story.kind.label(), + story.part_name + ))); + } if original_skeleton != edited_skeleton { return Err(Error::Other(format!( "{} story root shell changed in {}", @@ -925,7 +932,7 @@ fn compare_owned_story( for (index, (left, right)) in original_spans.iter().zip(&edited_spans).enumerate() { let left_xml = &original[left.clone()]; let right_xml = &edited[right.clone()]; - if owner_start_signature(left_xml)? != owner_start_signature(right_xml)? { + if original_owners[index].first() != edited_owners[index].first() { return Err(Error::Other(format!( "{} owner shell changed at {}[{index}]", story.kind.label(), @@ -937,7 +944,7 @@ fn compare_owned_story( ComparisonStoryKind::Footnote | ComparisonStoryKind::Endnote ) && !normal_note_owner(left_xml)? { - if left_xml != right_xml { + if original_owners[index] != edited_owners[index] { return Err(Error::Other(format!( "{} separator shell changed at {}[{index}]", story.kind.label(), @@ -1956,11 +1963,118 @@ fn story_skeleton(xml: &str, spans: &[Range]) -> String { skeleton } -fn owner_start_signature(xml: &str) -> Result { - let end = xml - .find('>') - .ok_or_else(|| Error::Other("comparison owner has no start tag".to_owned()))?; - Ok(xml[..=end].to_owned()) +/// An owned story read so that two serializations of one tree compare equal: +/// the part with each owner as one placeholder, then each owner, whose first +/// token is its start tag. +/// +/// Names resolve to their namespaces, attributes compare as a sorted set and +/// an empty element reads as a start and an end. The XML declaration, +/// comments, processing instructions, whitespace-only text, namespace +/// declarations and Markup Compatibility attributes are left out. They say +/// how the part is written, not what it holds, and the redline keeps the +/// original bytes. +fn canonical_owned_story(xml: &str, owner_local: &str) -> Result<(Vec, Vec>)> { + let scan_error = |error: &dyn std::fmt::Display| { + Error::Other(format!("comparison story shell scan failed: {error}")) + }; + let mut reader = NsReader::from_reader(xml.as_bytes()); + reader.config_mut().trim_text(false); + let mut skeleton = Vec::new(); + let mut owners = Vec::>::new(); + let mut in_owner = false; + let mut depth = 0usize; + let mut buffer = Vec::new(); + loop { + let event = reader + .read_event_into(&mut buffer) + .map_err(|error| scan_error(&error))?; + let tokens = match (in_owner, owners.last_mut()) { + (true, Some(owner)) => owner, + _ => &mut skeleton, + }; + let (element, empty) = match event { + Event::Start(element) => (element, false), + Event::Empty(element) => (element, true), + Event::End(_) => { + tokens.push("end".to_owned()); + depth = depth.saturating_sub(1); + in_owner &= depth > 1; + buffer.clear(); + continue; + } + Event::Text(text) if !text.iter().all(u8::is_ascii_whitespace) => { + tokens.push(format!("text {:?}", String::from_utf8_lossy(&text))); + buffer.clear(); + continue; + } + Event::CData(text) => { + tokens.push(format!("text {:?}", String::from_utf8_lossy(&text))); + buffer.clear(); + continue; + } + Event::GeneralRef(reference) => { + tokens.push(format!( + "reference {:?}", + String::from_utf8_lossy(&reference) + )); + buffer.clear(); + continue; + } + Event::Eof => break, + _ => { + buffer.clear(); + continue; + } + }; + let resolver = reader.resolver(); + let (namespace, local) = resolver.resolve_element(element.name()); + let mut attributes = Vec::new(); + for attribute in element.attributes() { + let attribute = attribute.map_err(|error| scan_error(&error))?; + let key = attribute.key.as_ref(); + if key == b"xmlns" || key.starts_with(b"xmlns:") { + continue; + } + let (attribute_namespace, attribute_local) = resolver.resolve_attribute(attribute.key); + if matches!( + attribute_namespace, + ResolveResult::Bound(Namespace(uri)) if uri == oxml_core::xml::MC_NS.as_bytes() + ) { + continue; + } + let value = attribute + .decoded_and_normalized_value(XmlVersion::Implicit1_0, element.decoder()) + .map_err(|error| scan_error(&error))?; + attributes.push(format!( + "{attribute_namespace:?} {:?} {value:?}", + String::from_utf8_lossy(attribute_local.as_ref()) + )); + } + attributes.sort(); + let start = format!( + "start {namespace:?} {:?} {attributes:?}", + String::from_utf8_lossy(local.as_ref()) + ); + let is_owner = depth == 1 + && local.as_ref() == owner_local.as_bytes() + && matches!(namespace, ResolveResult::Bound(Namespace(uri)) if uri == W_NS.as_bytes()); + let tokens = if is_owner { + skeleton.push("owner".to_owned()); + in_owner = !empty; + owners.push(Vec::new()); + owners.last_mut().expect("owner was just pushed") + } else { + tokens + }; + tokens.push(start); + if empty { + tokens.push("end".to_owned()); + } else { + depth += 1; + } + buffer.clear(); + } + Ok((skeleton, owners)) } fn normal_note_owner(xml: &str) -> Result { @@ -6669,8 +6783,8 @@ fn utf8_error(error: impl std::fmt::Display) -> Error { mod tests { use super::{ ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING, - attributed_run_units, comparison_postcondition_error, complex_field_result, story_document, - word_fragments, + attributed_run_units, canonical_owned_story, comparison_postcondition_error, + complex_field_result, story_document, word_fragments, }; use crate::Document; use rdocx_oxml::document::BodyContent; @@ -6792,6 +6906,27 @@ mod tests { } } + #[test] + fn an_owned_story_shell_reads_the_same_in_every_serialization() { + let compatibility = oxml_core::xml::MC_NS; + let word = format!( + r#" + +"# + ); + let other = |author: &str| { + format!( + r#""# + ) + }; + let canonical = |xml: &str| canonical_owned_story(xml, "comment").unwrap(); + assert_eq!(canonical(&word), canonical(&other("Ada"))); + let (skeleton, owners) = canonical(&word); + assert_eq!(skeleton.iter().filter(|token| *token == "owner").count(), 1); + assert_eq!(owners.len(), 1); + assert_ne!(canonical(&word), canonical(&other("Bob"))); + } + #[test] fn staged_comparison_postcondition_failure_preserves_bytes_and_layout_cache() { let mut original = Document::new(); diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 997aa66dd..19798b3bb 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -31247,6 +31247,157 @@ mod compare_producer_noise { ); } } + + const MARKUP_COMPATIBILITY: &str = + "http://schemas.openxmlformats.org/markup-compatibility/2006"; + const RELATIONSHIPS: &str = + "http://schemas.openxmlformats.org/officeDocument/2006/relationships"; + const WORD_2010: &str = "http://schemas.microsoft.com/office/word/2010/wordml"; + const WORD_2012: &str = "http://schemas.microsoft.com/office/word/2012/wordml"; + + fn document_with_comments_part(comments: &str) -> Document { + let mut seed = Document::new(); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + package.set_part("/word/comments.xml", comments.as_bytes().to_vec()); + package.content_types.add_override( + "/word/comments.xml", + "application/vnd.openxmlformats-officedocument.wordprocessingml.comments+xml", + ); + package + .get_or_create_part_rels("/word/document.xml") + .add(oxml_opc::relationship::rel_types::COMMENTS, "comments.xml"); + package.set_part( + "/word/document.xml", + wrap_word_body("Lorem ipsum.").into_bytes(), + ); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + fn with_part(mut document: Document, part_name: &str, xml: &str) -> Document { + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(document.to_bytes().unwrap())) + .unwrap(); + package.set_part(part_name, xml.as_bytes().to_vec()); + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + Document::from_bytes(bytes.get_ref()).unwrap() + } + + /// A comments, notes or header part written again by another producer, + /// rdocx included, keeps its content: the declaration, namespace + /// declarations, attribute order and empty-element form are not. + #[test] + fn a_reserialized_story_shell_is_not_a_change() { + // #160 section 3: the empty comments part of a Google Docs export, + // and the same part as a no-op save used to write it. + let google_docs = format!( + r#" +"# + ); + let rewritten = format!( + r#" +"# + ); + let mut saved = document_with_comments_part(&google_docs); + let saved = Document::from_bytes(&saved.to_bytes().unwrap()).unwrap(); + for edited in [saved, document_with_comments_part(&rewritten)] { + let mut compared = document_with_comments_part(&google_docs); + let diagnostics = compared.compare(&edited, "R", TIMESTAMP).unwrap(); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + assert_eq!(revision_kinds(&compared), []); + } + + // The owner start tags in Word's order and in rdocx's, next to a + // separator note and a header root written again. + let comments = |attributes: &str, text: &str| { + format!( + r#"{text} comment"# + ) + }; + let rdocx_order = + r#"w:author="Ada" w:date="2026-09-04T09:00:00Z" w:initials="AL" w:id="0""#; + let footnotes = format!( + r#" same footnote"# + ); + let header = format!( + r#"edited header"# + ); + for text in ["same", "edited"] { + let original = document_with_comparison_stories("same"); + let mut edited = with_part( + document_with_comparison_stories("same"), + "/word/comments.xml", + &comments(rdocx_order, text), + ); + edited = with_part(edited, "/word/footnotes.xml", &footnotes); + edited = with_part(edited, "/word/header1.xml", &header); + let mut compared = document_with_comparison_stories("same"); + let diagnostics = compared + .compare(&edited, "R", TIMESTAMP) + .unwrap_or_else(|error| panic!("{text}: {error}")); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + let tracked = compared.to_bytes().unwrap(); + let redline = comparison_part_xml(&mut compared, "/word/comments.xml"); + assert_eq!( + redline.contains(" rdocx::Result, + &edited, + ), + (Document::reject_all, &original), + ] { + let mut resolved = Document::from_bytes(&tracked).unwrap(); + resolve(&mut resolved).unwrap(); + let diagnostics = resolved + .compare(expected, "postcondition", TIMESTAMP) + .unwrap(); + assert!(diagnostics.is_empty(), "{diagnostics:?}"); + } + } + } + + /// A shell that differs in content still refuses the pair. A re-dated + /// comment waits for the redline to carry the edited comment threads. + #[test] + fn a_changed_story_shell_still_refuses() { + let comments = |root_child: &str, date: &str| { + format!( + r#"{root_child}same comment"# + ) + }; + for (edited, expected) in [ + ( + comments("", "2026-09-05T09:00:00Z"), + "comments owner shell changed at /word/comments.xml[0]", + ), + ( + comments( + r#""#, + "2026-09-04T09:00:00Z", + ), + "comments story root shell changed in /word/comments.xml", + ), + ] { + let edited = with_part( + document_with_comparison_stories("same"), + "/word/comments.xml", + &edited, + ); + let mut compared = document_with_comparison_stories("same"); + let error = compared.compare(&edited, "R", TIMESTAMP).unwrap_err(); + assert!(error.to_string().contains(expected), "{error}"); + } + } } #[test] diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 273b8af13..fb4340373 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -910,8 +910,11 @@ section properties emit property revisions that retain the original property sidecars. Unsupported formatting differences retain the original bytes and produce stable `ComparisonDiagnostic` values at the actual story path. Inputs with existing modeled revisions or differing story shells are rejected unless -their story category is ignored. A content control's shell is its type and -data binding, and a difference there is rejected too. Its `w:id` is producer +their story category is ignored. The root and owner start tags of a comment +or note story compare as namespace-resolved trees, so a part written again +with other declarations, attribute order, or empty-element forms keeps its +shell. A content control's shell is its type and data binding, and a +difference there is rejected too. Its `w:id` is producer identity and ignored. Its tag, alias, lock, placeholder, and document-part gallery are metadata, so controls that differ only by those align, compare, keep the original `w:sdtPr`, and report one `content-control differs` From 537fbc5177e1554f0ea86785f2ba36f15396d72c Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:35:27 +0200 Subject: [PATCH 09/10] Rename only w:t when a compared field result is deleted compare() wraps an old field result, or a field whose instruction changed, in w:del and turns its text into deleted text. It did so by renaming every " and the w:textInput of a legacy text form became w:delTextextInput. The accept and reject checks read a field as its owner only, so the broken element reached the redline unseen. A refreshed result that holds a tab takes this path for a field written one run per part, and since the packed-field fix for a field packed in one run as well. The rename now matches the w:t start, end and empty tags alone. GitHub issue #160. --- crates/rdocx/src/comparison.rs | 29 +++++++++++++++-- crates/rdocx/tests/regression_test.rs | 46 +++++++++++++++++++++++++++ 2 files changed, 72 insertions(+), 3 deletions(-) diff --git a/crates/rdocx/src/comparison.rs b/crates/rdocx/src/comparison.rs index e749b77ef..df049ceaa 100644 --- a/crates/rdocx/src/comparison.rs +++ b/crates/rdocx/src/comparison.rs @@ -5413,9 +5413,22 @@ fn tracked_field_result( Ok(format!("{deleted}{inserted}")) } +/// Rename each `w:t` element to `w:delText`, and no other element whose +/// name starts the same way, such as `w:tab`. fn deleted_text_xml(xml: &str) -> String { - xml.replace("", "") + let mut output = String::with_capacity(xml.len()); + let mut rest = xml; + while let Some(at) = rest.find("w:t") { + let (before, after) = rest.split_at(at); + let after = &after["w:t".len()..]; + let is_tag = (before.ends_with('<') || before.ends_with("' | '/') || next.is_whitespace()); + output.push_str(before); + output.push_str(if is_tag { "w:delText" } else { "w:t" }); + rest = after; + } + output.push_str(rest); + output } fn complex_field_result(xml: &str) -> Result<(String, String, String)> { @@ -6784,7 +6797,7 @@ mod tests { use super::{ ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING, attributed_run_units, canonical_owned_story, comparison_postcondition_error, - complex_field_result, story_document, word_fragments, + complex_field_result, deleted_text_xml, story_document, word_fragments, }; use crate::Document; use rdocx_oxml::document::BodyContent; @@ -6906,6 +6919,16 @@ mod tests { } } + #[test] + fn only_text_elements_become_deleted_text() { + assert_eq!( + deleted_text_xml( + r#"a b "# + ), + r#"a b "# + ); + } + #[test] fn an_owned_story_shell_reads_the_same_in_every_serialization() { let compatibility = oxml_core::xml::MC_NS; diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 19798b3bb..68a12ad8f 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -17772,6 +17772,52 @@ fn comparison_deletes_text_without_corrupting_tabs() { ); } +#[test] +fn comparison_deletes_field_results_without_corrupting_tabs() { + let field = |instruction: &str, result: &str| { + wrap_word_body(&format!( + r#" {instruction} Page{result}"# + )) + }; + // A refreshed result, then a new instruction that replaces the field. + for (original_xml, edited_xml) in [ + (field("PAGE", "1"), field("PAGE", "2")), + (field("PAGE", "1"), field("NUMPAGES", "1")), + ] { + let original = document_with_content_controls(&original_xml); + let edited = document_with_content_controls(&edited_xml); + let mut compared = document_with_content_controls(&original_xml); + compared + .compare(&edited, "Ada", "2026-08-21T09:30:00Z") + .unwrap(); + let tracked = document_xml(&mut compared); + let deleted = &tracked[tracked.find("").unwrap()]; + assert!( + deleted.contains("Page"), + "{tracked}" + ); + assert!(!tracked.contains("delTextab"), "{tracked}"); + + let mut accepted = Document::from_bytes(&compared.to_bytes().unwrap()).unwrap(); + accepted.accept_all().unwrap(); + assert!( + accepted + .compare(&edited, "postcondition", "2026-08-21T09:31:00Z") + .unwrap() + .is_empty() + ); + let mut rejected = Document::from_bytes(&compared.to_bytes().unwrap()).unwrap(); + rejected.reject_all().unwrap(); + assert!( + rejected + .compare(&original, "postcondition", "2026-08-21T09:31:00Z") + .unwrap() + .is_empty() + ); + assert!(document_xml(&mut rejected).contains("Page")); + } +} + #[test] fn comparison_reports_formatting_inside_matched_table_rows() { let original_xml = wrap_word_body( From 4e0be1e8ea60165ed4d92e72f0e657c5926d5878 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 21:43:25 +0200 Subject: [PATCH 10/10] Re-record the rdocx archive measurement The comparison changes, their doc comments and the new unit tests grow the rdocx package, so the crates.io archive row in README.md and its ARCHIVE_MEASUREMENTS entry are re-measured on top of the stacked comparison branch. GitHub issues #159, #160 and #161. --- README.md | 2 +- scripts/readme_doctests.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 412db0df8..2cd613bc9 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,097,973 compressed bytes, 6,523,992 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,107,322 compressed bytes, 6,568,455 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index 25c233a49..11e2d86bb 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -383,7 +383,7 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_097_973, 6_523_992, 36), + "rdocx": (1_107_322, 6_568_455, 36), "rdocx-cli": (33_805, 145_256, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15),