diff --git a/README.md b/README.md index 479b462ef..995c35056 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,095,346 compressed bytes, 6,513,066 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-cli/README.md b/crates/rdocx-cli/README.md index 25ecc0fe5..288f9fbc2 100644 --- a/crates/rdocx-cli/README.md +++ b/crates/rdocx-cli/README.md @@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-cli | 34,666 compressed bytes, 148,140 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when @@ -83,4 +83,9 @@ fragment on each occupied page. exactly `N`. A mismatch exits unsuccessfully without creating or replacing the requested output. +`validate` exits unsuccessfully on a structural error: a relationship to a +missing part, a part without a content type, or a prefix that `mc:Ignorable` +or `mc:MustUnderstand` lists without a namespace declaration. Empty +paragraphs, heading level gaps, and missing metadata are warnings only. + Run `rdocx --help` or `rdocx --help` for the complete option set. diff --git a/crates/rdocx-cli/src/commands.rs b/crates/rdocx-cli/src/commands.rs index 93394f21e..6080da12f 100644 --- a/crates/rdocx-cli/src/commands.rs +++ b/crates/rdocx-cli/src/commands.rs @@ -1193,6 +1193,32 @@ pub fn validate(file: &Path) -> Result { } } + // Markup Compatibility requires every prefix that `mc:Ignorable` or + // `mc:MustUnderstand` lists to be declared, and a consumer may reject a + // part that breaks the rule. A part that does not parse as XML is + // outside this check. + let mut xml_parts = package + .parts + .iter() + .filter(|(part_name, _)| { + package + .content_types + .content_type_for(part_name) + .is_some_and(|content_type| content_type.ends_with("xml")) + }) + .collect::>(); + xml_parts.sort(); + for (part_name, xml) in xml_parts { + let Ok(findings) = rdocx_oxml::namespace::undeclared_compatibility_prefixes(xml) else { + continue; + }; + for (attribute, prefix) in findings { + errors.push(format!( + "part {part_name} lists undeclared prefix `{prefix}` in mc:{attribute}" + )); + } + } + // --- Advisory findings --- if doc.content_count() == 0 { diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs index e4cdd98fa..6cd8e88c7 100644 --- a/crates/rdocx-cli/tests/integration.rs +++ b/crates/rdocx-cli/tests/integration.rs @@ -376,6 +376,43 @@ fn validate_exit_status_is_a_verdict() { ); } +/// #160: rdocx 0.14 rewrote an empty comments root without its `w14` +/// declaration while `mc:Ignorable` still listed `w14`, and validate passed +/// the part. +#[test] +fn validate_reports_an_undeclared_ignorable_prefix() { + let temp = TempWorkspace::new("validate-compatibility"); + let valid = temp.path.join("valid.docx"); + let broken = temp.path.join("broken.docx"); + write_document(&valid, &["Valid content"]); + + let mut package = OpcPackage::open(&valid).unwrap(); + package.set_part( + "/word/comments.xml", + br#""#.to_vec(), + ); + package.content_types.add_override( + "/word/comments.xml", + "application/vnd.openxmlformats-officedocument.wordprocessingml.comments+xml", + ); + let document_part = package.main_document_part().unwrap(); + package + .get_or_create_part_rels(&document_part) + .add(rel_types::COMMENTS, "comments.xml"); + package.save(&broken).unwrap(); + + let output = cli(&["validate", path_text(&broken)]); + assert_eq!(output.status.code(), Some(1)); + assert!(output.stderr.is_empty()); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + format!( + "1 error(s) in {}:\n 1. part /word/comments.xml lists undeclared prefix `w14` in mc:Ignorable\n", + broken.display() + ) + ); +} + #[test] fn render_uses_the_bundled_font_deterministic_path() { let temp = TempWorkspace::new("render"); diff --git a/crates/rdocx-layout/README.md b/crates/rdocx-layout/README.md index 8e18a7d23..767727fc0 100644 --- a/crates/rdocx-layout/README.md +++ b/crates/rdocx-layout/README.md @@ -18,7 +18,7 @@ sections, and retains source provenance. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-layout | 255,752 compressed bytes, 1,385,701 member bytes, 15 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-layout` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx-layout | 255,757 compressed bytes, 1,385,791 member bytes, 15 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-layout` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-layout/src/table.rs b/crates/rdocx-layout/src/table.rs index 93e07ffb1..4ae7067dd 100644 --- a/crates/rdocx-layout/src/table.rs +++ b/crates/rdocx-layout/src/table.rs @@ -2180,6 +2180,7 @@ mod tests { extra_namespaces: Vec::new(), background_xml: None, background_extra_xml: Vec::new(), + root_attributes: Vec::new(), }, styles: styles.clone(), numbering: None, @@ -2371,6 +2372,7 @@ mod tests { extra_namespaces: Vec::new(), background_xml: None, background_extra_xml: Vec::new(), + root_attributes: Vec::new(), }, styles: styles.clone(), numbering: None, diff --git a/crates/rdocx-oxml/README.md b/crates/rdocx-oxml/README.md index 11fef2416..3bdb53a3d 100644 --- a/crates/rdocx-oxml/README.md +++ b/crates/rdocx-oxml/README.md @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-oxml | 369,949 compressed bytes, 2,391,227 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 | ## Use it when diff --git a/crates/rdocx-oxml/src/comments.rs b/crates/rdocx-oxml/src/comments.rs index 05edd9237..a6be64457 100644 --- a/crates/rdocx-oxml/src/comments.rs +++ b/crates/rdocx-oxml/src/comments.rs @@ -119,12 +119,24 @@ impl CT_Comments { Some("yes"), )))?; + // Retained root attributes keep their source order. The fixed `w` and + // `w14` declarations are added ahead of them only when the source + // lacks them, so a producer declaration listed in `mc:Ignorable` + // stays on the root whether or not a paragraph id still uses it. + let declared = |name: &str| { + self.root_attributes + .iter() + .any(|(candidate, _)| candidate == name) + }; let mut root = BytesStart::new("w:comments"); - root.push_attribute(("xmlns:w", W_NS)); - if self - .comments - .iter() - .any(|comment| comment.paragraph_ids.iter().any(Option::is_some)) + if !declared("xmlns:w") { + root.push_attribute(("xmlns:w", W_NS)); + } + if !declared("xmlns:w14") + && self + .comments + .iter() + .any(|comment| comment.paragraph_ids.iter().any(Option::is_some)) { root.push_attribute(("xmlns:w14", W14_NS)); } @@ -344,10 +356,14 @@ fn push_preserved_attributes( root: bool, ) { for (name, value) in attributes { - if root && (name == "xmlns:w" || name == "xmlns:w14") { - continue; - } - start.push_attribute((name.as_str(), value.as_str())); + // The serializer writes the fixed `w` and `w14` prefixes, so their + // root declarations bind the namespaces it writes. + let value = match name.as_str() { + "xmlns:w" if root => W_NS, + "xmlns:w14" if root => W14_NS, + _ => value.as_str(), + }; + start.push_attribute((name.as_str(), value)); } } @@ -465,4 +481,59 @@ mod tests { assert!(output.contains("w14:paraId=\"0000000A\"")); assert!(!output.contains("w14:paraId=\"foreign\"")); } + + /// #160: an empty Word comments root that lists `w14` in `mc:Ignorable` + /// lost the `w14` declaration once a comment without a paragraph id was + /// written into it, and the kept declarations moved behind `xmlns:w`. + #[test] + fn rewritten_root_keeps_its_declarations_in_source_order() { + let comment = |para_id: Option<&str>| { + let mut paragraph = CT_P::new(); + paragraph.add_run("note"); + CT_Comment { + id: 0, + author: Some("Ada".to_owned()), + date: None, + initials: None, + paragraphs: vec![paragraph], + paragraph_ids: vec![para_id.map(str::to_owned)], + extra_attributes: Vec::new(), + extra_xml: Vec::new(), + } + }; + let word_root = concat!( + r#""#, + ); + let mut comments = CT_Comments::from_xml(word_root.replace('>', "/>").as_bytes()).unwrap(); + comments.comments.push(comment(None)); + let output = String::from_utf8(comments.to_xml().unwrap()).unwrap(); + assert!( + output.contains(&format!("{word_root}"#, + ); + let mut comments = CT_Comments::from_xml(aliased_root.as_bytes()).unwrap(); + comments.comments.push(comment(Some("0000000A"))); + let output = String::from_utf8(comments.to_xml().unwrap()).unwrap(); + let expected = concat!( + r#">, + /// Non-namespace attributes of the original document element, such as + /// `mc:Ignorable`, in source order. + #[doc(hidden)] + pub root_attributes: Vec<(String, String)>, } #[allow(non_snake_case)] @@ -2823,6 +2827,7 @@ impl CT_Document { extra_namespaces: Vec::new(), background_xml: None, background_extra_xml: Vec::new(), + root_attributes: Vec::new(), } } @@ -2835,6 +2840,7 @@ impl CT_Document { let mut extra_namespaces = Vec::new(); let mut background_xml = None; let mut background_extra_xml = Vec::new(); + let mut root_attributes = Vec::new(); let mut buf = Vec::new(); let mut word_prefixes = Vec::new(); let mut document_open = false; @@ -2854,17 +2860,18 @@ impl CT_Document { } for attr in e.attributes().flatten() { let key = attr.key.as_ref(); - if (key.starts_with(b"xmlns:") || key == b"xmlns") - && !known_ns.contains(&key) - { - let key_str = std::str::from_utf8(key).unwrap_or("").to_string(); - let val_str = attr - .decoded_and_normalized_value( - XmlVersion::Implicit1_0, - e.decoder(), - )? - .into_owned(); + let is_namespace = key.starts_with(b"xmlns:") || key == b"xmlns"; + if is_namespace && known_ns.contains(&key) { + continue; + } + let key_str = std::str::from_utf8(key).unwrap_or("").to_string(); + let val_str = attr + .decoded_and_normalized_value(XmlVersion::Implicit1_0, e.decoder())? + .into_owned(); + if is_namespace { extra_namespaces.push((key_str, val_str)); + } else { + root_attributes.push((key_str, val_str)); } } document_open = true; @@ -2942,6 +2949,7 @@ impl CT_Document { extra_namespaces, background_xml, background_extra_xml, + root_attributes, }) } @@ -2979,8 +2987,9 @@ impl CT_Document { doc_start.push_attribute(("xmlns:wp", wp_ns)); } - // Replay captured extra namespaces - for (key, val) in &self.extra_namespaces { + // Replay captured extra namespaces, then the root attributes that + // may name their prefixes, such as `mc:Ignorable`. + for (key, val) in self.extra_namespaces.iter().chain(&self.root_attributes) { doc_start.push_attribute((key.as_str(), val.as_str())); } @@ -3210,6 +3219,36 @@ mod tests { } } + /// #160: a typed rewrite dropped `mc:Ignorable` and every other + /// non-namespace attribute of the document root. + #[test] + fn root_attributes_survive_a_rewrite_after_the_namespace_declarations() { + let xml = format!( + r#""# + ); + let parsed = CT_Document::from_xml(xml.as_bytes()).unwrap(); + assert_eq!( + parsed.root_attributes, + [ + ("mc:Ignorable".to_owned(), "w14".to_owned()), + ("x:root".to_owned(), "a & b".to_owned()), + ] + ); + + let written = String::from_utf8(parsed.to_xml().unwrap()).unwrap(); + let root = &written[written.find("').unwrap()]; + assert!( + root.ends_with( + r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml" xmlns:x="urn:producer" mc:Ignorable="w14" x:root="a & b">"# + ), + "{root}" + ); + let reparsed = CT_Document::from_xml(written.as_bytes()).unwrap(); + assert_eq!(reparsed.root_attributes, parsed.root_attributes); + assert_eq!(reparsed.to_xml().unwrap(), written.as_bytes()); + } + #[test] fn default_namespace_document_paragraph_properties_parse_in_scope() { let xml = format!( diff --git a/crates/rdocx-oxml/src/header_footer.rs b/crates/rdocx-oxml/src/header_footer.rs index 2ec2cb1b8..83c80379e 100644 --- a/crates/rdocx-oxml/src/header_footer.rs +++ b/crates/rdocx-oxml/src/header_footer.rs @@ -194,6 +194,9 @@ pub struct CT_HdrFtr { watermarks: Vec, /// Extra namespace declarations captured from the root element. pub extra_namespaces: Vec<(String, String)>, + /// Non-namespace attributes of the root element, such as `mc:Ignorable`, + /// in source order. + root_attributes: Vec<(String, String)>, /// Unknown child elements captured as raw XML. pub extra_xml: Vec>, } @@ -205,6 +208,7 @@ impl CT_HdrFtr { paragraphs: Vec::new(), watermarks: Vec::new(), extra_namespaces: Vec::new(), + root_attributes: Vec::new(), extra_xml: Vec::new(), } } @@ -231,6 +235,7 @@ impl CT_HdrFtr { let mut paragraphs = Vec::new(); let mut extra_namespaces = Vec::new(); + let mut root_attributes = Vec::new(); let mut extra_xml = Vec::new(); let mut buf = Vec::new(); let mut word_prefixes = Vec::new(); @@ -251,16 +256,25 @@ impl CT_HdrFtr { } else if is_word_element(name.as_ref(), b"hdr", &prefixes) || is_word_element(name.as_ref(), b"ftr", &prefixes) { - // Capture extra namespace declarations from root element + // Capture extra namespace declarations and the other + // attributes from root element for attr in e.attributes().flatten() { let key = attr.key.as_ref(); - if (key.starts_with(b"xmlns:") || key == b"xmlns") - && !known_ns.contains(&key) - { + let is_namespace = key.starts_with(b"xmlns:") || key == b"xmlns"; + if is_namespace && !known_ns.contains(&key) { let key_str = std::str::from_utf8(key).unwrap_or("").to_string(); let val_str = std::str::from_utf8(&attr.value).unwrap_or("").to_string(); extra_namespaces.push((key_str, val_str)); + } else if !is_namespace { + root_attributes.push(( + std::str::from_utf8(key)?.to_owned(), + attr.decoded_and_normalized_value( + XmlVersion::Implicit1_0, + e.decoder(), + )? + .into_owned(), + )); } } word_prefixes = prefixes; @@ -291,6 +305,7 @@ impl CT_HdrFtr { paragraphs, watermarks, extra_namespaces, + root_attributes, extra_xml, }) } @@ -334,8 +349,9 @@ impl CT_HdrFtr { start.push_attribute(("xmlns:wp", wp_ns)); } - // Replay captured extra namespaces - for (key, val) in &self.extra_namespaces { + // Replay captured extra namespaces, then the root attributes that + // may name their prefixes, such as `mc:Ignorable`. + for (key, val) in self.extra_namespaces.iter().chain(&self.root_attributes) { start.push_attribute((key.as_str(), val.as_str())); } @@ -1510,6 +1526,25 @@ mod tests { ); } + /// #160: a typed rewrite dropped `mc:Ignorable` from the part root. + #[test] + fn root_attributes_survive_a_rewrite_after_the_namespace_declarations() { + let xml = format!( + r#"Page"# + ); + let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap(); + let written = String::from_utf8(parsed.to_xml_footer().unwrap()).unwrap(); + assert!( + written.contains( + r#" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml" mc:Ignorable="w14">"# + ), + "{written}" + ); + let reparsed = CT_HdrFtr::from_xml(written.as_bytes()).unwrap(); + assert_eq!(reparsed.root_attributes, parsed.root_attributes); + assert_eq!(reparsed.to_xml_footer().unwrap(), written.as_bytes()); + } + #[test] fn non_watermark_w_pict_remains_opaque() { let xml = format!( diff --git a/crates/rdocx-oxml/src/namespace.rs b/crates/rdocx-oxml/src/namespace.rs index 03d559c22..d88c65122 100644 --- a/crates/rdocx-oxml/src/namespace.rs +++ b/crates/rdocx-oxml/src/namespace.rs @@ -1,5 +1,12 @@ //! OOXML namespace constants. +use quick_xml::XmlVersion; +use quick_xml::events::Event; +use quick_xml::name::{Namespace, QName, ResolveResult}; +use quick_xml::reader::NsReader; + +use crate::error::Result; + /// WordprocessingML main namespace pub const W_NS: &str = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"; /// WordprocessingML namespace prefix @@ -10,3 +17,71 @@ pub const M_NS: &str = "http://schemas.openxmlformats.org/officeDocument/2006/ma pub const M_PREFIX: &[u8] = b"m"; pub use oxml_core::xml::{MC_NS, R_NS, matches_local_name}; + +/// The prefixes that an `mc:Ignorable` or `mc:MustUnderstand` attribute in +/// `xml` lists without a namespace declaration in scope. +/// +/// Each finding is the attribute's local name and the prefix, once per pair, +/// in document order. Markup Compatibility (ECMA-376 Part 3) requires every +/// listed prefix to be declared, so a part with a finding is not conformant. +pub fn undeclared_compatibility_prefixes(xml: &[u8]) -> Result> { + let mut reader = NsReader::from_reader(xml); + let mut buffer = Vec::new(); + let mut findings: Vec<(String, String)> = Vec::new(); + loop { + match reader.read_event_into(&mut buffer)? { + Event::Start(element) | Event::Empty(element) => { + for attribute in element.attributes() { + let attribute = attribute?; + let (namespace, local) = reader.resolver().resolve_attribute(attribute.key); + let local = std::str::from_utf8(local.as_ref())?; + if namespace != ResolveResult::Bound(Namespace(MC_NS.as_bytes())) + || !matches!(local, "Ignorable" | "MustUnderstand") + { + continue; + } + let value = attribute + .decoded_and_normalized_value(XmlVersion::Implicit1_0, element.decoder())?; + for prefix in value.split_ascii_whitespace() { + let probe = format!("{prefix}:probe"); + let (bound, _) = reader.resolver().resolve_element(QName(probe.as_bytes())); + if !matches!(bound, ResolveResult::Bound(_)) + && !findings + .iter() + .any(|(name, known)| name == local && known == prefix) + { + findings.push((local.to_owned(), prefix.to_owned())); + } + } + } + } + Event::Eof => return Ok(findings), + _ => {} + } + buffer.clear(); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn compatibility_prefixes_must_be_declared_in_scope() { + let xml = br#""#; + assert_eq!( + undeclared_compatibility_prefixes(xml).unwrap(), + [ + ("Ignorable".to_owned(), "w14".to_owned()), + ("MustUnderstand".to_owned(), "w17".to_owned()), + ("Ignorable".to_owned(), "w16".to_owned()), + ] + ); + + // An attribute named Ignorable outside the Markup Compatibility + // namespace lists nothing. + let xml = br#""#; + assert_eq!(undeclared_compatibility_prefixes(xml).unwrap(), []); + assert!(undeclared_compatibility_prefixes(b"").is_err()); + } +} diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 5c4b2b468..4ce07673e 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -12053,8 +12053,20 @@ impl Document { self.flush_to_package() } + /// Stage every public output. A comments model that no longer matches its + /// part is written back even when no mutation marked it dirty. An + /// unchanged comments part keeps its producer bytes, as styles and the + /// other modelled parts do, so a save without a comment edit neither + /// rewrites it nor invalidates a package signature over it. pub(crate) fn prepare_staged_output(&mut self) -> Result<()> { - self.comments_dirty |= self.comments.is_some() && self.comments_part_name.is_some(); + if let (Some(comments), Some(part_name)) = (&self.comments, &self.comments_part_name) { + self.comments_dirty |= self + .package + .get_part(part_name) + .and_then(|xml| rdocx_oxml::comments::CT_Comments::from_xml(xml).ok()) + .as_ref() + != Some(comments); + } self.prepare_staged_package() } diff --git a/crates/rdocx/tests/integration_test.rs b/crates/rdocx/tests/integration_test.rs index 74d019501..8b68b2e18 100644 --- a/crates/rdocx/tests/integration_test.rs +++ b/crates/rdocx/tests/integration_test.rs @@ -11316,6 +11316,46 @@ fn comments_part_uses_its_existing_relationship_target() { let mut input = std::io::Cursor::new(Vec::new()); package.write_to(&mut input).unwrap(); let mut document = Document::from_bytes(input.get_ref()).unwrap(); + + // Without a comment edit, every output keeps the producer part byte for + // byte at its target, so the package signature over it stays valid. + // F-255 wrote the typed model here on every save, which made compare + // refuse a document against its own save (#160). + let saved = document.to_bytes().unwrap(); + let saved_package = OpcPackage::from_reader(std::io::Cursor::new(&saved)).unwrap(); + assert!(saved_package.get_part("/word/comments.xml").is_none()); + assert_eq!( + saved_package.get_part("/custom/comments-data.xml"), + Some(comments_xml.as_slice()) + ); + assert!( + !flat_opc_package_class_tests::has_package_signature_invalidation_marker(&saved_package) + ); + let flat = document.to_flat_opc_bytes().unwrap(); + let flat_xml = std::str::from_utf8(&flat).unwrap(); + assert!(flat_xml.contains("\n"; + + /// A document package with the given main part (or rdocx's own), plus + /// parts related from it as `(relationship id, part, kind, xml)`. + fn package_with( + document_xml: Option<&str>, + parts: &[(&str, &str, (&str, &str), &str)], + ) -> Vec { + let mut seed = Document::new(); + seed.add_paragraph("Lorem ipsum dolor."); + let mut package = + oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(seed.to_bytes().unwrap())) + .unwrap(); + if let Some(xml) = document_xml { + package.set_part("/word/document.xml", xml.as_bytes().to_vec()); + } + for (id, part, (content_type, relationship_type), xml) in parts { + package.set_part(part, xml.as_bytes().to_vec()); + package.content_types.add_override(part, content_type); + package + .get_or_create_part_rels("/word/document.xml") + .add_with_id(id, relationship_type, part.trim_start_matches("/word/")); + } + let mut bytes = std::io::Cursor::new(Vec::new()); + package.write_to(&mut bytes).unwrap(); + bytes.into_inner() + } + + fn part(package: &[u8], name: &str) -> String { + let package = oxml_opc::OpcPackage::from_reader(std::io::Cursor::new(package)).unwrap(); + String::from_utf8(package.get_part(name).unwrap().to_vec()).unwrap() + } + + /// The start tag of the root element of `xml`. + fn root_tag(xml: &str) -> &str { + let after_declaration = xml.find("?>").map_or(0, |end| end + 2); + let start = after_declaration + xml[after_declaration..].find('<').unwrap(); + &xml[start..=start + xml[start..].find('>').unwrap()] + } + + /// `root` keeps `mc:Ignorable="{ignorable}"` and declares every prefix it lists. + fn assert_ignorable_declared(root: &str, ignorable: &str) { + assert!( + root.contains(&format!(r#"mc:Ignorable="{ignorable}""#)), + "{root}" + ); + for prefix in ignorable.split_whitespace() { + assert!( + root.contains(&format!("xmlns:{prefix}=")), + "{prefix}: {root}" + ); + } + } + + /// The parts whose bytes differ between two packages with the same parts. + fn changed_parts(source: &[u8], saved: &[u8]) -> Vec { + let (source, saved) = (zip_entries(source), zip_entries(saved)); + assert_eq!( + source.keys().collect::>(), + saved.keys().collect::>() + ); + source + .into_iter() + .filter(|(name, bytes)| saved.get(name) != Some(bytes)) + .map(|(name, _)| name) + .collect() + } + + fn revision_kinds(original: &[u8], edited: &[u8]) -> Vec { + let mut compared = Document::from_bytes(original).unwrap(); + compared + .compare(&Document::from_bytes(edited).unwrap(), "R", TIMESTAMP) + .unwrap(); + compared + .revisions() + .iter() + .map(|revision| revision.kind()) + .collect() + } + + fn edited_save(source: &[u8], old: &str, new: &str, expected: usize) -> Vec { + let mut document = Document::from_bytes(source).unwrap(); + assert_eq!(document.try_replace_text(old, new).unwrap(), expected); + document.to_bytes().unwrap() + } + + /// The reproduction of #160 section 3, with the self-closed root it + /// reports, the open and close pair of the #158 report fixture, and that + /// fixture's unused default namespace. + #[test] + fn empty_comments_part_keeps_its_bytes_and_compares_against_its_own_save() { + for comments in [ + format!("{DECLARATION}"), + format!("{DECLARATION}"), + format!( + r#"{DECLARATION}"# + ), + ] { + let source = package_with( + None, + &[("rId99", "/word/comments.xml", COMMENTS, &comments)], + ); + + let saved = Document::from_bytes(&source).unwrap().to_bytes().unwrap(); + assert_eq!( + changed_parts(&source, &saved), + Vec::::new(), + "{comments}" + ); + assert_eq!(revision_kinds(&source, &saved), []); + + let edited = edited_save(&source, "Lorem", "LOREM", 1); + assert_eq!(part(&edited, "/word/comments.xml"), comments); + assert_eq!( + revision_kinds(&source, &edited), + [RevisionKind::Deletion, RevisionKind::Insertion] + ); + } + } + + /// A Word document with one comment. Word writes `w:id` first on the + /// comment and `w14:paraId` on its paragraph, and rdocx writes neither + /// that way. + fn word_comment_package() -> (Vec, String) { + let document = format!( + concat!( + "{}", + "Lorem", + "", + " ipsum dolor.", + "", + ), + DECLARATION, WORD_ROOT + ); + let comments = format!( + concat!( + "{}", + "Check this.", + "", + ), + DECLARATION, WORD_ROOT + ); + let source = package_with( + Some(&document), + &[("rId99", "/word/comments.xml", COMMENTS, &comments)], + ); + (source, comments) + } + + /// A rewrite of the unchanged part made compare refuse any Word file with + /// comments against its own save. + #[test] + fn word_comments_keep_their_bytes_through_a_body_edit() { + let (source, comments) = word_comment_package(); + + let saved = Document::from_bytes(&source).unwrap().to_bytes().unwrap(); + assert_eq!(changed_parts(&source, &saved), Vec::::new()); + assert_eq!(revision_kinds(&source, &saved), []); + + let edited = edited_save(&source, "dolor", "DOLOR", 1); + assert_eq!(part(&edited, "/word/comments.xml"), comments); + assert_eq!( + revision_kinds(&source, &edited), + [RevisionKind::Deletion, RevisionKind::Insertion] + ); + } + + /// Removing the last comment leaves no paragraph id, and the rewritten + /// root used to drop the `w14` declaration that `mc:Ignorable` lists. + #[test] + fn rewritten_comments_root_keeps_every_declaration_in_source_order() { + let (source, comments) = word_comment_package(); + let mut document = Document::from_bytes(&source).unwrap(); + assert!(document.remove_comment(0).unwrap()); + let xml = part(&document.to_bytes().unwrap(), "/word/comments.xml"); + assert!(!xml.contains("w:comment "), "{xml}"); + assert_eq!(root_tag(&xml), root_tag(&comments)); + assert_ignorable_declared(root_tag(&xml), "w14 w15"); + } + + /// The mirror case noted in PR #154: an edited `document.xml` lost + /// `mc:Ignorable` while its body kept every `w14:paraId`. + #[test] + fn edited_main_document_keeps_mc_ignorable_while_its_body_uses_w14() { + let document = format!( + concat!( + "{}", + "Lorem ipsum dolor.", + "Second paragraph.", + "", + ), + DECLARATION, WORD_ROOT + ); + let source = package_with(Some(&document), &[]); + + let edited = edited_save(&source, "Lorem", "LOREM", 1); + let xml = part(&edited, "/word/document.xml"); + assert_ignorable_declared(root_tag(&xml), "w14 w15"); + assert!(xml.contains(r#"w14:paraId="2A2B3C4D""#), "{xml}"); + assert_eq!( + revision_kinds(&source, &edited), + [RevisionKind::Deletion, RevisionKind::Insertion] + ); + + let again = edited_save(&edited, "Second", "SECOND", 1); + assert_eq!( + root_tag(&part(&again, "/word/document.xml")), + root_tag(&xml) + ); + } + + /// A replacement in a header or footer rewrites its part, and the + /// rewritten root dropped `mc:Ignorable`. + #[test] + fn rewritten_header_and_footer_keep_mc_ignorable_and_its_declarations() { + let document = format!( + concat!( + "{}Body", + "", + "", + "", + ), + DECLARATION, WORD_ROOT + ); + let story = |root: &str, text: &str| { + format!( + concat!( + "{}", + "{}", + ), + DECLARATION, root, WORD_ROOT, text, root + ) + }; + let source = package_with( + Some(&document), + &[ + ( + "rIdHeader", + "/word/header1.xml", + HEADER, + &story("hdr", "Header margin"), + ), + ( + "rIdFooter", + "/word/footer1.xml", + FOOTER, + &story("ftr", "Footer margin"), + ), + ], + ); + + let edited = edited_save(&source, "margin", "MARGIN", 2); + for name in ["/word/header1.xml", "/word/footer1.xml"] { + let xml = part(&edited, name); + assert!(xml.contains("MARGIN"), "{xml}"); + assert_ignorable_declared(root_tag(&xml), "w14 w15"); + assert!(xml.contains(r#"w14:paraId="3A2B3C4D""#), "{xml}"); + } + } +} + mod f265_run_property_regressions { use super::*; use rdocx::RunFontSlot; @@ -31453,6 +31745,7 @@ mod advanced_table_geometry_regressions { extra_namespaces: Vec::new(), background_xml: None, background_extra_xml: Vec::new(), + root_attributes: Vec::new(), }, styles: CT_Styles::new_default(), numbering: None, diff --git a/docs/hld/04-opc-and-packaging.md b/docs/hld/04-opc-and-packaging.md index b9c297692..2192154d3 100644 --- a/docs/hld/04-opc-and-packaging.md +++ b/docs/hld/04-opc-and-packaging.md @@ -427,6 +427,10 @@ rewriting the raw subtree bytes. Prefix aliases, nested shadows, and ordinary namespace URI escaping are resolved by the XML parser. Serialization fails closed when owner identity or a serializer prefix binding cannot be preserved safely, leaving the opened package bytes authoritative. +The main document, header, and footer roots also retain their other +attributes, such as `mc:Ignorable`, in source order. A typed rewrite writes +them after every namespace declaration it keeps, so a compatibility attribute +survives the rewrite and every prefix it lists stays declared. Modeled paragraph, run, and section-property owners retain every ordered root attribute, including producer identity, revision-session, foreign, and @@ -528,12 +532,15 @@ remain zero-width and keep their relative schema positions around literal text. The Word facade resolves an existing comments part through the main document's `COMMENTS` relationship and retains the normalized target. Saving serializes -the typed comments model back to that target with its content-type override. -The model preserves unmodelled attributes and children at their insertion -boundaries, while comment range and reference anchors remain ordered among -neighbouring paragraph and run XML. A document without a comments relationship -does not gain a comments part, relationship, or override during an ordinary -save. +the typed comments model back to that target with its content-type override +once the model no longer matches the part. Every public output shares that +test, so a save without a comment edit keeps the part byte for byte and leaves +a package signature over it valid. The model preserves unmodelled attributes +and children at their insertion boundaries, and a rewritten root keeps its +producer declarations and compatibility attributes in source order. Comment +range and reference anchors remain ordered among neighbouring paragraph and +run XML. A document without a comments relationship does not gain a comments +part, relationship, or override during an ordinary save. The Word facade resolves an existing settings part through the main document's `SETTINGS` relationship and retains the normalized target instead of assuming diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643be..3bc909b75 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -367,8 +367,10 @@ class ReadmeCase: ) MEASUREMENT_DATE = "2026-09-19" ARCHIVE_REMEASUREMENT_DATES = { - "rdocx": "2026-09-26", - "rdocx-layout": "2026-09-26", + "rdocx": "2026-09-27", + "rdocx-cli": "2026-09-27", + "rdocx-layout": "2026-09-27", + "rdocx-oxml": "2026-09-27", "rpptx": "2026-09-26", } MEASUREMENT_PLATFORM = "macOS 26.6.2, Apple M5 Max, arm64" @@ -383,12 +385,12 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), - "rdocx-cli": (33_805, 145_256, 8), + "rdocx": (1_095_346, 6_513_066, 36), + "rdocx-cli": (34_666, 148_140, 8), "rdocx-html": (15_486, 63_894, 11), - "rdocx-layout": (255_752, 1_385_701, 15), + "rdocx-layout": (255_757, 1_385_791, 15), "rdocx-opc": (3_655, 9_668, 6), - "rdocx-oxml": (367_500, 2_380_047, 32), + "rdocx-oxml": (369_949, 2_391_227, 32), "rdocx-pdf": (8_111, 26_758, 6), "rpptx": (407_658, 2_122_094, 16), "rpptx-chart": (6_648, 21_136, 6),