Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Crates.io archive: rdocx | 1,104,100 compressed bytes, 6,548,342 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-cli/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
| Crates.io archive: rdocx-cli | 35,147 compressed bytes, 149,734 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |

## Use it when

Expand Down
116 changes: 116 additions & 0 deletions crates/rdocx-cli/tests/integration.rs
Original file line number Diff line number Diff line change
Expand Up @@ -343,6 +343,122 @@ fn cli_replace_reports_namespace_preflight_errors_without_panicking() {
assert!(!output_path.exists());
}

/// `rdocx replace --expect 1` found none of the text that Google Docs and
/// Word keep in content controls: a run wrapped inside its paragraph, a
/// paragraph wrapped at body level, a control in a table cell, nested
/// controls, a control in a text box, and the table and controls of a
/// header or footer.
#[test]
fn replace_with_expect_counts_the_text_of_content_controls_everywhere() {
let temp = TempWorkspace::new("replace-content-controls");
let input = temp.path.join("controls.docx");
write_document(&input, &["seed"]);

let control = |tag: &str, content: &str| {
format!(
r#"<w:sdt><w:sdtPr><w:tag w:val="{tag}"/></w:sdtPr><w:sdtContent>{content}</w:sdtContent></w:sdt>"#
)
};
let run = |text: &str| format!("<w:r><w:t>{text}</w:t></w:r>");
let paragraph = |content: &str| format!("<w:p>{content}</w:p>");
let table = |cell: &str| {
format!(
r#"<w:tbl><w:tblGrid><w:gridCol w:w="4000"/></w:tblGrid><w:tr><w:tc>{cell}</w:tc></w:tr></w:tbl>"#
)
};
let text_box = format!(
r#"<w:r><w:pict><v:shape style="width:216pt;height:72pt"><v:textbox><w:txbxContent>{}</w:txbxContent></v:textbox></v:shape></w:pict></w:r>"#,
control("box", &paragraph(&run("{{box}}")))
);
let body = [
paragraph(&[run("Body "), control("goog_rdk_0", &run("{{inline}}"))].concat()),
control("goog_rdk_1", &paragraph(&run("{{block}}"))),
table(&control("cell", &paragraph(&run("{{cell}}")))),
control("outer", &paragraph(&control("inner", &run("{{nested}}")))),
paragraph(&[run("Host"), text_box].concat()),
]
.concat();
let word = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
let header = format!(
r#"<w:hdr xmlns:w="{word}">{}{}</w:hdr>"#,
table(&paragraph(&run("{{header_table}}"))),
control("header", &paragraph(&run("{{header_control}}")))
);
let footer = format!(
r#"<w:ftr xmlns:w="{word}">{}{}</w:ftr>"#,
control("page", &paragraph(&run("{{footer_control}}"))),
paragraph(&run("Confidential"))
);

let mut package =
OpcPackage::from_reader(std::io::Cursor::new(fs::read(&input).unwrap())).unwrap();
let mut references = String::new();
for (kind, xml, rel_type) in [
("header", header, rel_types::HEADER),
("footer", footer, rel_types::FOOTER),
] {
let part = format!("/word/{kind}1.xml");
package.set_part(&part, xml.into_bytes());
package.content_types.add_override(
&part,
&format!("application/vnd.openxmlformats-officedocument.wordprocessingml.{kind}+xml"),
);
let id = package
.get_or_create_part_rels("/word/document.xml")
.add(rel_type, &format!("{kind}1.xml"));
references.push_str(&format!(
r#"<w:{kind}Reference w:type="default" r:id="{id}"/>"#
));
}
package.set_part(
"/word/document.xml",
format!(
r#"<w:document xmlns:w="{word}" xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships" xmlns:v="urn:schemas-microsoft-com:vml"><w:body>{body}<w:sectPr>{references}</w:sectPr></w:body></w:document>"#
)
.into_bytes(),
);
package
.write_to(&mut fs::File::create(&input).unwrap())
.unwrap();

for name in [
"inline",
"block",
"cell",
"nested",
"box",
"header_table",
"header_control",
"footer_control",
] {
let tag = format!("{{{{{name}}}}}");
let replaced = temp.path.join(format!("{name}.docx"));
let output = cli(&[
"replace",
path_text(&input),
"--placeholder",
&tag,
"--value",
"done",
"--expect",
"1",
"--output",
path_text(&replaced),
]);
assert_success(&output, name);

let package = OpcPackage::open(&replaced).unwrap();
let saved = ["document", "header1", "footer1"]
.map(|part| {
let xml = package.get_part(&format!("/word/{part}.xml")).unwrap();
String::from_utf8(xml.to_vec()).unwrap()
})
.concat();
assert!(!saved.contains(&tag), "{name}: {saved}");
assert_eq!(saved.matches(">done<").count(), 1, "{name}: {saved}");
}
}

#[test]
fn validate_exit_status_is_a_verdict() {
let temp = TempWorkspace::new("validate");
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-oxml/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
| Crates.io archive: rdocx-oxml | 375,767 compressed bytes, 2,412,318 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |

## Use it when

Expand Down
21 changes: 21 additions & 0 deletions crates/rdocx-oxml/src/content_control.rs
Original file line number Diff line number Diff line change
Expand Up @@ -806,6 +806,27 @@ impl CT_Sdt {
}
}

/// Remove one content child, keeping the revisions and the source bytes
/// of the later runs at their content index.
pub(crate) fn remove_content(&mut self, index: usize) {
if index >= self.content.len() {
return;
}
self.content.remove(index);
for (boundary, _) in &mut self.revisions {
if *boundary > index {
*boundary -= 1;
}
}
self.inline_run_sources
.retain(|source| source.content_index != index);
for source in &mut self.inline_run_sources {
if source.content_index > index {
source.content_index -= 1;
}
}
}

pub(crate) fn word_prefixes(&self) -> &[String] {
&self.word_prefixes
}
Expand Down
79 changes: 74 additions & 5 deletions crates/rdocx-oxml/src/header_footer.rs
Original file line number Diff line number Diff line change
Expand Up @@ -196,6 +196,13 @@ pub struct CT_HdrFtr {
pub extra_namespaces: Vec<(String, String)>,
/// Unknown child elements captured as raw XML.
pub extra_xml: Vec<Vec<u8>>,
/// How many paragraphs precede each entry of `extra_xml`, so that a
/// rewrite puts a table or a content control back where it was. An entry
/// without a position is written after the last paragraph.
extra_xml_positions: Vec<usize>,
/// The namespace bindings of the root element, which a raw child is
/// parsed with when replacement reaches into it.
pub(crate) word_prefixes: Vec<String>,
}

#[allow(non_snake_case)]
Expand All @@ -206,6 +213,8 @@ impl CT_HdrFtr {
watermarks: Vec::new(),
extra_namespaces: Vec::new(),
extra_xml: Vec::new(),
extra_xml_positions: Vec::new(),
word_prefixes: vec!["w".to_owned()],
}
}

Expand All @@ -227,11 +236,16 @@ impl CT_HdrFtr {
pub fn from_xml(xml: &[u8]) -> Result<Self> {
let watermarks = parse_vml_watermarks(xml);
let mut reader = Reader::from_reader(xml);
reader.config_mut().trim_text(true);
// A raw child is captured with the text events of this reader, so
// trimming them would drop the edge spaces of its text, such as the
// one of "Page " before a page number. The text between the children
// of the root is skipped below.
reader.config_mut().trim_text(false);

let mut paragraphs = Vec::new();
let mut extra_namespaces = Vec::new();
let mut extra_xml = Vec::new();
let mut extra_xml_positions = Vec::new();
let mut buf = Vec::new();
let mut word_prefixes = Vec::new();

Expand Down Expand Up @@ -267,6 +281,7 @@ impl CT_HdrFtr {
} else {
// Capture unknown elements as raw XML
extra_xml.push(capture_element(&mut reader, e)?);
extra_xml_positions.push(paragraphs.len());
}
}
Ok(Event::Empty(ref e)) => {
Expand All @@ -278,6 +293,7 @@ impl CT_HdrFtr {
&& !matches_local_name(name.as_ref(), b"ftr")
{
extra_xml.push(capture_empty_element(e)?);
extra_xml_positions.push(paragraphs.len());
}
}
Ok(Event::Eof) => break,
Expand All @@ -292,6 +308,8 @@ impl CT_HdrFtr {
watermarks,
extra_namespaces,
extra_xml,
extra_xml_positions,
word_prefixes,
})
}

Expand Down Expand Up @@ -341,12 +359,24 @@ impl CT_HdrFtr {

writer.write_event(Event::Start(start))?;

for p in &self.paragraphs {
// Write each captured unknown element before the paragraph it
// preceded, and the rest after the last paragraph.
let mut raw_children = self
.extra_xml
.iter()
.enumerate()
.map(|(index, raw)| {
let position = self.extra_xml_positions.get(index).copied();
(position.unwrap_or(usize::MAX), raw)
})
.peekable();
for (index, p) in self.paragraphs.iter().enumerate() {
while let Some((_, raw)) = raw_children.next_if(|(position, _)| *position <= index) {
writer.get_mut().extend_from_slice(raw);
}
p.to_xml(&mut writer)?;
}

// Write captured unknown elements
for raw in &self.extra_xml {
for (_, raw) in raw_children {
writer.get_mut().extend_from_slice(raw);
}

Expand Down Expand Up @@ -1219,6 +1249,45 @@ mod tests {
assert_eq!(parsed.paragraphs.len(), 0);
}

/// A rewrite wrote every table and content control after the last
/// paragraph. A raw child a caller adds still goes after it.
#[test]
fn raw_children_keep_their_place_between_paragraphs() {
let xml = format!(
r#"<w:hdr xmlns:w="{W_NS}"><w:tbl><w:tblGrid/><w:tr><w:tc><w:p/></w:tc></w:tr></w:tbl><w:p><w:r><w:t>one</w:t></w:r></w:p><w:sdt><w:sdtContent><w:p/></w:sdtContent></w:sdt><w:p><w:r><w:t>two</w:t></w:r></w:p><w:bookmarkStart w:id="0" w:name="end"/></w:hdr>"#
);
let mut parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();
parsed
.extra_xml
.push(br#"<w:bookmarkEnd w:id="0"/>"#.to_vec());

let written = String::from_utf8(parsed.to_xml_header().unwrap()).unwrap();

let positions = [
"<w:tbl>",
">one<",
"<w:sdt>",
">two<",
"<w:bookmarkStart",
"<w:bookmarkEnd",
]
.map(|marker| written.find(marker).unwrap());
assert!(positions.is_sorted(), "{written}");
}

/// A raw child was captured with trimmed text events, so the page-number
/// control of a footer came back as "Page" once the part was rewritten.
#[test]
fn a_raw_child_keeps_the_edge_spaces_of_its_text() {
let control = r#"<w:sdt><w:sdtContent><w:p><w:r><w:t xml:space="preserve">Page </w:t></w:r><w:fldSimple w:instr=" PAGE "/></w:p></w:sdtContent></w:sdt>"#;
let xml = format!("<w:ftr xmlns:w=\"{W_NS}\">\n {control}\n <w:p/>\n</w:ftr>");

let parsed = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();

assert_eq!(parsed.extra_xml, [control.as_bytes()]);
assert_eq!(parsed.paragraphs.len(), 1);
}

#[test]
fn aliased_header_paragraph_properties_keep_root_scope() {
let xml = format!(
Expand Down
Loading
Loading