Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Crates.io archive: rdocx | 1,097,973 compressed bytes, 6,523,992 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-oxml/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
| Crates.io archive: rdocx-oxml | 367,851 compressed bytes, 2,381,304 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |

## Use it when

Expand Down
37 changes: 34 additions & 3 deletions crates/rdocx-oxml/src/placeholder.rs
Original file line number Diff line number Diff line change
Expand Up @@ -153,7 +153,9 @@ fn replace_in_single_run(
new_text.push_str(replacement);
new_text.push_str(&t.text[byte_end..]);
t.text = new_text;
t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' ');
// Keep the flag the producer wrote. Dropping it rewrites an unchanged
// run, which a later comparison against the source then reports.
t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' ');
}
}

Expand All @@ -176,7 +178,7 @@ fn replace_across_runs(
new_text.push_str(&t.text[..first_byte_offset]);
new_text.push_str(replacement);
t.text = new_text;
t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' ');
t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' ');
}

// Handle the last run: replace from start to match end within that content item.
Expand All @@ -189,7 +191,7 @@ fn replace_across_runs(
let ch_len = remaining.chars().next().map(|c| c.len_utf8()).unwrap_or(0);
let byte_end = last_byte_offset + ch_len;
t.text = t.text[byte_end..].to_string();
t.preserve_space = t.text.starts_with(' ') || t.text.ends_with(' ');
t.preserve_space = t.preserve_space || t.text.starts_with(' ') || t.text.ends_with(' ');
}

// Clear text content from runs strictly between first and last.
Expand Down Expand Up @@ -796,6 +798,35 @@ mod tests {
assert_eq!(p.runs[1].properties.as_ref().unwrap().italic, Some(true));
}

#[test]
fn replace_keeps_the_producer_space_flag() {
let mut preserved = CT_R::new("WORD");
if let RunContent::Text(text) = &mut preserved.content[0] {
text.preserve_space = true;
}
let mut p = CT_P::new();
p.runs.push(preserved.clone());
p.runs.push(preserved);
p.add_run("tail");

assert_eq!(replace_in_paragraph(&mut p, "WORD", "WORD"), 2);
assert_eq!(replace_in_paragraph(&mut p, "DWO", "D-WO"), 1);
assert_eq!(replace_in_paragraph(&mut p, "tail", " end"), 1);
let flags = p
.runs
.iter()
.map(|run| match &run.content[0] {
RunContent::Text(text) => (text.text.as_str(), text.preserve_space),
_ => unreachable!(),
})
.collect::<Vec<_>>();
assert_eq!(
flags,
[("WORD-WO", true), ("RD", true), (" end", true)],
"a rewritten run keeps a producer flag and gains one at a text edge"
);
}

#[test]
fn replace_multiple_occurrences() {
let mut p = make_para(&["{{x}} and {{x}}"]);
Expand Down
199 changes: 174 additions & 25 deletions crates/rdocx/src/comparison.rs
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ use rdocx_oxml::content_control::{CT_Sdt, SdtContent};
use rdocx_oxml::document::{BodyContent, CT_Document};
use rdocx_oxml::namespace::W_NS;
use rdocx_oxml::properties::CT_PPr;
use rdocx_oxml::shared::ST_PageOrientation;
use rdocx_oxml::table::{CT_Row, CT_Tbl, CT_TblPr, CT_Tc, CT_TrPr, CellContent};
use rdocx_oxml::text::{CT_P, CT_R, CT_Text, RunContent};
use sha2::{Digest, Sha256};
Expand All @@ -29,7 +30,6 @@ thread_local! {
type ControlPropertySignature<'a> = Option<(
Option<&'a str>,
Option<&'a str>,
Option<i32>,
Option<rdocx_oxml::content_control::SdtType>,
Option<&'a rdocx_oxml::content_control::CT_DataBinding>,
)>;
Expand Down Expand Up @@ -2985,8 +2985,16 @@ fn compare_granular_paragraph(
)));
}

let original_run_signatures = original.runs.iter().map(run_signature).collect::<Vec<_>>();
let edited_run_signatures = edited.runs.iter().map(run_signature).collect::<Vec<_>>();
let original_run_signatures = original
.runs
.iter()
.map(attributed_run_signature)
.collect::<Vec<_>>();
let edited_run_signatures = edited
.runs
.iter()
.map(attributed_run_signature)
.collect::<Vec<_>>();
if original_run_signatures == edited_run_signatures
&& original.content_controls == edited.content_controls
{
Expand Down Expand Up @@ -3558,12 +3566,30 @@ fn granular_text(text: &CT_Text, options: &ComparisonOptions) -> Vec<CT_Text> {
fragments
.into_iter()
.map(|value| CT_Text {
// A unit is written back as its own `w:t`, where edge whitespace
// needs the flag to survive. Whitespace in a unit then reads as
// whitespace whatever the source flag was.
preserve_space: text.preserve_space || has_edge_whitespace(&value),
text: value,
preserve_space: text.preserve_space,
})
.collect()
}

/// The whole-run signature that agrees with the attributed units.
///
/// It reads the space flag the way `granular_text` writes it on every unit,
/// so a run that differs only by the flag stays whole instead of being
/// rewritten one unit per run, which would move run-indexed bookmark ends.
fn attributed_run_signature(run: &CT_R) -> String {
let mut run = run.clone();
for content in &mut run.content {
if let RunContent::Text(text) | RunContent::DeletedText(text) = content {
text.preserve_space |= has_edge_whitespace(&text.text);
}
}
run_signature(&run)
}

fn whitespace_fragments(text: &str) -> Vec<String> {
let mut output = Vec::new();
let mut current = String::new();
Expand Down Expand Up @@ -3977,8 +4003,22 @@ fn nonempty_paragraph_properties(mut properties: CT_PPr) -> Option<CT_PPr> {
}

fn paragraph_properties_differ(original: Option<&CT_PPr>, edited: Option<&CT_PPr>) -> bool {
original.cloned().and_then(nonempty_paragraph_properties)
!= edited.cloned().and_then(nonempty_paragraph_properties)
let modeled = |properties: Option<&CT_PPr>| {
properties.cloned().and_then(|mut properties| {
if let Some(section) = properties.sect_pr.as_mut() {
clear_default_orientation(section);
}
nonempty_paragraph_properties(properties)
})
};
modeled(original) != modeled(edited)
}

/// Portrait is the schema default, which the section writer omits.
fn clear_default_orientation(section: &mut rdocx_oxml::document::CT_SectPr) {
if section.orientation == Some(ST_PageOrientation::Portrait) {
section.orientation = None;
}
}

fn section_properties_xml(
Expand Down Expand Up @@ -4008,6 +4048,8 @@ fn section_properties_xml(
original_modeled.change = None;
let mut edited_modeled = edited.clone();
edited_modeled.change = None;
clear_default_orientation(&mut original_modeled);
clear_default_orientation(&mut edited_modeled);
if original_modeled == edited_modeled {
return section_property_xml(original);
}
Expand Down Expand Up @@ -5099,24 +5141,69 @@ fn deleted_text_xml(xml: &str) -> String {
}

fn complex_field_result(xml: &str) -> Result<(String, String, String)> {
let separate = xml
.find("fldCharType=\"separate\"")
.or_else(|| xml.find("fldCharType='separate'"))
let separate = field_character(xml, 0, "separate")
.ok_or_else(|| Error::Other("complex field source has no separate boundary".to_owned()))?;
let (_, result_start) = containing_run(xml, separate)?;
let end_marker = xml[result_start..]
.find("fldCharType=\"end\"")
.or_else(|| xml[result_start..].find("fldCharType='end'"))
.map(|offset| result_start + offset)
// A producer may pack a whole field in one run, as Google Docs writes page
// fields. Ending a run after `separate` and starting one at `end` reads it
// as the same field written one run per part.
let xml = split_field_run(xml, separate, true)?;
let (_, result_start) = containing_run(&xml, separate)?;
let end_marker = field_character(&xml, result_start, "end")
.ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?;
let xml = split_field_run(&xml, end_marker, false)?;
// The split may have moved the `end` character further along.
let end_marker = field_character(&xml, result_start, "end")
.ok_or_else(|| Error::Other("complex field source has no end boundary".to_owned()))?;
let (result_end, _) = containing_run(xml, end_marker)?;
let (result_end, _) = containing_run(&xml, end_marker)?;
Ok((
xml[..result_start].to_owned(),
xml[result_start..result_end].to_owned(),
xml[result_end..].to_owned(),
))
}

fn field_character(xml: &str, from: usize, kind: &str) -> Option<usize> {
let tail = &xml[from..];
tail.find(&format!("fldCharType=\"{kind}\""))
.or_else(|| tail.find(&format!("fldCharType='{kind}'")))
.map(|offset| from + offset)
}

/// Split the run holding the field character at `marker` so that the
/// character ends its run (`after`) or starts it.
///
/// Both runs repeat the original start tag and run properties. A run with no
/// content on that side of the character is returned unchanged.
fn split_field_run(xml: &str, marker: usize, after: bool) -> Result<String> {
let (run_start, run_end) = containing_run(xml, marker)?;
let run = &xml[run_start..run_end];
let children = direct_element_spans(run)?;
let character = children
.iter()
.position(|child| child.contains(&(marker - run_start)))
.ok_or_else(|| Error::Other("complex field character is not a run child".to_owned()))?;
let is_properties = |child: &Range<usize>| {
let mut reader = Reader::from_reader(run[child.clone()].as_bytes());
matches!(
reader.read_event(),
Ok(Event::Start(element) | Event::Empty(element))
if element.local_name().as_ref() == b"rPr"
)
};
let first_content = usize::from(children.first().is_some_and(is_properties));
let split = if after { character + 1 } else { character };
if split <= first_content || split >= children.len() {
return Ok(xml.to_owned());
}
let head = &run[..children[first_content].start];
let close = run
.rfind("</")
.map(|at| &run[at..])
.ok_or_else(|| Error::Other("complex field run has no end tag".to_owned()))?;
let at = run_start + children[split].start;
Ok(format!("{}{close}{head}{}", &xml[..at], &xml[at..]))
}

fn containing_run(xml: &str, at: usize) -> Result<(usize, usize)> {
let mut reader = Reader::from_reader(xml.as_bytes());
reader.config_mut().trim_text(false);
Expand Down Expand Up @@ -5411,10 +5498,29 @@ fn run_content_signature(content: &RunContent) -> String {
match content {
RunContent::Field(_) => "field-owner".to_owned(),
RunContent::Drawing(drawing) => format!("Drawing({:?})", drawing_signature(drawing)),
RunContent::Text(text) => format!("Text({:?})", text_signature(text)),
RunContent::DeletedText(text) => format!("DeletedText({:?})", text_signature(text)),
content => format!("{content:?}"),
}
}

/// The text and whether its `xml:space="preserve"` changes how it reads.
///
/// The flag only protects whitespace at an edge of the text. Producers write
/// it on every `w:t` or only where needed, so elsewhere it is serialization.
fn text_signature(text: &CT_Text) -> (&str, bool) {
(
&text.text,
text.preserve_space && has_edge_whitespace(&text.text),
)
}

/// Whether XML whitespace starts or ends the text.
fn has_edge_whitespace(text: &str) -> bool {
let whitespace = |character: char| matches!(character, ' ' | '\t' | '\n' | '\r');
text.starts_with(whitespace) || text.ends_with(whitespace)
}

fn table_signature(table: &CT_Tbl) -> String {
format!(
"{:?}:{:?}:{:?}:{:?}",
Expand Down Expand Up @@ -5478,16 +5584,25 @@ fn control_signature(control: &CT_Sdt) -> String {
)
}

/// The content-control properties that alignment, refusal and the accept and
/// reject postconditions compare.
///
/// `w:id` is left out. Producers renumber it on save and it carries no
/// content, so a pair that differs only by it keeps the original's `w:sdtPr`.
/// A `w:sdtPr` with none of these properties reads like no `w:sdtPr`.
fn control_property_signature(control: &CT_Sdt) -> ControlPropertySignature<'_> {
control.properties.as_ref().map(|properties| {
(
properties.alias.as_deref(),
properties.tag.as_deref(),
properties.id,
properties.control_type,
properties.data_binding.as_ref(),
)
})
control
.properties
.as_ref()
.map(|properties| {
(
properties.alias.as_deref(),
properties.tag.as_deref(),
properties.control_type,
properties.data_binding.as_ref(),
)
})
.filter(|signature| !matches!(signature, (None, None, None, None)))
}

fn modeled_control_content(control: &CT_Sdt) -> Vec<&SdtContent> {
Expand Down Expand Up @@ -5757,6 +5872,9 @@ fn paragraph_formatting(paragraph: &CT_P) -> Option<CT_PPr> {
properties.numbering_revision_position = None;
properties.change = None;
properties.revision_xml.clear();
if let Some(section) = properties.sect_pr.as_mut() {
clear_default_orientation(section);
}
properties
})
}
Expand Down Expand Up @@ -6287,7 +6405,8 @@ fn utf8_error(error: impl std::fmt::Display) -> Error {
mod tests {
use super::{
ComparisonGranularity, ComparisonOptions, FAIL_AFTER_COMPARISON_STAGING,
attributed_run_units, comparison_postcondition_error, story_document, word_fragments,
attributed_run_units, comparison_postcondition_error, complex_field_result, story_document,
word_fragments,
};
use crate::Document;
use rdocx_oxml::document::BodyContent;
Expand Down Expand Up @@ -6379,6 +6498,36 @@ mod tests {
);
}

#[test]
fn a_packed_field_result_is_read_as_one_run_per_part() {
for w in ["w", "q"] {
let shell = format!(r#"<{w}:r {w}:rsidR="00AB12CD"><{w}:rPr><{w}:b/></{w}:rPr>"#);
let close = format!("</{w}:r>");
let character = |kind: &str| format!(r#"<{w}:fldChar {w}:fldCharType="{kind}"/>"#);
let (begin, separate, end) =
(character("begin"), character("separate"), character("end"));
let code = format!("<{w}:instrText>PAGE</{w}:instrText>");
let result = format!("<{w}:t>1</{w}:t>");
let expected = (
format!("{shell}{begin}{code}{separate}{close}"),
format!("{shell}{result}{close}"),
format!("{shell}{end}{close}"),
);
for field in [
format!("{shell}{begin}{code}{separate}{result}{end}{close}"),
format!("{shell}{begin}{code}{separate}{close}{shell}{result}{end}{close}"),
format!("{}{}{}", expected.0, expected.1, expected.2),
] {
assert_eq!(complex_field_result(&field).unwrap(), expected, "{field}");
}
let uncached = format!("{shell}{begin}{code}{separate}{end}{close}");
assert_eq!(
complex_field_result(&uncached).unwrap(),
(expected.0, String::new(), expected.2)
);
}
}

#[test]
fn staged_comparison_postcondition_failure_preserves_bytes_and_layout_cache() {
let mut original = Document::new();
Expand Down
Loading
Loading