Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Crates.io archive: rdocx | 1,095,349 compressed bytes, 6,509,769 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-oxml/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
| Crates.io archive: rdocx-oxml | 368,549 compressed bytes, 2,383,545 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 |

## Use it when

Expand Down
74 changes: 50 additions & 24 deletions crates/rdocx-oxml/src/drawing.rs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
//! Drawing elements for inline and anchor images: `CT_Drawing`, `CT_Inline`, `CT_Anchor`.

use quick_xml::events::{BytesEnd, BytesStart, BytesText, Event};
use quick_xml::name::{Namespace, ResolveResult};
use quick_xml::name::{Namespace, PrefixDeclaration, ResolveResult};
use quick_xml::reader::NsReader;
use quick_xml::{Reader, Writer, XmlVersion};

Expand Down Expand Up @@ -698,29 +698,7 @@ impl CT_Anchor {
Ok(Event::Start(ref ie))
if matches_local_name(ie.name().as_ref(), b"p") =>
{
let raw = capture_ns_element(reader, ie)?;
let mut paragraph_reader = Reader::from_reader(raw.as_slice());
let mut paragraph_buffer = Vec::new();
loop {
match paragraph_reader
.read_event_into(&mut paragraph_buffer)?
{
Event::Start(ref paragraph_start)
if matches_local_name(
paragraph_start.name().as_ref(),
b"p",
) =>
{
paragraphs.push(crate::text::CT_P::from_xml(
&mut paragraph_reader,
)?);
break;
}
Event::Eof => break,
_ => {}
}
paragraph_buffer.clear();
}
paragraphs.extend(text_box_paragraph(reader, ie)?);
}
Ok(Event::End(ref ie))
if matches_local_name(ie.name().as_ref(), b"txbxContent") =>
Expand Down Expand Up @@ -1384,6 +1362,54 @@ fn canonical_wp_element(
namespace_matches(&namespace, drawing_ns::WP, b"wp") && local.as_ref() == expected_local
}

/// Parse a text-box paragraph out of its anchor.
///
/// The paragraph is parsed on its own, so its start tag gets the bindings in
/// scope here, on top of the scope `CT_P::from_xml` assumes. A run attribute
/// under any prefix the part binds then resolves as it does in the body.
fn text_box_paragraph(
reader: &mut NsReader<&[u8]>,
start: &BytesStart<'_>,
) -> Result<Option<crate::text::CT_P>> {
let mut bindings = Vec::new();
for (prefix, namespace) in reader.resolver().bindings() {
let prefix = match prefix {
PrefixDeclaration::Default => "",
PrefixDeclaration::Named(prefix) => std::str::from_utf8(prefix)?,
};
let namespace = quick_xml::escape::unescape(std::str::from_utf8(namespace.as_ref())?)
.map_err(quick_xml::Error::from)?;
bindings.push((prefix.to_owned(), namespace.into_owned()));
}
let raw =
crate::text::raw_with_external_bindings(&capture_ns_element(reader, start)?, &bindings)?;
let mut paragraph_reader = Reader::from_reader(raw.as_slice());
let mut buffer = Vec::new();
loop {
match paragraph_reader.read_event_into(&mut buffer)? {
Event::Start(ref paragraph_start)
if matches_local_name(paragraph_start.name().as_ref(), b"p") =>
{
let prefixes = crate::numbering::word_prefixes_at(
paragraph_start,
&[
"w".to_owned(),
format!("\0r\0{}", crate::namespace::R_NS),
format!("\0mc\0{}", crate::namespace::MC_NS),
],
)?;
return Ok(Some(crate::text::CT_P::from_xml_with_prefixes(
&mut paragraph_reader,
&prefixes,
)?));
}
Event::Eof => return Ok(None),
_ => {}
}
buffer.clear();
}
}

fn capture_ns_element(reader: &mut NsReader<&[u8]>, start: &BytesStart<'_>) -> Result<Vec<u8>> {
let mut writer = Writer::new(Vec::new());
writer.write_event(Event::Start(start.to_owned()))?;
Expand Down
62 changes: 61 additions & 1 deletion crates/rdocx-oxml/src/text.rs
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,19 @@ pub(crate) fn capture_root_attribute_record(
start: &BytesStart<'_>,
prefixes: &[String],
) -> Result<Option<Vec<u8>>> {
let bindings = namespace_bindings(prefixes);
let mut bindings = namespace_bindings(prefixes);
// A plain scope entry names a Word prefix. `word_prefixes_at` adds one
// only beside its binding, but the default scope of the public `from_xml`
// entrypoints, `CT_P::from_xml` among them, names `w` by convention and
// binds nothing. A caller that parses a paragraph cut out of its part in
// that scope, as the text-box replacement and template walkers do, reaches
// this capture with it, so the Word prefix resolves here instead of
// failing on the first `w:rsidR`. An explicit binding always wins.
for prefix in prefixes {
if !prefix.starts_with('\0') && !bindings.iter().any(|(bound, _)| bound == prefix) {
bindings.push((prefix.clone(), crate::namespace::W_NS.to_owned()));
}
}
let mut attributes = Vec::new();
let mut expanded = HashSet::new();
let mut used_prefixes = Vec::new();
Expand Down Expand Up @@ -8722,6 +8734,54 @@ mod tests {
assert!(record.is_none(), "{record:?}");
}

#[test]
fn a_run_identity_resolves_the_word_prefix_the_default_scope_names() {
// `CT_P::from_xml` names `w` as the Word prefix without binding it,
// which is the scope a paragraph cut out of its part can be parsed in.
// A run identity under it used to fail the paragraph as unbound.
let paragraph = parse_paragraph(
r#"<w:r w:rsidR="00A1B2C3" w:rsidRPr="00D4E5F6"><w:t>entry</w:t></w:r>"#,
);
assert_eq!(paragraph.text(), "entry");
let mut output = Vec::new();
paragraph.to_xml(&mut Writer::new(&mut output)).unwrap();
let output = String::from_utf8(output).unwrap();
assert!(
output.contains(r#"<w:r w:rsidR="00A1B2C3" w:rsidRPr="00D4E5F6""#),
"{output}"
);
}

#[test]
fn the_default_word_prefix_records_what_its_explicit_binding_records() {
// The fallback must produce the record a part that binds `w` produces,
// must never override an explicit binding, and must not reach a
// prefix the scope does not name as Word.
let start = BytesStart::from_content(r#"w:r w:rsidR="00A1B2C3""#, "w:r".len());
let default_scope = capture_root_attribute_record(&start, &["w".to_owned()]).unwrap();
let bound = capture_root_attribute_record(
&start,
&[format!("\0w\0{}", crate::namespace::W_NS), "w".to_owned()],
)
.unwrap();
assert!(default_scope.is_some());
assert_eq!(default_scope, bound);

let rebound =
capture_root_attribute_record(&start, &["w".to_owned(), "\0w\0urn:other".to_owned()])
.unwrap()
.unwrap();
let rebound = String::from_utf8(rebound).unwrap();
assert!(rebound.contains(r#"xmlns:w="urn:other""#), "{rebound}");

let foreign = BytesStart::from_content(r#"w:r x:id="1""#, "w:r".len());
let error = capture_root_attribute_record(&foreign, &["w".to_owned()]).unwrap_err();
assert!(
error.to_string().contains("prefix `x` is unbound"),
"{error}"
);
}

#[test]
fn complex_hyperlink_field_exposes_its_target_and_cached_text() {
let p = parse_paragraph(concat!(
Expand Down
24 changes: 24 additions & 0 deletions crates/rdocx-py/tests/test_core.py
Original file line number Diff line number Diff line change
Expand Up @@ -939,6 +939,30 @@ def test_word_default_toc_switch_rebuilds_and_reports_ordered_diagnostics():
report.diagnostics = ()


def test_rebuild_toc_accepts_identity_attributes_on_the_field_runs():
import rdocx

# Word writes w:rsidR and w:rsidRPr on the runs it saves, and Google Docs
# exports write them on every run, the runs of the TOC field code included.
source = rdocx.Document()
source.add_paragraph("placeholder")
document = _replace_document_body(
source,
"""
<w:p><w:r w:rsidR="00A1B2C3"><w:fldChar w:fldCharType="begin"/></w:r><w:r w:rsidRPr="00A1B2C3"><w:instrText>TOC \\o "1-1"</w:instrText></w:r><w:r w:rsidDel="00A1B2C3"><w:fldChar w:fldCharType="separate"/></w:r></w:p>
<w:p><w:r><w:fldChar w:fldCharType="end"/></w:r></w:p>
<w:p><w:pPr><w:pStyle w:val="Heading1"/></w:pPr><w:r><w:t>Heading</w:t></w:r></w:p>
""",
)

assert document.rebuild_toc() == rdocx.TocRebuildReport(
entry_count=1, bookmark_count=1, diagnostics=()
)
saved = _document_xml(document)
for identity in (b'w:rsidR="00A1B2C3"', b'w:rsidRPr="00A1B2C3"', b'w:rsidDel="00A1B2C3"'):
assert identity in saved


def test_update_page_fields_writes_layout_page_numbers():
import rdocx

Expand Down
35 changes: 26 additions & 9 deletions crates/rdocx/src/field.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3298,6 +3298,7 @@ struct DynamicTocSpan {
result_end_position: TocRunPosition,
end_run_end: usize,
start_paragraph_name: String,
start_paragraph_namespaces: BTreeMap<String, String>,
separator_wrapper_names: Vec<String>,
instruction_runs: Vec<DynamicInstructionRun>,
end_paragraph_start: usize,
Expand All @@ -3317,6 +3318,7 @@ struct DynamicFieldScan {
result_start: Option<usize>,
result_start_position: Option<TocRunPosition>,
start_paragraph_name: Option<String>,
start_paragraph_namespaces: BTreeMap<String, String>,
separator_wrapper_names: Vec<String>,
instruction_runs: Vec<DynamicInstructionRun>,
}
Expand Down Expand Up @@ -4638,6 +4640,7 @@ fn update_dynamic_field_stack(
result_start: None,
result_start_position: None,
start_paragraph_name: None,
start_paragraph_namespaces: BTreeMap::new(),
separator_wrapper_names: Vec::new(),
instruction_runs: Vec::new(),
}),
Expand Down Expand Up @@ -4687,6 +4690,7 @@ fn update_dynamic_field_stack(
}
});
field.start_paragraph_name = Some(para.qualified_name.clone());
field.start_paragraph_namespaces = para.inherited_namespaces.clone();
let paragraph_position = elements
.iter()
.position(|element| std::ptr::eq(element, para))
Expand Down Expand Up @@ -4802,6 +4806,7 @@ fn update_dynamic_field_stack(
start_paragraph_name: field.start_paragraph_name.ok_or_else(|| {
Error::Other("table of contents field is missing its separator".to_owned())
})?,
start_paragraph_namespaces: field.start_paragraph_namespaces,
separator_wrapper_names: field.separator_wrapper_names,
instruction_runs: field.instruction_runs,
end_paragraph_start: end_para.start,
Expand Down Expand Up @@ -4851,7 +4856,15 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result<Field> {
.start_paragraph_name
.split_once(':')
.map_or("w", |(prefix, _)| prefix);
let mut source = xml[span.instruction_paragraph_start..span.result_start].to_vec();
// The instruction paragraph is cut out of its part, so its start tag gets
// the declarations it inherits there. Every run in it then resolves the
// prefixes it resolved when the document was read.
let mut source = Vec::new();
append_with_inherited_namespaces(
&mut source,
&xml[span.instruction_paragraph_start..span.result_start],
&span.start_paragraph_namespaces,
)?;
source.extend_from_slice(
format!("<{prefix}:r><{prefix}:fldChar {prefix}:fldCharType=\"end\"/></{prefix}:r>")
.as_bytes(),
Expand Down Expand Up @@ -4881,11 +4894,15 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result<Field> {
buffer.clear();
}
};
parse_paragraph(&source)?;
CT_P::from_xml_fragment(&source)?;

let mut projected = format!("<w:p xmlns:w=\"{W_NS}\">").into_bytes();
for run in &span.instruction_runs {
append_instruction_run_with_namespaces(&mut projected, &xml[run.start..run.end], run)?;
append_with_inherited_namespaces(
&mut projected,
&xml[run.start..run.end],
&run.inherited_namespaces,
)?;
}
projected.extend_from_slice(b"<w:r><w:fldChar w:fldCharType=\"end\"/></w:r></w:p>");
let paragraph = parse_paragraph(&projected)?;
Expand All @@ -4906,26 +4923,26 @@ fn parse_dynamic_toc_field(xml: &[u8], span: &DynamicTocSpan) -> Result<Field> {
})
}

fn append_instruction_run_with_namespaces(
fn append_with_inherited_namespaces(
output: &mut Vec<u8>,
raw: &[u8],
run: &DynamicInstructionRun,
inherited_namespaces: &BTreeMap<String, String>,
) -> Result<()> {
let mut reader = quick_xml::Reader::from_reader(raw);
reader.config_mut().trim_text(false);
let mut buffer = Vec::new();
let (insertion, local_namespaces) =
match reader.read_event_into(&mut buffer).map_err(|error| {
Error::Other(format!(
"invalid table of contents instruction run: {error}"
"invalid table of contents instruction XML: {error}"
))
})? {
Event::Start(start) | Event::Empty(start) => {
let mut local_namespaces = HashSet::new();
for attribute in start.attributes() {
let attribute = attribute.map_err(|error| {
Error::Other(format!(
"invalid table of contents instruction run: {error}"
"invalid table of contents instruction XML: {error}"
))
})?;
let key = attribute.key.as_ref();
Expand All @@ -4945,12 +4962,12 @@ fn append_instruction_run_with_namespaces(
}
_ => {
return Err(Error::Other(
"table of contents instruction run has no start tag".to_owned(),
"table of contents instruction XML has no start tag".to_owned(),
));
}
};
output.extend_from_slice(&raw[..insertion]);
for (prefix, namespace) in &run.inherited_namespaces {
for (prefix, namespace) in inherited_namespaces {
if prefix == "xml" || local_namespaces.contains(prefix) {
continue;
}
Expand Down
Loading
Loading