Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Crates.io archive: rdocx | 1,101,657 compressed bytes, 6,540,109 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-layout/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@ sections, and retains source provenance.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx-layout | 255,752 compressed bytes, 1,385,701 member bytes, 15 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-layout` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Crates.io archive: rdocx-layout | 256,067 compressed bytes, 1,386,937 member bytes, 15 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-layout` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
Expand Down
22 changes: 21 additions & 1 deletion crates/rdocx-layout/src/engine.rs
Original file line number Diff line number Diff line change
Expand Up @@ -3617,7 +3617,9 @@ fn table_is_cache_safe(table: &CT_Tbl, styles: &CT_Styles) -> bool {
properties.change.is_none() && properties.revision_xml.is_empty()
})
&& table.rows.iter().all(|row| {
row.extra_xml.is_empty()
row.extra_xml
.iter()
.all(|(position, raw)| CT_Row::raw_is_root_attributes(*position, raw))
&& row.content_controls.is_empty()
&& row.properties.as_ref().is_none_or(|properties| {
properties.revision_markers.is_empty() && properties.revision_xml.is_empty()
Expand Down Expand Up @@ -14416,6 +14418,24 @@ mod tests {
.extra_xml
.push((0, br#"<w:unknown/>"#.to_vec()));
assert!(!table_is_cache_safe(&preserved_cell, &input.styles));

// Word writes revision-save and paragraph identities on every row.
// They carry no content, so the row stays cache safe, unlike a raw
// child of the row.
let document = rdocx_oxml::CT_Document::from_xml(
br#"<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml"><w:body><w:tbl><w:tr w:rsidR="00A1B2C3" w:rsidTr="00A1B2C4" w14:paraId="1A2B3C4D"><w:tc><w:p><w:r><w:t>identified row</w:t></w:r></w:p></w:tc></w:tr></w:tbl></w:body></w:document>"#,
)
.expect("identified row document parses");
let Some(BodyContent::Table(mut identified_row)) = document.body.content.into_iter().next()
else {
panic!("identified row table parses");
};
assert!(!identified_row.rows[0].extra_xml.is_empty());
assert!(table_is_cache_safe(&identified_row, &input.styles));
identified_row.rows[0]
.extra_xml
.push((0, br#"<w:unknown/>"#.to_vec()));
assert!(!table_is_cache_safe(&identified_row, &input.styles));
}

fn restart_input() -> LayoutInput {
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-oxml/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@ schema order, and retains unmodelled XML alongside typed edits.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx-oxml | 367,500 compressed bytes, 2,380,047 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 |
| Crates.io archive: rdocx-oxml | 372,614 compressed bytes, 2,400,815 member bytes, 32 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-oxml` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-27 |

## Use it when

Expand Down
19 changes: 17 additions & 2 deletions crates/rdocx-oxml/src/comments.rs
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@ use crate::namespace::{W_NS, matches_local_name};
use crate::numbering::word_prefixes_at;
use crate::properties::is_word_element;
use crate::raw_xml::{capture_element, capture_empty_element};
use crate::text::CT_P;
use crate::text::{CT_P, declare_w14_on_part_root};

const W14_NS: &str = "http://schemas.microsoft.com/office/word/2010/wordml";

Expand Down Expand Up @@ -137,7 +137,9 @@ impl CT_Comments {
write_raw_at(&mut writer, &self.extra_xml, index + 1)?;
}
writer.write_event(Event::End(BytesEnd::new("w:comments")))?;
Ok(writer.into_inner())
let mut xml = writer.into_inner();
declare_w14_on_part_root(&mut xml)?;
Ok(xml)
}
}

Expand Down Expand Up @@ -447,6 +449,19 @@ mod tests {
assert!(output.contains(r#"ext:item="kept""#));
}

#[test]
fn a_retained_text_identity_stays_bound_without_a_paragraph_identity() {
// The root declares `w14` for an authored paragraph identity only, so
// a paragraph that carries `w14:textId` alone relies on the check of
// the written part.
let xml = br#"<w:comments xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml"><w:comment w:id="1"><w:p w14:textId="5E6F7A8B"><w:r><w:t>thread</w:t></w:r></w:p></w:comment></w:comments>"#;
let comments = CT_Comments::from_xml(xml).unwrap();
let output = comments.to_xml().unwrap();
let reparsed = CT_Comments::from_xml(&output)
.unwrap_or_else(|error| panic!("{error}: {}", String::from_utf8_lossy(&output)));
assert_eq!(reparsed.to_xml().unwrap(), output);
}

#[test]
fn malformed_comment_id_is_rejected() {
let xml = br#"<w:comments xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"><w:comment w:id="not-a-number"><w:p/></w:comment></w:comments>"#;
Expand Down
3 changes: 2 additions & 1 deletion crates/rdocx-oxml/src/content_control.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1242,6 +1242,7 @@ fn parse_content(
reader,
&prefixes,
&owner_bindings,
Some(&child),
)?,
));
} else if is_word_element(child.name().as_ref(), b"tc", &prefixes) {
Expand Down Expand Up @@ -1298,7 +1299,7 @@ fn parse_content(
} else if is_word_element(child.name().as_ref(), b"tbl", &prefixes) {
content.push(SdtContent::Table(CT_Tbl::new()));
} else if is_word_element(child.name().as_ref(), b"tr", &prefixes) {
content.push(SdtContent::Row(CT_Row::new()));
content.push(SdtContent::Row(CT_Row::from_empty_root(&child, &prefixes)?));
} else if is_word_element(child.name().as_ref(), b"tc", &prefixes) {
content.push(SdtContent::Cell(CT_Tc {
properties: None,
Expand Down
7 changes: 5 additions & 2 deletions crates/rdocx-oxml/src/document.rs
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,8 @@ use crate::revision::CT_Revision;
use crate::shared::{ST_PageOrientation, ST_SectionType};
use crate::table::{CT_Tbl, ST_VerticalJc};
use crate::text::{
CT_P, capture_root_attribute_record, is_root_attribute_record, push_root_attribute_record,
CT_P, capture_root_attribute_record, declare_w14_on_part_root, is_root_attribute_record,
push_root_attribute_record,
};
use crate::units::Twips;

Expand Down Expand Up @@ -2998,7 +2999,9 @@ impl CT_Document {

writer.write_event(Event::End(BytesEnd::new("w:document")))?;

Ok(writer.into_inner())
let mut xml = writer.into_inner();
declare_w14_on_part_root(&mut xml)?;
Ok(xml)
}
}

Expand Down
74 changes: 50 additions & 24 deletions crates/rdocx-oxml/src/drawing.rs
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
//! Drawing elements for inline and anchor images: `CT_Drawing`, `CT_Inline`, `CT_Anchor`.

use quick_xml::events::{BytesEnd, BytesStart, BytesText, Event};
use quick_xml::name::{Namespace, ResolveResult};
use quick_xml::name::{Namespace, PrefixDeclaration, ResolveResult};
use quick_xml::reader::NsReader;
use quick_xml::{Reader, Writer, XmlVersion};

Expand Down Expand Up @@ -698,29 +698,7 @@ impl CT_Anchor {
Ok(Event::Start(ref ie))
if matches_local_name(ie.name().as_ref(), b"p") =>
{
let raw = capture_ns_element(reader, ie)?;
let mut paragraph_reader = Reader::from_reader(raw.as_slice());
let mut paragraph_buffer = Vec::new();
loop {
match paragraph_reader
.read_event_into(&mut paragraph_buffer)?
{
Event::Start(ref paragraph_start)
if matches_local_name(
paragraph_start.name().as_ref(),
b"p",
) =>
{
paragraphs.push(crate::text::CT_P::from_xml(
&mut paragraph_reader,
)?);
break;
}
Event::Eof => break,
_ => {}
}
paragraph_buffer.clear();
}
paragraphs.extend(text_box_paragraph(reader, ie)?);
}
Ok(Event::End(ref ie))
if matches_local_name(ie.name().as_ref(), b"txbxContent") =>
Expand Down Expand Up @@ -1384,6 +1362,54 @@ fn canonical_wp_element(
namespace_matches(&namespace, drawing_ns::WP, b"wp") && local.as_ref() == expected_local
}

/// Parse a text-box paragraph out of its anchor.
///
/// The paragraph is parsed on its own, so its start tag gets the bindings in
/// scope here, on top of the scope `CT_P::from_xml` assumes. A run attribute
/// under any prefix the part binds then resolves as it does in the body.
fn text_box_paragraph(
reader: &mut NsReader<&[u8]>,
start: &BytesStart<'_>,
) -> Result<Option<crate::text::CT_P>> {
let mut bindings = Vec::new();
for (prefix, namespace) in reader.resolver().bindings() {
let prefix = match prefix {
PrefixDeclaration::Default => "",
PrefixDeclaration::Named(prefix) => std::str::from_utf8(prefix)?,
};
let namespace = quick_xml::escape::unescape(std::str::from_utf8(namespace.as_ref())?)
.map_err(quick_xml::Error::from)?;
bindings.push((prefix.to_owned(), namespace.into_owned()));
}
let raw =
crate::text::raw_with_external_bindings(&capture_ns_element(reader, start)?, &bindings)?;
let mut paragraph_reader = Reader::from_reader(raw.as_slice());
let mut buffer = Vec::new();
loop {
match paragraph_reader.read_event_into(&mut buffer)? {
Event::Start(ref paragraph_start)
if matches_local_name(paragraph_start.name().as_ref(), b"p") =>
{
let prefixes = crate::numbering::word_prefixes_at(
paragraph_start,
&[
"w".to_owned(),
format!("\0r\0{}", crate::namespace::R_NS),
format!("\0mc\0{}", crate::namespace::MC_NS),
],
)?;
return Ok(Some(crate::text::CT_P::from_xml_with_prefixes(
&mut paragraph_reader,
&prefixes,
)?));
}
Event::Eof => return Ok(None),
_ => {}
}
buffer.clear();
}
}

fn capture_ns_element(reader: &mut NsReader<&[u8]>, start: &BytesStart<'_>) -> Result<Vec<u8>> {
let mut writer = Writer::new(Vec::new());
writer.write_event(Event::Start(start.to_owned()))?;
Expand Down
18 changes: 16 additions & 2 deletions crates/rdocx-oxml/src/footnotes.rs
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ use crate::error::Result;
use crate::namespace::{W_NS, matches_local_name};
use crate::numbering::word_prefixes_at;
use crate::properties::is_word_element;
use crate::text::CT_P;
use crate::text::{CT_P, declare_w14_on_part_root};

/// `ST_FtnEdn` — what a note in the stream is for.
///
Expand Down Expand Up @@ -212,7 +212,9 @@ impl CT_Footnotes {

writer.write_event(Event::End(BytesEnd::new(root_tag)))?;

Ok(writer.into_inner())
let mut xml = writer.into_inner();
declare_w14_on_part_root(&mut xml)?;
Ok(xml)
}
}

Expand Down Expand Up @@ -273,6 +275,18 @@ fn parse_footnote_content(
mod tests {
use super::*;

#[test]
fn a_note_paragraph_identity_stays_bound_under_the_written_root() {
// The written root declares only `w` and `r`, while Word declares
// `w14` on the root of the notes part it writes.
let xml = br#"<w:footnotes xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main" xmlns:w14="http://schemas.microsoft.com/office/word/2010/wordml"><w:footnote w:id="1"><w:p w14:paraId="1A2B3C4D" w14:textId="5E6F7A8B"><w:r><w:t>note</w:t></w:r></w:p></w:footnote></w:footnotes>"#;
let footnotes = CT_Footnotes::from_xml(xml).unwrap();
let output = footnotes.to_xml_footnotes().unwrap();
let reparsed = CT_Footnotes::from_xml(&output)
.unwrap_or_else(|error| panic!("{error}: {}", String::from_utf8_lossy(&output)));
assert_eq!(reparsed.to_xml_footnotes().unwrap(), output);
}

#[test]
fn parse_footnotes_xml() {
let xml = br#"<?xml version="1.0" encoding="UTF-8" standalone="yes"?>
Expand Down
20 changes: 18 additions & 2 deletions crates/rdocx-oxml/src/header_footer.rs
Original file line number Diff line number Diff line change
Expand Up @@ -10,7 +10,7 @@ use crate::namespace::{W_NS, matches_local_name};
use crate::numbering::{namespace_bindings, word_prefixes_at};
use crate::properties::is_word_element;
use crate::raw_xml::{capture_element, capture_empty_element};
use crate::text::CT_P;
use crate::text::{CT_P, declare_w14_on_part_root};

const VML_NS: &str = "urn:schemas-microsoft-com:vml";
const OFFICE_NS: &str = "urn:schemas-microsoft-com:office:office";
Expand Down Expand Up @@ -352,7 +352,9 @@ impl CT_HdrFtr {

writer.write_event(Event::End(BytesEnd::new(root_tag)))?;

Ok(writer.into_inner())
let mut xml = writer.into_inner();
declare_w14_on_part_root(&mut xml)?;
Ok(xml)
}
}

Expand Down Expand Up @@ -1186,6 +1188,20 @@ pub struct HdrFtrRef {
mod tests {
use super::*;

#[test]
fn a_paragraph_identity_stays_bound_when_only_the_paragraph_declares_w14() {
let w14 = "http://schemas.microsoft.com/office/word/2010/wordml";
let xml = format!(
r#"<w:hdr xmlns:w="{W_NS}"><w:p xmlns:w14="{w14}" w14:paraId="1A2B3C4D"><w:r><w:t>header</w:t></w:r></w:p></w:hdr>"#
);
let header = CT_HdrFtr::from_xml(xml.as_bytes()).unwrap();
let output = String::from_utf8(header.to_xml_header().unwrap()).unwrap();
let root = &output[output.find("<w:hdr").unwrap()..];
let root = &root[..root.find('>').unwrap()];
assert!(root.contains(&format!(r#"xmlns:w14="{w14}""#)), "{output}");
assert!(output.contains(r#"w14:paraId="1A2B3C4D""#), "{output}");
}

#[test]
fn round_trip_header() {
let mut hdr = CT_HdrFtr::new();
Expand Down
Loading
Loading