From 5067e2142eb6886d822906d6aa58bc4bdd47bccb Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:10:45 +0200 Subject: [PATCH 1/3] Read content controls and nested tables in Document::text Document::text, and so plain `rdocx text`, skipped every content control. A body-level control, which Google Docs writes around whole paragraphs with a goog_rdk tag, dropped its paragraph from the output. Cell controls, row controls around cells and table controls around rows dropped theirs from the row line. Paragraphs of a table nested in a cell were skipped as well, although the doc comment promises table cell text. The walker now recurses through controls at every level and builds each row line with the existing visit_row visitor. A body paragraph still ends with a newline, a table row is still one line, and every paragraph of its cells, nested tables included, ends with a tab. Documents without content controls or nested tables give exactly the text they gave before. GitHub issue #160. --- crates/rdocx-cli/tests/integration.rs | 27 ++++++++ crates/rdocx/src/document.rs | 77 +++++++++++++++++------ crates/rdocx/tests/regression_test.rs | 90 +++++++++++++++++++++++++++ docs/hld/03-architecture.md | 6 +- 4 files changed, 180 insertions(+), 20 deletions(-) diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs index e4cdd98f..320be069 100644 --- a/crates/rdocx-cli/tests/integration.rs +++ b/crates/rdocx-cli/tests/integration.rs @@ -177,6 +177,33 @@ fn text_prints_body_and_table_content_in_document_order() { ); } +#[test] +fn text_prints_paragraphs_wrapped_by_a_body_content_control() { + let temp = TempWorkspace::new("text-content-control"); + let input = temp.path.join("control.docx"); + write_document(&input, &["Body text"]); + + let mut package = + OpcPackage::from_reader(std::io::Cursor::new(fs::read(&input).unwrap())).unwrap(); + let xml = std::str::from_utf8(package.get_part("/word/document.xml").unwrap()) + .unwrap() + .replacen( + "", + r#"Wrapped paragraph"#, + 1, + ); + package.set_part("/word/document.xml", xml.into_bytes()); + let mut file = fs::File::create(&input).unwrap(); + package.write_to(&mut file).unwrap(); + + let output = cli(&["text", path_text(&input)]); + assert_success(&output, "text"); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + "Wrapped paragraph\nBody text\n" + ); +} + #[test] fn convert_writes_valid_formats_and_uses_the_shared_default_output() { let temp = TempWorkspace::new("convert"); diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 5c4b2b46..79fdfcc9 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -9389,6 +9389,57 @@ fn visit_sdt(control: &CT_Sdt, visitor: &mut impl FnMut(&CT_P)) { } } +/// Append a block-level paragraph to [`Document::text`] as one line. +fn push_paragraph_line(paragraph: &CT_P, text: &mut String) { + text.push_str(¶graph.text()); + text.push('\n'); +} + +/// Append a table to [`Document::text`] as one line per row, rows wrapped by +/// table-level content controls included. +fn push_table_text(table: &CT_Tbl, text: &mut String) { + for index in 0..=table.rows.len() { + for (_, _, control) in table + .content_controls + .iter() + .filter(|(at, _, _)| *at == index) + { + push_control_text(control, text); + } + if let Some(row) = table.rows.get(index) { + push_row_text(row, text); + } + } +} + +/// Append a row to [`Document::text`] as one line in which every paragraph of +/// its cells ends with a tab, cell controls and nested tables included. +fn push_row_text(row: &CT_Row, text: &mut String) { + visit_row(row, &mut |paragraph| push_cell_paragraph(paragraph, text)); + text.push('\n'); +} + +fn push_cell_paragraph(paragraph: &CT_P, text: &mut String) { + text.push_str(¶graph.text()); + text.push('\t'); +} + +/// Append what a block, table or row content control wraps to [`Document::text`]. +fn push_control_text(control: &CT_Sdt, text: &mut String) { + for item in &control.content { + match item { + SdtContent::Paragraph(paragraph) => push_paragraph_line(paragraph, text), + SdtContent::Table(table) => push_table_text(table, text), + SdtContent::Row(row) => push_row_text(row, text), + SdtContent::Cell(cell) => { + visit_cell(cell, &mut |paragraph| push_cell_paragraph(paragraph, text)); + } + SdtContent::Run(_) | SdtContent::RawXml(_) => {} + SdtContent::ContentControl(nested) => push_control_text(nested, text), + } + } +} + fn visit_body_paragraphs_mut(content: &mut [BodyContent], visitor: &mut impl FnMut(&mut CT_P)) { for item in content { match item { @@ -14633,28 +14684,18 @@ impl Document { } /// Get the plain text of body paragraphs and table cells in document order. + /// + /// A body paragraph ends with a newline. A table row is one line in which + /// every cell paragraph ends with a tab, paragraphs of nested tables + /// included. Content controls at every level contribute the paragraphs, + /// rows and cells they wrap at the position they occupy. pub fn text(&self) -> String { let mut result = String::new(); for content in &self.document.body.content { match content { - BodyContent::Paragraph(paragraph) => { - result.push_str(¶graph.text()); - result.push('\n'); - } - BodyContent::Table(table) => { - for row in &table.rows { - for cell in &row.cells { - for content in &cell.content { - if let CellContent::Paragraph(paragraph) = content { - result.push_str(¶graph.text()); - result.push('\t'); - } - } - } - result.push('\n'); - } - } - BodyContent::ContentControl(_) => {} + BodyContent::Paragraph(paragraph) => push_paragraph_line(paragraph, &mut result), + BodyContent::Table(table) => push_table_text(table, &mut result), + BodyContent::ContentControl(control) => push_control_text(control, &mut result), BodyContent::RawXml(_) => {} } } diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c7..365637e1 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -12339,6 +12339,96 @@ fn namespace_classification_metadata_exists_only_for_raw_children() { )); } +/// The body read walkers see through content controls (GitHub issue #160). +mod content_control_read_walker_regressions { + use super::*; + + fn control(tag: &str, content: &str) -> String { + format!( + r#"{content}"# + ) + } + + fn table(rows: &str) -> String { + format!(r#"{rows}"#) + } + + fn row(cells: &str) -> String { + format!("{cells}") + } + + fn cell(content: &str) -> String { + format!("{content}") + } + + fn paragraph(text: &str) -> String { + format!(r#"{text}"#) + } + + #[test] + fn text_reads_every_content_control_location_in_document_order() { + let body = [ + paragraph("first"), + control("goog_rdk_1", ¶graph("block")), + table(&format!( + "{}{}", + control("rows", &row(&cell(¶graph("wrapped row")))), + row(&format!( + "{}{}{}{}", + cell(¶graph("plain")), + control("cells", &cell(¶graph("wrapped cell"))), + cell(&control("goog_rdk_2", ¶graph("cell control"))), + cell(&format!( + "{}{}", + table(&row(&format!( + "{}{}", + cell(¶graph("inner a")), + cell(¶graph("inner b")) + ))), + paragraph("after inner") + )), + )) + )), + control("outer", &control("goog_rdk_3", ¶graph("nested"))), + format!( + r#"{} tail"#, + control("goog_rdk_4", r#"inline"#) + ), + paragraph("last"), + ] + .concat(); + let document = document_with_content_controls(&wrap_word_body(&body)); + + assert_eq!( + document.text(), + "first\nblock\nwrapped row\t\nplain\twrapped cell\tcell control\tinner a\tinner b\tafter inner\t\nnested\ninline tail\nlast\n" + ); + } + + #[test] + fn nested_table_cells_contribute_text_inside_their_outer_row() { + let body = table(&row(&format!( + "{}{}", + cell(¶graph("outer")), + cell(&format!( + "{}{}", + table(&format!( + "{}{}", + row(&cell(¶graph("first inner row"))), + row(&cell(¶graph("second inner row"))) + )), + paragraph("after") + )) + ))); + let document = document_with_content_controls(&wrap_word_body(&body)); + + assert_eq!( + document.text(), + "outer\tfirst inner row\tsecond inner row\tafter\t\n" + ); + } +} + fn ordered_reader_fixture() -> &'static str { r#" Date: Sun, 27 Sep 2026 17:11:31 +0200 Subject: [PATCH 2/3] Reach content controls from images, word counts, headings and links Document::images, word_count, headings and links skipped content controls, and so did document_outline and the accessibility audit built on them. A picture or a heading that Google Docs wrapped in a goog_rdk control was not reported, and word_count missed the words of every wrapped paragraph. images now walks visit_all_drawings and word_count walks visit_body_paragraphs, the existing visitors that interleave body, table, row, cell and inline controls in document order. headings and links keep their scope, body paragraphs without table cells, and now collect it through body-level and nested controls. A table that such a control wraps is still not searched, so wrapping content in a control never changes what they report. The private collectors of images and word_count are removed. Results for documents without content controls are unchanged. MHTML export pairs the tags of the HTML emitter with picture sizes by position and refuses a count mismatch. The emitter still drops content controls, so sizes taken from the new images() would make the export fail. The writer now collects them from the paragraphs the emitter reaches, which is the reach images() had before. A picture in a content control is still dropped with the existing loss diagnostic, as it was before this change. GitHub issue #160. --- crates/rdocx/src/document.rs | 225 +++++++++++--------------- crates/rdocx/src/html.rs | 135 ++++++++++++++-- crates/rdocx/tests/regression_test.rs | 152 +++++++++++++++++ docs/hld/03-architecture.md | 6 +- 4 files changed, 376 insertions(+), 142 deletions(-) diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 79fdfcc9..9af51da1 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -9389,6 +9389,40 @@ fn visit_sdt(control: &CT_Sdt, visitor: &mut impl FnMut(&CT_P)) { } } +/// Collect the body paragraphs in document order, those that body-level +/// content controls wrap included. Tables are not entered, with or without a +/// control around them. This is what [`Document::headings`] and +/// [`Document::links`] read. +fn block_paragraphs(content: &[BodyContent]) -> Vec<&CT_P> { + let mut paragraphs = Vec::new(); + for item in content { + match item { + BodyContent::Paragraph(paragraph) => paragraphs.push(paragraph), + BodyContent::ContentControl(control) => { + collect_block_control_paragraphs(control, &mut paragraphs); + } + BodyContent::Table(_) | BodyContent::RawXml(_) => {} + } + } + paragraphs +} + +fn collect_block_control_paragraphs<'a>(control: &'a CT_Sdt, paragraphs: &mut Vec<&'a CT_P>) { + for item in &control.content { + match item { + SdtContent::Paragraph(paragraph) => paragraphs.push(paragraph), + SdtContent::ContentControl(nested) => { + collect_block_control_paragraphs(nested, paragraphs); + } + SdtContent::Table(_) + | SdtContent::Row(_) + | SdtContent::Cell(_) + | SdtContent::Run(_) + | SdtContent::RawXml(_) => {} + } + } +} + /// Append a block-level paragraph to [`Document::text`] as one line. fn push_paragraph_line(paragraph: &CT_P, text: &mut String) { text.push_str(¶graph.text()); @@ -22741,13 +22775,13 @@ impl Document { /// Get all headings in the document as (level, text) pairs. /// - /// Detects heading paragraphs by their style ID (e.g. "Heading1", "Heading2"). + /// Detects heading paragraphs by their style ID (e.g. "Heading1", "Heading2") + /// among the body paragraphs, including those that body-level content + /// controls wrap. Table cells are not searched. pub fn headings(&self) -> Vec<(u32, String)> { let mut result = Vec::new(); - for content in &self.document.body.content { - if let BodyContent::Paragraph(p) = content - && let Some(level) = Self::detect_heading_level_for_toc(p) - { + for p in block_paragraphs(&self.document.body.content) { + if let Some(level) = Self::detect_heading_level_for_toc(p) { result.push((level, p.text())); } } @@ -22765,75 +22799,39 @@ impl Document { /// Get information about all images in the document. /// - /// Returns metadata for each inline and anchored image found in body paragraphs. + /// Returns metadata for each inline and anchored image found in body + /// paragraphs, table cells and content controls, in document order. pub fn images(&self) -> Vec { let mut result = Vec::new(); - - for content in &self.document.body.content { - Self::collect_images_from_content(content, &mut result); - } - result - } - - fn collect_images_from_content(content: &BodyContent, result: &mut Vec) { - match content { - BodyContent::Paragraph(p) => Self::collect_images_from_paragraph(p, result), - BodyContent::Table(tbl) => Self::collect_images_from_table(tbl, result), - BodyContent::ContentControl(_) => {} - BodyContent::RawXml(_) => {} - } - } - - fn collect_images_from_paragraph(p: &CT_P, result: &mut Vec) { - for run in &p.runs { - for rc in &run.content { - let RunContent::Drawing(drawing) = rc else { - continue; - }; - if let Some(inline) = &drawing.inline { - result.push(ImageInfo { - embed_id: inline.embed_id.clone(), - name: inline.name.clone(), - description: inline.description.clone(), - width_emu: inline.extent_cx.0, - height_emu: inline.extent_cy.0, - is_anchor: false, - }); - } - if let Some(anchor) = &drawing.anchor { - result.push(ImageInfo { - embed_id: anchor.embed_id.clone(), - name: anchor.name.clone(), - description: anchor.description.clone(), - width_emu: anchor.extent_cx.0, - height_emu: anchor.extent_cy.0, - is_anchor: true, - }); - } + visit_all_drawings(&self.document.body.content, &mut |drawing| { + if let Some(inline) = &drawing.inline { + result.push(ImageInfo { + embed_id: inline.embed_id.clone(), + name: inline.name.clone(), + description: inline.description.clone(), + width_emu: inline.extent_cx.0, + height_emu: inline.extent_cy.0, + is_anchor: false, + }); } - } - } - - fn collect_images_from_table(tbl: &CT_Tbl, result: &mut Vec) { - use rdocx_oxml::table::CellContent; - - for row in &tbl.rows { - for cell in &row.cells { - for cc in &cell.content { - match cc { - CellContent::Paragraph(p) => Self::collect_images_from_paragraph(p, result), - CellContent::Table(nested) => { - Self::collect_images_from_table(nested, result) - } - CellContent::ContentControl(_) => {} - } - } + if let Some(anchor) = &drawing.anchor { + result.push(ImageInfo { + embed_id: anchor.embed_id.clone(), + name: anchor.name.clone(), + description: anchor.description.clone(), + width_emu: anchor.extent_cx.0, + height_emu: anchor.extent_cy.0, + is_anchor: true, + }); } - } + }); + result } /// Get information about all hyperlinks in the document. /// + /// Reads the body paragraphs, including those that body-level content + /// controls wrap. Table cells are not searched. /// Resolves hyperlink relationship IDs to their target URLs where possible. pub fn links(&self) -> Vec { use oxml_opc::relationship::rel_types; @@ -22851,35 +22849,33 @@ impl Document { } let mut result = Vec::new(); - for content in &self.document.body.content { - if let BodyContent::Paragraph(p) = content { - for hl in &p.hyperlinks { - // `HyperlinkSpan`'s bounds are public and can be set by - // hand, so clamp rather than slice-panic on a bad range. - let start = hl.run_start.min(p.runs.len()); - let end = hl.run_end.clamp(start, p.runs.len()); - let text: String = p.runs[start..end].iter().map(|r| r.text()).collect(); - - let url = hl.rel_id.as_ref().and_then(|id| url_map.get(id)).cloned(); - - result.push(LinkInfo { - text, - url, - anchor: hl.anchor.clone(), - rel_id: hl.rel_id.clone(), - }); - } - for field in p.complex_field_hyperlinks() { - let start = field.run_start.min(p.runs.len()); - let end = field.run_end.clamp(start, p.runs.len()); - let text: String = p.runs[start..end].iter().map(|run| run.text()).collect(); - result.push(LinkInfo { - text, - url: Some(field.target), - anchor: None, - rel_id: None, - }); - } + for p in block_paragraphs(&self.document.body.content) { + for hl in &p.hyperlinks { + // `HyperlinkSpan`'s bounds are public and can be set by + // hand, so clamp rather than slice-panic on a bad range. + let start = hl.run_start.min(p.runs.len()); + let end = hl.run_end.clamp(start, p.runs.len()); + let text: String = p.runs[start..end].iter().map(|r| r.text()).collect(); + + let url = hl.rel_id.as_ref().and_then(|id| url_map.get(id)).cloned(); + + result.push(LinkInfo { + text, + url, + anchor: hl.anchor.clone(), + rel_id: hl.rel_id.clone(), + }); + } + for field in p.complex_field_hyperlinks() { + let start = field.run_start.min(p.runs.len()); + let end = field.run_end.clamp(start, p.runs.len()); + let text: String = p.runs[start..end].iter().map(|run| run.text()).collect(); + result.push(LinkInfo { + text, + url: Some(field.target), + anchor: None, + rel_id: None, + }); } } result @@ -22888,43 +22884,12 @@ impl Document { /// Count the number of words in the document. /// /// Counts whitespace-separated tokens across all paragraphs (including - /// paragraphs inside table cells). + /// paragraphs inside table cells and content controls). pub fn word_count(&self) -> usize { let mut count = 0; - for content in &self.document.body.content { - count += Self::word_count_in_content(content); - } - count - } - - fn word_count_in_content(content: &BodyContent) -> usize { - match content { - BodyContent::Paragraph(p) => p.text().split_whitespace().count(), - BodyContent::Table(tbl) => Self::word_count_in_table(tbl), - BodyContent::ContentControl(_) => 0, - BodyContent::RawXml(_) => 0, - } - } - - fn word_count_in_table(tbl: &CT_Tbl) -> usize { - use rdocx_oxml::table::CellContent; - - let mut count = 0; - for row in &tbl.rows { - for cell in &row.cells { - for cc in &cell.content { - match cc { - CellContent::Paragraph(p) => { - count += p.text().split_whitespace().count(); - } - CellContent::Table(nested) => { - count += Self::word_count_in_table(nested); - } - CellContent::ContentControl(_) => {} - } - } - } - } + visit_body_paragraphs(&self.document.body.content, &mut |paragraph| { + count += paragraph.text().split_whitespace().count(); + }); count } diff --git a/crates/rdocx/src/html.rs b/crates/rdocx/src/html.rs index 6ec00fb4..ade6c90e 100644 --- a/crates/rdocx/src/html.rs +++ b/crates/rdocx/src/html.rs @@ -11,7 +11,7 @@ use base64::Engine as _; use base64::engine::general_purpose::STANDARD as BASE64; use rdocx_oxml::document::BodyContent; use rdocx_oxml::table::{ - CT_Row, CT_Tbl, CT_TblGrid, CT_TblGridCol, CT_TblPr, CT_TblWidth, CT_Tc, VMerge, + CT_Row, CT_Tbl, CT_TblGrid, CT_TblGridCol, CT_TblPr, CT_TblWidth, CT_Tc, CellContent, VMerge, }; use rdocx_oxml::text::{CT_P, RunContent}; use rdocx_oxml::units::Twips; @@ -616,15 +616,58 @@ fn base64_lines(bytes: &[u8]) -> String { output } +/// Collect the body paragraphs, table cell paragraphs and nested table +/// paragraphs in document order, leaving content controls out as the +/// rdocx-html emitter does. This is the reach [`Document::images`] had before +/// it read content controls, so the emitter's `` tags still pair with the +/// pictures of these paragraphs by position. +fn emitted_paragraphs(content: &[BodyContent]) -> Vec<&CT_P> { + let mut paragraphs = Vec::new(); + for item in content { + match item { + BodyContent::Paragraph(paragraph) => paragraphs.push(paragraph), + BodyContent::Table(table) => collect_emitted_table_paragraphs(table, &mut paragraphs), + BodyContent::ContentControl(_) | BodyContent::RawXml(_) => {} + } + } + paragraphs +} + +fn collect_emitted_table_paragraphs<'a>(table: &'a CT_Tbl, paragraphs: &mut Vec<&'a CT_P>) { + for cell in table.rows.iter().flat_map(|row| &row.cells) { + for item in &cell.content { + match item { + CellContent::Paragraph(paragraph) => paragraphs.push(paragraph), + CellContent::Table(nested) => collect_emitted_table_paragraphs(nested, paragraphs), + CellContent::ContentControl(_) => {} + } + } + } +} + fn mhtml_export_html(document: &Document) -> Result<(String, Vec)> { let html = document.to_html(); - let image_sizes = document - .images() - .into_iter() - .filter(|image| { - !image.embed_id.is_empty() && document.image_data(&image.embed_id).is_some() - }) - .collect::>(); + let mut image_sizes = Vec::new(); + for paragraph in emitted_paragraphs(&document.document.body.content) { + for content in paragraph.runs.iter().flat_map(|run| &run.content) { + let RunContent::Drawing(drawing) = content else { + continue; + }; + let inline = drawing + .inline + .as_ref() + .map(|image| (&image.embed_id, image.extent_cx.0, image.extent_cy.0)); + let anchor = drawing + .anchor + .as_ref() + .map(|image| (&image.embed_id, image.extent_cx.0, image.extent_cy.0)); + for (embed_id, width_emu, height_emu) in inline.into_iter().chain(anchor) { + if !embed_id.is_empty() && document.image_data(embed_id).is_some() { + image_sizes.push((width_emu, height_emu)); + } + } + } + } let mut image_size_index = 0_usize; let mut output = String::with_capacity(html.len()); let mut remainder = html.as_str(); @@ -681,7 +724,7 @@ fn mhtml_export_html(document: &Document) -> Result<(String, Vec) by_digest.insert(digest, index); index }; - let image = image_sizes.get(image_size_index).ok_or_else(|| { + let (width_emu, height_emu) = *image_sizes.get(image_size_index).ok_or_else(|| { mhtml_error( None, 0, @@ -691,8 +734,8 @@ fn mhtml_export_html(document: &Document) -> Result<(String, Vec) image_size_index += 1; output.push_str(&format!( "").unwrap(); + let run_end = xml.rfind("").unwrap() + "".len(); + xml.insert_str(run_end, ""); + xml.insert_str( + run_start, + r#""#, + ); + let xml = xml + .replacen( + "", + "", + 1, + ) + .replacen("", "", 1); + package.set_part("/word/document.xml", xml.into_bytes()); + let mut saved = Cursor::new(Vec::new()); + package.write_to(&mut saved).unwrap(); + let document = Document::from_bytes(saved.get_ref()).unwrap(); + assert_eq!(document.images().len(), 3); + + let written = document + .to_mhtml_bytes() + .expect("pictures in content controls are dropped, not refused"); + assert_eq!( + written + .diagnostics + .iter() + .map(|diagnostic| (diagnostic.location.as_str(), diagnostic.message.as_str())) + .collect::>(), + vec![ + ("body[0]", "dropped Word body content control"), + ( + "body[2]/paragraph/item[0]", + "dropped Word paragraph content control" + ), + ] + ); + let reopened = Document::from_mhtml_bytes(&written.bytes).unwrap(); + assert_eq!( + reopened + .document + .images() + .iter() + .map(|image| (image.width_emu, image.height_emu)) + .collect::>(), + [(19_050, 28_575)] + ); + } + #[test] fn mhtml_loss_records_do_not_hide_supported_siblings() { let parsed = diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index 365637e1..e4766ff8 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -12427,6 +12427,158 @@ mod content_control_read_walker_regressions { "outer\tfirst inner row\tsecond inner row\tafter\t\n" ); } + + fn picture(id: usize, name: &str) -> String { + format!( + r#""# + ) + } + + /// A heading paragraph holding a word, an internal link and a picture. + fn probe_paragraph() -> String { + format!( + r#"probe link{}"#, + picture(7, "probe picture") + ) + } + + /// The same heading with its word and picture runs passed through `wrap`. + fn inline_probe_paragraph(wrap: impl Fn(&str) -> String) -> String { + let runs = format!( + r#"probe {}"#, + picture(7, "probe picture") + ); + format!( + r#"{}link"#, + wrap(&runs) + ) + } + + #[derive(Debug, PartialEq)] + struct Walked { + text: String, + images: Vec<(String, Option)>, + word_count: usize, + headings: Vec<(u32, String)>, + links: Vec<(String, Option)>, + } + + fn walk(body: &str) -> Walked { + let document = document_with_content_controls(&wrap_word_body(body)); + Walked { + text: document.text(), + images: document + .images() + .into_iter() + .map(|image| (image.embed_id, image.name)) + .collect(), + word_count: document.word_count(), + headings: document.headings(), + links: document + .links() + .into_iter() + .map(|link| (link.text, link.anchor)) + .collect(), + } + } + + #[test] + fn a_content_control_hides_nothing_from_any_body_read_walker() { + let probe = probe_paragraph(); + let in_cell = |content: &str| table(&row(&cell(content))); + // (location, wrapped body, the same body without the control, body level) + let locations = [ + ( + "body block control", + control("goog_rdk_1", &probe), + probe.clone(), + true, + ), + ( + "nested body control", + control("outer", &control("goog_rdk_2", &probe)), + probe.clone(), + true, + ), + ( + "inline control", + inline_probe_paragraph(|runs| control("goog_rdk_3", runs)), + inline_probe_paragraph(str::to_owned), + true, + ), + ( + "nested inline control", + inline_probe_paragraph(|runs| control("outer", &control("goog_rdk_4", runs))), + inline_probe_paragraph(str::to_owned), + true, + ), + ( + "cell control", + in_cell(&control("goog_rdk_5", &probe)), + in_cell(&probe), + false, + ), + ( + "nested cell control", + in_cell(&control("outer", &control("goog_rdk_6", &probe))), + in_cell(&probe), + false, + ), + ( + "row control around a cell", + table(&row(&control("goog_rdk_7", &cell(&probe)))), + in_cell(&probe), + false, + ), + ( + "table control around a row", + table(&control("goog_rdk_8", &row(&cell(&probe)))), + in_cell(&probe), + false, + ), + ( + "control in a nested table", + in_cell(&format!( + "{}", + in_cell(&control("goog_rdk_9", &probe)) + )), + in_cell(&format!("{}", in_cell(&probe))), + false, + ), + ( + "body control around a table", + control("goog_rdk_10", &in_cell(&probe)), + in_cell(&probe), + false, + ), + ]; + for (location, wrapped, unwrapped, body_level) in locations { + let body = format!("{}{wrapped}{}", paragraph("before"), paragraph("end")); + let plain = format!("{}{unwrapped}{}", paragraph("before"), paragraph("end")); + let seen = walk(&body); + assert_eq!(seen, walk(&plain), "{location}"); + + assert!(seen.text.contains("probe link"), "{location}: text"); + assert_eq!(seen.text.matches("probe").count(), 1, "{location}: text"); + assert_eq!( + seen.images, + [("rId7".to_owned(), Some("probe picture".to_owned()))], + "{location}: images" + ); + assert_eq!(seen.word_count, 4, "{location}: word count"); + let (headings, links) = if body_level { + ( + vec![(2, "probe link".to_owned())], + vec![("link".to_owned(), Some("target".to_owned()))], + ) + } else { + // Headings and links do not search table cells. + (Vec::new(), Vec::new()) + }; + assert_eq!(seen.headings, headings, "{location}: headings"); + assert_eq!(seen.links, links, "{location}: links"); + } + } } fn ordered_reader_fixture() -> &'static str { diff --git a/docs/hld/03-architecture.md b/docs/hld/03-architecture.md index 358d004a..a9da3c85 100644 --- a/docs/hld/03-architecture.md +++ b/docs/hld/03-architecture.md @@ -1359,7 +1359,11 @@ surface is introduced. `Document::text` traverses body paragraphs and table cells in document order. Nested tables and the content controls at every level contribute their -paragraphs in place. +paragraphs in place. `Document::images` and `Document::word_count` reach the +same content. `Document::headings` and `Document::links` read the body +paragraphs and those that body-level content controls wrap, and do not search +table cells. MHTML export sizes its images from the paragraphs the HTML emitter +reaches, which leaves content controls out. The WASM binding uses `Document::text` for its existing `getText` method and otherwise owns one complete `Document`. It never reaches into `rdocx-oxml` or maintains a second package representation. From 367c48ace834c955926db8c1796cbd2941199850 Mon Sep 17 00:00:00 2001 From: Hadrien Mary Date: Sun, 27 Sep 2026 17:12:00 +0200 Subject: [PATCH 3/3] Re-record the archive measurements of rdocx and rdocx-cli The read walker fix and its regression tests grow the rdocx package, and the new `rdocx text` integration test grows the rdocx-cli package, so both README rows and their ARCHIVE_MEASUREMENTS entries are re-measured. GitHub issue #160. --- README.md | 2 +- crates/rdocx-cli/README.md | 2 +- scripts/readme_doctests.py | 4 ++-- 3 files changed, 4 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 479b462e..c2f746d6 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,095,513 compressed bytes, 6,512,875 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-cli/README.md b/crates/rdocx-cli/README.md index 25ecc0fe..4619c5aa 100644 --- a/crates/rdocx-cli/README.md +++ b/crates/rdocx-cli/README.md @@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-cli | 33,940 compressed bytes, 146,296 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/scripts/readme_doctests.py b/scripts/readme_doctests.py index e24c643b..17249e7b 100644 --- a/scripts/readme_doctests.py +++ b/scripts/readme_doctests.py @@ -383,8 +383,8 @@ class ReadmeCase: "oxml-opc": (92_122, 355_510, 12), "oxml-pdf": (66_015, 304_432, 14), "oxml-sml": (12_511, 49_803, 6), - "rdocx": (1_092_256, 6_498_484, 36), - "rdocx-cli": (33_805, 145_256, 8), + "rdocx": (1_095_513, 6_512_875, 36), + "rdocx-cli": (33_940, 146_296, 8), "rdocx-html": (15_486, 63_894, 11), "rdocx-layout": (255_752, 1_385_701, 15), "rdocx-opc": (3_655, 9_668, 6),