diff --git a/README.md b/README.md index 479b462e..c2f746d6 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | +| Crates.io archive: rdocx | 1,095,513 compressed bytes, 6,512,875 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 | | Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | | Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 | | Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 | diff --git a/crates/rdocx-cli/README.md b/crates/rdocx-cli/README.md index 25ecc0fe..4619c5aa 100644 --- a/crates/rdocx-cli/README.md +++ b/crates/rdocx-cli/README.md @@ -22,7 +22,7 @@ and produces fixed or flow output without an Office host. | Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on | |---|---|---|---|---|---|---|---|---| -| Crates.io archive: rdocx-cli | 33,805 compressed bytes, 145,256 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | +| Crates.io archive: rdocx-cli | 33,940 compressed bytes, 146,296 member bytes, 8 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx-cli` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-19 | ## Use it when diff --git a/crates/rdocx-cli/tests/integration.rs b/crates/rdocx-cli/tests/integration.rs index e4cdd98f..320be069 100644 --- a/crates/rdocx-cli/tests/integration.rs +++ b/crates/rdocx-cli/tests/integration.rs @@ -177,6 +177,33 @@ fn text_prints_body_and_table_content_in_document_order() { ); } +#[test] +fn text_prints_paragraphs_wrapped_by_a_body_content_control() { + let temp = TempWorkspace::new("text-content-control"); + let input = temp.path.join("control.docx"); + write_document(&input, &["Body text"]); + + let mut package = + OpcPackage::from_reader(std::io::Cursor::new(fs::read(&input).unwrap())).unwrap(); + let xml = std::str::from_utf8(package.get_part("/word/document.xml").unwrap()) + .unwrap() + .replacen( + "", + r#"Wrapped paragraph"#, + 1, + ); + package.set_part("/word/document.xml", xml.into_bytes()); + let mut file = fs::File::create(&input).unwrap(); + package.write_to(&mut file).unwrap(); + + let output = cli(&["text", path_text(&input)]); + assert_success(&output, "text"); + assert_eq!( + String::from_utf8(output.stdout).unwrap(), + "Wrapped paragraph\nBody text\n" + ); +} + #[test] fn convert_writes_valid_formats_and_uses_the_shared_default_output() { let temp = TempWorkspace::new("convert"); diff --git a/crates/rdocx/src/document.rs b/crates/rdocx/src/document.rs index 5c4b2b46..9af51da1 100644 --- a/crates/rdocx/src/document.rs +++ b/crates/rdocx/src/document.rs @@ -9389,6 +9389,91 @@ fn visit_sdt(control: &CT_Sdt, visitor: &mut impl FnMut(&CT_P)) { } } +/// Collect the body paragraphs in document order, those that body-level +/// content controls wrap included. Tables are not entered, with or without a +/// control around them. This is what [`Document::headings`] and +/// [`Document::links`] read. +fn block_paragraphs(content: &[BodyContent]) -> Vec<&CT_P> { + let mut paragraphs = Vec::new(); + for item in content { + match item { + BodyContent::Paragraph(paragraph) => paragraphs.push(paragraph), + BodyContent::ContentControl(control) => { + collect_block_control_paragraphs(control, &mut paragraphs); + } + BodyContent::Table(_) | BodyContent::RawXml(_) => {} + } + } + paragraphs +} + +fn collect_block_control_paragraphs<'a>(control: &'a CT_Sdt, paragraphs: &mut Vec<&'a CT_P>) { + for item in &control.content { + match item { + SdtContent::Paragraph(paragraph) => paragraphs.push(paragraph), + SdtContent::ContentControl(nested) => { + collect_block_control_paragraphs(nested, paragraphs); + } + SdtContent::Table(_) + | SdtContent::Row(_) + | SdtContent::Cell(_) + | SdtContent::Run(_) + | SdtContent::RawXml(_) => {} + } + } +} + +/// Append a block-level paragraph to [`Document::text`] as one line. +fn push_paragraph_line(paragraph: &CT_P, text: &mut String) { + text.push_str(¶graph.text()); + text.push('\n'); +} + +/// Append a table to [`Document::text`] as one line per row, rows wrapped by +/// table-level content controls included. +fn push_table_text(table: &CT_Tbl, text: &mut String) { + for index in 0..=table.rows.len() { + for (_, _, control) in table + .content_controls + .iter() + .filter(|(at, _, _)| *at == index) + { + push_control_text(control, text); + } + if let Some(row) = table.rows.get(index) { + push_row_text(row, text); + } + } +} + +/// Append a row to [`Document::text`] as one line in which every paragraph of +/// its cells ends with a tab, cell controls and nested tables included. +fn push_row_text(row: &CT_Row, text: &mut String) { + visit_row(row, &mut |paragraph| push_cell_paragraph(paragraph, text)); + text.push('\n'); +} + +fn push_cell_paragraph(paragraph: &CT_P, text: &mut String) { + text.push_str(¶graph.text()); + text.push('\t'); +} + +/// Append what a block, table or row content control wraps to [`Document::text`]. +fn push_control_text(control: &CT_Sdt, text: &mut String) { + for item in &control.content { + match item { + SdtContent::Paragraph(paragraph) => push_paragraph_line(paragraph, text), + SdtContent::Table(table) => push_table_text(table, text), + SdtContent::Row(row) => push_row_text(row, text), + SdtContent::Cell(cell) => { + visit_cell(cell, &mut |paragraph| push_cell_paragraph(paragraph, text)); + } + SdtContent::Run(_) | SdtContent::RawXml(_) => {} + SdtContent::ContentControl(nested) => push_control_text(nested, text), + } + } +} + fn visit_body_paragraphs_mut(content: &mut [BodyContent], visitor: &mut impl FnMut(&mut CT_P)) { for item in content { match item { @@ -14633,28 +14718,18 @@ impl Document { } /// Get the plain text of body paragraphs and table cells in document order. + /// + /// A body paragraph ends with a newline. A table row is one line in which + /// every cell paragraph ends with a tab, paragraphs of nested tables + /// included. Content controls at every level contribute the paragraphs, + /// rows and cells they wrap at the position they occupy. pub fn text(&self) -> String { let mut result = String::new(); for content in &self.document.body.content { match content { - BodyContent::Paragraph(paragraph) => { - result.push_str(¶graph.text()); - result.push('\n'); - } - BodyContent::Table(table) => { - for row in &table.rows { - for cell in &row.cells { - for content in &cell.content { - if let CellContent::Paragraph(paragraph) = content { - result.push_str(¶graph.text()); - result.push('\t'); - } - } - } - result.push('\n'); - } - } - BodyContent::ContentControl(_) => {} + BodyContent::Paragraph(paragraph) => push_paragraph_line(paragraph, &mut result), + BodyContent::Table(table) => push_table_text(table, &mut result), + BodyContent::ContentControl(control) => push_control_text(control, &mut result), BodyContent::RawXml(_) => {} } } @@ -22700,13 +22775,13 @@ impl Document { /// Get all headings in the document as (level, text) pairs. /// - /// Detects heading paragraphs by their style ID (e.g. "Heading1", "Heading2"). + /// Detects heading paragraphs by their style ID (e.g. "Heading1", "Heading2") + /// among the body paragraphs, including those that body-level content + /// controls wrap. Table cells are not searched. pub fn headings(&self) -> Vec<(u32, String)> { let mut result = Vec::new(); - for content in &self.document.body.content { - if let BodyContent::Paragraph(p) = content - && let Some(level) = Self::detect_heading_level_for_toc(p) - { + for p in block_paragraphs(&self.document.body.content) { + if let Some(level) = Self::detect_heading_level_for_toc(p) { result.push((level, p.text())); } } @@ -22724,75 +22799,39 @@ impl Document { /// Get information about all images in the document. /// - /// Returns metadata for each inline and anchored image found in body paragraphs. + /// Returns metadata for each inline and anchored image found in body + /// paragraphs, table cells and content controls, in document order. pub fn images(&self) -> Vec { let mut result = Vec::new(); - - for content in &self.document.body.content { - Self::collect_images_from_content(content, &mut result); - } - result - } - - fn collect_images_from_content(content: &BodyContent, result: &mut Vec) { - match content { - BodyContent::Paragraph(p) => Self::collect_images_from_paragraph(p, result), - BodyContent::Table(tbl) => Self::collect_images_from_table(tbl, result), - BodyContent::ContentControl(_) => {} - BodyContent::RawXml(_) => {} - } - } - - fn collect_images_from_paragraph(p: &CT_P, result: &mut Vec) { - for run in &p.runs { - for rc in &run.content { - let RunContent::Drawing(drawing) = rc else { - continue; - }; - if let Some(inline) = &drawing.inline { - result.push(ImageInfo { - embed_id: inline.embed_id.clone(), - name: inline.name.clone(), - description: inline.description.clone(), - width_emu: inline.extent_cx.0, - height_emu: inline.extent_cy.0, - is_anchor: false, - }); - } - if let Some(anchor) = &drawing.anchor { - result.push(ImageInfo { - embed_id: anchor.embed_id.clone(), - name: anchor.name.clone(), - description: anchor.description.clone(), - width_emu: anchor.extent_cx.0, - height_emu: anchor.extent_cy.0, - is_anchor: true, - }); - } + visit_all_drawings(&self.document.body.content, &mut |drawing| { + if let Some(inline) = &drawing.inline { + result.push(ImageInfo { + embed_id: inline.embed_id.clone(), + name: inline.name.clone(), + description: inline.description.clone(), + width_emu: inline.extent_cx.0, + height_emu: inline.extent_cy.0, + is_anchor: false, + }); } - } - } - - fn collect_images_from_table(tbl: &CT_Tbl, result: &mut Vec) { - use rdocx_oxml::table::CellContent; - - for row in &tbl.rows { - for cell in &row.cells { - for cc in &cell.content { - match cc { - CellContent::Paragraph(p) => Self::collect_images_from_paragraph(p, result), - CellContent::Table(nested) => { - Self::collect_images_from_table(nested, result) - } - CellContent::ContentControl(_) => {} - } - } + if let Some(anchor) = &drawing.anchor { + result.push(ImageInfo { + embed_id: anchor.embed_id.clone(), + name: anchor.name.clone(), + description: anchor.description.clone(), + width_emu: anchor.extent_cx.0, + height_emu: anchor.extent_cy.0, + is_anchor: true, + }); } - } + }); + result } /// Get information about all hyperlinks in the document. /// + /// Reads the body paragraphs, including those that body-level content + /// controls wrap. Table cells are not searched. /// Resolves hyperlink relationship IDs to their target URLs where possible. pub fn links(&self) -> Vec { use oxml_opc::relationship::rel_types; @@ -22810,35 +22849,33 @@ impl Document { } let mut result = Vec::new(); - for content in &self.document.body.content { - if let BodyContent::Paragraph(p) = content { - for hl in &p.hyperlinks { - // `HyperlinkSpan`'s bounds are public and can be set by - // hand, so clamp rather than slice-panic on a bad range. - let start = hl.run_start.min(p.runs.len()); - let end = hl.run_end.clamp(start, p.runs.len()); - let text: String = p.runs[start..end].iter().map(|r| r.text()).collect(); - - let url = hl.rel_id.as_ref().and_then(|id| url_map.get(id)).cloned(); - - result.push(LinkInfo { - text, - url, - anchor: hl.anchor.clone(), - rel_id: hl.rel_id.clone(), - }); - } - for field in p.complex_field_hyperlinks() { - let start = field.run_start.min(p.runs.len()); - let end = field.run_end.clamp(start, p.runs.len()); - let text: String = p.runs[start..end].iter().map(|run| run.text()).collect(); - result.push(LinkInfo { - text, - url: Some(field.target), - anchor: None, - rel_id: None, - }); - } + for p in block_paragraphs(&self.document.body.content) { + for hl in &p.hyperlinks { + // `HyperlinkSpan`'s bounds are public and can be set by + // hand, so clamp rather than slice-panic on a bad range. + let start = hl.run_start.min(p.runs.len()); + let end = hl.run_end.clamp(start, p.runs.len()); + let text: String = p.runs[start..end].iter().map(|r| r.text()).collect(); + + let url = hl.rel_id.as_ref().and_then(|id| url_map.get(id)).cloned(); + + result.push(LinkInfo { + text, + url, + anchor: hl.anchor.clone(), + rel_id: hl.rel_id.clone(), + }); + } + for field in p.complex_field_hyperlinks() { + let start = field.run_start.min(p.runs.len()); + let end = field.run_end.clamp(start, p.runs.len()); + let text: String = p.runs[start..end].iter().map(|run| run.text()).collect(); + result.push(LinkInfo { + text, + url: Some(field.target), + anchor: None, + rel_id: None, + }); } } result @@ -22847,43 +22884,12 @@ impl Document { /// Count the number of words in the document. /// /// Counts whitespace-separated tokens across all paragraphs (including - /// paragraphs inside table cells). + /// paragraphs inside table cells and content controls). pub fn word_count(&self) -> usize { let mut count = 0; - for content in &self.document.body.content { - count += Self::word_count_in_content(content); - } - count - } - - fn word_count_in_content(content: &BodyContent) -> usize { - match content { - BodyContent::Paragraph(p) => p.text().split_whitespace().count(), - BodyContent::Table(tbl) => Self::word_count_in_table(tbl), - BodyContent::ContentControl(_) => 0, - BodyContent::RawXml(_) => 0, - } - } - - fn word_count_in_table(tbl: &CT_Tbl) -> usize { - use rdocx_oxml::table::CellContent; - - let mut count = 0; - for row in &tbl.rows { - for cell in &row.cells { - for cc in &cell.content { - match cc { - CellContent::Paragraph(p) => { - count += p.text().split_whitespace().count(); - } - CellContent::Table(nested) => { - count += Self::word_count_in_table(nested); - } - CellContent::ContentControl(_) => {} - } - } - } - } + visit_body_paragraphs(&self.document.body.content, &mut |paragraph| { + count += paragraph.text().split_whitespace().count(); + }); count } diff --git a/crates/rdocx/src/html.rs b/crates/rdocx/src/html.rs index 6ec00fb4..ade6c90e 100644 --- a/crates/rdocx/src/html.rs +++ b/crates/rdocx/src/html.rs @@ -11,7 +11,7 @@ use base64::Engine as _; use base64::engine::general_purpose::STANDARD as BASE64; use rdocx_oxml::document::BodyContent; use rdocx_oxml::table::{ - CT_Row, CT_Tbl, CT_TblGrid, CT_TblGridCol, CT_TblPr, CT_TblWidth, CT_Tc, VMerge, + CT_Row, CT_Tbl, CT_TblGrid, CT_TblGridCol, CT_TblPr, CT_TblWidth, CT_Tc, CellContent, VMerge, }; use rdocx_oxml::text::{CT_P, RunContent}; use rdocx_oxml::units::Twips; @@ -616,15 +616,58 @@ fn base64_lines(bytes: &[u8]) -> String { output } +/// Collect the body paragraphs, table cell paragraphs and nested table +/// paragraphs in document order, leaving content controls out as the +/// rdocx-html emitter does. This is the reach [`Document::images`] had before +/// it read content controls, so the emitter's `` tags still pair with the +/// pictures of these paragraphs by position. +fn emitted_paragraphs(content: &[BodyContent]) -> Vec<&CT_P> { + let mut paragraphs = Vec::new(); + for item in content { + match item { + BodyContent::Paragraph(paragraph) => paragraphs.push(paragraph), + BodyContent::Table(table) => collect_emitted_table_paragraphs(table, &mut paragraphs), + BodyContent::ContentControl(_) | BodyContent::RawXml(_) => {} + } + } + paragraphs +} + +fn collect_emitted_table_paragraphs<'a>(table: &'a CT_Tbl, paragraphs: &mut Vec<&'a CT_P>) { + for cell in table.rows.iter().flat_map(|row| &row.cells) { + for item in &cell.content { + match item { + CellContent::Paragraph(paragraph) => paragraphs.push(paragraph), + CellContent::Table(nested) => collect_emitted_table_paragraphs(nested, paragraphs), + CellContent::ContentControl(_) => {} + } + } + } +} + fn mhtml_export_html(document: &Document) -> Result<(String, Vec)> { let html = document.to_html(); - let image_sizes = document - .images() - .into_iter() - .filter(|image| { - !image.embed_id.is_empty() && document.image_data(&image.embed_id).is_some() - }) - .collect::>(); + let mut image_sizes = Vec::new(); + for paragraph in emitted_paragraphs(&document.document.body.content) { + for content in paragraph.runs.iter().flat_map(|run| &run.content) { + let RunContent::Drawing(drawing) = content else { + continue; + }; + let inline = drawing + .inline + .as_ref() + .map(|image| (&image.embed_id, image.extent_cx.0, image.extent_cy.0)); + let anchor = drawing + .anchor + .as_ref() + .map(|image| (&image.embed_id, image.extent_cx.0, image.extent_cy.0)); + for (embed_id, width_emu, height_emu) in inline.into_iter().chain(anchor) { + if !embed_id.is_empty() && document.image_data(embed_id).is_some() { + image_sizes.push((width_emu, height_emu)); + } + } + } + } let mut image_size_index = 0_usize; let mut output = String::with_capacity(html.len()); let mut remainder = html.as_str(); @@ -681,7 +724,7 @@ fn mhtml_export_html(document: &Document) -> Result<(String, Vec) by_digest.insert(digest, index); index }; - let image = image_sizes.get(image_size_index).ok_or_else(|| { + let (width_emu, height_emu) = *image_sizes.get(image_size_index).ok_or_else(|| { mhtml_error( None, 0, @@ -691,8 +734,8 @@ fn mhtml_export_html(document: &Document) -> Result<(String, Vec) image_size_index += 1; output.push_str(&format!( "").unwrap(); + let run_end = xml.rfind("").unwrap() + "".len(); + xml.insert_str(run_end, ""); + xml.insert_str( + run_start, + r#""#, + ); + let xml = xml + .replacen( + "", + "", + 1, + ) + .replacen("", "", 1); + package.set_part("/word/document.xml", xml.into_bytes()); + let mut saved = Cursor::new(Vec::new()); + package.write_to(&mut saved).unwrap(); + let document = Document::from_bytes(saved.get_ref()).unwrap(); + assert_eq!(document.images().len(), 3); + + let written = document + .to_mhtml_bytes() + .expect("pictures in content controls are dropped, not refused"); + assert_eq!( + written + .diagnostics + .iter() + .map(|diagnostic| (diagnostic.location.as_str(), diagnostic.message.as_str())) + .collect::>(), + vec![ + ("body[0]", "dropped Word body content control"), + ( + "body[2]/paragraph/item[0]", + "dropped Word paragraph content control" + ), + ] + ); + let reopened = Document::from_mhtml_bytes(&written.bytes).unwrap(); + assert_eq!( + reopened + .document + .images() + .iter() + .map(|image| (image.width_emu, image.height_emu)) + .collect::>(), + [(19_050, 28_575)] + ); + } + #[test] fn mhtml_loss_records_do_not_hide_supported_siblings() { let parsed = diff --git a/crates/rdocx/tests/regression_test.rs b/crates/rdocx/tests/regression_test.rs index d143b3c7..e4766ff8 100644 --- a/crates/rdocx/tests/regression_test.rs +++ b/crates/rdocx/tests/regression_test.rs @@ -12339,6 +12339,248 @@ fn namespace_classification_metadata_exists_only_for_raw_children() { )); } +/// The body read walkers see through content controls (GitHub issue #160). +mod content_control_read_walker_regressions { + use super::*; + + fn control(tag: &str, content: &str) -> String { + format!( + r#"{content}"# + ) + } + + fn table(rows: &str) -> String { + format!(r#"{rows}"#) + } + + fn row(cells: &str) -> String { + format!("{cells}") + } + + fn cell(content: &str) -> String { + format!("{content}") + } + + fn paragraph(text: &str) -> String { + format!(r#"{text}"#) + } + + #[test] + fn text_reads_every_content_control_location_in_document_order() { + let body = [ + paragraph("first"), + control("goog_rdk_1", ¶graph("block")), + table(&format!( + "{}{}", + control("rows", &row(&cell(¶graph("wrapped row")))), + row(&format!( + "{}{}{}{}", + cell(¶graph("plain")), + control("cells", &cell(¶graph("wrapped cell"))), + cell(&control("goog_rdk_2", ¶graph("cell control"))), + cell(&format!( + "{}{}", + table(&row(&format!( + "{}{}", + cell(¶graph("inner a")), + cell(¶graph("inner b")) + ))), + paragraph("after inner") + )), + )) + )), + control("outer", &control("goog_rdk_3", ¶graph("nested"))), + format!( + r#"{} tail"#, + control("goog_rdk_4", r#"inline"#) + ), + paragraph("last"), + ] + .concat(); + let document = document_with_content_controls(&wrap_word_body(&body)); + + assert_eq!( + document.text(), + "first\nblock\nwrapped row\t\nplain\twrapped cell\tcell control\tinner a\tinner b\tafter inner\t\nnested\ninline tail\nlast\n" + ); + } + + #[test] + fn nested_table_cells_contribute_text_inside_their_outer_row() { + let body = table(&row(&format!( + "{}{}", + cell(¶graph("outer")), + cell(&format!( + "{}{}", + table(&format!( + "{}{}", + row(&cell(¶graph("first inner row"))), + row(&cell(¶graph("second inner row"))) + )), + paragraph("after") + )) + ))); + let document = document_with_content_controls(&wrap_word_body(&body)); + + assert_eq!( + document.text(), + "outer\tfirst inner row\tsecond inner row\tafter\t\n" + ); + } + + fn picture(id: usize, name: &str) -> String { + format!( + r#""# + ) + } + + /// A heading paragraph holding a word, an internal link and a picture. + fn probe_paragraph() -> String { + format!( + r#"probe link{}"#, + picture(7, "probe picture") + ) + } + + /// The same heading with its word and picture runs passed through `wrap`. + fn inline_probe_paragraph(wrap: impl Fn(&str) -> String) -> String { + let runs = format!( + r#"probe {}"#, + picture(7, "probe picture") + ); + format!( + r#"{}link"#, + wrap(&runs) + ) + } + + #[derive(Debug, PartialEq)] + struct Walked { + text: String, + images: Vec<(String, Option)>, + word_count: usize, + headings: Vec<(u32, String)>, + links: Vec<(String, Option)>, + } + + fn walk(body: &str) -> Walked { + let document = document_with_content_controls(&wrap_word_body(body)); + Walked { + text: document.text(), + images: document + .images() + .into_iter() + .map(|image| (image.embed_id, image.name)) + .collect(), + word_count: document.word_count(), + headings: document.headings(), + links: document + .links() + .into_iter() + .map(|link| (link.text, link.anchor)) + .collect(), + } + } + + #[test] + fn a_content_control_hides_nothing_from_any_body_read_walker() { + let probe = probe_paragraph(); + let in_cell = |content: &str| table(&row(&cell(content))); + // (location, wrapped body, the same body without the control, body level) + let locations = [ + ( + "body block control", + control("goog_rdk_1", &probe), + probe.clone(), + true, + ), + ( + "nested body control", + control("outer", &control("goog_rdk_2", &probe)), + probe.clone(), + true, + ), + ( + "inline control", + inline_probe_paragraph(|runs| control("goog_rdk_3", runs)), + inline_probe_paragraph(str::to_owned), + true, + ), + ( + "nested inline control", + inline_probe_paragraph(|runs| control("outer", &control("goog_rdk_4", runs))), + inline_probe_paragraph(str::to_owned), + true, + ), + ( + "cell control", + in_cell(&control("goog_rdk_5", &probe)), + in_cell(&probe), + false, + ), + ( + "nested cell control", + in_cell(&control("outer", &control("goog_rdk_6", &probe))), + in_cell(&probe), + false, + ), + ( + "row control around a cell", + table(&row(&control("goog_rdk_7", &cell(&probe)))), + in_cell(&probe), + false, + ), + ( + "table control around a row", + table(&control("goog_rdk_8", &row(&cell(&probe)))), + in_cell(&probe), + false, + ), + ( + "control in a nested table", + in_cell(&format!( + "{}", + in_cell(&control("goog_rdk_9", &probe)) + )), + in_cell(&format!("{}", in_cell(&probe))), + false, + ), + ( + "body control around a table", + control("goog_rdk_10", &in_cell(&probe)), + in_cell(&probe), + false, + ), + ]; + for (location, wrapped, unwrapped, body_level) in locations { + let body = format!("{}{wrapped}{}", paragraph("before"), paragraph("end")); + let plain = format!("{}{unwrapped}{}", paragraph("before"), paragraph("end")); + let seen = walk(&body); + assert_eq!(seen, walk(&plain), "{location}"); + + assert!(seen.text.contains("probe link"), "{location}: text"); + assert_eq!(seen.text.matches("probe").count(), 1, "{location}: text"); + assert_eq!( + seen.images, + [("rId7".to_owned(), Some("probe picture".to_owned()))], + "{location}: images" + ); + assert_eq!(seen.word_count, 4, "{location}: word count"); + let (headings, links) = if body_level { + ( + vec![(2, "probe link".to_owned())], + vec![("link".to_owned(), Some("target".to_owned()))], + ) + } else { + // Headings and links do not search table cells. + (Vec::new(), Vec::new()) + }; + assert_eq!(seen.headings, headings, "{location}: headings"); + assert_eq!(seen.links, links, "{location}: links"); + } + } +} + fn ordered_reader_fixture() -> &'static str { r#"