Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,7 +39,7 @@ rows are the enforced release-mode bounds plus one dated observation.

| Measurement | Value | Version | Platform | Build mode | Input | Command | Statistic | Measured on |
|---|---|---|---|---|---|---|---|---|
| Crates.io archive: rdocx | 1,092,256 compressed bytes, 6,498,484 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Crates.io archive: rdocx | 1,095,450 compressed bytes, 6,512,553 member bytes, 36 members | 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | `cargo package --locked --no-verify` | Tracked `rdocx` package inventory | `python3 scripts/readme_doctests.py --record-measurements` | gzip archive bytes, tar member bytes, tar member count | 2026-09-26 |
| Large-document layout throughput | minimum 250 pages/s, observed 31,019.1 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
| Large-document layout peak allocation | maximum 64 MiB, observed 29.03 MiB | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 one-page paragraphs with deterministic fonts | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | peak live allocation | 2026-09-19 |
| Large-document PDF throughput | minimum 1,000 pages/s, observed 60,058.0 pages/s | rdocx 0.14.0 | macOS 26.6.2, Apple M5 Max, arm64 | release, one test thread | 1,000 deterministic layout pages | `cargo test -p rdocx --test regression_test --release a_thousand_page_document_paginates_and_renders_within_the_declared_limits -- --ignored --exact --nocapture --test-threads=1` | pages per wall-clock second | 2026-09-19 |
Expand Down
2 changes: 1 addition & 1 deletion crates/rdocx-py/python/rdocx/_rdocx.pyi
Original file line number Diff line number Diff line change
Expand Up @@ -428,7 +428,7 @@ class Document:
self, story: Story, relationship_id: str, data: bytes
) -> None: ...
def split_run(
self, body_index: int, run_index: int, character_offset: int
self, body_index: int | Paragraph, run_index: int, character_offset: int
) -> int: ...
def to_pdf(self) -> bytes: ...
def render_page_to_png(self, page_index: int, dpi: float = 150.0) -> bytes | None: ...
Expand Down
87 changes: 72 additions & 15 deletions crates/rdocx-py/src/document.rs
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ use pyo3::prelude::*;
use pyo3::types::{PyAny, PyBytes, PyList, PyTuple};
use smallvec::smallvec;

use crate::paragraph::{PyParagraph, PyParagraphCollection};
use crate::paragraph::{ParagraphLocation, PyParagraph, PyParagraphCollection};
use crate::rdocx_to_pyerr;
use crate::table::{PyTable, PyTableCollection};

Expand Down Expand Up @@ -979,6 +979,32 @@ impl PyDocument {
)))
}

/// Split a run of the direct body paragraph at `body_index`. The revision
/// advances only when a continuation run is created.
fn split_body_run(
&mut self,
py: Python<'_>,
body_index: usize,
run_index: usize,
character_offset: usize,
) -> PyResult<usize> {
let paragraph_index = self.inner.paragraph_index_of_content(body_index);
let run_count = |document: &rdocx::Document| {
paragraph_index
.and_then(|index| document.paragraph(index))
.map(|paragraph| paragraph.run_count())
};
let before = run_count(&self.inner);
let boundary = self
.inner
.split_run(body_index, run_index, character_offset)
.map_err(|error| rdocx_to_pyerr(py, error))?;
if run_count(&self.inner) != before {
self.revisions.bump();
}
Ok(boundary)
}

/// Run a native mutation that reports how many things it changed.
///
/// The GIL is released while it runs, and live handles are staled only
Expand Down Expand Up @@ -1193,26 +1219,57 @@ impl PyDocument {
}

fn split_run(
&mut self,
slf: Py<Self>,
py: Python<'_>,
body_index: usize,
body_index: &Bound<'_, PyAny>,
run_index: usize,
character_offset: usize,
) -> PyResult<usize> {
let before = self
.inner
.paragraph(body_index)
.map(|paragraph| paragraph.run_count());
let boundary = self
let Ok(paragraph) = body_index.cast::<PyParagraph>() else {
// An integer is the direct body index, as in `RunPosition`.
let body_index = body_index.extract::<usize>().map_err(|error| {
if error.is_instance_of::<PyTypeError>(py) {
PyTypeError::new_err("body_index must be an int or a Paragraph handle")
} else {
error
}
})?;
return slf
.borrow_mut(py)
.split_body_run(py, body_index, run_index, character_offset);
};
let paragraph = paragraph.borrow();
if !paragraph.belongs_to(py, &slf) {
return Err(PyValueError::new_err(
"paragraph handle belongs to a different document",
));
}
// A cell handle counts paragraphs inside cell content controls, which
// `Cell::paragraph_mut` does not, so it could name another paragraph.
let ParagraphLocation::Body(paragraph_index) = paragraph.validate(py)? else {
return Err(PyValueError::new_err(
"split_run does not accept a table cell paragraph handle",
));
};
let mut document = slf.borrow_mut(py);
if let Some(body_index) = document.inner.content_index_of_paragraph(paragraph_index) {
return document.split_body_run(py, body_index, run_index, character_offset);
}
// A paragraph inside a block content control has no direct body index.
let run_count = |document: &rdocx::Document| {
document
.paragraph(paragraph_index)
.map(|paragraph| paragraph.run_count())
};
let before = run_count(&document.inner);
let boundary = document
.inner
.split_run(body_index, run_index, character_offset)
.paragraph_mut(paragraph_index)
.ok_or_else(|| PyIndexError::new_err("paragraph index out of range"))?
.split_run(run_index, character_offset)
.map_err(|error| rdocx_to_pyerr(py, error))?;
let after = self
.inner
.paragraph(body_index)
.map(|paragraph| paragraph.run_count());
if before != after {
self.revisions.bump();
if run_count(&document.inner) != before {
document.revisions.bump();
}
Ok(boundary)
}
Expand Down
126 changes: 126 additions & 0 deletions crates/rdocx-py/tests/test_core.py
Original file line number Diff line number Diff line change
Expand Up @@ -1291,6 +1291,132 @@ def test_split_run_rejects_bad_coordinates_without_changing_the_document():
assert document.to_bytes() == before


def test_split_run_takes_the_direct_body_index_after_a_table():
import rdocx

# GitHub issue #163: the index find_content_index returns must address
# the same paragraph in split_run.
document = rdocx.Document()
document.add_paragraph("Alpha paragraph before the table.")
document.add_table(1, 1).cell(0, 0).text = "cell"
document.add_paragraph("Beta paragraph after the table.")
document.add_paragraph("Gamma paragraph at the end.")
target = next(p for p in document.paragraphs if p.text.startswith("Beta"))
body_index = document.find_content_index(target)
assert body_index == 2

assert document.split_run(body_index, 0, 4) == 1
assert [[run.text for run in p.runs] for p in document.paragraphs] == [
["Alpha paragraph before the table."],
["Beta", " paragraph after the table."],
["Gamma paragraph at the end."],
]

before = document.to_bytes()
with pytest.raises(rdocx.RdocxError, match="is a table, not a paragraph"):
document.split_run(1, 0, 1)
with pytest.raises(TypeError, match="must be an int or a Paragraph handle"):
document.split_run("Beta", 0, 1)
with pytest.raises(OverflowError):
document.split_run(-1, 0, 1)
assert document.to_bytes() == before


def test_split_run_accepts_a_paragraph_handle_in_a_control_or_the_body():
import rdocx

document = _replace_document_body(
rdocx.Document(),
"<w:p><w:r><w:t>Alpha.</w:t></w:r></w:p>"
"<w:sdt><w:sdtContent>"
"<w:p><w:r><w:t>Control one.</w:t></w:r></w:p>"
"<w:p><w:r><w:t>Control two.</w:t></w:r></w:p>"
"</w:sdtContent></w:sdt>"
'<w:tbl><w:tblGrid><w:gridCol w:w="4000"/></w:tblGrid>'
"<w:tr><w:tc>"
"<w:sdt><w:sdtContent>"
"<w:p><w:r><w:t>In control.</w:t></w:r></w:p>"
"</w:sdtContent></w:sdt>"
"<w:p><w:r><w:t>Cell text.</w:t></w:r></w:p>"
"</w:tc></w:tr>"
"</w:tbl>"
"<w:p><w:r><w:t>Beta.</w:t></w:r></w:p>",
)
control_two = document.paragraphs[2]
assert control_two.text == "Control two."
run = control_two.runs[0]
before_no_op = document.to_bytes()
assert document.split_run(control_two, 0, 0) == 0
assert document.to_bytes() == before_no_op
assert run.text == "Control two."

assert document.split_run(control_two, 0, 7) == 1
with pytest.raises(rdocx.StaleElementError):
run.text
with pytest.raises(rdocx.StaleElementError):
document.split_run(control_two, 0, 1)
assert [r.text for r in document.paragraphs[2].runs] == ["Control", " two."]

# A cell handle is refused, because its index counts the paragraph
# inside the cell's content control and the cell writer does not.
cell = document.tables[0].rows[0].cells[0]
assert [p.text for p in cell.paragraphs] == ["In control.", "Cell text."]
before_cell = document.to_bytes()
for cell_paragraph in cell.paragraphs:
with pytest.raises(ValueError, match="table cell paragraph handle"):
document.split_run(cell_paragraph, 0, 2)
assert document.to_bytes() == before_cell

beta = document.paragraphs[3]
assert document.find_content_index(beta) == 3
beta_run = beta.runs[0]
assert document.split_run(beta, 0, 5) == 1
assert beta_run.text == "Beta."
assert document.split_run(beta, 0, 4) == 1
assert [r.text for r in document.paragraphs[3].runs] == ["Beta", "."]

other = rdocx.Document()
other.add_paragraph("elsewhere")
before = document.to_bytes()
with pytest.raises(ValueError, match="different document"):
document.split_run(other.paragraphs[0], 0, 1)
assert document.to_bytes() == before


def test_story_comment_after_a_block_content_control_anchors_on_its_paragraph():
import rdocx

document = _replace_document_body(
rdocx.Document(),
"<w:p><w:r><w:t>Alpha.</w:t></w:r></w:p>"
"<w:sdt><w:sdtContent>"
"<w:p><w:r><w:t>Control one.</w:t></w:r></w:p>"
"</w:sdtContent></w:sdt>"
"<w:p><w:r><w:t>Beta paragraph.</w:t></w:r></w:p>",
)
item = next(
item
for item in document.story_items
if item.story.kind == "body"
and item.kind == "paragraph"
and item.text == "Beta paragraph."
)
comment_id = document.add_comment(
rdocx.StoryRunRange(
start=rdocx.StoryRunPosition(item=item, run_index=0),
end=rdocx.StoryRunPosition(item=item, run_index=1),
),
author="Ada",
text="Here",
)
xml = _document_xml(document).decode()
start = xml.index(f'commentRangeStart w:id="{comment_id}"')
end = xml.index(f'commentRangeEnd w:id="{comment_id}"')
assert re.findall(r"<w:t[^>]*>([^<]*)</w:t>", xml[start:end]) == [
"Beta paragraph."
]


def test_python_round_three_authoring_and_inspection_is_typed_and_lossless():
import rdocx

Expand Down
1 change: 1 addition & 0 deletions crates/rdocx-py/tests/typing_smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,7 @@ def exercise_rdocx_types(path: Path) -> None:
paragraph_format: ParagraphFormat = paragraph.paragraph_format
paragraph_format.keep_together = None
split_boundary: int = document.split_run(0, 0, 1)
handle_boundary: int = document.split_run(paragraph, 0, 1)
paragraphs: ParagraphCollection = document.paragraphs
first: Paragraph = paragraphs[0]
sliced: list[Paragraph] = paragraphs[:]
Expand Down
Loading
Loading