Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
29 changes: 26 additions & 3 deletions crates/rdocx-py/python/rdocx/_rdocx.pyi
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
import datetime as _datetime
import os as _os
from collections.abc import Iterator as _Iterator
from collections.abc import Iterator as _Iterator, Sequence as _Sequence
from typing import Literal as _Literal, NoReturn as _Never, final as _final, overload as _overload

from . import shared as _shared
Expand Down Expand Up @@ -443,8 +443,31 @@ class Document:
pages: list[int] | None = None,
) -> list[bytes] | bytes: ...
def compare(
self, edited: Document, author: str, timestamp: str
) -> tuple[ComparisonDiagnostic, ...]: ...
self,
edited: Document,
author: str,
timestamp: str,
*,
granularity: _Literal["run", "word", "character"] = "run",
ignore_formatting: bool = False,
ignore_whitespace: bool = False,
ignore_fields: bool = False,
ignore_comments: bool = False,
ignored_stories: _Sequence[
_Literal["body", "header", "footer", "comment", "text_box", "footnote", "endnote"]
] | None = None,
) -> tuple[ComparisonDiagnostic, ...]:
"""Record how ``edited`` differs from this document as tracked changes.

``granularity`` selects the unit of a text change. The default ``"run"``
replaces a changed run whole, while ``"word"`` and ``"character"`` mark
only the changed words or characters. ``ignored_stories`` takes
``Story.kind`` names. The ignore options and ``ignored_stories`` are
left-biased: an ignored difference or story keeps this document's
content. With ``ignore_comments=True`` the result keeps this document's
comments and anchors, and the comments of ``edited`` are not carried
over. An unknown option value raises ``RdocxError``.
"""
@property
def comments(self) -> tuple[Comment, ...]: ...
@property
Expand Down
63 changes: 62 additions & 1 deletion crates/rdocx-py/src/document.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1275,17 +1275,50 @@ impl PyDocument {
}
}

#[pyo3(signature = (
edited,
author,
timestamp,
*,
granularity = "run",
ignore_formatting = false,
ignore_whitespace = false,
ignore_fields = false,
ignore_comments = false,
ignored_stories = None,
))]
#[allow(clippy::too_many_arguments)]
fn compare<'py>(
&mut self,
edited: &PyDocument,
author: &str,
timestamp: &str,
py: Python<'py>,
granularity: &str,
ignore_formatting: bool,
ignore_whitespace: bool,
ignore_fields: bool,
ignore_comments: bool,
ignored_stories: Option<Vec<String>>,
) -> PyResult<Bound<'py, PyTuple>> {
let (diagnostics, changed) = py
.detach(|| {
let options = rdocx::ComparisonOptions {
granularity: parse_comparison_granularity(granularity)?,
ignore_formatting,
ignore_whitespace,
ignore_fields,
ignore_comments,
ignored_stories: ignored_stories
.iter()
.flatten()
.map(|name| parse_comparison_story(name))
.collect::<rdocx::Result<_>>()?,
};
let before = self.inner.to_bytes()?;
let diagnostics = self.inner.compare(&edited.inner, author, timestamp)?;
let diagnostics =
self.inner
.compare_with_options(&edited.inner, author, timestamp, &options)?;
let changed = self.inner.to_bytes()? != before;
Ok::<_, rdocx::Error>((diagnostics, changed))
})
Expand Down Expand Up @@ -1978,3 +2011,31 @@ fn parse_raster_format(
))),
}
}

fn parse_comparison_granularity(name: &str) -> rdocx::Result<rdocx::ComparisonGranularity> {
match name {
"run" => Ok(rdocx::ComparisonGranularity::Run),
"word" => Ok(rdocx::ComparisonGranularity::Word),
"character" => Ok(rdocx::ComparisonGranularity::Character),
other => Err(rdocx::Error::Other(format!(
"unknown comparison granularity {other:?}, expected run, word, or character"
))),
}
}

/// Story names follow `Story.kind`, so `body` selects the main story.
fn parse_comparison_story(name: &str) -> rdocx::Result<rdocx::ComparisonStoryKind> {
match name {
"body" => Ok(rdocx::ComparisonStoryKind::Main),
"header" => Ok(rdocx::ComparisonStoryKind::Header),
"footer" => Ok(rdocx::ComparisonStoryKind::Footer),
"comment" => Ok(rdocx::ComparisonStoryKind::Comment),
"text_box" => Ok(rdocx::ComparisonStoryKind::TextBox),
"footnote" => Ok(rdocx::ComparisonStoryKind::Footnote),
"endnote" => Ok(rdocx::ComparisonStoryKind::Endnote),
other => Err(rdocx::Error::Other(format!(
"unknown comparison story {other:?}, expected body, header, footer, comment, \
text_box, footnote, or endnote"
))),
}
}
205 changes: 205 additions & 0 deletions crates/rdocx-py/tests/test_core.py
Original file line number Diff line number Diff line change
Expand Up @@ -914,6 +914,211 @@ def test_priority_word_operations_return_typed_snapshots_and_remain_atomic():
assert live_after_noop.text == "no table of contents"


_COMPARE_TIMESTAMP = "2026-09-27T12:00:00Z"
_LOREM = (
"Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod "
"tempor incididunt ut labore et dolore magna aliqua. Ut enim ad minim "
"veniam, quis nostrud exercitation ullamco."
)


def _tracked_texts(xml):
xml = xml.decode()
deleted = [
"".join(re.findall(r"<w:delText(?: [^>]*)?>([^<]*)</w:delText>", wrapper))
for wrapper in re.findall(r"<w:del\b[^>]*(?<!/)>.*?</w:del>", xml)
]
inserted = [
"".join(re.findall(r"<w:t(?: [^>]*)?>([^<]*)</w:t>", wrapper))
for wrapper in re.findall(r"<w:ins\b[^>]*(?<!/)>.*?</w:ins>", xml)
]
return deleted, inserted


def _package_part(document, name):
with zipfile.ZipFile(io.BytesIO(document.to_bytes())) as archive:
return archive.read(name)


def test_compare_granularity_marks_only_the_changed_word():
import rdocx

edited_text = _LOREM.replace("magna", "MAGNA")

def redline(**options):
original = rdocx.Document()
original.add_paragraph(_LOREM)
edited = rdocx.Document()
edited.add_paragraph(edited_text)
assert original.compare(edited, "Ada", _COMPARE_TIMESTAMP, **options) == ()
return original

whole_run = ([_LOREM], [edited_text])
for options, expected in [
({}, whole_run),
({"granularity": "run"}, whole_run),
({"granularity": "word"}, (["magna"], ["MAGNA"])),
({"granularity": "character"}, (["magna"], ["MAGNA"])),
]:
compared = redline(**options)
assert _tracked_texts(_document_xml(compared)) == expected, options
accepted = rdocx.Document.from_bytes(compared.to_bytes())
accepted.accept_all()
assert accepted.paragraphs[0].text == edited_text
rejected = rdocx.Document.from_bytes(compared.to_bytes())
rejected.reject_all()
assert rejected.paragraphs[0].text == _LOREM

explicit_defaults = redline(
granularity="run",
ignore_formatting=False,
ignore_whitespace=False,
ignore_fields=False,
ignore_comments=False,
ignored_stories=(),
)
assert explicit_defaults.to_bytes() == redline().to_bytes()


def test_compare_ignore_options_keep_the_original_side():
import rdocx

def compared(original, edited, **options):
work = rdocx.Document.from_bytes(original.to_bytes())
assert work.compare(edited, "Ada", _COMPARE_TIMESTAMP, **options) == ()
return work

plain = rdocx.Document()
plain.add_paragraph("plain")
bold = rdocx.Document.from_bytes(plain.to_bytes())
bold.paragraphs[0].runs[0].font.bold = True
assert [item.kind for item in compared(plain, bold).revisions] == [
"run_property_change"
]
unformatted = compared(plain, bold, ignore_formatting=True)
assert unformatted.revisions == ()
assert unformatted.paragraphs[0].runs[0].font.bold is None

spaced = rdocx.Document()
spaced.add_paragraph("old tail")
single = rdocx.Document()
single.add_paragraph("old tail")
assert [item.kind for item in compared(spaced, single).revisions] == [
"deletion",
"insertion",
]
unspaced = compared(spaced, single, ignore_whitespace=True)
assert unspaced.revisions == ()
assert unspaced.paragraphs[0].text == "old tail"

def page_field(result):
document = rdocx.Document()
document.add_paragraph("placeholder")
return _replace_document_body(
document,
'<w:p><w:r><w:fldChar w:fldCharType="begin"/></w:r>'
'<w:r><w:instrText xml:space="preserve"> PAGE </w:instrText></w:r>'
'<w:r><w:fldChar w:fldCharType="separate"/></w:r>'
f"<w:r><w:t>{result}</w:t></w:r>"
'<w:r><w:fldChar w:fldCharType="end"/></w:r></w:p>',
)

first_page, second_page = page_field("1"), page_field("2")
assert [item.kind for item in compared(first_page, second_page).revisions] == [
"deletion",
"insertion",
]
unfielded = compared(first_page, second_page, ignore_fields=True)
assert unfielded.revisions == ()
assert b"<w:t>1</w:t>" in _document_xml(unfielded)

reviewed = rdocx.Document()
reviewed.add_paragraph("review this")
reviewed.add_paragraph("old ending")
commented = rdocx.Document.from_bytes(reviewed.to_bytes())
commented.add_comment(
rdocx.RunRange(
start=rdocx.RunPosition(body_index=0, run_index=0),
end=rdocx.RunPosition(body_index=0, run_index=1),
),
author="Bo",
text="edited side note",
)
commented.paragraphs[1].runs[0].text = "new ending"
redline = compared(reviewed, commented, ignore_comments=True)
assert redline.comments == ()
assert b"<w:comment" not in _document_xml(redline)
assert _tracked_texts(_document_xml(redline)) == (["old ending"], ["new ending"])

headed = rdocx.Document()
headed.add_paragraph("old body")
headed.set_header("old header")
rewritten = rdocx.Document.from_bytes(headed.to_bytes())
for story in ("body", "header"):
item = next(
item
for item in rewritten.story_items
if item.story.kind == story and item.text
)
rewritten.set_story_text(item, item.text.replace("old", "new"))
header_part = next(
story.part_name for story in headed.stories if story.kind == "header"
).lstrip("/")
tracked = compared(headed, rewritten)
assert _tracked_texts(_package_part(tracked, header_part)) == (
["old header"],
["new header"],
)
ignored = compared(headed, rewritten, ignored_stories=("header",))
assert _package_part(ignored, header_part) == _package_part(headed, header_part)
assert _tracked_texts(_document_xml(ignored)) == (["old body"], ["new body"])
unbodied = compared(headed, rewritten, ignored_stories=("body",))
assert _document_xml(unbodied) == _document_xml(headed)
assert _tracked_texts(_package_part(unbodied, header_part)) == (
["old header"],
["new header"],
)
for story in ("footer", "comment", "text_box", "footnote", "endnote"):
untouched = compared(headed, rewritten, ignored_stories=(story,))
assert _tracked_texts(_document_xml(untouched)) == (
["old body"],
["new body"],
), story
assert _tracked_texts(_package_part(untouched, header_part)) == (
["old header"],
["new header"],
), story


def test_compare_rejects_unknown_options_before_mutation():
import rdocx

original = rdocx.Document()
original.add_paragraph("before")
edited = rdocx.Document()
edited.add_paragraph("after")
live = original.paragraphs[0]
before = original.to_bytes()
for options, error, message in [
({"granularity": "words"}, rdocx.RdocxError, 'granularity "words"'),
({"ignored_stories": ["main"]}, rdocx.RdocxError, 'story "main"'),
({"ignored_stories": ["table_cell"]}, rdocx.RdocxError, 'story "table_cell"'),
(
{"ignored_stories": ["header", "header"]},
rdocx.RdocxError,
"duplicate ignored story",
),
({"ignored_stories": "header"}, TypeError, None),
]:
with pytest.raises(error, match=message):
original.compare(edited, "Ada", _COMPARE_TIMESTAMP, **options)
assert original.to_bytes() == before
assert live.text == "before"
with pytest.raises(TypeError):
original.compare(edited, "Ada", _COMPARE_TIMESTAMP, "word")
assert live.text == "before"


def test_word_default_toc_switch_rebuilds_and_reports_ordered_diagnostics():
import rdocx

Expand Down
13 changes: 13 additions & 0 deletions crates/rdocx-py/tests/typing_smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -119,6 +119,17 @@ def exercise_rdocx_types(path: Path) -> None:
diagnostics: tuple[ComparisonDiagnostic, ...] = document.compare(
opened, author="Ada", timestamp="2026-09-14T09:00:00Z"
)
optioned: tuple[ComparisonDiagnostic, ...] = document.compare(
opened,
"Ada",
"2026-09-14T09:00:00Z",
granularity="word",
ignore_formatting=True,
ignore_whitespace=True,
ignore_fields=True,
ignore_comments=True,
ignored_stories=("header", "text_box"),
)
fragments: tuple[LayoutFragment, ...] = document.layout()
maybe_layout_page: LayoutPage | None = document.layout_page(0)
report: TocRebuildReport = document.rebuild_toc()
Expand Down Expand Up @@ -212,3 +223,5 @@ def exercise_rdocx_types(path: Path) -> None:
RunCollection() # type: ignore[call-arg]
Table() # type: ignore[call-arg]
TableCollection() # type: ignore[call-arg]
Document().compare(Document(), "Ada", "2026-09-14T09:00:00Z", granularity="words") # type: ignore[arg-type]
Document().compare(Document(), "Ada", "2026-09-14T09:00:00Z", ignored_stories="header") # type: ignore[arg-type]
13 changes: 11 additions & 2 deletions docs/hld/10-bindings-spec.md
Original file line number Diff line number Diff line change
Expand Up @@ -1295,8 +1295,17 @@ and sibling fields from one physical run share that owner.
It emits same-story moves and supported run, paragraph, table, and section
property revisions. Diagnostic locations retain the actual story identity and
stable owner path. `rdocx-cli compare` exposes the source-compatible whole-run
comparison with explicit author, RFC 3339 timestamp, and output. Python and
WASM preserve comparison output when they save their owned document.
comparison with explicit author, RFC 3339 timestamp, and output. Python
`Document.compare` takes the `ComparisonOptions` fields as keyword-only
arguments. `granularity` is `"run"`, `"word"`, or `"character"` and defaults to
the native `"run"`. `ignore_formatting`, `ignore_whitespace`, `ignore_fields`,
and `ignore_comments` default to false. `ignored_stories` takes `Story.kind`
names, where `body` selects the main story and `table_cell` is not a comparison
category. An unknown granularity or story name raises `RdocxError` before the
document changes, and a duplicated story keeps the native rejection.
`ignore_comments` also leaves comment anchors to the original, while ignoring
the `comment` story excludes only the comments part. WASM preserves comparison
output when it saves its owned document.

Native Word rendering exposes `rdocx::RevisionView` and the concrete
`rdocx::RenderOptions`, whose default selects the accepted view. Additive
Expand Down
Loading