From 08e73eac50f241b89ff03fe122c7712e3287e157 Mon Sep 17 00:00:00 2001 From: Vincent Gao Date: Mon, 3 Aug 2026 15:19:26 +0200 Subject: [PATCH 1/2] Tokenize backslash escapes by parity in the splitter The mark iterator used a fixed-width `(? Iterator[re.Match]: + """The marks of `pattern` in `string`, starting at `pos`, without the escape pairs. + + `pos` must not be within a run of backslashes, as the pairing starts there. + """ + return (m for m in pattern.finditer(string, pos) if m.group(0)[0] != "\\") class Splitter: @@ -137,7 +150,7 @@ def _open_block(self, m: re.Match) -> None: if self.bibstr[m.end()] == "(": self._closing_delimiter = ")" # `(` is not a mark, hence the block-specific marks start right after it. - self._markiter = _PAREN_BLOCK_MARK_PATTERN.finditer(self.bibstr, m.end() + 1) + self._markiter = _iter_marks(_PAREN_BLOCK_MARK_PATTERN, self.bibstr, m.end() + 1) else: self._closing_delimiter = "}" # The `{` is a mark (guaranteed to be the next one by the block start regex) @@ -152,7 +165,7 @@ def _close_block(self) -> None: resume_index = self._unaccepted_mark.end() else: resume_index = self._current_char_index + 1 - self._markiter = _MARK_PATTERN.finditer(self.bibstr, resume_index) + self._markiter = _iter_marks(_MARK_PATTERN, self.bibstr, resume_index) self._closing_delimiter = "}" def _move_to_closing_delimiter(self, track_quotes: bool) -> int: @@ -334,7 +347,7 @@ def split(self) -> Library: Returns: A new library containing the split blocks. """ - self._markiter = _MARK_PATTERN.finditer(self.bibstr) + self._markiter = _iter_marks(_MARK_PATTERN, self.bibstr) library = Library() diff --git a/tests/splitter_tests/test_splitter_backslash_parity.py b/tests/splitter_tests/test_splitter_backslash_parity.py new file mode 100644 index 0000000..bc97e31 --- /dev/null +++ b/tests/splitter_tests/test_splitter_backslash_parity.py @@ -0,0 +1,144 @@ +"""Tests that the mark tokenizer treats backslash runs by parity. + +A backslash escapes the character right after it, so a backslash which is itself +escaped cannot escape the next one. Only an odd-length run hides the delimiter +behind it; an even-length run leaves it in force. +""" + +import pytest as pytest + +import bibtexparser +from bibtexparser.library import Library +from bibtexparser.model import Entry +from bibtexparser.model import Field +from bibtexparser.splitter import Splitter + +BS = "\\" + +EVEN_RUNS = [0, 2, 4] +ODD_RUNS = [1, 3] + +# Value shapes, each with a `%s` slot for the backslash run right before the +# delimiter that ends (or is contained in) the value. +CLOSED_VALUES = {"braced": "{v%s}", "quoted": '"v%s"'} +OPEN_VALUES = {"bare": "v%s", "nested-group": "{a%s{b} c}"} +ALL_VALUES = {**CLOSED_VALUES, **OPEN_VALUES} + + +def _parse_entry(value: str) -> Library: + return Splitter("@article{key,\n title = " + value + ",\n year = {2024}\n}").split() + + +@pytest.mark.parametrize("run", EVEN_RUNS) +@pytest.mark.parametrize("template", ALL_VALUES.values(), ids=ALL_VALUES.keys()) +def test_even_backslash_run_leaves_delimiter_in_force(template: str, run: int): + value = template % (BS * run) + library = _parse_entry(value) + + assert library.failed_blocks == [] + fields = library.entries[0].fields_dict + assert set(fields) == {"title", "year"} + assert fields["title"].value == value + assert fields["year"].value == "{2024}" + + +@pytest.mark.parametrize("run", ODD_RUNS) +@pytest.mark.parametrize("template", CLOSED_VALUES.values(), ids=CLOSED_VALUES.keys()) +def test_odd_backslash_run_escapes_the_closing_enclosing(template: str, run: int): + """The value is never closed, so the block is reported as failed.""" + library = _parse_entry(template % (BS * run)) + + assert len(library.failed_blocks) == 1 + assert library.entries == [] + + +@pytest.mark.parametrize("run", ODD_RUNS) +@pytest.mark.parametrize("template", OPEN_VALUES.values(), ids=OPEN_VALUES.keys()) +def test_odd_backslash_run_escapes_the_field_separator(template: str, run: int): + """The following comma is escaped, so the rest of the entry is part of the value.""" + library = _parse_entry(template % (BS * run)) + + assert library.failed_blocks == [] + assert set(library.entries[0].fields_dict) == {"title"} + + +@pytest.mark.parametrize("run", EVEN_RUNS) +def test_even_backslash_run_closes_string_block(run: int): + library = Splitter("@string{s = {v" + BS * run + "}}").split() + + assert library.failed_blocks == [] + assert library.strings[0].value == "{v" + BS * run + "}" + + +@pytest.mark.parametrize("run", EVEN_RUNS) +def test_even_backslash_run_closes_preamble_block(run: int): + library = Splitter("@preamble{{v" + BS * run + "}}").split() + + assert library.failed_blocks == [] + assert library.preambles[0].value == "{v" + BS * run + "}" + + +@pytest.mark.parametrize("run", EVEN_RUNS) +def test_even_backslash_run_closes_explicit_comment_block(run: int): + library = Splitter("@comment{c" + BS * run + "}\n@article{key, year = {2024}}").split() + + assert library.failed_blocks == [] + assert library.comments[0].comment == "c" + BS * run + assert library.entries[0].key == "key" + + +@pytest.mark.parametrize("run", ODD_RUNS + EVEN_RUNS) +def test_backslashes_at_end_of_line_do_not_shift_line_numbers(run: int): + """A LaTeX line break (`\\\\`) at the end of a line is not a line continuation.""" + bibtex = ( + "@article{first,\n" + " abstract = {First line." + BS * run + "\n" + " Second line.},\n" + " year = {2024}\n" + "}\n" + "\n" + "@book{second,\n" + " year = {1999}\n" + "}" + ) + + library = Splitter(bibtex).split() + + assert [block.start_line for block in library.blocks] == [0, 6] + assert library.entries[0].fields_dict["year"].start_line == 3 + + +@pytest.mark.parametrize("run", EVEN_RUNS) +def test_written_backslash_run_is_read_back_unchanged(run: int): + """The writer emits these values, so the splitter has to accept them again.""" + value = "C:" + BS * run + library = Library() + library.add(Entry("article", "key", [Field("title", value), Field("year", "2024")])) + + reparsed = bibtexparser.parse_string(bibtexparser.write_string(library)) + + assert reparsed.failed_blocks == [] + assert reparsed.entries[0].fields_dict["title"].value == value + assert reparsed.entries[0].fields_dict["year"].value == "2024" + + +def test_escaped_delimiter_in_an_entry_key_is_not_a_mark(): + library = Splitter("@article{ke" + BS + "{y, title = {v}}").split() + + assert library.failed_blocks == [] + assert library.entries[0].key == "ke" + BS + "{y" + assert library.entries[0].fields_dict["title"].value == "{v}" + + +def test_escaped_delimiter_in_a_field_key_is_not_a_mark(): + library = Splitter("@article{key, ti" + BS + "=tle = {v}, year = {2024}}").split() + + assert library.failed_blocks == [] + assert set(library.entries[0].fields_dict) == {"ti" + BS + "=tle", "year"} + + +def test_escaped_delimiter_in_a_string_key_is_not_a_mark(): + library = Splitter("@string{s" + BS + "{x = {v}}").split() + + assert library.failed_blocks == [] + assert library.strings[0].key == "s" + BS + "{x" From 81bafd629e9c1ee37edf0d684f95b0ce4f037f5a Mon Sep 17 00:00:00 2001 From: Michael Weiss Date: Fri, 2 Oct 2026 18:05:11 +0200 Subject: [PATCH 2/2] =?UTF-8?q?=F0=9F=90=9B=20Use=20backslash=20parity=20i?= =?UTF-8?q?n=20the=20enclosing=20middleware=20too?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RemoveEnclosingMiddleware mirrors the splitter's escape convention, which is now parity-based. With the old single-backslash rule, a value like `{C:\\}` that the splitter reads fine kept its braces here, and gained another pair on every parse/write cycle. --- bibtexparser/middlewares/enclosing.py | 27 ++++++++++++++++-------- tests/middleware_tests/test_enclosing.py | 14 ++++++------ 2 files changed, 26 insertions(+), 15 deletions(-) diff --git a/bibtexparser/middlewares/enclosing.py b/bibtexparser/middlewares/enclosing.py index 91a85e4..da017ab 100644 --- a/bibtexparser/middlewares/enclosing.py +++ b/bibtexparser/middlewares/enclosing.py @@ -1,4 +1,5 @@ import re +from collections.abc import Iterator from bibtexparser.library import Library from bibtexparser.model import Entry @@ -10,10 +11,17 @@ REMOVED_ENCLOSING_KEY = "removed_enclosing" # Delimiters relevant when scanning a value, using the splitter's escaping -# convention: a delimiter is escaped iff it is directly preceded by a backslash. -_BRACES = re.compile(r"(? Iterator[re.Match]: + """The delimiters of `pattern` in `value`, without the escape pairs.""" + return (m for m in pattern.finditer(value) if m.group()[0] != "\\") def _is_writable_unenclosed(value: str) -> bool: @@ -26,7 +34,7 @@ def _is_writable_unenclosed(value: str) -> bool: """ depth = 0 in_quotes = False - for match in _UNENCLOSED_MARKS.finditer(value): + for match in _unescaped(_UNENCLOSED_MARKS, value): char = match.group() if char == "{": depth += 1 @@ -98,11 +106,12 @@ def _is_enclosed_in_braces(value: str) -> bool: """ inner = value[1:-1] if "{" not in inner and "}" not in inner: - # Fast path for the common case of a plainly braced value. - return not inner.endswith("\\") + # Fast path for the common case of a plainly braced value: + # enclosed unless an odd run of backslashes escapes the closing brace. + return (len(inner) - len(inner.rstrip("\\"))) % 2 == 0 depth = 0 last_index = len(value) - 1 - for match in _BRACES.finditer(value): + for match in _unescaped(_BRACES, value): if match.group() == "{": depth += 1 continue @@ -126,7 +135,7 @@ def _is_enclosed_in_quotes(value: str) -> bool: # Fast path for the common case of a value without inner quotes. return True depth = 0 - for match in _BRACES_AND_QUOTE.finditer(value): + for match in _unescaped(_BRACES_AND_QUOTE, value): index = match.start() if index == 0 or index == last_index: continue diff --git a/tests/middleware_tests/test_enclosing.py b/tests/middleware_tests/test_enclosing.py index 5eb4f56..24228d2 100644 --- a/tests/middleware_tests/test_enclosing.py +++ b/tests/middleware_tests/test_enclosing.py @@ -537,18 +537,20 @@ def test_concatenation_roundtrip(): @pytest.mark.parametrize( "value, expected_stripped, expected_enclosing", [ - pytest.param(r"{\\}a}", r"\\}a", "{", id="doubled_backslash_before_brace"), - pytest.param(r"{a\\{}", r"a\\{", "{", id="doubled_backslash_before_open_brace"), - pytest.param(r'"\\""', r'\\"', '"', id="doubled_backslash_before_quote"), + pytest.param(r"{\\}a}", r"{\\}a}", "no-enclosing", id="doubled_backslash_before_brace"), + pytest.param( + r"{a\\{}", r"{a\\{}", "no-enclosing", id="doubled_backslash_before_open_brace" + ), + pytest.param(r'"\\""', r'"\\""', "no-enclosing", id="doubled_backslash_before_quote"), + pytest.param(r"{\\\}a}", r"\\\}a", "{", id="tripled_backslash_before_brace"), ], ) def test_escaping_follows_the_splitter_convention( value: str, expected_stripped: str, expected_enclosing: str ): - """A delimiter is escaped iff directly preceded by a backslash. + """A delimiter is escaped iff preceded by an odd number of backslashes. - This is the convention of the splitter's mark regex, which skips such a - delimiter regardless of how many backslashes precede it. The two must agree, + This is the convention of the splitter's mark regex. The two must agree, or values the parser read as a single group are not stripped here. """ assert RemoveEnclosingMiddleware._strip_enclosing(value) == (