Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 18 additions & 9 deletions bibtexparser/middlewares/enclosing.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import re
from collections.abc import Iterator

from bibtexparser.library import Library
from bibtexparser.model import Entry
Expand All @@ -10,10 +11,17 @@
REMOVED_ENCLOSING_KEY = "removed_enclosing"

# Delimiters relevant when scanning a value, using the splitter's escaping
# convention: a delimiter is escaped iff it is directly preceded by a backslash.
_BRACES = re.compile(r"(?<!\\)[{}]")
_BRACES_AND_QUOTE = re.compile(r"(?<!\\)[{}\"]")
_UNENCLOSED_MARKS = re.compile(r"(?<!\\)[{}\",=\n]")
# convention: a backslash escapes the character right after it, so a delimiter is
# escaped iff it is preceded by an odd number of backslashes. As in the splitter,
# escape pairs are matched (which gets that parity right) and then dropped.
_BRACES = re.compile(r"\\[\\{}]|[{}]")
_BRACES_AND_QUOTE = re.compile(r"\\[\\{}\"]|[{}\"]")
_UNENCLOSED_MARKS = re.compile(r"\\[\\{}\",=]|[{}\",=\n]")


def _unescaped(pattern: re.Pattern, value: str) -> Iterator[re.Match]:
"""The delimiters of `pattern` in `value`, without the escape pairs."""
return (m for m in pattern.finditer(value) if m.group()[0] != "\\")


def _is_writable_unenclosed(value: str) -> bool:
Expand All @@ -26,7 +34,7 @@ def _is_writable_unenclosed(value: str) -> bool:
"""
depth = 0
in_quotes = False
for match in _UNENCLOSED_MARKS.finditer(value):
for match in _unescaped(_UNENCLOSED_MARKS, value):
char = match.group()
if char == "{":
depth += 1
Expand Down Expand Up @@ -98,11 +106,12 @@ def _is_enclosed_in_braces(value: str) -> bool:
"""
inner = value[1:-1]
if "{" not in inner and "}" not in inner:
# Fast path for the common case of a plainly braced value.
return not inner.endswith("\\")
# Fast path for the common case of a plainly braced value:
# enclosed unless an odd run of backslashes escapes the closing brace.
return (len(inner) - len(inner.rstrip("\\"))) % 2 == 0
depth = 0
last_index = len(value) - 1
for match in _BRACES.finditer(value):
for match in _unescaped(_BRACES, value):
if match.group() == "{":
depth += 1
continue
Expand All @@ -126,7 +135,7 @@ def _is_enclosed_in_quotes(value: str) -> bool:
# Fast path for the common case of a value without inner quotes.
return True
depth = 0
for match in _BRACES_AND_QUOTE.finditer(value):
for match in _unescaped(_BRACES_AND_QUOTE, value):
index = match.start()
if index == 0 or index == last_index:
continue
Expand Down
23 changes: 18 additions & 5 deletions bibtexparser/splitter.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
import logging
import re
from collections.abc import Iterator

from .exceptions import BlockAbortedException
from .exceptions import ParserStateException
Expand All @@ -20,10 +21,22 @@
# The opening delimiter is not part of the block start mark, so that a `{` after an
# `@` within a value (e.g. `LeQua @ {CLEF}`) is still counted as a mark of its own.
_BLOCK_START = r"@[\w]*( |\t)*(?=[{(])"
_MARK_PATTERN = re.compile(r"(?<!\\)[\{\}\",=\n]|" + _BLOCK_START)
# A backslash escapes the character right after it, so a backslash that is itself
# escaped cannot escape the next one. Matching the escape pairs (and then dropping
# them, see `_iter_marks`) gets that parity right, which a look-behind cannot.
# Newlines are not escapable: they are marks for line counting only.
_MARK_PATTERN = re.compile(r"\\[\\\{\}\",=]|[\{\}\",=\n]|" + _BLOCK_START)
# Inside a `(`-delimited block, the closing `)` is a mark too.
# It is not a mark elsewhere, so `)` in `{`-delimited blocks needs no special handling.
_PAREN_BLOCK_MARK_PATTERN = re.compile(r"(?<!\\)[\{\}\",=\n)]|" + _BLOCK_START)
_PAREN_BLOCK_MARK_PATTERN = re.compile(r"\\[\\\{\}\",=)]|[\{\}\",=\n)]|" + _BLOCK_START)


def _iter_marks(pattern: re.Pattern, string: str, pos: int = 0) -> Iterator[re.Match]:
"""The marks of `pattern` in `string`, starting at `pos`, without the escape pairs.

`pos` must not be within a run of backslashes, as the pairing starts there.
"""
return (m for m in pattern.finditer(string, pos) if m.group(0)[0] != "\\")


class Splitter:
Expand Down Expand Up @@ -137,7 +150,7 @@ def _open_block(self, m: re.Match) -> None:
if self.bibstr[m.end()] == "(":
self._closing_delimiter = ")"
# `(` is not a mark, hence the block-specific marks start right after it.
self._markiter = _PAREN_BLOCK_MARK_PATTERN.finditer(self.bibstr, m.end() + 1)
self._markiter = _iter_marks(_PAREN_BLOCK_MARK_PATTERN, self.bibstr, m.end() + 1)
else:
self._closing_delimiter = "}"
# The `{` is a mark (guaranteed to be the next one by the block start regex)
Expand All @@ -152,7 +165,7 @@ def _close_block(self) -> None:
resume_index = self._unaccepted_mark.end()
else:
resume_index = self._current_char_index + 1
self._markiter = _MARK_PATTERN.finditer(self.bibstr, resume_index)
self._markiter = _iter_marks(_MARK_PATTERN, self.bibstr, resume_index)
self._closing_delimiter = "}"

def _move_to_closing_delimiter(self, track_quotes: bool) -> int:
Expand Down Expand Up @@ -334,7 +347,7 @@ def split(self) -> Library:
Returns:
A new library containing the split blocks.
"""
self._markiter = _MARK_PATTERN.finditer(self.bibstr)
self._markiter = _iter_marks(_MARK_PATTERN, self.bibstr)

library = Library()

Expand Down
14 changes: 8 additions & 6 deletions tests/middleware_tests/test_enclosing.py
Original file line number Diff line number Diff line change
Expand Up @@ -537,18 +537,20 @@ def test_concatenation_roundtrip():
@pytest.mark.parametrize(
"value, expected_stripped, expected_enclosing",
[
pytest.param(r"{\\}a}", r"\\}a", "{", id="doubled_backslash_before_brace"),
pytest.param(r"{a\\{}", r"a\\{", "{", id="doubled_backslash_before_open_brace"),
pytest.param(r'"\\""', r'\\"', '"', id="doubled_backslash_before_quote"),
pytest.param(r"{\\}a}", r"{\\}a}", "no-enclosing", id="doubled_backslash_before_brace"),
pytest.param(
r"{a\\{}", r"{a\\{}", "no-enclosing", id="doubled_backslash_before_open_brace"
),
pytest.param(r'"\\""', r'"\\""', "no-enclosing", id="doubled_backslash_before_quote"),
pytest.param(r"{\\\}a}", r"\\\}a", "{", id="tripled_backslash_before_brace"),
],
)
def test_escaping_follows_the_splitter_convention(
value: str, expected_stripped: str, expected_enclosing: str
):
"""A delimiter is escaped iff directly preceded by a backslash.
"""A delimiter is escaped iff preceded by an odd number of backslashes.

This is the convention of the splitter's mark regex, which skips such a
delimiter regardless of how many backslashes precede it. The two must agree,
This is the convention of the splitter's mark regex. The two must agree,
or values the parser read as a single group are not stripped here.
"""
assert RemoveEnclosingMiddleware._strip_enclosing(value) == (
Expand Down
144 changes: 144 additions & 0 deletions tests/splitter_tests/test_splitter_backslash_parity.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,144 @@
"""Tests that the mark tokenizer treats backslash runs by parity.

A backslash escapes the character right after it, so a backslash which is itself
escaped cannot escape the next one. Only an odd-length run hides the delimiter
behind it; an even-length run leaves it in force.
"""

import pytest as pytest

import bibtexparser
from bibtexparser.library import Library
from bibtexparser.model import Entry
from bibtexparser.model import Field
from bibtexparser.splitter import Splitter

BS = "\\"

EVEN_RUNS = [0, 2, 4]
ODD_RUNS = [1, 3]

# Value shapes, each with a `%s` slot for the backslash run right before the
# delimiter that ends (or is contained in) the value.
CLOSED_VALUES = {"braced": "{v%s}", "quoted": '"v%s"'}
OPEN_VALUES = {"bare": "v%s", "nested-group": "{a%s{b} c}"}
ALL_VALUES = {**CLOSED_VALUES, **OPEN_VALUES}


def _parse_entry(value: str) -> Library:
return Splitter("@article{key,\n title = " + value + ",\n year = {2024}\n}").split()


@pytest.mark.parametrize("run", EVEN_RUNS)
@pytest.mark.parametrize("template", ALL_VALUES.values(), ids=ALL_VALUES.keys())
def test_even_backslash_run_leaves_delimiter_in_force(template: str, run: int):
value = template % (BS * run)
library = _parse_entry(value)

assert library.failed_blocks == []
fields = library.entries[0].fields_dict
assert set(fields) == {"title", "year"}
assert fields["title"].value == value
assert fields["year"].value == "{2024}"


@pytest.mark.parametrize("run", ODD_RUNS)
@pytest.mark.parametrize("template", CLOSED_VALUES.values(), ids=CLOSED_VALUES.keys())
def test_odd_backslash_run_escapes_the_closing_enclosing(template: str, run: int):
"""The value is never closed, so the block is reported as failed."""
library = _parse_entry(template % (BS * run))

assert len(library.failed_blocks) == 1
assert library.entries == []


@pytest.mark.parametrize("run", ODD_RUNS)
@pytest.mark.parametrize("template", OPEN_VALUES.values(), ids=OPEN_VALUES.keys())
def test_odd_backslash_run_escapes_the_field_separator(template: str, run: int):
"""The following comma is escaped, so the rest of the entry is part of the value."""
library = _parse_entry(template % (BS * run))

assert library.failed_blocks == []
assert set(library.entries[0].fields_dict) == {"title"}


@pytest.mark.parametrize("run", EVEN_RUNS)
def test_even_backslash_run_closes_string_block(run: int):
library = Splitter("@string{s = {v" + BS * run + "}}").split()

assert library.failed_blocks == []
assert library.strings[0].value == "{v" + BS * run + "}"


@pytest.mark.parametrize("run", EVEN_RUNS)
def test_even_backslash_run_closes_preamble_block(run: int):
library = Splitter("@preamble{{v" + BS * run + "}}").split()

assert library.failed_blocks == []
assert library.preambles[0].value == "{v" + BS * run + "}"


@pytest.mark.parametrize("run", EVEN_RUNS)
def test_even_backslash_run_closes_explicit_comment_block(run: int):
library = Splitter("@comment{c" + BS * run + "}\n@article{key, year = {2024}}").split()

assert library.failed_blocks == []
assert library.comments[0].comment == "c" + BS * run
assert library.entries[0].key == "key"


@pytest.mark.parametrize("run", ODD_RUNS + EVEN_RUNS)
def test_backslashes_at_end_of_line_do_not_shift_line_numbers(run: int):
"""A LaTeX line break (`\\\\`) at the end of a line is not a line continuation."""
bibtex = (
"@article{first,\n"
" abstract = {First line." + BS * run + "\n"
" Second line.},\n"
" year = {2024}\n"
"}\n"
"\n"
"@book{second,\n"
" year = {1999}\n"
"}"
)

library = Splitter(bibtex).split()

assert [block.start_line for block in library.blocks] == [0, 6]
assert library.entries[0].fields_dict["year"].start_line == 3


@pytest.mark.parametrize("run", EVEN_RUNS)
def test_written_backslash_run_is_read_back_unchanged(run: int):
"""The writer emits these values, so the splitter has to accept them again."""
value = "C:" + BS * run
library = Library()
library.add(Entry("article", "key", [Field("title", value), Field("year", "2024")]))

reparsed = bibtexparser.parse_string(bibtexparser.write_string(library))

assert reparsed.failed_blocks == []
assert reparsed.entries[0].fields_dict["title"].value == value
assert reparsed.entries[0].fields_dict["year"].value == "2024"


def test_escaped_delimiter_in_an_entry_key_is_not_a_mark():
library = Splitter("@article{ke" + BS + "{y, title = {v}}").split()

assert library.failed_blocks == []
assert library.entries[0].key == "ke" + BS + "{y"
assert library.entries[0].fields_dict["title"].value == "{v}"


def test_escaped_delimiter_in_a_field_key_is_not_a_mark():
library = Splitter("@article{key, ti" + BS + "=tle = {v}, year = {2024}}").split()

assert library.failed_blocks == []
assert set(library.entries[0].fields_dict) == {"ti" + BS + "=tle", "year"}


def test_escaped_delimiter_in_a_string_key_is_not_a_mark():
library = Splitter("@string{s" + BS + "{x = {v}}").split()

assert library.failed_blocks == []
assert library.strings[0].key == "s" + BS + "{x"
Loading