From 18bd1e221983b4bddad4669329263bf84e4579fc Mon Sep 17 00:00:00 2001 From: Michael Weiss Date: Sat, 3 Oct 2026 20:24:46 +0200 Subject: [PATCH] =?UTF-8?q?=E2=9C=A8=20Support=20biber-style=20%-comments?= =?UTF-8?q?=20within=20entries?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Where a field key is expected, `%` up to the end of the line is now parsed as a comment (as in biber; BibTeX does not allow `%` in field keys). Previously, such lines became bogus field keys, were merged into the next field's key, or made the entry fail to parse. Comments are kept: `Field.comments` holds those before a field, `Entry.trailing_comments` those after the last field, and the writer writes them back. Within values and entry keys, `%` stays literal. Closes #372 --- bibtexparser/model.py | 49 ++++++++++- bibtexparser/splitter.py | 60 ++++++++++--- bibtexparser/writer.py | 18 +++- docs/source/biber.rst | 23 +++++ .../test_splitter_percent_comments.py | 87 +++++++++++++++++++ tests/test_entrypoint.py | 8 ++ tests/test_model.py | 13 +++ tests/test_writer.py | 15 ++++ 8 files changed, 258 insertions(+), 15 deletions(-) create mode 100644 tests/splitter_tests/test_splitter_percent_comments.py diff --git a/bibtexparser/model.py b/bibtexparser/model.py index e801bc93..a06455c5 100644 --- a/bibtexparser/model.py +++ b/bibtexparser/model.py @@ -1,4 +1,5 @@ import abc +from collections.abc import Iterable from collections.abc import Iterator from typing import Any @@ -13,6 +14,13 @@ def _validated_enclosing(enclosing: str | None) -> str | None: return enclosing +def _validated_comments(comments: Iterable[str]) -> tuple[str, ...] | None: + if isinstance(comments, str): + raise TypeError("comments must be an iterable of strings, not a string") + # `None` if empty, as that is much cheaper to deep-copy than an empty tuple + return tuple(comments) or None + + class Block(abc.ABC): """An abstract superclass of all top-level building blocks of a bibtex file. @@ -246,11 +254,14 @@ def __init__( value: Any, start_line: int | None = None, enclosing: str | None = None, + comments: Iterable[str] = (), ): self._start_line = start_line self._key = key self._value = value self._enclosing = _validated_enclosing(enclosing) + # Skipping the call for empty comments is notably faster when parsing large files + self._comments = _validated_comments(comments) if comments else None @property def key(self) -> str: @@ -291,6 +302,18 @@ def enclosing(self) -> str | None: def enclosing(self, enclosing: str | None): self._enclosing = _validated_enclosing(enclosing) + @property + def comments(self) -> tuple[str, ...]: + """The ``%``-comment lines right before this field, as supported by biber. + + Each comment is the text after the ``%``, e.g. ``year = 2020`` for ``%year = 2020``. + When writing, they are written on the lines above the field.""" + return self._comments or () + + @comments.setter + def comments(self, value: Iterable[str]): + self._comments = _validated_comments(value) + @property def start_line(self) -> int: """The line number of the first line of this field in the originally parsed string.""" @@ -316,7 +339,8 @@ def __str__(self) -> str: def __repr__(self) -> str: return ( f"Field(key=`{self.key}`, value=`{self.value}`, " - f"start_line={self.start_line}, enclosing={self._enclosing!r})" + f"start_line={self.start_line}, enclosing={self._enclosing!r}, " + f"comments={self.comments!r})" ) @@ -330,11 +354,15 @@ def __init__( fields: list[Field], start_line: int | None = None, raw: str | None = None, + trailing_comments: Iterable[str] = (), ): super().__init__(start_line, raw) self._entry_type = entry_type self._key = key self._fields = fields + self._trailing_comments = ( + _validated_comments(trailing_comments) if trailing_comments else None + ) @property def entry_type(self) -> str: @@ -363,6 +391,17 @@ def fields(self) -> list[Field]: def fields(self, value: list[Field]): self._fields = value + @property + def trailing_comments(self) -> tuple[str, ...]: + """The ``%``-comment lines after the last field, as supported by biber. + + Comments before a field are attached to that field instead (see ``Field.comments``).""" + return self._trailing_comments or () + + @trailing_comments.setter + def trailing_comments(self, value: Iterable[str]): + self._trailing_comments = _validated_comments(value) + @property def fields_dict(self) -> dict[str, Field]: """A dict of fields, with field keys as keys. @@ -426,6 +465,7 @@ def __setitem__(self, key: str, value: Any): This serves for partial v1.x backwards compatibility, as well as for a shorthand for `set_field`. + The comments of a replaced field are kept. Mirroring ``__getitem__``, the keys ``ENTRYTYPE`` and ``ID`` set ``entry_type`` and ``key`` instead of a field. @@ -435,7 +475,9 @@ def __setitem__(self, key: str, value: Any): elif key == "ID": self.key = value else: - self.set_field(Field(key, value)) + replaced = self.get(key) + comments = replaced.comments if replaced is not None else () + self.set_field(Field(key, value, comments=comments)) def __delitem__(self, key: str) -> None: """Dict-mimicking index. @@ -476,7 +518,8 @@ def __str__(self) -> str: def __repr__(self) -> str: return ( f"Entry(entry_type=`{self.entry_type}`, key=`{self.key}`, " - f"fields=`{self.fields.__repr__()}`, start_line={self.start_line})" + f"fields=`{self.fields.__repr__()}`, start_line={self.start_line}, " + f"trailing_comments={self.trailing_comments!r})" ) diff --git a/bibtexparser/splitter.py b/bibtexparser/splitter.py index 217c72f0..0cb85bdd 100644 --- a/bibtexparser/splitter.py +++ b/bibtexparser/splitter.py @@ -29,6 +29,9 @@ # Inside a `(`-delimited block, the closing `)` is a mark too. # It is not a mark elsewhere, so `)` in `{`-delimited blocks needs no special handling. _PAREN_BLOCK_MARK_PATTERN = re.compile(r"\\[\\\{\}\",=)]|[\{\}\",=\n)]|" + _BLOCK_START) +# Biber-style comment (`%` up to the end of the line) where a field key is expected. +# `%` is not a mark: comments are only looked for there, see `_skip_comments`. +_COMMENT_PATTERN = re.compile(r"\s*%([^\n]*)") def _iter_marks(pattern: re.Pattern, string: str, pos: int = 0) -> Iterator[re.Match]: @@ -277,8 +280,32 @@ def _is_escaped(): end_index=next_mark.start() - 1, ) - def _move_to_end_of_entry(self, first_key_start: int) -> tuple[list[Field], int, set[str]]: - """Move to the end of the entry and return the fields and the end index.""" + def _skip_comments(self, pos: int) -> tuple[tuple[str, ...], int]: + """The comments starting at `pos`, and the index after them. + + In BibTeX, `%` is not allowed in field keys, hence, where a field key is expected, + `%` starts a comment (as in biber). Within field values, `%` is a literal. + """ + comments = [] + m = _COMMENT_PATTERN.match(self.bibstr, pos) + while m is not None: + comments.append(m.group(1).rstrip()) + pos = m.end() + m = _COMMENT_PATTERN.match(self.bibstr, pos) + + if comments: + # Consume the marks within the comments (e.g. a `,`), counting their lines. + mark = self._next_mark(accept_eof=False) + while mark.start() < pos: + mark = self._next_mark(accept_eof=False) + self._unaccepted_mark = mark + return tuple(comments), pos + + def _move_to_end_of_entry( + self, first_key_start: int + ) -> tuple[list[Field], int, set[str], tuple[str, ...]]: + """Move to the end of the entry and return the fields, the end index, + the duplicate field keys and the comments after the last field.""" result = [] keys = set() duplicate_keys = set() @@ -286,16 +313,26 @@ def _move_to_end_of_entry(self, first_key_start: int) -> tuple[list[Field], int, key_start = first_key_start while True: equals_mark = self._next_mark(accept_eof=False) + key = self.bibstr[key_start : equals_mark.start()] + if "%" in key: + # Rare: re-read the key after the comments (which may contain marks). + self._unaccepted_mark = equals_mark + comments, key_start = self._skip_comments(key_start) + equals_mark = self._next_mark(accept_eof=False) + key = self.bibstr[key_start : equals_mark.start()] + else: + comments = () + key = key.strip() + if equals_mark.group(0) == self._closing_delimiter: - dangling_key = self.bibstr[key_start : equals_mark.start()].strip() - if dangling_key: + if key: raise BlockAbortedException( - abort_reason=f"Expected a `=` after entry key `{dangling_key}`, " + abort_reason=f"Expected a `=` after entry key `{key}`, " f"but found the end of the entry (`{self._closing_delimiter}`).", end_index=equals_mark.end(), ) # End of entry - return result, equals_mark.end(), duplicate_keys + return result, equals_mark.end(), duplicate_keys, comments if equals_mark.group(0) != "=": self._unaccepted_mark = equals_mark @@ -308,20 +345,18 @@ def _move_to_end_of_entry(self, first_key_start: int) -> tuple[list[Field], int, # We follow the convention that the field start line # is where the `=` between key and value is. start_line = self._current_line - key_end = equals_mark.start() value_start = equals_mark.end() value_end = self._move_to_comma_or_closing_delimiter( currently_quote_escaped=False, num_open_curls=0 ) - key = self.bibstr[key_start:key_end].strip() value = self.bibstr[value_start:value_end].strip() if key in keys: duplicate_keys.add(key) keys.add(key) - result.append(Field(start_line=start_line, key=key, value=value)) + result.append(Field(start_line=start_line, key=key, value=value, comments=comments)) # If next mark is a comma, continue after_field_mark = self._next_mark(accept_eof=False) @@ -445,7 +480,7 @@ def _handle_entry(self, m, m_val) -> Entry | ParsingFailedBlock: # This is an entry without any comma after the key, and with no fields # Used e.g. by RefTeX (see issue #384) key = self.bibstr[m.end() + 1 : comma_mark.start()].strip() - fields, end_index, duplicate_keys = [], comma_mark.end(), [] + fields, end_index, duplicate_keys, trailing_comments = [], comma_mark.end(), [], () elif comma_mark.group(0) != ",": self._unaccepted_mark = comma_mark raise BlockAbortedException( @@ -454,7 +489,9 @@ def _handle_entry(self, m, m_val) -> Entry | ParsingFailedBlock: ) else: key = self.bibstr[m.end() + 1 : comma_mark.start()].strip() - fields, end_index, duplicate_keys = self._move_to_end_of_entry(comma_mark.end()) + fields, end_index, duplicate_keys, trailing_comments = self._move_to_end_of_entry( + comma_mark.end() + ) entry = Entry( start_line=start_line, @@ -462,6 +499,7 @@ def _handle_entry(self, m, m_val) -> Entry | ParsingFailedBlock: key=key, fields=fields, raw=self.bibstr[m.start() : end_index], + trailing_comments=trailing_comments, ) # If there were duplicate field keys, we return a DuplicateFieldKeyBlock wrapping diff --git a/bibtexparser/writer.py b/bibtexparser/writer.py index ace8c699..327bc0bf 100644 --- a/bibtexparser/writer.py +++ b/bibtexparser/writer.py @@ -1,3 +1,4 @@ +import re from copy import deepcopy from typing import Optional @@ -11,6 +12,8 @@ from .model import String VAL_SEP = " = " +# Line breaks as recognized when reading files (universal newlines) +_LINE_BREAK = re.compile(r"\r\n?|\n") PARSING_FAILED_COMMENT = "% WARNING Parsing failed for the following {n} lines." @@ -18,18 +21,31 @@ def _treat_entry(block: Entry, bibtex_format) -> list[str]: res = ["@", block.entry_type, "{", block.key, ",\n"] field: Field for i, field in enumerate(block.fields): + if field.comments: + res.extend(_comment_lines(field.comments, bibtex_format)) res.append(bibtex_format.indent) res.append(field.key) res.append(_val_indent_string(bibtex_format, field.key)) res.append(VAL_SEP) res.append(field.value) - if bibtex_format.trailing_comma or i < len(block.fields) - 1: + # Comments after the last field are only parsed as such after a comma + if bibtex_format.trailing_comma or i < len(block.fields) - 1 or block.trailing_comments: res.append(",") res.append("\n") + res.extend(_comment_lines(block.trailing_comments, bibtex_format)) res.append("}\n") return res +def _comment_lines(comments: tuple[str, ...], bibtex_format: "BibtexFormat") -> list[str]: + # A line break would end the comment, hence each line is commented separately + return [ + f"{bibtex_format.indent}%{line}\n" + for comment in comments + for line in _LINE_BREAK.split(comment) + ] + + def _val_indent_string(bibtex_format: "BibtexFormat", key: str) -> str: """The spaces which have to be added after the ` = `.""" length = bibtex_format.value_column - len(key) - len(VAL_SEP) diff --git a/docs/source/biber.rst b/docs/source/biber.rst index fb58c3fc..60630666 100644 --- a/docs/source/biber.rst +++ b/docs/source/biber.rst @@ -8,3 +8,26 @@ with ease, as they all share the same general syntax. That said, we did not explicitly check against all of biber and biblatex features. Should you detect anything which is not supported, please open an issue or send a pull request. + +Comments within entries +======================= + +Biber allows ``%``-comments within entries, e.g. to comment out a field: + +.. code-block:: bibtex + + @article{Cesar2013, + author = {Jean César}, + % title = {An amazing title}, + } + +Where a field key is expected, bibtexparser reads ``%`` up to the end of the line as such a comment. +Comments before a field are available as ``field.comments``, +comments after the last field as ``entry.trailing_comments``. +When writing, they are written on the lines above their field, or above the closing brace of the entry. + +A comment extends to the end of the line, as in biber, +hence a closing brace of the entry on the same line is commented out as well. +Within field values and entry keys, ``%`` is a literal character, as in BibTeX. +Note that a comment is only recognized after the comma ending the previous field: +in ``year = 2020 % comment``, without a comma before the ``%``, the comment becomes part of the value. diff --git a/tests/splitter_tests/test_splitter_percent_comments.py b/tests/splitter_tests/test_splitter_percent_comments.py new file mode 100644 index 00000000..3da43dda --- /dev/null +++ b/tests/splitter_tests/test_splitter_percent_comments.py @@ -0,0 +1,87 @@ +"""Tests the parsing of biber-style `%`-comments within entries (issue #372).""" + +import pytest + +from bibtexparser.splitter import Splitter + + +@pytest.mark.parametrize( + "bibtex_str, expected_fields, expected_trailing_comments", + [ + pytest.param( + "@article{key,\n %somecomment\n author = {A},\n title = {T},\n}", + [("author", "{A}", ["somecomment"]), ("title", "{T}", [])], + [], + id="before_field", + ), + pytest.param( + "@article{key,\n author = {A},\n % title = {T},\n}", + [("author", "{A}", [])], + [" title = {T},"], + id="after_last_field", + ), + pytest.param( + '@article{key,\n % a, b = {c "d @misc{e,\n\n % second\n year = 2020\n}', + [("year", "2020", [' a, b = {c "d @misc{e,', " second"])], + [], + id="syntax_in_comments", + ), + pytest.param( + "@article{key,\n author = {A}, % same line\n year = 2020\n}", + [("author", "{A}", []), ("year", "2020", [" same line"])], + [], + id="after_comma_on_same_line", + ), + pytest.param( + "@article{key,\r\n %c1\r\n year = 2020,\r\n %c2\r\n}", + [("year", "2020", ["c1"])], + ["c2"], + id="crlf", + ), + pytest.param( + "@article(key,\n % (a)\n year = 2020,\n % b)\n)", + [("year", "2020", [" (a)"])], + [" b)"], + id="parenthesis_block", + ), + pytest.param( + "@article{key,\n % no fields\n}", + [], + [" no fields"], + id="no_fields", + ), + ], +) +def test_comments_within_entry(bibtex_str, expected_fields, expected_trailing_comments): + library = Splitter(bibtex_str).split() + + assert len(library.failed_blocks) == 0 + entry = library.entries[0] + assert [(f.key, f.value, list(f.comments)) for f in entry.fields] == expected_fields + assert list(entry.trailing_comments) == expected_trailing_comments + + +@pytest.mark.parametrize( + "bibtex_str, expected_key, expected_value", + [ + pytest.param("@misc{key, note = {a\n % b}}", "key", "{a\n % b}", id="curly"), + pytest.param('@misc{key, note = "50% off"}', "key", '"50% off"', id="quotes"), + pytest.param("@misc{50%off, note = {a}}", "50%off", "{a}", id="entry_key"), + ], +) +def test_percent_outside_of_comment_position_is_literal(bibtex_str, expected_key, expected_value): + library = Splitter(bibtex_str).split() + + assert len(library.failed_blocks) == 0 + entry = library.entries[0] + assert entry.key == expected_key + assert [(f.key, f.value, f.comments) for f in entry.fields] == [("note", expected_value, ())] + + +def test_comments_keep_line_numbers(): + # The `,` is a mark within the comment, which must be consumed while counting lines + bibtex_str = "@article{key,\n % a, b\n % c\n year = 2020,\n}\n\n@misc{other, title = {T}}" + library = Splitter(bibtex_str).split() + + assert library.entries[0].fields[0].start_line == 3 + assert library.entries[1].start_line == 6 diff --git a/tests/test_entrypoint.py b/tests/test_entrypoint.py index 5edce996..b8dacb4c 100644 --- a/tests/test_entrypoint.py +++ b/tests/test_entrypoint.py @@ -106,6 +106,14 @@ def test_write_file_roundtrip_gbk(): os.unlink(temp_path) +def test_entry_comments_roundtrip(): + bibtex_str = "@article{key,\n % a\n author = {A},\n %title = {T},\n}" + library = parse_string(write_string(parse_string(bibtex_str))) + entry = library.entries[0] + assert [(f.key, f.value, f.comments) for f in entry.fields] == [("author", "A", (" a",))] + assert entry.trailing_comments == ("title = {T},",) + + # Deprecation warning tests for write_file and write_string def test_write_file_deprecated_parse_stack_parameter(): """Test that using deprecated 'parse_stack' parameter issues a warning.""" diff --git a/tests/test_model.py b/tests/test_model.py index fad6500c..ae17caf0 100644 --- a/tests/test_model.py +++ b/tests/test_model.py @@ -120,6 +120,19 @@ def test_entry_setitem_field(): assert [f.key for f in entry.fields] == ["field", "other"] +def test_entry_setitem_keeps_field_comments(): + entry = Entry("article", "key", [Field("field", "value", 1, comments=["note"])], 1, "raw") + entry["field"] = "new_value" + assert entry.fields[0].comments == ("note",) + + +def test_comments_must_not_be_a_single_string(): + with pytest.raises(TypeError): + Field("field", "value", comments="note") + with pytest.raises(TypeError): + Entry("article", "key", []).trailing_comments = "note" + + def test_entry_setitem_entrytype_and_id(): entry = Entry("article", "key", [Field("field", "value", 1)], 1, "raw") entry["ENTRYTYPE"] = "book" diff --git a/tests/test_writer.py b/tests/test_writer.py index ec4aa994..e5cafeb9 100644 --- a/tests/test_writer.py +++ b/tests/test_writer.py @@ -81,6 +81,21 @@ def test_write_entry_with_trailing_comma(trailing_comma): ) +def test_write_entry_with_comments(): + entry_block = Entry( + entry_type="article", + key="myKey", + fields=[Field(key="title", value='"myTitle"', comments=["a", " b\nc\rd"])], + trailing_comments=["year = 2020"], + ) + string = writer.write(Library(blocks=[entry_block])) + # The comma after the last field is needed for the trailing comments to be parsed as such + assert ( + string + == '@article{myKey,\n\t%a\n\t% b\n\t%c\n\t%d\n\ttitle = "myTitle",\n\t%year = 2020\n}\n' + ) + + @pytest.mark.parametrize("value_column", [None, 10, "auto"]) def test_entry_value_column(value_column): entry_block = _dummy_entry()