Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
49 changes: 46 additions & 3 deletions bibtexparser/model.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import abc
from collections.abc import Iterable
from collections.abc import Iterator
from typing import Any

Expand All @@ -13,6 +14,13 @@ def _validated_enclosing(enclosing: str | None) -> str | None:
return enclosing


def _validated_comments(comments: Iterable[str]) -> tuple[str, ...] | None:
if isinstance(comments, str):
raise TypeError("comments must be an iterable of strings, not a string")
# `None` if empty, as that is much cheaper to deep-copy than an empty tuple
return tuple(comments) or None


class Block(abc.ABC):
"""An abstract superclass of all top-level building blocks of a bibtex file.

Expand Down Expand Up @@ -246,11 +254,14 @@ def __init__(
value: Any,
start_line: int | None = None,
enclosing: str | None = None,
comments: Iterable[str] = (),
):
self._start_line = start_line
self._key = key
self._value = value
self._enclosing = _validated_enclosing(enclosing)
# Skipping the call for empty comments is notably faster when parsing large files
self._comments = _validated_comments(comments) if comments else None

@property
def key(self) -> str:
Expand Down Expand Up @@ -291,6 +302,18 @@ def enclosing(self) -> str | None:
def enclosing(self, enclosing: str | None):
self._enclosing = _validated_enclosing(enclosing)

@property
def comments(self) -> tuple[str, ...]:
"""The ``%``-comment lines right before this field, as supported by biber.

Each comment is the text after the ``%``, e.g. ``year = 2020`` for ``%year = 2020``.
When writing, they are written on the lines above the field."""
return self._comments or ()

@comments.setter
def comments(self, value: Iterable[str]):
self._comments = _validated_comments(value)

@property
def start_line(self) -> int:
"""The line number of the first line of this field in the originally parsed string."""
Expand All @@ -316,7 +339,8 @@ def __str__(self) -> str:
def __repr__(self) -> str:
return (
f"Field(key=`{self.key}`, value=`{self.value}`, "
f"start_line={self.start_line}, enclosing={self._enclosing!r})"
f"start_line={self.start_line}, enclosing={self._enclosing!r}, "
f"comments={self.comments!r})"
)


Expand All @@ -330,11 +354,15 @@ def __init__(
fields: list[Field],
start_line: int | None = None,
raw: str | None = None,
trailing_comments: Iterable[str] = (),
):
super().__init__(start_line, raw)
self._entry_type = entry_type
self._key = key
self._fields = fields
self._trailing_comments = (
_validated_comments(trailing_comments) if trailing_comments else None
)

@property
def entry_type(self) -> str:
Expand Down Expand Up @@ -363,6 +391,17 @@ def fields(self) -> list[Field]:
def fields(self, value: list[Field]):
self._fields = value

@property
def trailing_comments(self) -> tuple[str, ...]:
"""The ``%``-comment lines after the last field, as supported by biber.

Comments before a field are attached to that field instead (see ``Field.comments``)."""
return self._trailing_comments or ()

@trailing_comments.setter
def trailing_comments(self, value: Iterable[str]):
self._trailing_comments = _validated_comments(value)

@property
def fields_dict(self) -> dict[str, Field]:
"""A dict of fields, with field keys as keys.
Expand Down Expand Up @@ -426,6 +465,7 @@ def __setitem__(self, key: str, value: Any):

This serves for partial v1.x backwards compatibility,
as well as for a shorthand for `set_field`.
The comments of a replaced field are kept.

Mirroring ``__getitem__``, the keys ``ENTRYTYPE`` and ``ID``
set ``entry_type`` and ``key`` instead of a field.
Expand All @@ -435,7 +475,9 @@ def __setitem__(self, key: str, value: Any):
elif key == "ID":
self.key = value
else:
self.set_field(Field(key, value))
replaced = self.get(key)
comments = replaced.comments if replaced is not None else ()
self.set_field(Field(key, value, comments=comments))

def __delitem__(self, key: str) -> None:
"""Dict-mimicking index.
Expand Down Expand Up @@ -476,7 +518,8 @@ def __str__(self) -> str:
def __repr__(self) -> str:
return (
f"Entry(entry_type=`{self.entry_type}`, key=`{self.key}`, "
f"fields=`{self.fields.__repr__()}`, start_line={self.start_line})"
f"fields=`{self.fields.__repr__()}`, start_line={self.start_line}, "
f"trailing_comments={self.trailing_comments!r})"
)


Expand Down
60 changes: 49 additions & 11 deletions bibtexparser/splitter.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,9 @@
# Inside a `(`-delimited block, the closing `)` is a mark too.
# It is not a mark elsewhere, so `)` in `{`-delimited blocks needs no special handling.
_PAREN_BLOCK_MARK_PATTERN = re.compile(r"\\[\\\{\}\",=)]|[\{\}\",=\n)]|" + _BLOCK_START)
# Biber-style comment (`%` up to the end of the line) where a field key is expected.
# `%` is not a mark: comments are only looked for there, see `_skip_comments`.
_COMMENT_PATTERN = re.compile(r"\s*%([^\n]*)")


def _iter_marks(pattern: re.Pattern, string: str, pos: int = 0) -> Iterator[re.Match]:
Expand Down Expand Up @@ -277,25 +280,59 @@ def _is_escaped():
end_index=next_mark.start() - 1,
)

def _move_to_end_of_entry(self, first_key_start: int) -> tuple[list[Field], int, set[str]]:
"""Move to the end of the entry and return the fields and the end index."""
def _skip_comments(self, pos: int) -> tuple[tuple[str, ...], int]:
"""The comments starting at `pos`, and the index after them.

In BibTeX, `%` is not allowed in field keys, hence, where a field key is expected,
`%` starts a comment (as in biber). Within field values, `%` is a literal.
"""
comments = []
m = _COMMENT_PATTERN.match(self.bibstr, pos)
while m is not None:
comments.append(m.group(1).rstrip())
pos = m.end()
m = _COMMENT_PATTERN.match(self.bibstr, pos)

if comments:
# Consume the marks within the comments (e.g. a `,`), counting their lines.
mark = self._next_mark(accept_eof=False)
while mark.start() < pos:
mark = self._next_mark(accept_eof=False)
self._unaccepted_mark = mark
return tuple(comments), pos

def _move_to_end_of_entry(
self, first_key_start: int
) -> tuple[list[Field], int, set[str], tuple[str, ...]]:
"""Move to the end of the entry and return the fields, the end index,
the duplicate field keys and the comments after the last field."""
result = []
keys = set()
duplicate_keys = set()

key_start = first_key_start
while True:
equals_mark = self._next_mark(accept_eof=False)
key = self.bibstr[key_start : equals_mark.start()]
if "%" in key:
# Rare: re-read the key after the comments (which may contain marks).
self._unaccepted_mark = equals_mark
comments, key_start = self._skip_comments(key_start)
equals_mark = self._next_mark(accept_eof=False)
key = self.bibstr[key_start : equals_mark.start()]
else:
comments = ()
key = key.strip()

if equals_mark.group(0) == self._closing_delimiter:
dangling_key = self.bibstr[key_start : equals_mark.start()].strip()
if dangling_key:
if key:
raise BlockAbortedException(
abort_reason=f"Expected a `=` after entry key `{dangling_key}`, "
abort_reason=f"Expected a `=` after entry key `{key}`, "
f"but found the end of the entry (`{self._closing_delimiter}`).",
end_index=equals_mark.end(),
)
# End of entry
return result, equals_mark.end(), duplicate_keys
return result, equals_mark.end(), duplicate_keys, comments

if equals_mark.group(0) != "=":
self._unaccepted_mark = equals_mark
Expand All @@ -308,20 +345,18 @@ def _move_to_end_of_entry(self, first_key_start: int) -> tuple[list[Field], int,
# We follow the convention that the field start line
# is where the `=` between key and value is.
start_line = self._current_line
key_end = equals_mark.start()
value_start = equals_mark.end()
value_end = self._move_to_comma_or_closing_delimiter(
currently_quote_escaped=False, num_open_curls=0
)

key = self.bibstr[key_start:key_end].strip()
value = self.bibstr[value_start:value_end].strip()

if key in keys:
duplicate_keys.add(key)

keys.add(key)
result.append(Field(start_line=start_line, key=key, value=value))
result.append(Field(start_line=start_line, key=key, value=value, comments=comments))

# If next mark is a comma, continue
after_field_mark = self._next_mark(accept_eof=False)
Expand Down Expand Up @@ -445,7 +480,7 @@ def _handle_entry(self, m, m_val) -> Entry | ParsingFailedBlock:
# This is an entry without any comma after the key, and with no fields
# Used e.g. by RefTeX (see issue #384)
key = self.bibstr[m.end() + 1 : comma_mark.start()].strip()
fields, end_index, duplicate_keys = [], comma_mark.end(), []
fields, end_index, duplicate_keys, trailing_comments = [], comma_mark.end(), [], ()
elif comma_mark.group(0) != ",":
self._unaccepted_mark = comma_mark
raise BlockAbortedException(
Expand All @@ -454,14 +489,17 @@ def _handle_entry(self, m, m_val) -> Entry | ParsingFailedBlock:
)
else:
key = self.bibstr[m.end() + 1 : comma_mark.start()].strip()
fields, end_index, duplicate_keys = self._move_to_end_of_entry(comma_mark.end())
fields, end_index, duplicate_keys, trailing_comments = self._move_to_end_of_entry(
comma_mark.end()
)

entry = Entry(
start_line=start_line,
entry_type=entry_type,
key=key,
fields=fields,
raw=self.bibstr[m.start() : end_index],
trailing_comments=trailing_comments,
)

# If there were duplicate field keys, we return a DuplicateFieldKeyBlock wrapping
Expand Down
18 changes: 17 additions & 1 deletion bibtexparser/writer.py
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
import re
from copy import deepcopy
from typing import Optional

Expand All @@ -11,25 +12,40 @@
from .model import String

VAL_SEP = " = "
# Line breaks as recognized when reading files (universal newlines)
_LINE_BREAK = re.compile(r"\r\n?|\n")
PARSING_FAILED_COMMENT = "% WARNING Parsing failed for the following {n} lines."


def _treat_entry(block: Entry, bibtex_format) -> list[str]:
res = ["@", block.entry_type, "{", block.key, ",\n"]
field: Field
for i, field in enumerate(block.fields):
if field.comments:
res.extend(_comment_lines(field.comments, bibtex_format))
res.append(bibtex_format.indent)
res.append(field.key)
res.append(_val_indent_string(bibtex_format, field.key))
res.append(VAL_SEP)
res.append(field.value)
if bibtex_format.trailing_comma or i < len(block.fields) - 1:
# Comments after the last field are only parsed as such after a comma
if bibtex_format.trailing_comma or i < len(block.fields) - 1 or block.trailing_comments:
res.append(",")
res.append("\n")
res.extend(_comment_lines(block.trailing_comments, bibtex_format))
res.append("}\n")
return res


def _comment_lines(comments: tuple[str, ...], bibtex_format: "BibtexFormat") -> list[str]:
# A line break would end the comment, hence each line is commented separately
return [
f"{bibtex_format.indent}%{line}\n"
for comment in comments
for line in _LINE_BREAK.split(comment)
]


def _val_indent_string(bibtex_format: "BibtexFormat", key: str) -> str:
"""The spaces which have to be added after the ` = `."""
length = bibtex_format.value_column - len(key) - len(VAL_SEP)
Expand Down
23 changes: 23 additions & 0 deletions docs/source/biber.rst
Original file line number Diff line number Diff line change
Expand Up @@ -8,3 +8,26 @@ with ease, as they all share the same general syntax.

That said, we did not explicitly check against all of biber and biblatex features. Should you detect anything which is not supported,
please open an issue or send a pull request.

Comments within entries
=======================

Biber allows ``%``-comments within entries, e.g. to comment out a field:

.. code-block:: bibtex

@article{Cesar2013,
author = {Jean César},
% title = {An amazing title},
}

Where a field key is expected, bibtexparser reads ``%`` up to the end of the line as such a comment.
Comments before a field are available as ``field.comments``,
comments after the last field as ``entry.trailing_comments``.
When writing, they are written on the lines above their field, or above the closing brace of the entry.

A comment extends to the end of the line, as in biber,
hence a closing brace of the entry on the same line is commented out as well.
Within field values and entry keys, ``%`` is a literal character, as in BibTeX.
Note that a comment is only recognized after the comma ending the previous field:
in ``year = 2020 % comment``, without a comma before the ``%``, the comment becomes part of the value.
Loading
Loading