diff --git a/babel/messages/extract.py b/babel/messages/extract.py index 6fad84304..63b4ce886 100644 --- a/babel/messages/extract.py +++ b/babel/messages/extract.py @@ -710,7 +710,19 @@ def _parse_python_string(value: str, encoding: str, future_flags: int) -> str | if isinstance(code, ast.Expression): body = code.body if isinstance(body, ast.Constant): - return body.value + if isinstance(body.value, str): + return body.value + if isinstance(body.value, bytes): + # A bytes literal used to propagate out of here and crash the + # extraction frontend (GH #1190). Warn rather than fail, and + # return None so the message is skipped. + warnings.warn( + f"Bytes literal {value!r} passed to a gettext function; " + "it will be skipped during message extraction. " + "Use a str literal instead.", + SyntaxWarning, + stacklevel=2, + ) if isinstance(body, ast.JoinedStr): # f-string if all(isinstance(node, ast.Constant) for node in body.values): return ''.join(node.value for node in body.values) diff --git a/tests/messages/test_extract.py b/tests/messages/test_extract.py index 41eda8903..12ac0b197 100644 --- a/tests/messages/test_extract.py +++ b/tests/messages/test_extract.py @@ -180,3 +180,19 @@ def test_issue_1195_2(): 'NOTE: This should still be considered, even if', 'the text is far away', ] + + +def test_bytes_literal_warning(): + buf = BytesIO(b""" +t = _(b'hello') +""") + with pytest.warns(SyntaxWarning, match="Bytes literal"): + messages = list(extract.extract('python', buf, extract.DEFAULT_KEYWORDS, [], {})) + assert len(messages) == 0 + + +# Note: a non-string constant such as _(42) is not covered here. It reaches the +# tokenizer as a NUMBER token, and _parse_python_string() is only called for +# STRING and FSTRING_START tokens, so no code path inspects it. Warning about it +# would mean handling NUMBER inside a translator call, which is a wider change +# than the bytes crash in GH #1190 that this fixes.