From 035dd156afac82a165de94e2693757c0ae9645cc Mon Sep 17 00:00:00 2001 From: jg Date: Sat, 3 Oct 2026 15:27:06 +0100 Subject: [PATCH] gh-158649: Extend BINARY_OP_SUBSCR_USTR_INT to latin-1 characters The specialized instruction returned an interned one-character string only for ASCII characters and exited to the generic path for anything else, so every accented character in otherwise latin-1 text missed its guard. The runtime also interns the 128 latin-1 characters, so widen the guard to 256 and pick the result from either table. Add assert_specialization_stable to test_opcache, which compares the adaptive counters of a function's specialized instructions before and after extra calls, and use it to check that reading a latin-1 string hits on every character. --- Lib/test/test_opcache.py | 50 +++++++++++++++++++ ...-10-03-14-27-06.gh-issue-158649.bGp4Bw.rst | 4 ++ Modules/_testinternalcapi/test_cases.c.h | 6 ++- Python/bytecodes.c | 8 +-- Python/executor_cases.c.h | 6 ++- Python/generated_cases.c.h | 6 ++- 6 files changed, 71 insertions(+), 9 deletions(-) create mode 100644 Misc/NEWS.d/next/Core_and_Builtins/2026-10-03-14-27-06.gh-issue-158649.bGp4Bw.rst diff --git a/Lib/test/test_opcache.py b/Lib/test/test_opcache.py index 60879e2774e707..2971ca87216018 100644 --- a/Lib/test/test_opcache.py +++ b/Lib/test/test_opcache.py @@ -38,6 +38,43 @@ def assert_no_opcode(self, f, opname): opnames = {instruction.opname for instruction in instructions} self.assertNotIn(opname, opnames) + def adaptive_counters(self, f): + """Map each specialized instruction in f to its adaptive counter.""" + counters = {} + for instruction in dis.get_instructions(f, adaptive=True): + if instruction.opname == instruction.baseopname: + continue + if instruction.baseopname in ("RESUME", "JUMP_BACKWARD"): + continue + cache = {name: data for name, _, data in instruction.cache_info} + if "counter" in cache: + counters[instruction.offset] = (instruction.opname, + cache["counter"]) + return counters + + def assert_specialization_stable(self, f, *args, calls=10): + """Assert that no specialized instruction in f misses its guard. + + A guard miss advances the instruction's adaptive counter. f should + have a fresh code object (see reset_code()), otherwise leftover + specializations from earlier runs can show up as misses. + """ + before = self.adaptive_counters(f) + self.assertTrue(before, f"{f.__qualname__} has no specialized " + "instructions") + for _ in range(calls): + f(*args) + after = self.adaptive_counters(f) + # Ignore instructions that only specialize during these calls. + moved = [] + for off, (op, counter) in before.items(): + if after.get(off) != (op, counter): + now = after[off][1] if off in after else "unspecialized" + moved.append(f"{op} at offset {off}: counter {counter} -> {now}") + self.assertEqual(moved, [], + f"specialized instructions in {f.__qualname__} " + f"missed their guard during {calls} calls") + class TestLoadSuperAttrCache(unittest.TestCase): def test_descriptor_not_double_executed_on_spec_fail(self): @@ -1977,6 +2014,19 @@ def binary_subscr_str_int_non_compact(): self.assert_specialized(binary_subscr_str_int_non_compact, "BINARY_OP_SUBSCR_USTR_INT") self.assert_no_opcode(binary_subscr_str_int_non_compact, "BINARY_OP_SUBSCR_STR_INT") + @reset_code + def binary_subscr_str_int_latin1(): + # Latin-1 characters are interned too, so reads must not miss. + for _ in range(_testinternalcapi.SPECIALIZATION_THRESHOLD): + a = "olá mundo, ça va?" + for idx, expected in enumerate(a): + self.assertEqual(a[idx], expected) + + binary_subscr_str_int_latin1() + self.assert_specialized(binary_subscr_str_int_latin1, "BINARY_OP_SUBSCR_USTR_INT") + self.assert_no_opcode(binary_subscr_str_int_latin1, "BINARY_OP_SUBSCR_STR_INT") + self.assert_specialization_stable(binary_subscr_str_int_latin1) + def binary_subscr_getitems(): class C: def __init__(self, val): diff --git a/Misc/NEWS.d/next/Core_and_Builtins/2026-10-03-14-27-06.gh-issue-158649.bGp4Bw.rst b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-03-14-27-06.gh-issue-158649.bGp4Bw.rst new file mode 100644 index 00000000000000..bfdb1038ba57ee --- /dev/null +++ b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-03-14-27-06.gh-issue-158649.bGp4Bw.rst @@ -0,0 +1,4 @@ +Speed up subscripting a non-ASCII string with an integer when the character +read is latin-1. ``BINARY_OP_SUBSCR_USTR_INT`` now returns the interned +one-character string for characters below 256 instead of only ASCII, so +reads of accented characters no longer fall back to the generic path. diff --git a/Modules/_testinternalcapi/test_cases.c.h b/Modules/_testinternalcapi/test_cases.c.h index 3bdc16437e2bc6..b4c0b4d5e099ba 100644 --- a/Modules/_testinternalcapi/test_cases.c.h +++ b/Modules/_testinternalcapi/test_cases.c.h @@ -1225,13 +1225,15 @@ JUMP_TO_PREDICTED(BINARY_OP); } Py_UCS4 c = PyUnicode_READ_CHAR(str, index); - if (Py_ARRAY_LENGTH(_Py_SINGLETON(strings).ascii) <= c) { + if (c >= 256) { UPDATE_MISS_STATS(BINARY_OP); assert(_PyOpcode_Deopt[opcode] == (BINARY_OP)); JUMP_TO_PREDICTED(BINARY_OP); } STAT_INC(BINARY_OP, hit); - PyObject *res_o = (PyObject*)&_Py_SINGLETON(strings).ascii[c]; + PyObject *res_o = (c < 128) + ? (PyObject*)&_Py_SINGLETON(strings).ascii[c] + : (PyObject*)&_Py_SINGLETON(strings).latin1[c - 128]; s = str_st; i = sub_st; res = PyStackRef_FromPyObjectBorrow(res_o); diff --git a/Python/bytecodes.c b/Python/bytecodes.c index 31eaeab0d67841..ec83f5f76a46d0 100644 --- a/Python/bytecodes.c +++ b/Python/bytecodes.c @@ -1224,11 +1224,13 @@ dummy_func( EXIT_IF(!_PyLong_IsNonNegativeCompact((PyLongObject*)sub)); Py_ssize_t index = ((PyLongObject*)sub)->long_value.ob_digit[0]; EXIT_IF(PyUnicode_GET_LENGTH(str) <= index); - // Specialize for reading an ASCII character from any string: + // Interned one-character strings exist for latin-1 only: Py_UCS4 c = PyUnicode_READ_CHAR(str, index); - EXIT_IF(Py_ARRAY_LENGTH(_Py_SINGLETON(strings).ascii) <= c); + EXIT_IF(c >= 256); STAT_INC(BINARY_OP, hit); - PyObject *res_o = (PyObject*)&_Py_SINGLETON(strings).ascii[c]; + PyObject *res_o = (c < 128) + ? (PyObject*)&_Py_SINGLETON(strings).ascii[c] + : (PyObject*)&_Py_SINGLETON(strings).latin1[c - 128]; s = str_st; i = sub_st; INPUTS_DEAD(); diff --git a/Python/executor_cases.c.h b/Python/executor_cases.c.h index c5b2dfcf5f618f..e487d876ec90ff 100644 --- a/Python/executor_cases.c.h +++ b/Python/executor_cases.c.h @@ -7165,7 +7165,7 @@ JUMP_TO_JUMP_TARGET(); } Py_UCS4 c = PyUnicode_READ_CHAR(str, index); - if (Py_ARRAY_LENGTH(_Py_SINGLETON(strings).ascii) <= c) { + if (c >= 256) { UOP_STAT_INC(uopcode, miss); _tos_cache1 = sub_st; _tos_cache0 = str_st; @@ -7173,7 +7173,9 @@ JUMP_TO_JUMP_TARGET(); } STAT_INC(BINARY_OP, hit); - PyObject *res_o = (PyObject*)&_Py_SINGLETON(strings).ascii[c]; + PyObject *res_o = (c < 128) + ? (PyObject*)&_Py_SINGLETON(strings).ascii[c] + : (PyObject*)&_Py_SINGLETON(strings).latin1[c - 128]; s = str_st; i = sub_st; res = PyStackRef_FromPyObjectBorrow(res_o); diff --git a/Python/generated_cases.c.h b/Python/generated_cases.c.h index dd0ce41e4b06b4..154dbe2cb9c961 100644 --- a/Python/generated_cases.c.h +++ b/Python/generated_cases.c.h @@ -1225,13 +1225,15 @@ JUMP_TO_PREDICTED(BINARY_OP); } Py_UCS4 c = PyUnicode_READ_CHAR(str, index); - if (Py_ARRAY_LENGTH(_Py_SINGLETON(strings).ascii) <= c) { + if (c >= 256) { UPDATE_MISS_STATS(BINARY_OP); assert(_PyOpcode_Deopt[opcode] == (BINARY_OP)); JUMP_TO_PREDICTED(BINARY_OP); } STAT_INC(BINARY_OP, hit); - PyObject *res_o = (PyObject*)&_Py_SINGLETON(strings).ascii[c]; + PyObject *res_o = (c < 128) + ? (PyObject*)&_Py_SINGLETON(strings).ascii[c] + : (PyObject*)&_Py_SINGLETON(strings).latin1[c - 128]; s = str_st; i = sub_st; res = PyStackRef_FromPyObjectBorrow(res_o);