From c4bacc68c69de4be8f43457aa8fbaa189ee1d8fd Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sat, 3 Oct 2026 20:21:37 +0200 Subject: [PATCH 1/2] Add output_length to _Py_EncodeLocaleEx() Add output_length to _Py_EncodeLocaleEx(), _Py_EncodeUTF8Ex(), encode_current_locale() and encode_ascii(). So unicode_encode_locale() and wstr_to_utf8() can use the output_length, instead of having to compute strlen(). * Add encode_current_locale_impl() to simplify encode_current_locale(). * _Py_EncodeLocaleEx() now sets error_pos and reason if it fails with -1 or -3. * Add tests on Py_EncodeLocale() and Py_DecodeLocale() functions in test_codecs. * Remove reason parameter of _Py_EncodeUTF8Ex(), encode_current_locale() and encode_ascii(). --- Include/internal/pycore_fileutils.h | 3 +- Lib/test/test_codecs.py | 59 ++++-- Modules/_testinternalcapi.c | 29 ++- Modules/_testlimitedcapi/codec.c | 76 +++++++- Objects/unicodeobject.c | 34 ++-- Python/fileutils.c | 286 ++++++++++++++++------------ Python/initconfig.c | 6 +- 7 files changed, 332 insertions(+), 161 deletions(-) diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index 128790823aa8794..0c7e024da87feaa 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -36,6 +36,7 @@ PyAPI_FUNC(int) _Py_DecodeLocaleEx( PyAPI_FUNC(int) _Py_EncodeLocaleEx( const wchar_t *text, char **str, + size_t *output_length, size_t *error_pos, const char **reason, int current_locale, @@ -201,8 +202,8 @@ extern int _Py_DecodeUTF8Ex( extern int _Py_EncodeUTF8Ex( const wchar_t *text, char **str, + size_t *output_length, size_t *error_pos, - const char **reason, int raw_malloc, _Py_error_handler errors); diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 1916ee5507e9726..2785756d02f9c57 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4057,6 +4057,7 @@ def test_pickle(self): pickle.dumps(sr, proto) +@unittest.skipIf(_testlimitedcapi is None, 'need _testlimitedcapi module') @unittest.skipIf(_testinternalcapi is None, 'need _testinternalcapi module') class LocaleCodecTest(unittest.TestCase): """ @@ -4070,7 +4071,12 @@ class LocaleCodecTest(unittest.TestCase): BYTES_STRINGS = (b"blatin1:\xa7\xe9", b"b255:\xff") SURROGATES = "\uDC80\uDCFF" - def encode(self, text, errors="strict"): + def encode_locale(self, text): + # Test Py_EncodeLocale(): use the "surrogateescape" error handler + return _testlimitedcapi.encode_locale(text) + + def encode_locale_ex(self, text, errors="strict"): + # Test _Py_EncodeLocaleEx() return _testinternalcapi.EncodeLocaleEx(text, 0, errors) def check_encode_strings(self, errors): @@ -4079,12 +4085,30 @@ def check_encode_strings(self, errors): try: expected = text.encode(self.ENCODING, errors) except UnicodeEncodeError: + for error_pos in range(len(text)): + try: + text[error_pos].encode(self.ENCODING, errors) + except UnicodeEncodeError: + break + else: + self.fail("failed to compute error_pos") + + if errors == "surrogateescape": + with self.assertRaises(ValueError) as cm: + self.encode_locale(text) + errmsg = str(cm.exception) + self.assertRegex(errmsg, f"Py_EncodeLocale failed: error_pos={error_pos}") + with self.assertRaises(RuntimeError) as cm: - self.encode(text, errors) + self.encode_locale_ex(text, errors) errmsg = str(cm.exception) - self.assertRegex(errmsg, r"encode error: pos=[0-9]+, reason=") + self.assertRegex(errmsg, f"encode error: pos={error_pos}, reason=encoding error") else: - encoded = self.encode(text, errors) + if errors in ("strict", "surrogateescape"): + encoded = self.encode_locale(text) + self.assertEqual(encoded, expected) + + encoded = self.encode_locale_ex(text, errors) self.assertEqual(encoded, expected) def test_encode_strict(self): @@ -4095,7 +4119,7 @@ def test_encode_surrogateescape(self): def test_encode_surrogatepass(self): try: - self.encode('', 'surrogatepass') + self.encode_locale_ex('', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} encoder doesn't support " @@ -4107,12 +4131,17 @@ def test_encode_surrogatepass(self): def test_encode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.encode('', 'backslashreplace') + self.encode_locale_ex('', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') - def decode(self, encoded, errors="strict"): + def decode_locale_ex(self, encoded, errors="strict"): + # Test _Py_DecodeLocaleEx() return _testinternalcapi.DecodeLocaleEx(encoded, 0, errors) + def decode_locale(self, encoded): + # Test DecodeLocale(): use the "surrogateescape" error handler + return _testlimitedcapi.decode_locale(encoded) + def check_decode_strings(self, errors): is_utf8 = (self.ENCODING == "utf-8") if is_utf8: @@ -4139,12 +4168,20 @@ def check_decode_strings(self, errors): try: expected = encoded.decode(self.ENCODING, errors) except UnicodeDecodeError: + if errors == "surrogateescape": + with self.assertRaises(ValueError): + self.decode_locale(encoded) + with self.assertRaises(RuntimeError) as cm: - self.decode(encoded, errors) + self.decode_locale_ex(encoded, errors) errmsg = str(cm.exception) self.assertStartsWith(errmsg, "decode error: ") else: - decoded = self.decode(encoded, errors) + if errors == ("strict", "surrogateescape"): + decoded = self.decode_locale(encoded) + self.assertEqual(decoded, expected) + + decoded = self.decode_locale_ex(encoded, errors) self.assertEqual(decoded, expected) def test_decode_strict(self): @@ -4155,7 +4192,7 @@ def test_decode_surrogateescape(self): def test_decode_surrogatepass(self): try: - self.decode(b'', 'surrogatepass') + self.decode_locale_ex(b'', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} decoder doesn't support " @@ -4167,7 +4204,7 @@ def test_decode_surrogatepass(self): def test_decode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.decode(b'', 'backslashreplace') + self.decode_locale_ex(b'', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index a2ce266eb35a655..67254bf97b902f9 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1046,48 +1046,64 @@ get_getpath_codeobject(PyObject *self, PyObject *Py_UNUSED(args)) { } +// Test _Py_EncodeLocaleEx() static PyObject * encode_locale_ex(PyObject *self, PyObject *args) { PyObject *unicode; int current_locale = 0; - wchar_t *wstr; PyObject *res = NULL; const char *errors = NULL; if (!PyArg_ParseTuple(args, "U|is", &unicode, ¤t_locale, &errors)) { return NULL; } - wstr = PyUnicode_AsWideCharString(unicode, NULL); + + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, NULL); if (wstr == NULL) { return NULL; } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); char *str = NULL; - size_t error_pos; - const char *reason = NULL; + size_t error_pos_canary = (size_t)-123; + size_t error_pos = error_pos_canary; + size_t output_length = (size_t)-123; + const char *reason_canary = "canary"; + const char *reason = reason_canary; int ret = _Py_EncodeLocaleEx(wstr, - &str, &error_pos, &reason, + &str, &output_length, &error_pos, &reason, current_locale, error_handler); PyMem_Free(wstr); switch(ret) { case 0: - res = PyBytes_FromString(str); + res = PyBytes_FromStringAndSize(str, output_length); PyMem_RawFree(str); break; case -1: + assert(output_length == 0); + assert(error_pos == 0); + assert(reason == NULL); PyErr_NoMemory(); break; case -2: + assert(output_length == 0); + assert(error_pos != error_pos_canary); + assert(reason != reason_canary); PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu, reason=%s", error_pos, reason); break; case -3: + assert(output_length == 0); + assert(error_pos == 0); + assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: + assert(output_length == 0); + assert(error_pos == 0); + assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unknown error code"); break; } @@ -1095,6 +1111,7 @@ encode_locale_ex(PyObject *self, PyObject *args) } +// Test _Py_DecodeLocaleEx() static PyObject * decode_locale_ex(PyObject *self, PyObject *args) { diff --git a/Modules/_testlimitedcapi/codec.c b/Modules/_testlimitedcapi/codec.c index 44eecf99f678bb8..d0b85e04a2e6b64 100644 --- a/Modules/_testlimitedcapi/codec.c +++ b/Modules/_testlimitedcapi/codec.c @@ -2,8 +2,8 @@ #ifdef Py_GIL_DISABLED # define Py_TARGET_ABI3T 0x030f0000 #else - // Need limited C API version 3.5 for PyCodec_NameReplaceErrors() -# define Py_LIMITED_API 0x03050000 + // Need limited C API version 3.13 for PyMem_RawFree() +# define Py_LIMITED_API 0x030d0000 #endif #include "parts.h" @@ -15,16 +15,80 @@ codec_namereplace_errors(PyObject *Py_UNUSED(module), PyObject *exc) return PyCodec_NameReplaceErrors(exc); } + +// Test Py_DecodeLocale() +static PyObject * +decode_locale(PyObject *Py_UNUSED(module), PyObject *arg) +{ + const char *str; + if (PyArg_Parse(arg, "y", &str) < 0) { + return NULL; + } + + size_t wstr_len = (size_t)-123; + wchar_t *wstr = Py_DecodeLocale(str, &wstr_len); + + if (str == NULL) { + if (wstr_len == (size_t)-1) { + PyErr_NoMemory(); + } + else if (wstr_len == (size_t)-2) { + PyErr_SetString(PyExc_ValueError, "decode error"); + } + else { + PyErr_Format(PyExc_SystemError, + "unknown Py_DecodeLocale() return value: %zd", + (Py_ssize_t)wstr_len); + } + return NULL; + } + + PyObject *result = PyUnicode_FromWideChar(wstr, wstr_len); + PyMem_RawFree(wstr); + return result; +} + + +// Test Py_EncodeLocale() +static PyObject * +encode_locale(PyObject *Py_UNUSED(module), PyObject *arg) +{ + PyObject *unicode; + if (PyArg_Parse(arg, "U", &unicode) < 0) { + return NULL; + } + + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, NULL); + if (wstr == NULL) { + return NULL; + } + + size_t error_pos = (size_t)-123; + char *str = Py_EncodeLocale(wstr, &error_pos); + PyMem_Free(wstr); + + if (str == NULL) { + return PyErr_Format(PyExc_ValueError, + "Py_EncodeLocale failed: error_pos=%zd", + error_pos); + } + assert(error_pos == (size_t)-123); + + PyObject *result = PyBytes_FromString(str); + PyMem_Free(str); + return result; +} + + static PyMethodDef test_methods[] = { {"codec_namereplace_errors", codec_namereplace_errors, METH_O}, + {"decode_locale", decode_locale, METH_O}, + {"encode_locale", encode_locale, METH_O}, {NULL}, }; int _PyTestLimitedCAPI_Init_Codec(PyObject *module) { - if (PyModule_AddFunctions(module, test_methods) < 0) { - return -1; - } - return 0; + return PyModule_AddFunctions(module, test_methods); } diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index e630dbb77f87251..aaaece0f05a6a5d 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3764,9 +3764,10 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, } char *str; + size_t str_len; size_t error_pos; const char *reason; - int res = _Py_EncodeLocaleEx(wstr, &str, &error_pos, &reason, + int res = _Py_EncodeLocaleEx(wstr, &str, &str_len, &error_pos, &reason, current_locale, error_handler); PyMem_Free(wstr); @@ -3792,7 +3793,7 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, return NULL; } - PyObject *bytes = PyBytes_FromString(str); + PyObject *bytes = PyBytes_FromStringAndSize(str, str_len); PyMem_RawFree(str); return bytes; } @@ -5628,17 +5629,18 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, PyMem_Free() to free the memory) into *str. On encoding failure, return -2 and write the position of the invalid - surrogate character into *error_pos (if error_pos is set) and the decoding - error message into *reason (if reason is set). + surrogate character into *error_pos (if error_pos is set). On memory allocation failure, return -1. */ int -_Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, - const char **reason, int raw_malloc, _Py_error_handler errors) +_Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { + assert(str != NULL); + assert(output_length != NULL); + const Py_ssize_t max_char_size = 4; Py_ssize_t len = wcslen(text); - assert(len >= 0); int surrogateescape = 0; @@ -5689,7 +5691,6 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, if (ch < 0x80) { /* Encode ASCII */ *p++ = (char) ch; - } else if (ch < 0x0800) { /* Encode Latin-1 */ @@ -5699,12 +5700,9 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, else if (Py_UNICODE_IS_SURROGATE(ch) && !surrogatepass) { /* surrogateescape error handler */ if (!surrogateescape || !(0xDC80 <= ch && ch <= 0xDCFF)) { - if (error_pos != NULL) { + if (error_pos) { *error_pos = (size_t)ch_pos; } - if (reason != NULL) { - *reason = "encoding error"; - } if (raw_malloc) { PyMem_RawFree(bytes); } @@ -5740,9 +5738,6 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, bytes2 = PyMem_Realloc(bytes, final_size); } if (bytes2 == NULL) { - if (error_pos != NULL) { - *error_pos = (size_t)-1; - } if (raw_malloc) { PyMem_RawFree(bytes); } @@ -5752,6 +5747,7 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, return -1; } *str = bytes2; + *output_length = final_size - 1; // -1 for the trailing NUL byte return 0; } @@ -15229,8 +15225,11 @@ unicode_iter(PyObject *seq) static int encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) { - int res; - res = _Py_EncodeUTF8Ex(wstr, str, NULL, NULL, 1, _Py_ERROR_STRICT); + assert(str != NULL); + + size_t output_length; + int res = _Py_EncodeUTF8Ex(wstr, str, &output_length, + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { PyErr_Format(PyExc_RuntimeError, "cannot encode %s", name); return -1; @@ -15239,6 +15238,7 @@ encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) PyErr_NoMemory(); return -1; } + assert(strlen(*str) == output_length); return 0; } diff --git a/Python/fileutils.c b/Python/fileutils.c index 404fec83385e970..4a52cd1c85c300b 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -354,10 +354,12 @@ _Py_ResetForceASCII(void) static int -encode_ascii(const wchar_t *text, char **str, - size_t *error_pos, const char **reason, - int raw_malloc, _Py_error_handler errors) +encode_ascii(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { + assert(str != NULL); + assert(output_length != NULL); + char *result = NULL, *out; size_t len, i; wchar_t ch; @@ -368,6 +370,7 @@ encode_ascii(const wchar_t *text, char **str, } len = wcslen(text); + *output_length = len; /* +1 for NULL byte */ if (raw_malloc) { @@ -381,7 +384,7 @@ encode_ascii(const wchar_t *text, char **str, } out = result; - for (i=0; i= 0xdc80 && c <= 0xdcff) { - if (!surrogateescape) { - goto encode_error; - } - /* UTF-8b surrogate */ - if (bytes != NULL) { - *bytes++ = c - 0xdc00; - size--; - } - else { - size++; + + for (size_t i=0; i < len; i++) { + wchar_t c = text[i]; + if (c >= 0xdc80 && c <= 0xdcff) { + if (!surrogateescape) { + if (error_pos != NULL) { + *error_pos = i; } - continue; + return DECODE_ERROR; + } + /* UTF-8b surrogate */ + if (bytes != NULL) { + *bytes++ = c - 0xdc00; + size--; } else { - buf[0] = c; - if (bytes != NULL) { - converted = wcstombs(bytes, buf, size); - } - else { - converted = wcstombs(NULL, buf, 0); - } - if (converted == DECODE_ERROR) { - goto encode_error; - } - if (bytes != NULL) { - bytes += converted; - size -= converted; - } - else { - size += converted; - } + size++; } } - if (result != NULL) { - *bytes = '\0'; - break; - } - - size += 1; /* nul byte at the end */ - if (raw_malloc) { - result = PyMem_RawMalloc(size); - } else { - result = PyMem_Malloc(size); - } - if (result == NULL) { - return -1; + buf[0] = c; + size_t converted; + if (bytes != NULL) { + converted = wcstombs(bytes, buf, size); + } + else { + converted = wcstombs(NULL, buf, 0); + } + if (converted == DECODE_ERROR) { + if (error_pos != NULL) { + *error_pos = i; + } + return DECODE_ERROR; + } + if (bytes != NULL) { + bytes += converted; + size -= converted; + } + else { + size += converted; + } } - bytes = result; } + if (bytes) { + *bytes = '\0'; + } + return size; +} + + +static int +encode_current_locale(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + _Py_error_handler errors) +{ + assert(str != NULL); + assert(output_length != NULL); + + int surrogateescape; + if (get_surrogateescape(errors, &surrogateescape) < 0) { + return -3; + } + + // First, compute the output length + char *result = NULL; + const size_t len = wcslen(text); + size_t size = encode_current_locale_impl(text, len, surrogateescape, + NULL, 0, error_pos); + if (size == DECODE_ERROR) { + goto encode_error; + } + + *output_length = size; + if (raw_malloc) { + result = PyMem_RawMalloc(size + 1); + } + else { + result = PyMem_Malloc(size + 1); + } + if (result == NULL) { + return -1; + } + + // Second, encode characters + size = encode_current_locale_impl(text, len, surrogateescape, + result, size, error_pos); + if (size == DECODE_ERROR) { + goto encode_error; + } + assert(size == 0); + *str = result; return 0; @@ -783,50 +808,28 @@ encode_current_locale(const wchar_t *text, char **str, else { PyMem_Free(result); } - if (error_pos != NULL) { - *error_pos = i; - } - if (reason) { - *reason = "encoding error"; - } return -2; } -/* Encode a string to the locale encoding. - - Parameters: - - * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead - of PyMem_Malloc(). - * current_locale: if non-zero, use the current LC_CTYPE, otherwise use - Python filesystem encoding. - * errors: error handler like "strict" or "surrogateescape". - - Return value: - - 0: success, *str is set to a newly allocated decoded string. - -1: memory allocation failure - -2: encoding error, set *error_pos and *reason (if set). - -3: the error handler 'errors' is not supported. - */ static int -encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, - const char **reason, - int raw_malloc, int current_locale, _Py_error_handler errors) +encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + int current_locale, _Py_error_handler errors) { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8Ex(text, str, output_length, + error_pos, raw_malloc, errors); #else - return encode_current_locale(text, str, error_pos, reason, - raw_malloc, errors); + return encode_current_locale(text, str, output_length, + error_pos, raw_malloc, errors); #endif } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, + return _Py_EncodeUTF8Ex(text, str, output_length, + error_pos, raw_malloc, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); @@ -834,8 +837,8 @@ encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8Ex(text, str, output_length, + error_pos, raw_malloc, errors); } #ifdef USE_FORCE_ASCII @@ -844,30 +847,76 @@ encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, } if (force_ascii) { - return encode_ascii(text, str, error_pos, reason, - raw_malloc, errors); + return encode_ascii(text, str, output_length, + error_pos, raw_malloc, errors); } #endif - return encode_current_locale(text, str, error_pos, reason, - raw_malloc, errors); + return encode_current_locale(text, str, output_length, + error_pos, raw_malloc, errors); #endif /* _Py_FORCE_UTF8_FS_ENCODING */ } +/* Encode a string to the locale encoding. + + Parameters: + + * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead + of PyMem_Malloc(). + * current_locale: if non-zero, use the current LC_CTYPE, otherwise use + Python filesystem encoding. + * errors: error handler like "strict" or "surrogateescape". + + Return value: + + 0: success, *str is set to a newly allocated decoded string. + -1: memory allocation failure + -2: encoding error, set *error_pos and *reason (if set). + -3: the error handler 'errors' is not supported. + */ +static int +encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, const char **reason, + int raw_malloc, int current_locale, _Py_error_handler errors) +{ + int res = encode_locale_inner(text, str, output_length, + error_pos, raw_malloc, + current_locale, errors); + if (res < 0) { + if (output_length) { + *output_length = 0; + } + if (res == -2) { + if (reason) { + *reason = "encoding error"; + } + } + else { + if (error_pos) { + *error_pos = 0; + } + if (reason) { + *reason = NULL; + } + } + } + return res; +} + static char* encode_locale(const wchar_t *text, size_t *error_pos, int raw_malloc, int current_locale) { char *str; - int res = encode_locale_ex(text, &str, error_pos, NULL, - raw_malloc, current_locale, - _Py_ERROR_SURROGATEESCAPE); - if (res != -2 && error_pos) { - *error_pos = (size_t)-1; - } + size_t output_length; + int res = encode_locale_impl(text, &str, &output_length, + error_pos, NULL, + raw_malloc, current_locale, + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { return NULL; } + assert(strlen(str) == output_length); return str; } @@ -900,12 +949,13 @@ _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) int -_Py_EncodeLocaleEx(const wchar_t *text, char **str, +_Py_EncodeLocaleEx(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, const char **reason, int current_locale, _Py_error_handler errors) { - return encode_locale_ex(text, str, error_pos, reason, 1, - current_locale, errors); + return encode_locale_impl(text, str, output_length, + error_pos, reason, 1, + current_locale, errors); } diff --git a/Python/initconfig.c b/Python/initconfig.c index 178e6f992fa11fc..d5b28c6f3be4bd5 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4153,7 +4153,9 @@ static char* wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) { char *utf8; - int res = _Py_EncodeUTF8Ex(wstr, &utf8, NULL, NULL, 1, _Py_ERROR_STRICT); + size_t utf8_len; + int res = _Py_EncodeUTF8Ex(wstr, &utf8, &utf8_len, + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { initconfig_set_error(config, "encoding error"); return NULL; @@ -4164,7 +4166,7 @@ wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) } // Copy to use the malloc() memory allocator - size_t size = strlen(utf8) + 1; + size_t size = utf8_len + 1; char *str = malloc(size); if (str == NULL) { PyMem_RawFree(utf8); From 783aff1deceabf53a91e4b7f922e4593ef98f2f8 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 01:28:44 +0200 Subject: [PATCH 2/2] Wrap long lines Also revert an useless change --- Lib/test/test_codecs.py | 7 +++++-- Objects/unicodeobject.c | 2 +- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 2785756d02f9c57..e07ac65d98456d8 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4097,12 +4097,15 @@ def check_encode_strings(self, errors): with self.assertRaises(ValueError) as cm: self.encode_locale(text) errmsg = str(cm.exception) - self.assertRegex(errmsg, f"Py_EncodeLocale failed: error_pos={error_pos}") + regex = f"Py_EncodeLocale failed: error_pos={error_pos}" + self.assertRegex(errmsg, regex) with self.assertRaises(RuntimeError) as cm: self.encode_locale_ex(text, errors) errmsg = str(cm.exception) - self.assertRegex(errmsg, f"encode error: pos={error_pos}, reason=encoding error") + regex = (f"encode error: pos={error_pos}, " + "reason=encoding error") + self.assertRegex(errmsg, regex) else: if errors in ("strict", "surrogateescape"): encoded = self.encode_locale(text) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index aaaece0f05a6a5d..f9cf743550acf5d 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -5700,7 +5700,7 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, else if (Py_UNICODE_IS_SURROGATE(ch) && !surrogatepass) { /* surrogateescape error handler */ if (!surrogateescape || !(0xDC80 <= ch && ch <= 0xDCFF)) { - if (error_pos) { + if (error_pos != NULL) { *error_pos = (size_t)ch_pos; } if (raw_malloc) {