From c4bacc68c69de4be8f43457aa8fbaa189ee1d8fd Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sat, 3 Oct 2026 20:21:37 +0200 Subject: [PATCH 01/16] Add output_length to _Py_EncodeLocaleEx() Add output_length to _Py_EncodeLocaleEx(), _Py_EncodeUTF8Ex(), encode_current_locale() and encode_ascii(). So unicode_encode_locale() and wstr_to_utf8() can use the output_length, instead of having to compute strlen(). * Add encode_current_locale_impl() to simplify encode_current_locale(). * _Py_EncodeLocaleEx() now sets error_pos and reason if it fails with -1 or -3. * Add tests on Py_EncodeLocale() and Py_DecodeLocale() functions in test_codecs. * Remove reason parameter of _Py_EncodeUTF8Ex(), encode_current_locale() and encode_ascii(). --- Include/internal/pycore_fileutils.h | 3 +- Lib/test/test_codecs.py | 59 ++++-- Modules/_testinternalcapi.c | 29 ++- Modules/_testlimitedcapi/codec.c | 76 +++++++- Objects/unicodeobject.c | 34 ++-- Python/fileutils.c | 286 ++++++++++++++++------------ Python/initconfig.c | 6 +- 7 files changed, 332 insertions(+), 161 deletions(-) diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index 128790823aa8794..0c7e024da87feaa 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -36,6 +36,7 @@ PyAPI_FUNC(int) _Py_DecodeLocaleEx( PyAPI_FUNC(int) _Py_EncodeLocaleEx( const wchar_t *text, char **str, + size_t *output_length, size_t *error_pos, const char **reason, int current_locale, @@ -201,8 +202,8 @@ extern int _Py_DecodeUTF8Ex( extern int _Py_EncodeUTF8Ex( const wchar_t *text, char **str, + size_t *output_length, size_t *error_pos, - const char **reason, int raw_malloc, _Py_error_handler errors); diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 1916ee5507e9726..2785756d02f9c57 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4057,6 +4057,7 @@ def test_pickle(self): pickle.dumps(sr, proto) +@unittest.skipIf(_testlimitedcapi is None, 'need _testlimitedcapi module') @unittest.skipIf(_testinternalcapi is None, 'need _testinternalcapi module') class LocaleCodecTest(unittest.TestCase): """ @@ -4070,7 +4071,12 @@ class LocaleCodecTest(unittest.TestCase): BYTES_STRINGS = (b"blatin1:\xa7\xe9", b"b255:\xff") SURROGATES = "\uDC80\uDCFF" - def encode(self, text, errors="strict"): + def encode_locale(self, text): + # Test Py_EncodeLocale(): use the "surrogateescape" error handler + return _testlimitedcapi.encode_locale(text) + + def encode_locale_ex(self, text, errors="strict"): + # Test _Py_EncodeLocaleEx() return _testinternalcapi.EncodeLocaleEx(text, 0, errors) def check_encode_strings(self, errors): @@ -4079,12 +4085,30 @@ def check_encode_strings(self, errors): try: expected = text.encode(self.ENCODING, errors) except UnicodeEncodeError: + for error_pos in range(len(text)): + try: + text[error_pos].encode(self.ENCODING, errors) + except UnicodeEncodeError: + break + else: + self.fail("failed to compute error_pos") + + if errors == "surrogateescape": + with self.assertRaises(ValueError) as cm: + self.encode_locale(text) + errmsg = str(cm.exception) + self.assertRegex(errmsg, f"Py_EncodeLocale failed: error_pos={error_pos}") + with self.assertRaises(RuntimeError) as cm: - self.encode(text, errors) + self.encode_locale_ex(text, errors) errmsg = str(cm.exception) - self.assertRegex(errmsg, r"encode error: pos=[0-9]+, reason=") + self.assertRegex(errmsg, f"encode error: pos={error_pos}, reason=encoding error") else: - encoded = self.encode(text, errors) + if errors in ("strict", "surrogateescape"): + encoded = self.encode_locale(text) + self.assertEqual(encoded, expected) + + encoded = self.encode_locale_ex(text, errors) self.assertEqual(encoded, expected) def test_encode_strict(self): @@ -4095,7 +4119,7 @@ def test_encode_surrogateescape(self): def test_encode_surrogatepass(self): try: - self.encode('', 'surrogatepass') + self.encode_locale_ex('', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} encoder doesn't support " @@ -4107,12 +4131,17 @@ def test_encode_surrogatepass(self): def test_encode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.encode('', 'backslashreplace') + self.encode_locale_ex('', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') - def decode(self, encoded, errors="strict"): + def decode_locale_ex(self, encoded, errors="strict"): + # Test _Py_DecodeLocaleEx() return _testinternalcapi.DecodeLocaleEx(encoded, 0, errors) + def decode_locale(self, encoded): + # Test DecodeLocale(): use the "surrogateescape" error handler + return _testlimitedcapi.decode_locale(encoded) + def check_decode_strings(self, errors): is_utf8 = (self.ENCODING == "utf-8") if is_utf8: @@ -4139,12 +4168,20 @@ def check_decode_strings(self, errors): try: expected = encoded.decode(self.ENCODING, errors) except UnicodeDecodeError: + if errors == "surrogateescape": + with self.assertRaises(ValueError): + self.decode_locale(encoded) + with self.assertRaises(RuntimeError) as cm: - self.decode(encoded, errors) + self.decode_locale_ex(encoded, errors) errmsg = str(cm.exception) self.assertStartsWith(errmsg, "decode error: ") else: - decoded = self.decode(encoded, errors) + if errors == ("strict", "surrogateescape"): + decoded = self.decode_locale(encoded) + self.assertEqual(decoded, expected) + + decoded = self.decode_locale_ex(encoded, errors) self.assertEqual(decoded, expected) def test_decode_strict(self): @@ -4155,7 +4192,7 @@ def test_decode_surrogateescape(self): def test_decode_surrogatepass(self): try: - self.decode(b'', 'surrogatepass') + self.decode_locale_ex(b'', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} decoder doesn't support " @@ -4167,7 +4204,7 @@ def test_decode_surrogatepass(self): def test_decode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.decode(b'', 'backslashreplace') + self.decode_locale_ex(b'', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index a2ce266eb35a655..67254bf97b902f9 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1046,48 +1046,64 @@ get_getpath_codeobject(PyObject *self, PyObject *Py_UNUSED(args)) { } +// Test _Py_EncodeLocaleEx() static PyObject * encode_locale_ex(PyObject *self, PyObject *args) { PyObject *unicode; int current_locale = 0; - wchar_t *wstr; PyObject *res = NULL; const char *errors = NULL; if (!PyArg_ParseTuple(args, "U|is", &unicode, ¤t_locale, &errors)) { return NULL; } - wstr = PyUnicode_AsWideCharString(unicode, NULL); + + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, NULL); if (wstr == NULL) { return NULL; } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); char *str = NULL; - size_t error_pos; - const char *reason = NULL; + size_t error_pos_canary = (size_t)-123; + size_t error_pos = error_pos_canary; + size_t output_length = (size_t)-123; + const char *reason_canary = "canary"; + const char *reason = reason_canary; int ret = _Py_EncodeLocaleEx(wstr, - &str, &error_pos, &reason, + &str, &output_length, &error_pos, &reason, current_locale, error_handler); PyMem_Free(wstr); switch(ret) { case 0: - res = PyBytes_FromString(str); + res = PyBytes_FromStringAndSize(str, output_length); PyMem_RawFree(str); break; case -1: + assert(output_length == 0); + assert(error_pos == 0); + assert(reason == NULL); PyErr_NoMemory(); break; case -2: + assert(output_length == 0); + assert(error_pos != error_pos_canary); + assert(reason != reason_canary); PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu, reason=%s", error_pos, reason); break; case -3: + assert(output_length == 0); + assert(error_pos == 0); + assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: + assert(output_length == 0); + assert(error_pos == 0); + assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unknown error code"); break; } @@ -1095,6 +1111,7 @@ encode_locale_ex(PyObject *self, PyObject *args) } +// Test _Py_DecodeLocaleEx() static PyObject * decode_locale_ex(PyObject *self, PyObject *args) { diff --git a/Modules/_testlimitedcapi/codec.c b/Modules/_testlimitedcapi/codec.c index 44eecf99f678bb8..d0b85e04a2e6b64 100644 --- a/Modules/_testlimitedcapi/codec.c +++ b/Modules/_testlimitedcapi/codec.c @@ -2,8 +2,8 @@ #ifdef Py_GIL_DISABLED # define Py_TARGET_ABI3T 0x030f0000 #else - // Need limited C API version 3.5 for PyCodec_NameReplaceErrors() -# define Py_LIMITED_API 0x03050000 + // Need limited C API version 3.13 for PyMem_RawFree() +# define Py_LIMITED_API 0x030d0000 #endif #include "parts.h" @@ -15,16 +15,80 @@ codec_namereplace_errors(PyObject *Py_UNUSED(module), PyObject *exc) return PyCodec_NameReplaceErrors(exc); } + +// Test Py_DecodeLocale() +static PyObject * +decode_locale(PyObject *Py_UNUSED(module), PyObject *arg) +{ + const char *str; + if (PyArg_Parse(arg, "y", &str) < 0) { + return NULL; + } + + size_t wstr_len = (size_t)-123; + wchar_t *wstr = Py_DecodeLocale(str, &wstr_len); + + if (str == NULL) { + if (wstr_len == (size_t)-1) { + PyErr_NoMemory(); + } + else if (wstr_len == (size_t)-2) { + PyErr_SetString(PyExc_ValueError, "decode error"); + } + else { + PyErr_Format(PyExc_SystemError, + "unknown Py_DecodeLocale() return value: %zd", + (Py_ssize_t)wstr_len); + } + return NULL; + } + + PyObject *result = PyUnicode_FromWideChar(wstr, wstr_len); + PyMem_RawFree(wstr); + return result; +} + + +// Test Py_EncodeLocale() +static PyObject * +encode_locale(PyObject *Py_UNUSED(module), PyObject *arg) +{ + PyObject *unicode; + if (PyArg_Parse(arg, "U", &unicode) < 0) { + return NULL; + } + + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, NULL); + if (wstr == NULL) { + return NULL; + } + + size_t error_pos = (size_t)-123; + char *str = Py_EncodeLocale(wstr, &error_pos); + PyMem_Free(wstr); + + if (str == NULL) { + return PyErr_Format(PyExc_ValueError, + "Py_EncodeLocale failed: error_pos=%zd", + error_pos); + } + assert(error_pos == (size_t)-123); + + PyObject *result = PyBytes_FromString(str); + PyMem_Free(str); + return result; +} + + static PyMethodDef test_methods[] = { {"codec_namereplace_errors", codec_namereplace_errors, METH_O}, + {"decode_locale", decode_locale, METH_O}, + {"encode_locale", encode_locale, METH_O}, {NULL}, }; int _PyTestLimitedCAPI_Init_Codec(PyObject *module) { - if (PyModule_AddFunctions(module, test_methods) < 0) { - return -1; - } - return 0; + return PyModule_AddFunctions(module, test_methods); } diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index e630dbb77f87251..aaaece0f05a6a5d 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3764,9 +3764,10 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, } char *str; + size_t str_len; size_t error_pos; const char *reason; - int res = _Py_EncodeLocaleEx(wstr, &str, &error_pos, &reason, + int res = _Py_EncodeLocaleEx(wstr, &str, &str_len, &error_pos, &reason, current_locale, error_handler); PyMem_Free(wstr); @@ -3792,7 +3793,7 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, return NULL; } - PyObject *bytes = PyBytes_FromString(str); + PyObject *bytes = PyBytes_FromStringAndSize(str, str_len); PyMem_RawFree(str); return bytes; } @@ -5628,17 +5629,18 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, PyMem_Free() to free the memory) into *str. On encoding failure, return -2 and write the position of the invalid - surrogate character into *error_pos (if error_pos is set) and the decoding - error message into *reason (if reason is set). + surrogate character into *error_pos (if error_pos is set). On memory allocation failure, return -1. */ int -_Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, - const char **reason, int raw_malloc, _Py_error_handler errors) +_Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { + assert(str != NULL); + assert(output_length != NULL); + const Py_ssize_t max_char_size = 4; Py_ssize_t len = wcslen(text); - assert(len >= 0); int surrogateescape = 0; @@ -5689,7 +5691,6 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, if (ch < 0x80) { /* Encode ASCII */ *p++ = (char) ch; - } else if (ch < 0x0800) { /* Encode Latin-1 */ @@ -5699,12 +5700,9 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, else if (Py_UNICODE_IS_SURROGATE(ch) && !surrogatepass) { /* surrogateescape error handler */ if (!surrogateescape || !(0xDC80 <= ch && ch <= 0xDCFF)) { - if (error_pos != NULL) { + if (error_pos) { *error_pos = (size_t)ch_pos; } - if (reason != NULL) { - *reason = "encoding error"; - } if (raw_malloc) { PyMem_RawFree(bytes); } @@ -5740,9 +5738,6 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, bytes2 = PyMem_Realloc(bytes, final_size); } if (bytes2 == NULL) { - if (error_pos != NULL) { - *error_pos = (size_t)-1; - } if (raw_malloc) { PyMem_RawFree(bytes); } @@ -5752,6 +5747,7 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, return -1; } *str = bytes2; + *output_length = final_size - 1; // -1 for the trailing NUL byte return 0; } @@ -15229,8 +15225,11 @@ unicode_iter(PyObject *seq) static int encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) { - int res; - res = _Py_EncodeUTF8Ex(wstr, str, NULL, NULL, 1, _Py_ERROR_STRICT); + assert(str != NULL); + + size_t output_length; + int res = _Py_EncodeUTF8Ex(wstr, str, &output_length, + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { PyErr_Format(PyExc_RuntimeError, "cannot encode %s", name); return -1; @@ -15239,6 +15238,7 @@ encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) PyErr_NoMemory(); return -1; } + assert(strlen(*str) == output_length); return 0; } diff --git a/Python/fileutils.c b/Python/fileutils.c index 404fec83385e970..4a52cd1c85c300b 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -354,10 +354,12 @@ _Py_ResetForceASCII(void) static int -encode_ascii(const wchar_t *text, char **str, - size_t *error_pos, const char **reason, - int raw_malloc, _Py_error_handler errors) +encode_ascii(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { + assert(str != NULL); + assert(output_length != NULL); + char *result = NULL, *out; size_t len, i; wchar_t ch; @@ -368,6 +370,7 @@ encode_ascii(const wchar_t *text, char **str, } len = wcslen(text); + *output_length = len; /* +1 for NULL byte */ if (raw_malloc) { @@ -381,7 +384,7 @@ encode_ascii(const wchar_t *text, char **str, } out = result; - for (i=0; i= 0xdc80 && c <= 0xdcff) { - if (!surrogateescape) { - goto encode_error; - } - /* UTF-8b surrogate */ - if (bytes != NULL) { - *bytes++ = c - 0xdc00; - size--; - } - else { - size++; + + for (size_t i=0; i < len; i++) { + wchar_t c = text[i]; + if (c >= 0xdc80 && c <= 0xdcff) { + if (!surrogateescape) { + if (error_pos != NULL) { + *error_pos = i; } - continue; + return DECODE_ERROR; + } + /* UTF-8b surrogate */ + if (bytes != NULL) { + *bytes++ = c - 0xdc00; + size--; } else { - buf[0] = c; - if (bytes != NULL) { - converted = wcstombs(bytes, buf, size); - } - else { - converted = wcstombs(NULL, buf, 0); - } - if (converted == DECODE_ERROR) { - goto encode_error; - } - if (bytes != NULL) { - bytes += converted; - size -= converted; - } - else { - size += converted; - } + size++; } } - if (result != NULL) { - *bytes = '\0'; - break; - } - - size += 1; /* nul byte at the end */ - if (raw_malloc) { - result = PyMem_RawMalloc(size); - } else { - result = PyMem_Malloc(size); - } - if (result == NULL) { - return -1; + buf[0] = c; + size_t converted; + if (bytes != NULL) { + converted = wcstombs(bytes, buf, size); + } + else { + converted = wcstombs(NULL, buf, 0); + } + if (converted == DECODE_ERROR) { + if (error_pos != NULL) { + *error_pos = i; + } + return DECODE_ERROR; + } + if (bytes != NULL) { + bytes += converted; + size -= converted; + } + else { + size += converted; + } } - bytes = result; } + if (bytes) { + *bytes = '\0'; + } + return size; +} + + +static int +encode_current_locale(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + _Py_error_handler errors) +{ + assert(str != NULL); + assert(output_length != NULL); + + int surrogateescape; + if (get_surrogateescape(errors, &surrogateescape) < 0) { + return -3; + } + + // First, compute the output length + char *result = NULL; + const size_t len = wcslen(text); + size_t size = encode_current_locale_impl(text, len, surrogateescape, + NULL, 0, error_pos); + if (size == DECODE_ERROR) { + goto encode_error; + } + + *output_length = size; + if (raw_malloc) { + result = PyMem_RawMalloc(size + 1); + } + else { + result = PyMem_Malloc(size + 1); + } + if (result == NULL) { + return -1; + } + + // Second, encode characters + size = encode_current_locale_impl(text, len, surrogateescape, + result, size, error_pos); + if (size == DECODE_ERROR) { + goto encode_error; + } + assert(size == 0); + *str = result; return 0; @@ -783,50 +808,28 @@ encode_current_locale(const wchar_t *text, char **str, else { PyMem_Free(result); } - if (error_pos != NULL) { - *error_pos = i; - } - if (reason) { - *reason = "encoding error"; - } return -2; } -/* Encode a string to the locale encoding. - - Parameters: - - * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead - of PyMem_Malloc(). - * current_locale: if non-zero, use the current LC_CTYPE, otherwise use - Python filesystem encoding. - * errors: error handler like "strict" or "surrogateescape". - - Return value: - - 0: success, *str is set to a newly allocated decoded string. - -1: memory allocation failure - -2: encoding error, set *error_pos and *reason (if set). - -3: the error handler 'errors' is not supported. - */ static int -encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, - const char **reason, - int raw_malloc, int current_locale, _Py_error_handler errors) +encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + int current_locale, _Py_error_handler errors) { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8Ex(text, str, output_length, + error_pos, raw_malloc, errors); #else - return encode_current_locale(text, str, error_pos, reason, - raw_malloc, errors); + return encode_current_locale(text, str, output_length, + error_pos, raw_malloc, errors); #endif } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, + return _Py_EncodeUTF8Ex(text, str, output_length, + error_pos, raw_malloc, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); @@ -834,8 +837,8 @@ encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8Ex(text, str, output_length, + error_pos, raw_malloc, errors); } #ifdef USE_FORCE_ASCII @@ -844,30 +847,76 @@ encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, } if (force_ascii) { - return encode_ascii(text, str, error_pos, reason, - raw_malloc, errors); + return encode_ascii(text, str, output_length, + error_pos, raw_malloc, errors); } #endif - return encode_current_locale(text, str, error_pos, reason, - raw_malloc, errors); + return encode_current_locale(text, str, output_length, + error_pos, raw_malloc, errors); #endif /* _Py_FORCE_UTF8_FS_ENCODING */ } +/* Encode a string to the locale encoding. + + Parameters: + + * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead + of PyMem_Malloc(). + * current_locale: if non-zero, use the current LC_CTYPE, otherwise use + Python filesystem encoding. + * errors: error handler like "strict" or "surrogateescape". + + Return value: + + 0: success, *str is set to a newly allocated decoded string. + -1: memory allocation failure + -2: encoding error, set *error_pos and *reason (if set). + -3: the error handler 'errors' is not supported. + */ +static int +encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, const char **reason, + int raw_malloc, int current_locale, _Py_error_handler errors) +{ + int res = encode_locale_inner(text, str, output_length, + error_pos, raw_malloc, + current_locale, errors); + if (res < 0) { + if (output_length) { + *output_length = 0; + } + if (res == -2) { + if (reason) { + *reason = "encoding error"; + } + } + else { + if (error_pos) { + *error_pos = 0; + } + if (reason) { + *reason = NULL; + } + } + } + return res; +} + static char* encode_locale(const wchar_t *text, size_t *error_pos, int raw_malloc, int current_locale) { char *str; - int res = encode_locale_ex(text, &str, error_pos, NULL, - raw_malloc, current_locale, - _Py_ERROR_SURROGATEESCAPE); - if (res != -2 && error_pos) { - *error_pos = (size_t)-1; - } + size_t output_length; + int res = encode_locale_impl(text, &str, &output_length, + error_pos, NULL, + raw_malloc, current_locale, + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { return NULL; } + assert(strlen(str) == output_length); return str; } @@ -900,12 +949,13 @@ _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) int -_Py_EncodeLocaleEx(const wchar_t *text, char **str, +_Py_EncodeLocaleEx(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, const char **reason, int current_locale, _Py_error_handler errors) { - return encode_locale_ex(text, str, error_pos, reason, 1, - current_locale, errors); + return encode_locale_impl(text, str, output_length, + error_pos, reason, 1, + current_locale, errors); } diff --git a/Python/initconfig.c b/Python/initconfig.c index 178e6f992fa11fc..d5b28c6f3be4bd5 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4153,7 +4153,9 @@ static char* wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) { char *utf8; - int res = _Py_EncodeUTF8Ex(wstr, &utf8, NULL, NULL, 1, _Py_ERROR_STRICT); + size_t utf8_len; + int res = _Py_EncodeUTF8Ex(wstr, &utf8, &utf8_len, + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { initconfig_set_error(config, "encoding error"); return NULL; @@ -4164,7 +4166,7 @@ wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) } // Copy to use the malloc() memory allocator - size_t size = strlen(utf8) + 1; + size_t size = utf8_len + 1; char *str = malloc(size); if (str == NULL) { PyMem_RawFree(utf8); From 783aff1deceabf53a91e4b7f922e4593ef98f2f8 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 01:28:44 +0200 Subject: [PATCH 02/16] Wrap long lines Also revert an useless change --- Lib/test/test_codecs.py | 7 +++++-- Objects/unicodeobject.c | 2 +- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 2785756d02f9c57..e07ac65d98456d8 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4097,12 +4097,15 @@ def check_encode_strings(self, errors): with self.assertRaises(ValueError) as cm: self.encode_locale(text) errmsg = str(cm.exception) - self.assertRegex(errmsg, f"Py_EncodeLocale failed: error_pos={error_pos}") + regex = f"Py_EncodeLocale failed: error_pos={error_pos}" + self.assertRegex(errmsg, regex) with self.assertRaises(RuntimeError) as cm: self.encode_locale_ex(text, errors) errmsg = str(cm.exception) - self.assertRegex(errmsg, f"encode error: pos={error_pos}, reason=encoding error") + regex = (f"encode error: pos={error_pos}, " + "reason=encoding error") + self.assertRegex(errmsg, regex) else: if errors in ("strict", "surrogateescape"): encoded = self.encode_locale(text) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index aaaece0f05a6a5d..f9cf743550acf5d 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -5700,7 +5700,7 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, else if (Py_UNICODE_IS_SURROGATE(ch) && !surrogatepass) { /* surrogateescape error handler */ if (!surrogateescape || !(0xDC80 <= ch && ch <= 0xDCFF)) { - if (error_pos) { + if (error_pos != NULL) { *error_pos = (size_t)ch_pos; } if (raw_malloc) { From faab2a87430a18bb5f5d0866a8514d69048f5d50 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 17:00:53 +0200 Subject: [PATCH 03/16] Add constants for encode errors _Py_EncodeUTF8Ex() now uses size_t instead of Py_ssize_t to iterate on the input string. On error, encode_current_locale_impl() now returns ENCODE_ERROR (new constant) instead of DECODE_ERROR. Add comments on the 3 encode functions. --- Include/internal/pycore_fileutils.h | 5 ++ Objects/unicodeobject.c | 44 +++++----- Python/fileutils.c | 128 +++++++++++++++++----------- 3 files changed, 106 insertions(+), 71 deletions(-) diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index 0c7e024da87feaa..507da4f79d04219 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -191,6 +191,11 @@ extern int _Py_open_osfhandle(void *handle, int flags); ? _PyStatus_ERR("cannot decode " NAME) \ : _PyStatus_NO_MEMORY() +#define _Py_CODEC_MEMORY_ERROR -1 +#define _Py_CODEC_DECODE_ERROR -2 +#define _Py_CODEC_ENCODE_ERROR -2 +#define _Py_CODEC_UNSUPPORTED_ERROR_HANDLER -3 + extern int _Py_DecodeUTF8Ex( const char *arg, Py_ssize_t arglen, diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index f9cf743550acf5d..e8ce8bd074c3669 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -5628,10 +5628,11 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, On success, return 0 and write the newly allocated character string (use PyMem_Free() to free the memory) into *str. - On encoding failure, return -2 and write the position of the invalid - surrogate character into *error_pos (if error_pos is set). + On encoding failure, return _Py_CODEC_ENCODE_ERROR (-2) and write the + position of the invalid surrogate character into *error_pos (if error_pos is + set). - On memory allocation failure, return -1. */ + On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). */ int _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, _Py_error_handler errors) @@ -5639,9 +5640,9 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, assert(str != NULL); assert(output_length != NULL); - const Py_ssize_t max_char_size = 4; - Py_ssize_t len = wcslen(text); - assert(len >= 0); + // U+10ffff encoded to UTF-8 takes 4 bytes + const size_t max_char_size = 4; + size_t len = wcslen(text); int surrogateescape = 0; int surrogatepass = 0; @@ -5656,11 +5657,11 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, surrogatepass = 1; break; default: - return -3; + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } - if (len > PY_SSIZE_T_MAX / max_char_size - 1) { - return -1; + if (len > (size_t)PY_SSIZE_T_MAX / max_char_size - 1) { + return _Py_CODEC_MEMORY_ERROR; } char *bytes; if (raw_malloc) { @@ -5670,12 +5671,11 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, bytes = PyMem_Malloc((len + 1) * max_char_size); } if (bytes == NULL) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } char *p = bytes; - Py_ssize_t i; - for (i = 0; i < len; ) { + for (size_t i = 0; i < len; ) { Py_ssize_t ch_pos = i; Py_UCS4 ch = text[i]; i++; @@ -5709,7 +5709,7 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, else { PyMem_Free(bytes); } - return -2; + return _Py_CODEC_ENCODE_ERROR; } *p++ = (char)(ch & 0xff); } @@ -5727,27 +5727,29 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, *p++ = (char)(0x80 | (ch & 0x3f)); } } - *p++ = '\0'; + *p++ = '\0'; // trailing NUL byte size_t final_size = (p - bytes); - char *bytes2; + assert(final_size >= 1); + *output_length = final_size - 1; // -1 for the trailing NUL byte + + char *result; if (raw_malloc) { - bytes2 = PyMem_RawRealloc(bytes, final_size); + result = PyMem_RawRealloc(bytes, final_size); } else { - bytes2 = PyMem_Realloc(bytes, final_size); + result = PyMem_Realloc(bytes, final_size); } - if (bytes2 == NULL) { + if (result == NULL) { if (raw_malloc) { PyMem_RawFree(bytes); } else { PyMem_Free(bytes); } - return -1; + return _Py_CODEC_MEMORY_ERROR; } - *str = bytes2; - *output_length = final_size - 1; // -1 for the trailing NUL byte + *str = result; return 0; } diff --git a/Python/fileutils.c b/Python/fileutils.c index 4a52cd1c85c300b..c49974c1a2ee18b 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -51,6 +51,8 @@ extern int winerror_to_errno(int); int _Py_open_cloexec_works = -1; #endif +// wcstombs() error +static const size_t ENCODE_ERROR = ((size_t)-1); // mbstowcs() and mbrtowc() errors static const size_t DECODE_ERROR = ((size_t)-1); #ifdef HAVE_MBRTOWC @@ -58,6 +60,10 @@ static const size_t INCOMPLETE_CHARACTER = (size_t)-2; #endif +// Get the error handler as 'int surrogateescape'. +// Set '*surrogateescape' and return 0 on success. +// Return -1 if the error handler is not supported: other than "strict" and +// "surrogateescape". static int get_surrogateescape(_Py_error_handler errors, int *surrogateescape) { @@ -227,8 +233,7 @@ extern int _Py_normalize_encoding(const char *, char *, size_t, int); Values of force_ascii: 1: the workaround is used: Py_EncodeLocale() uses - encode_ascii_surrogateescape() and Py_DecodeLocale() uses - decode_ascii() + encode_ascii() and Py_DecodeLocale() uses decode_ascii() 0: the workaround is not used: Py_EncodeLocale() uses wcstombs() and Py_DecodeLocale() uses mbstowcs() -1: unknown, need to call check_force_ascii() to get the value @@ -353,6 +358,13 @@ _Py_ResetForceASCII(void) } +// Encode a wide string to the ASCII encoding. If errors is +// _Py_ERROR_SURROGATEESCAPE, handle non-ASCII character using the +// surrogateescape error handler. In that case, the output is non-ASCII. +// +// Set *str to a newly allocated string on success. +// Return a negative number on error: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_ENCODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. static int encode_ascii(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, _Py_error_handler errors) @@ -360,32 +372,25 @@ encode_ascii(const wchar_t *text, char **str, size_t *output_length, assert(str != NULL); assert(output_length != NULL); - char *result = NULL, *out; - size_t len, i; - wchar_t ch; - int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } - len = wcslen(text); + size_t len = wcslen(text); *output_length = len; /* +1 for NULL byte */ - if (raw_malloc) { - result = PyMem_RawMalloc(len + 1); - } - else { - result = PyMem_Malloc(len + 1); - } + char *result; + result = raw_malloc ? PyMem_RawMalloc(len + 1) : PyMem_Malloc(len + 1); if (result == NULL) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } - out = result; - for (i=0; i < len; i++) { - ch = text[i]; + char *out = result; + for (size_t i=0; i < len; i++) { + wchar_t ch = text[i]; if (ch <= 0x7f) { /* ASCII character */ @@ -405,10 +410,12 @@ encode_ascii(const wchar_t *text, char **str, size_t *output_length, if (error_pos != NULL) { *error_pos = i; } - return -2; + return _Py_CODEC_ENCODE_ERROR; } } - *out = '\0'; + *out++ = '\0'; + assert((out - result) == (*output_length + 1)); + *str = result; return 0; } @@ -424,7 +431,7 @@ _Py_ResetForceASCII(void) { /* nothing to do */ } -#endif /* !defined(_Py_FORCE_UTF8_FS_ENCODING) && !defined(MS_WINDOWS) */ +#endif /* defined(_Py_FORCE_UTF8_FS_ENCODING) || defined(MS_WINDOWS) */ #if !defined(HAVE_MBRTOWC) || defined(USE_FORCE_ASCII) @@ -439,15 +446,16 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } if (argsize > PY_SSIZE_T_MAX / sizeof(wchar_t)) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } res = PyMem_RawMalloc(argsize * sizeof(wchar_t)); if (!res) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } out = res; @@ -465,7 +473,7 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, if (reason) { *reason = "decoding error"; } - return -2; + return _Py_CODEC_DECODE_ERROR; } *out++ = 0xdc00 + ch; } @@ -495,7 +503,8 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } #ifdef HAVE_BROKEN_MBSTOWCS @@ -509,11 +518,11 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, #endif if (argsize != DECODE_ERROR) { if (argsize > PY_SSIZE_T_MAX / sizeof(wchar_t) - 1) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } res = (wchar_t *)PyMem_RawMalloc((argsize + 1) * sizeof(wchar_t)); if (!res) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } count = _Py_mbstowcs(res, arg, argsize + 1); @@ -535,11 +544,11 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, actual output could use less memory. */ argsize = strlen(arg) + 1; if (argsize > PY_SSIZE_T_MAX / sizeof(wchar_t)) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } res = (wchar_t*)PyMem_RawMalloc(argsize * sizeof(wchar_t)); if (!res) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } in = (unsigned char*)arg; @@ -587,7 +596,7 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, if (reason) { *reason = "decoding error"; } - return -2; + return _Py_CODEC_DECODE_ERROR; #else /* HAVE_MBRTOWC */ /* Cannot use C locale for escaping; manually escape as if charset is ASCII (i.e. escape all bytes > 128. This will still roundtrip @@ -609,13 +618,14 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, *wstr (use PyMem_RawFree() to free the memory). If wlen is not NULL, write the number of wide characters excluding the null character into *wlen. - On memory allocation failure, return -1. + On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). - On decoding error, return -2. If wlen is not NULL, write the start of - invalid byte sequence in the input string into *wlen. If reason is not NULL, - write the decoding error message into *reason. + On decoding error, return _Py_CODEC_DECODE_ERROR (-2). If wlen is not NULL, + write the start of invalid byte sequence in the input string into *wlen. If + reason is not NULL, write the decoding error message into *reason. - Return -3 if the error handler 'errors' is not supported. + Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3) if the 'errors' error + handler is not supported: other than "strict" and "surrogateescape". Use the Py_EncodeLocaleEx() function to encode the character string back to a byte string. */ @@ -699,6 +709,10 @@ Py_DecodeLocale(const char* arg, size_t *wlen) } +// If bytes is NULL, return the length in 'bytes' of the encoded string. +// Otherwise, encode the wide string into 'bytes', decrements 'size', and +// return 0 on success. +// On encoding error, set 'error_pos' (if set) and return ENCODE_ERROR. static size_t encode_current_locale_impl(const wchar_t *text, const size_t len, int surrogateescape, @@ -706,6 +720,7 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, size_t *error_pos) { wchar_t buf[2]; + // The second character is always empty buf[1] = 0; for (size_t i=0; i < len; i++) { @@ -715,7 +730,7 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, if (error_pos != NULL) { *error_pos = i; } - return DECODE_ERROR; + return ENCODE_ERROR; } /* UTF-8b surrogate */ if (bytes != NULL) { @@ -727,6 +742,7 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, } } else { + // Encode a single character using wcstombs() buf[0] = c; size_t converted; if (bytes != NULL) { @@ -735,11 +751,11 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, else { converted = wcstombs(NULL, buf, 0); } - if (converted == DECODE_ERROR) { + if (converted == ENCODE_ERROR) { if (error_pos != NULL) { *error_pos = i; } - return DECODE_ERROR; + return ENCODE_ERROR; } if (bytes != NULL) { bytes += converted; @@ -753,10 +769,16 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, if (bytes) { *bytes = '\0'; } + // Sanity check, it cannot happen in practice + assert(size != ENCODE_ERROR); return size; } +// Encode a wide string to the current locale encoding. +// Set *str to a newly allocated string on success. +// Return a negative number on error: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_ENCODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. static int encode_current_locale(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, @@ -767,15 +789,18 @@ encode_current_locale(const wchar_t *text, char **str, size_t *output_length, int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } // First, compute the output length char *result = NULL; const size_t len = wcslen(text); + // Sanity check, it cannot happen in practice + assert(len != ENCODE_ERROR); size_t size = encode_current_locale_impl(text, len, surrogateescape, NULL, 0, error_pos); - if (size == DECODE_ERROR) { + if (size == ENCODE_ERROR) { goto encode_error; } @@ -787,13 +812,13 @@ encode_current_locale(const wchar_t *text, char **str, size_t *output_length, result = PyMem_Malloc(size + 1); } if (result == NULL) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } // Second, encode characters size = encode_current_locale_impl(text, len, surrogateescape, result, size, error_pos); - if (size == DECODE_ERROR) { + if (size == ENCODE_ERROR) { goto encode_error; } assert(size == 0); @@ -808,7 +833,7 @@ encode_current_locale(const wchar_t *text, char **str, size_t *output_length, else { PyMem_Free(result); } - return -2; + return _Py_CODEC_ENCODE_ERROR; } @@ -867,12 +892,15 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, Python filesystem encoding. * errors: error handler like "strict" or "surrogateescape". - Return value: + Set *str to a newly allocated decoded string and return 0 on success. + Return a negative result on error: - 0: success, *str is set to a newly allocated decoded string. - -1: memory allocation failure - -2: encoding error, set *error_pos and *reason (if set). - -3: the error handler 'errors' is not supported. + * _Py_CODEC_MEMORY_ERROR (-1): memory allocation failure + * _Py_CODEC_ENCODE_ERROR (-2): encoding error, set *error_pos + and *reason (if set). + * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3): the 'errors' error handler + is not supported. Most functions only support "strict" and + "surrogateescape". The UTF-8 encoder also supports "surrogatepass". */ static int encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, @@ -886,7 +914,7 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, if (output_length) { *output_length = 0; } - if (res == -2) { + if (res == _Py_CODEC_ENCODE_ERROR) { if (reason) { *reason = "encoding error"; } From 6d7139cf1a5a299bd31660d53809b8363306aa83 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 17:37:30 +0200 Subject: [PATCH 04/16] Add more tests on all Encode/Decode output parameters Add more comments. --- Modules/_testinternalcapi.c | 51 +++++--- Modules/_testlimitedcapi/codec.c | 6 +- Objects/unicodeobject.c | 52 ++++++--- Python/fileutils.c | 193 +++++++++++++++++++++++-------- Python/initconfig.c | 6 +- 5 files changed, 223 insertions(+), 85 deletions(-) diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 67254bf97b902f9..d29777f978bc58b 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1065,11 +1065,13 @@ encode_locale_ex(PyObject *self, PyObject *args) } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); - char *str = NULL; + const char *str_canary = (const char*)0x1234; + char *str = (char*)str_canary; size_t error_pos_canary = (size_t)-123; size_t error_pos = error_pos_canary; - size_t output_length = (size_t)-123; - const char *reason_canary = "canary"; + const size_t output_length_canary = (size_t)-456; + size_t output_length = output_length_canary; + const char *reason_canary = (const char*)0x123456; const char *reason = reason_canary; int ret = _Py_EncodeLocaleEx(wstr, &str, &output_length, &error_pos, &reason, @@ -1078,29 +1080,37 @@ encode_locale_ex(PyObject *self, PyObject *args) switch(ret) { case 0: + assert(str != NULL && str != str_canary); + assert(output_length != output_length_canary); + assert(error_pos == error_pos_canary); + assert(reason == reason_canary); res = PyBytes_FromStringAndSize(str, output_length); PyMem_RawFree(str); break; - case -1: + case _Py_CODEC_MEMORY_ERROR: + assert(str == NULL); assert(output_length == 0); assert(error_pos == 0); assert(reason == NULL); PyErr_NoMemory(); break; - case -2: + case _Py_CODEC_ENCODE_ERROR: + assert(str == NULL); assert(output_length == 0); assert(error_pos != error_pos_canary); - assert(reason != reason_canary); + assert(reason != NULL && reason != reason_canary); PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu, reason=%s", error_pos, reason); break; - case -3: + case _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: + assert(str == NULL); assert(output_length == 0); assert(error_pos == 0); assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: + assert(str == NULL); assert(output_length == 0); assert(error_pos == 0); assert(reason == NULL); @@ -1125,26 +1135,41 @@ decode_locale_ex(PyObject *self, PyObject *args) } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); - wchar_t *wstr = NULL; - size_t wlen = 0; - const char *reason = NULL; + const wchar_t *wstr_canary = (const wchar_t*)0x12345; + wchar_t *wstr = (wchar_t*)wstr_canary; + const size_t wlen_canary = (size_t)-123; + size_t wlen = wlen_canary; + const char *reason_canary = (const char*)0x123456; + const char *reason = reason_canary; int ret = _Py_DecodeLocaleEx(str, &wstr, &wlen, &reason, current_locale, error_handler); switch(ret) { case 0: + assert(wstr != NULL && wstr != wstr_canary); + assert(wlen != wlen_canary); + assert(reason == reason_canary); res = PyUnicode_FromWideChar(wstr, wlen); PyMem_RawFree(wstr); break; - case -1: + case _Py_CODEC_MEMORY_ERROR: + assert(wstr == NULL); + assert(wlen == 0); + assert(reason == NULL); PyErr_NoMemory(); break; - case -2: + case _Py_CODEC_DECODE_ERROR: + assert(wstr == NULL); + assert(wlen != wlen_canary); + assert(reason != NULL && reason != reason_canary); PyErr_Format(PyExc_RuntimeError, "decode error: pos=%zu, reason=%s", wlen, reason); break; - case -3: + case _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: + assert(wstr == NULL); + assert(wlen == 0); + assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: diff --git a/Modules/_testlimitedcapi/codec.c b/Modules/_testlimitedcapi/codec.c index d0b85e04a2e6b64..bc2531fb01f00f8 100644 --- a/Modules/_testlimitedcapi/codec.c +++ b/Modules/_testlimitedcapi/codec.c @@ -63,16 +63,18 @@ encode_locale(PyObject *Py_UNUSED(module), PyObject *arg) return NULL; } - size_t error_pos = (size_t)-123; + const size_t error_pos_canary = (size_t)-123; + size_t error_pos = error_pos_canary; char *str = Py_EncodeLocale(wstr, &error_pos); PyMem_Free(wstr); if (str == NULL) { + assert(error_pos != error_pos_canary); return PyErr_Format(PyExc_ValueError, "Py_EncodeLocale failed: error_pos=%zd", error_pos); } - assert(error_pos == (size_t)-123); + assert(error_pos == error_pos_canary); PyObject *result = PyBytes_FromString(str); PyMem_Free(str); diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index e8ce8bd074c3669..6447fd904293b49 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -5482,22 +5482,31 @@ PyUnicode_DecodeUTF8Stateful(const char *s, } -/* UTF-8 decoder: use surrogateescape error handler if 'surrogateescape' is - non-zero, use strict error handler otherwise. - - On success, write a pointer to a newly allocated wide character string into - *wstr (use PyMem_RawFree() to free the memory) and write the output length - (in number of wchar_t units) into *wlen (if wlen is set). - - On memory allocation failure, return -1. - - On decoding error (if surrogateescape is zero), return -2. If wlen is - non-NULL, write the start of the illegal byte sequence into *wlen. If reason - is not NULL, write the decoding error message into *reason. */ +// UTF-8 decoder. +// +// Supported error handlers: "strict", "surrogateescape" and "surrogatepass". +// +// On success, write a pointer to a newly allocated wide character string into +// *wstr (use PyMem_RawFree() to free the memory), write the output length +// (in number of wchar_t units) into *wlen (if wlen is set), and return 0. +// +// On error, return a negative number. +// +// On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. +// +// On decoding error (if surrogateescape is zero), return -2. If wlen is +// non-NULL, write the start of the illegal byte sequence into *wlen. If reason +// is not NULL, write the decoding error message into *reason. +// +// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if 'errors' error handler is not +// supported. int _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, const char **reason, _Py_error_handler errors) { + assert(s != NULL); + assert(wstr != NULL); + const char *orig_s = s; const char *e; wchar_t *unicode; @@ -5516,18 +5525,18 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, surrogatepass = 1; break; default: - return -3; + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } /* Note: size will always be longer than the resulting Unicode character count */ if (PY_SSIZE_T_MAX / (Py_ssize_t)sizeof(wchar_t) - 1 < size) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } unicode = PyMem_RawMalloc((size + 1) * sizeof(wchar_t)); if (!unicode) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } /* Unpack UTF-8 encoded data */ @@ -5589,7 +5598,7 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, if (wlen != NULL) { *wlen = s - orig_s; } - return -2; + return _Py_CODEC_DECODE_ERROR; } } } @@ -5603,6 +5612,10 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, } +// Decode from UTF-8 with the "surrogateescape" error handler. +// On success, set *wlen and return a newly allocated string. +// On error, set *wlen to the error (_Py_CODEC_MEMORY_ERROR or +// _Py_CODEC_DECODE_ERROR) and the return NULL wchar_t* _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, size_t *wlen) @@ -5613,7 +5626,7 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, NULL, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { /* _Py_DecodeUTF8Ex() must support _Py_ERROR_SURROGATEESCAPE */ - assert(res != -3); + assert(res != _Py_CODEC_UNSUPPORTED_ERROR_HANDLER); if (wlen) { *wlen = (size_t)res; } @@ -5632,7 +5645,10 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, position of the invalid surrogate character into *error_pos (if error_pos is set). - On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). */ + On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). + + str and output_length must not be NULL +*/ int _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, _Py_error_handler errors) diff --git a/Python/fileutils.c b/Python/fileutils.c index c49974c1a2ee18b..706f0e0edf6164f 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -365,6 +365,8 @@ _Py_ResetForceASCII(void) // Set *str to a newly allocated string on success. // Return a negative number on error: _Py_CODEC_MEMORY_ERROR, // _Py_CODEC_ENCODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// +// str and output_length must not be NULL static int encode_ascii(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, _Py_error_handler errors) @@ -414,7 +416,7 @@ encode_ascii(const wchar_t *text, char **str, size_t *output_length, } } *out++ = '\0'; - assert((out - result) == (*output_length + 1)); + assert((size_t)(out - result) == (*output_length + 1)); *str = result; return 0; @@ -435,10 +437,21 @@ _Py_ResetForceASCII(void) #if !defined(HAVE_MBRTOWC) || defined(USE_FORCE_ASCII) +// Decode a bytes string from ASCII. If 'errors' is "surrogateescape", handle +// non-ASCII with "surrogateescape" error handler and so the output is +// non-ASCII. +// +// On success, set *wstr and *wlen (if set) and return 0. +// On error, return a negative number: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_DECODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// On decode, set *wlen (if set) and *reason (if set). static int decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, const char **reason, _Py_error_handler errors) { + assert(arg != NULL); + assert(wstr != NULL); + wchar_t *res; unsigned char *in; wchar_t *out; @@ -461,7 +474,7 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, out = res; for (in = (unsigned char*)arg; *in; in++) { unsigned char ch = *in; - if (ch < 128) { + if (ch <= 127) { *out++ = ch; } else { @@ -488,10 +501,18 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, } #endif /* !HAVE_MBRTOWC */ +// Decode a bytes string from the current locale. +// On success, set *wstr and *wlen (if set) and return 0. +// On error, return a negative number: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_DECODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// On decode, set *wlen (if set) and *reason (if set). static int decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, const char **reason, _Py_error_handler errors) { + assert(arg != NULL); + assert(wstr != NULL); + wchar_t *res; size_t argsize; size_t count; @@ -606,31 +627,8 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, } -/* Decode a byte string from the locale encoding. - - Use the strict error handler if 'surrogateescape' is zero. Use the - surrogateescape error handler if 'surrogateescape' is non-zero: undecodable - bytes are decoded as characters in range U+DC80..U+DCFF. If a byte sequence - can be decoded as a surrogate character, escape the bytes using the - surrogateescape error handler instead of decoding them. - - On success, return 0 and write the newly allocated wide character string into - *wstr (use PyMem_RawFree() to free the memory). If wlen is not NULL, write - the number of wide characters excluding the null character into *wlen. - - On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). - - On decoding error, return _Py_CODEC_DECODE_ERROR (-2). If wlen is not NULL, - write the start of invalid byte sequence in the input string into *wlen. If - reason is not NULL, write the decoding error message into *reason. - - Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3) if the 'errors' error - handler is not supported: other than "strict" and "surrogateescape". - - Use the Py_EncodeLocaleEx() function to encode the character string back to - a byte string. */ -int -_Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, +static int +decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, const char **reason, int current_locale, _Py_error_handler errors) { @@ -672,6 +670,80 @@ _Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, } +// Decode a byte string from the locale encoding. +// +// Use the strict error handler if 'surrogateescape' is zero. Use the +// surrogateescape error handler if 'surrogateescape' is non-zero: undecodable +// bytes are decoded as characters in range U+DC80..U+DCFF. If a byte sequence +// can be decoded as a surrogate character, escape the bytes using the +// surrogateescape error handler instead of decoding them. +// +// On success, return 0 and write the newly allocated wide character string into +// *wstr (use PyMem_RawFree() to free the memory). If wlen is not NULL, write +// the number of wide characters excluding the null character into *wlen. +// +// On error, return a negative number. +// +// On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). +// +// On decoding error, return _Py_CODEC_DECODE_ERROR (-2). If wlen is not NULL, +// write the start of invalid byte sequence in the input string into *wlen. If +// reason is not NULL, write the decoding error message into *reason. +// +// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3) if the 'errors' error +// handler is not supported: other than "strict" and "surrogateescape". +// +// Use the Py_EncodeLocaleEx() function to encode the character string back to +// a byte string. +// +// arg and wstr must not be NULL. +int +_Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, + const char **reason, + int current_locale, _Py_error_handler errors) +{ + assert(arg != NULL); + assert(wstr != NULL); + +#ifdef Py_DEBUG + size_t wlen_canary = (size_t)-2; + if (wlen) { + *wlen = wlen_canary; + } +#endif + + int res = decode_locale_impl(arg, wstr, wlen, + reason, current_locale, errors); + if (res < 0) { + // Error + *wstr = NULL; + if (res == _Py_CODEC_DECODE_ERROR) { +#ifdef Py_DEBUG + assert(wlen == NULL || *wlen != wlen_canary); +#endif + assert(reason == NULL || *reason != NULL); + } + else { + if (wlen) { + *wlen = 0; + } + if (reason) { + *reason = NULL; + } + } + } + else { + // Success + assert(*wstr != NULL); +#ifdef Py_DEBUG + assert(wlen == NULL || *wlen != wlen_canary); +#endif + // *reason is left unchanged + } + return res; +} + + /* Decode a byte string from the locale encoding with the surrogateescape error handler: undecodable bytes are decoded as characters in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate @@ -720,7 +792,7 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, size_t *error_pos) { wchar_t buf[2]; - // The second character is always empty + // The second character is always the NUL character buf[1] = 0; for (size_t i=0; i < len; i++) { @@ -779,6 +851,8 @@ encode_current_locale_impl(const wchar_t *text, const size_t len, // Set *str to a newly allocated string on success. // Return a negative number on error: _Py_CODEC_MEMORY_ERROR, // _Py_CODEC_ENCODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// +// str and output_length must not be NULL static int encode_current_locale(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, @@ -882,38 +956,49 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, #endif /* _Py_FORCE_UTF8_FS_ENCODING */ } -/* Encode a string to the locale encoding. - - Parameters: - - * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead - of PyMem_Malloc(). - * current_locale: if non-zero, use the current LC_CTYPE, otherwise use - Python filesystem encoding. - * errors: error handler like "strict" or "surrogateescape". - Set *str to a newly allocated decoded string and return 0 on success. - Return a negative result on error: - - * _Py_CODEC_MEMORY_ERROR (-1): memory allocation failure - * _Py_CODEC_ENCODE_ERROR (-2): encoding error, set *error_pos - and *reason (if set). - * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3): the 'errors' error handler - is not supported. Most functions only support "strict" and - "surrogateescape". The UTF-8 encoder also supports "surrogatepass". - */ +// Encode a wide string to the locale encoding. +// +// Parameters: +// +// * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead +// of PyMem_Malloc(). +// * current_locale: if non-zero, use the current LC_CTYPE, otherwise use +// Python filesystem encoding. +// * errors: error handler like "strict" or "surrogateescape". +// +// Set *str to a newly allocated decoded string and return 0 on success. +// Return a negative result on error: +// +// * _Py_CODEC_MEMORY_ERROR (-1): memory allocation failure +// * _Py_CODEC_ENCODE_ERROR (-2): encoding error, set *error_pos +// and *reason (if set). +// * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3): the 'errors' error handler +// is not supported. Most functions only support "strict" and +// "surrogateescape". The UTF-8 encoder also supports "surrogatepass". +// +// text, str and output_length must not be NULL. static int encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, const char **reason, int raw_malloc, int current_locale, _Py_error_handler errors) { + assert(text != NULL); + assert(str != NULL); + assert(output_length != NULL); + +#ifdef Py_DEBUG + const size_t output_length_canary = (size_t)-2; + *output_length = output_length_canary; +#endif + int res = encode_locale_inner(text, str, output_length, error_pos, raw_malloc, current_locale, errors); if (res < 0) { - if (output_length) { - *output_length = 0; - } + // Error + *str = NULL; + *output_length = 0; if (res == _Py_CODEC_ENCODE_ERROR) { if (reason) { *reason = "encoding error"; @@ -928,6 +1013,14 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, } } } + else { + assert(*str != NULL); +#ifdef Py_DEBUG + assert(*output_length != output_length_canary); +#endif + // *error_pos is left unchanged + // *reason is left unchanged + } return res; } diff --git a/Python/initconfig.c b/Python/initconfig.c index d5b28c6f3be4bd5..e35e2318f76e7a8 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4325,12 +4325,14 @@ utf8_to_wstr(PyInitConfig *config, const char *str) { wchar_t *wstr; size_t wlen; - int res = _Py_DecodeUTF8Ex(str, strlen(str), &wstr, &wlen, NULL, _Py_ERROR_STRICT); - if (res == -2) { + int res = _Py_DecodeUTF8Ex(str, strlen(str), &wstr, &wlen, + NULL, _Py_ERROR_STRICT); + if (res == _Py_CODEC_DECODE_ERROR) { initconfig_set_error(config, "decoding error"); return NULL; } if (res < 0) { + assert(res == _Py_CODEC_MEMORY_ERROR); config->status = _PyStatus_NO_MEMORY(); return NULL; } From 6933b1df4ab86b73e82b7873bc05ba9760857d71 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:06:59 +0200 Subject: [PATCH 05/16] Remove reason parameter Check error pos in decode tests. --- Include/internal/pycore_fileutils.h | 3 -- Lib/test/test_codecs.py | 29 ++++++++---- Modules/_testinternalcapi.c | 23 ++------- Objects/unicodeobject.c | 46 +++++++----------- Python/fileutils.c | 72 +++++++++-------------------- Python/initconfig.c | 2 +- 6 files changed, 62 insertions(+), 113 deletions(-) diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index 507da4f79d04219..de5dfa8f2ee4ee1 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -28,7 +28,6 @@ PyAPI_FUNC(int) _Py_DecodeLocaleEx( const char *arg, wchar_t **wstr, size_t *wlen, - const char **reason, int current_locale, _Py_error_handler errors); @@ -38,7 +37,6 @@ PyAPI_FUNC(int) _Py_EncodeLocaleEx( char **str, size_t *output_length, size_t *error_pos, - const char **reason, int current_locale, _Py_error_handler errors); @@ -201,7 +199,6 @@ extern int _Py_DecodeUTF8Ex( Py_ssize_t arglen, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors); extern int _Py_EncodeUTF8Ex( diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index e07ac65d98456d8..78d1cb929d87b12 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4096,16 +4096,13 @@ def check_encode_strings(self, errors): if errors == "surrogateescape": with self.assertRaises(ValueError) as cm: self.encode_locale(text) - errmsg = str(cm.exception) - regex = f"Py_EncodeLocale failed: error_pos={error_pos}" - self.assertRegex(errmsg, regex) + errmsg = f"Py_EncodeLocale failed: error_pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) with self.assertRaises(RuntimeError) as cm: self.encode_locale_ex(text, errors) - errmsg = str(cm.exception) - regex = (f"encode error: pos={error_pos}, " - "reason=encoding error") - self.assertRegex(errmsg, regex) + errmsg = f"encode error: pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) else: if errors in ("strict", "surrogateescape"): encoded = self.encode_locale(text) @@ -4171,14 +4168,26 @@ def check_decode_strings(self, errors): try: expected = encoded.decode(self.ENCODING, errors) except UnicodeDecodeError: + for error_pos in range(len(text) - 1, -1, -1): + try: + encoded[:error_pos].decode(self.ENCODING, errors) + except UnicodeDecodeError: + pass + else: + break + else: + self.fail("failed to compute error_pos") + if errors == "surrogateescape": - with self.assertRaises(ValueError): + with self.assertRaises(ValueError) as cm: self.decode_locale(encoded) + errmsg = f"Py_DecodeLocale failed: error_pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) with self.assertRaises(RuntimeError) as cm: self.decode_locale_ex(encoded, errors) - errmsg = str(cm.exception) - self.assertStartsWith(errmsg, "decode error: ") + errmsg = f"decode error: pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) else: if errors == ("strict", "surrogateescape"): decoded = self.decode_locale(encoded) diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index d29777f978bc58b..8bb7d14c7018f4d 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1072,9 +1072,8 @@ encode_locale_ex(PyObject *self, PyObject *args) const size_t output_length_canary = (size_t)-456; size_t output_length = output_length_canary; const char *reason_canary = (const char*)0x123456; - const char *reason = reason_canary; int ret = _Py_EncodeLocaleEx(wstr, - &str, &output_length, &error_pos, &reason, + &str, &output_length, &error_pos, current_locale, error_handler); PyMem_Free(wstr); @@ -1083,7 +1082,6 @@ encode_locale_ex(PyObject *self, PyObject *args) assert(str != NULL && str != str_canary); assert(output_length != output_length_canary); assert(error_pos == error_pos_canary); - assert(reason == reason_canary); res = PyBytes_FromStringAndSize(str, output_length); PyMem_RawFree(str); break; @@ -1091,29 +1089,24 @@ encode_locale_ex(PyObject *self, PyObject *args) assert(str == NULL); assert(output_length == 0); assert(error_pos == 0); - assert(reason == NULL); PyErr_NoMemory(); break; case _Py_CODEC_ENCODE_ERROR: assert(str == NULL); assert(output_length == 0); assert(error_pos != error_pos_canary); - assert(reason != NULL && reason != reason_canary); - PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu, reason=%s", - error_pos, reason); + PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu", error_pos); break; case _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: assert(str == NULL); assert(output_length == 0); assert(error_pos == 0); - assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: assert(str == NULL); assert(output_length == 0); assert(error_pos == 0); - assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unknown error code"); break; } @@ -1139,37 +1132,29 @@ decode_locale_ex(PyObject *self, PyObject *args) wchar_t *wstr = (wchar_t*)wstr_canary; const size_t wlen_canary = (size_t)-123; size_t wlen = wlen_canary; - const char *reason_canary = (const char*)0x123456; - const char *reason = reason_canary; - int ret = _Py_DecodeLocaleEx(str, - &wstr, &wlen, &reason, + int ret = _Py_DecodeLocaleEx(str, &wstr, &wlen, current_locale, error_handler); switch(ret) { case 0: assert(wstr != NULL && wstr != wstr_canary); assert(wlen != wlen_canary); - assert(reason == reason_canary); res = PyUnicode_FromWideChar(wstr, wlen); PyMem_RawFree(wstr); break; case _Py_CODEC_MEMORY_ERROR: assert(wstr == NULL); assert(wlen == 0); - assert(reason == NULL); PyErr_NoMemory(); break; case _Py_CODEC_DECODE_ERROR: assert(wstr == NULL); assert(wlen != wlen_canary); - assert(reason != NULL && reason != reason_canary); - PyErr_Format(PyExc_RuntimeError, "decode error: pos=%zu, reason=%s", - wlen, reason); + PyErr_Format(PyExc_RuntimeError, "decode error: pos=%zu", wlen); break; case _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: assert(wstr == NULL); assert(wlen == 0); - assert(reason == NULL); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 6447fd904293b49..d80e036949e9991 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3766,28 +3766,29 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, char *str; size_t str_len; size_t error_pos; - const char *reason; - int res = _Py_EncodeLocaleEx(wstr, &str, &str_len, &error_pos, &reason, + int res = _Py_EncodeLocaleEx(wstr, &str, &str_len, &error_pos, current_locale, error_handler); PyMem_Free(wstr); if (res != 0) { - if (res == -2) { + if (res == _Py_CODEC_ENCODE_ERROR) { PyObject *exc; + assert(error_pos <= (PY_SSIZE_T_MAX - 1)); exc = PyObject_CallFunction(PyExc_UnicodeEncodeError, "sOnns", "locale", unicode, (Py_ssize_t)error_pos, (Py_ssize_t)(error_pos+1), - reason); + "encoding error"); if (exc != NULL) { PyCodec_StrictErrors(exc); Py_DECREF(exc); } } - else if (res == -3) { + else if (res == _Py_CODEC_UNSUPPORTED_ERROR_HANDLER) { PyErr_SetString(PyExc_ValueError, "unsupported error handler"); } else { + assert(res == _Py_CODEC_MEMORY_ERROR); PyErr_NoMemory(); } return NULL; @@ -3984,26 +3985,27 @@ unicode_decode_locale(const char *str, Py_ssize_t len, wchar_t *wstr; size_t wlen; - const char *reason; - int res = _Py_DecodeLocaleEx(str, &wstr, &wlen, &reason, + int res = _Py_DecodeLocaleEx(str, &wstr, &wlen, current_locale, errors); if (res != 0) { - if (res == -2) { + if (res == _Py_CODEC_DECODE_ERROR) { PyObject *exc; + assert(wlen <= (PY_SSIZE_T_MAX - 1)); exc = PyObject_CallFunction(PyExc_UnicodeDecodeError, "sy#nns", "locale", str, len, (Py_ssize_t)wlen, (Py_ssize_t)(wlen + 1), - reason); + "decoding error"); if (exc != NULL) { PyCodec_StrictErrors(exc); Py_DECREF(exc); } } - else if (res == -3) { + else if (res == _Py_CODEC_UNSUPPORTED_ERROR_HANDLER) { PyErr_SetString(PyExc_ValueError, "unsupported error handler"); } else { + assert(res == _Py_CODEC_MEMORY_ERROR); PyErr_NoMemory(); } return NULL; @@ -5495,14 +5497,13 @@ PyUnicode_DecodeUTF8Stateful(const char *s, // On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. // // On decoding error (if surrogateescape is zero), return -2. If wlen is -// non-NULL, write the start of the illegal byte sequence into *wlen. If reason -// is not NULL, write the decoding error message into *reason. +// non-NULL, write the start of the illegal byte sequence into *wlen. // // Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if 'errors' error handler is not // supported. int _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors) + _Py_error_handler errors) { assert(s != NULL); assert(wstr != NULL); @@ -5581,20 +5582,6 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, } else { PyMem_RawFree(unicode ); - if (reason != NULL) { - switch (ch) { - case 0: - *reason = "unexpected end of data"; - break; - case 1: - *reason = "invalid start byte"; - break; - /* 2, 3, 4 */ - default: - *reason = "invalid continuation byte"; - break; - } - } if (wlen != NULL) { *wlen = s - orig_s; } @@ -5621,9 +5608,8 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, size_t *wlen) { wchar_t *wstr; - int res = _Py_DecodeUTF8Ex(arg, arglen, - &wstr, wlen, - NULL, _Py_ERROR_SURROGATEESCAPE); + int res = _Py_DecodeUTF8Ex(arg, arglen, &wstr, wlen, + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { /* _Py_DecodeUTF8Ex() must support _Py_ERROR_SURROGATEESCAPE */ assert(res != _Py_CODEC_UNSUPPORTED_ERROR_HANDLER); diff --git a/Python/fileutils.c b/Python/fileutils.c index 706f0e0edf6164f..de30d38e73ed945 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -444,10 +444,10 @@ _Py_ResetForceASCII(void) // On success, set *wstr and *wlen (if set) and return 0. // On error, return a negative number: _Py_CODEC_MEMORY_ERROR, // _Py_CODEC_DECODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. -// On decode, set *wlen (if set) and *reason (if set). +// On decode, set *wlen (if set). static int decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors) + _Py_error_handler errors) { assert(arg != NULL); assert(wstr != NULL); @@ -483,9 +483,6 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, if (wlen) { *wlen = in - (unsigned char*)arg; } - if (reason) { - *reason = "decoding error"; - } return _Py_CODEC_DECODE_ERROR; } *out++ = 0xdc00 + ch; @@ -505,10 +502,10 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, // On success, set *wstr and *wlen (if set) and return 0. // On error, return a negative number: _Py_CODEC_MEMORY_ERROR, // _Py_CODEC_DECODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. -// On decode, set *wlen (if set) and *reason (if set). +// On decode, set *wlen (if set). static int decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors) + _Py_error_handler errors) { assert(arg != NULL); assert(wstr != NULL); @@ -614,44 +611,38 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, if (wlen) { *wlen = in - (unsigned char*)arg; } - if (reason) { - *reason = "decoding error"; - } return _Py_CODEC_DECODE_ERROR; #else /* HAVE_MBRTOWC */ /* Cannot use C locale for escaping; manually escape as if charset is ASCII (i.e. escape all bytes > 128. This will still roundtrip correctly in the locale's charset, which must be an ASCII superset. */ - return decode_ascii(arg, wstr, wlen, reason, errors); + return decode_ascii(arg, wstr, wlen, errors); #endif /* HAVE_MBRTOWC */ } static int decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, - const char **reason, int current_locale, _Py_error_handler errors) { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, reason, + return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, errors); #else - return decode_current_locale(arg, wstr, wlen, reason, errors); + return decode_current_locale(arg, wstr, wlen, errors); #endif } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, reason, - errors); + return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); #ifdef MS_WINDOWS use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, reason, - errors); + return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, errors); } #ifdef USE_FORCE_ASCII @@ -661,11 +652,11 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, if (force_ascii) { /* force ASCII encoding to workaround mbstowcs() issue */ - return decode_ascii(arg, wstr, wlen, reason, errors); + return decode_ascii(arg, wstr, wlen, errors); } #endif - return decode_current_locale(arg, wstr, wlen, reason, errors); + return decode_current_locale(arg, wstr, wlen, errors); #endif /* !_Py_FORCE_UTF8_FS_ENCODING */ } @@ -699,7 +690,6 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, // arg and wstr must not be NULL. int _Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, - const char **reason, int current_locale, _Py_error_handler errors) { assert(arg != NULL); @@ -712,8 +702,7 @@ _Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, } #endif - int res = decode_locale_impl(arg, wstr, wlen, - reason, current_locale, errors); + int res = decode_locale_impl(arg, wstr, wlen, current_locale, errors); if (res < 0) { // Error *wstr = NULL; @@ -721,15 +710,11 @@ _Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, #ifdef Py_DEBUG assert(wlen == NULL || *wlen != wlen_canary); #endif - assert(reason == NULL || *reason != NULL); } else { if (wlen) { *wlen = 0; } - if (reason) { - *reason = NULL; - } } } else { @@ -738,7 +723,6 @@ _Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, #ifdef Py_DEBUG assert(wlen == NULL || *wlen != wlen_canary); #endif - // *reason is left unchanged } return res; } @@ -767,8 +751,7 @@ wchar_t* Py_DecodeLocale(const char* arg, size_t *wlen) { wchar_t *wstr; - int res = _Py_DecodeLocaleEx(arg, &wstr, wlen, - NULL, 0, + int res = _Py_DecodeLocaleEx(arg, &wstr, wlen, 0, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { assert(res == -1 || res == -2); @@ -971,8 +954,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, // Return a negative result on error: // // * _Py_CODEC_MEMORY_ERROR (-1): memory allocation failure -// * _Py_CODEC_ENCODE_ERROR (-2): encoding error, set *error_pos -// and *reason (if set). +// * _Py_CODEC_ENCODE_ERROR (-2): encoding error, set *error_pos. // * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3): the 'errors' error handler // is not supported. Most functions only support "strict" and // "surrogateescape". The UTF-8 encoder also supports "surrogatepass". @@ -980,8 +962,8 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, // text, str and output_length must not be NULL. static int encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, - size_t *error_pos, const char **reason, - int raw_malloc, int current_locale, _Py_error_handler errors) + size_t *error_pos, int raw_malloc, + int current_locale, _Py_error_handler errors) { assert(text != NULL); assert(str != NULL); @@ -1000,17 +982,11 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, *str = NULL; *output_length = 0; if (res == _Py_CODEC_ENCODE_ERROR) { - if (reason) { - *reason = "encoding error"; - } } else { if (error_pos) { *error_pos = 0; } - if (reason) { - *reason = NULL; - } } } else { @@ -1019,7 +995,6 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, assert(*output_length != output_length_canary); #endif // *error_pos is left unchanged - // *reason is left unchanged } return res; } @@ -1031,8 +1006,7 @@ encode_locale(const wchar_t *text, size_t *error_pos, char *str; size_t output_length; int res = encode_locale_impl(text, &str, &output_length, - error_pos, NULL, - raw_malloc, current_locale, + error_pos, raw_malloc, current_locale, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { return NULL; @@ -1071,12 +1045,11 @@ _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) int _Py_EncodeLocaleEx(const wchar_t *text, char **str, size_t *output_length, - size_t *error_pos, const char **reason, - int current_locale, _Py_error_handler errors) + size_t *error_pos, int current_locale, + _Py_error_handler errors) { return encode_locale_impl(text, str, output_length, - error_pos, reason, 1, - current_locale, errors); + error_pos, 1, current_locale, errors); } @@ -1115,9 +1088,8 @@ _Py_GetLocaleEncoding(void) } wchar_t *wstr; - int res = decode_current_locale(encoding, &wstr, NULL, - NULL, _Py_ERROR_SURROGATEESCAPE); - if (res < 0) { + if (decode_current_locale(encoding, &wstr, NULL, + _Py_ERROR_SURROGATEESCAPE) < 0) { return NULL; } return wstr; diff --git a/Python/initconfig.c b/Python/initconfig.c index e35e2318f76e7a8..bdc7bcec427ea71 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4326,7 +4326,7 @@ utf8_to_wstr(PyInitConfig *config, const char *str) wchar_t *wstr; size_t wlen; int res = _Py_DecodeUTF8Ex(str, strlen(str), &wstr, &wlen, - NULL, _Py_ERROR_STRICT); + _Py_ERROR_STRICT); if (res == _Py_CODEC_DECODE_ERROR) { initconfig_set_error(config, "decoding error"); return NULL; From 282745355fca91d8c88e1bbb862965a621895ab1 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:19:17 +0200 Subject: [PATCH 06/16] Rename functions * Rename _Py_EncodeLocaleEx() to _Py_EncodeLocale() * Rename _Py_DecodeLocaleEx() to _Py_DecodeLocale() * Rename _Py_DecodeUTF8Ex() to _Py_DecodeUTF8() * Rename _Py_EncodeUTF8Ex() to _Py_EncodeUTF8() --- Include/internal/pycore_fileutils.h | 8 ++++---- Modules/_testinternalcapi.c | 8 ++++---- Objects/unicodeobject.c | 14 +++++++------- Python/fileutils.c | 18 +++++++++--------- Python/initconfig.c | 4 ++-- 5 files changed, 26 insertions(+), 26 deletions(-) diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index de5dfa8f2ee4ee1..5419c24cf059db5 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -24,7 +24,7 @@ extern "C" { PyAPI_FUNC(_Py_error_handler) _Py_GetErrorHandler(const char *errors); // Export for '_testinternalcapi' shared extension -PyAPI_FUNC(int) _Py_DecodeLocaleEx( +PyAPI_FUNC(int) _Py_DecodeLocale( const char *arg, wchar_t **wstr, size_t *wlen, @@ -32,7 +32,7 @@ PyAPI_FUNC(int) _Py_DecodeLocaleEx( _Py_error_handler errors); // Export for '_testinternalcapi' shared extension -PyAPI_FUNC(int) _Py_EncodeLocaleEx( +PyAPI_FUNC(int) _Py_EncodeLocale( const wchar_t *text, char **str, size_t *output_length, @@ -194,14 +194,14 @@ extern int _Py_open_osfhandle(void *handle, int flags); #define _Py_CODEC_ENCODE_ERROR -2 #define _Py_CODEC_UNSUPPORTED_ERROR_HANDLER -3 -extern int _Py_DecodeUTF8Ex( +extern int _Py_DecodeUTF8( const char *arg, Py_ssize_t arglen, wchar_t **wstr, size_t *wlen, _Py_error_handler errors); -extern int _Py_EncodeUTF8Ex( +extern int _Py_EncodeUTF8( const wchar_t *text, char **str, size_t *output_length, diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 8bb7d14c7018f4d..4cb8f44ab4f7fdf 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1046,7 +1046,7 @@ get_getpath_codeobject(PyObject *self, PyObject *Py_UNUSED(args)) { } -// Test _Py_EncodeLocaleEx() +// Test _Py_EncodeLocale() static PyObject * encode_locale_ex(PyObject *self, PyObject *args) { @@ -1072,7 +1072,7 @@ encode_locale_ex(PyObject *self, PyObject *args) const size_t output_length_canary = (size_t)-456; size_t output_length = output_length_canary; const char *reason_canary = (const char*)0x123456; - int ret = _Py_EncodeLocaleEx(wstr, + int ret = _Py_EncodeLocale(wstr, &str, &output_length, &error_pos, current_locale, error_handler); PyMem_Free(wstr); @@ -1114,7 +1114,7 @@ encode_locale_ex(PyObject *self, PyObject *args) } -// Test _Py_DecodeLocaleEx() +// Test _Py_DecodeLocale() static PyObject * decode_locale_ex(PyObject *self, PyObject *args) { @@ -1132,7 +1132,7 @@ decode_locale_ex(PyObject *self, PyObject *args) wchar_t *wstr = (wchar_t*)wstr_canary; const size_t wlen_canary = (size_t)-123; size_t wlen = wlen_canary; - int ret = _Py_DecodeLocaleEx(str, &wstr, &wlen, + int ret = _Py_DecodeLocale(str, &wstr, &wlen, current_locale, error_handler); switch(ret) { diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index d80e036949e9991..9529826438735a1 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3766,7 +3766,7 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, char *str; size_t str_len; size_t error_pos; - int res = _Py_EncodeLocaleEx(wstr, &str, &str_len, &error_pos, + int res = _Py_EncodeLocale(wstr, &str, &str_len, &error_pos, current_locale, error_handler); PyMem_Free(wstr); @@ -3985,7 +3985,7 @@ unicode_decode_locale(const char *str, Py_ssize_t len, wchar_t *wstr; size_t wlen; - int res = _Py_DecodeLocaleEx(str, &wstr, &wlen, + int res = _Py_DecodeLocale(str, &wstr, &wlen, current_locale, errors); if (res != 0) { if (res == _Py_CODEC_DECODE_ERROR) { @@ -5502,7 +5502,7 @@ PyUnicode_DecodeUTF8Stateful(const char *s, // Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if 'errors' error handler is not // supported. int -_Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, +_Py_DecodeUTF8(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, _Py_error_handler errors) { assert(s != NULL); @@ -5608,10 +5608,10 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, size_t *wlen) { wchar_t *wstr; - int res = _Py_DecodeUTF8Ex(arg, arglen, &wstr, wlen, + int res = _Py_DecodeUTF8(arg, arglen, &wstr, wlen, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { - /* _Py_DecodeUTF8Ex() must support _Py_ERROR_SURROGATEESCAPE */ + /* _Py_DecodeUTF8() must support _Py_ERROR_SURROGATEESCAPE */ assert(res != _Py_CODEC_UNSUPPORTED_ERROR_HANDLER); if (wlen) { *wlen = (size_t)res; @@ -5636,7 +5636,7 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, str and output_length must not be NULL */ int -_Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *output_length, +_Py_EncodeUTF8(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int raw_malloc, _Py_error_handler errors) { assert(str != NULL); @@ -15232,7 +15232,7 @@ encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) assert(str != NULL); size_t output_length; - int res = _Py_EncodeUTF8Ex(wstr, str, &output_length, + int res = _Py_EncodeUTF8(wstr, str, &output_length, NULL, 1, _Py_ERROR_STRICT); if (res == -2) { PyErr_Format(PyExc_RuntimeError, "cannot encode %s", name); diff --git a/Python/fileutils.c b/Python/fileutils.c index de30d38e73ed945..bf2b71060221966 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -627,7 +627,7 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); #else return decode_current_locale(arg, wstr, wlen, errors); @@ -635,14 +635,14 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, errors); + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); #ifdef MS_WINDOWS use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, errors); + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); } #ifdef USE_FORCE_ASCII @@ -689,7 +689,7 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, // // arg and wstr must not be NULL. int -_Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, +_Py_DecodeLocale(const char* arg, wchar_t **wstr, size_t *wlen, int current_locale, _Py_error_handler errors) { assert(arg != NULL); @@ -751,7 +751,7 @@ wchar_t* Py_DecodeLocale(const char* arg, size_t *wlen) { wchar_t *wstr; - int res = _Py_DecodeLocaleEx(arg, &wstr, wlen, 0, + int res = _Py_DecodeLocale(arg, &wstr, wlen, 0, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { assert(res == -1 || res == -2); @@ -901,7 +901,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_EncodeUTF8Ex(text, str, output_length, + return _Py_EncodeUTF8(text, str, output_length, error_pos, raw_malloc, errors); #else return encode_current_locale(text, str, output_length, @@ -910,7 +910,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_EncodeUTF8Ex(text, str, output_length, + return _Py_EncodeUTF8(text, str, output_length, error_pos, raw_malloc, errors); #else @@ -919,7 +919,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_EncodeUTF8Ex(text, str, output_length, + return _Py_EncodeUTF8(text, str, output_length, error_pos, raw_malloc, errors); } @@ -1044,7 +1044,7 @@ _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) int -_Py_EncodeLocaleEx(const wchar_t *text, char **str, size_t *output_length, +_Py_EncodeLocale(const wchar_t *text, char **str, size_t *output_length, size_t *error_pos, int current_locale, _Py_error_handler errors) { diff --git a/Python/initconfig.c b/Python/initconfig.c index bdc7bcec427ea71..7a5ff699414c7ea 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4154,7 +4154,7 @@ wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) { char *utf8; size_t utf8_len; - int res = _Py_EncodeUTF8Ex(wstr, &utf8, &utf8_len, + int res = _Py_EncodeUTF8(wstr, &utf8, &utf8_len, NULL, 1, _Py_ERROR_STRICT); if (res == -2) { initconfig_set_error(config, "encoding error"); @@ -4325,7 +4325,7 @@ utf8_to_wstr(PyInitConfig *config, const char *str) { wchar_t *wstr; size_t wlen; - int res = _Py_DecodeUTF8Ex(str, strlen(str), &wstr, &wlen, + int res = _Py_DecodeUTF8(str, strlen(str), &wstr, &wlen, _Py_ERROR_STRICT); if (res == _Py_CODEC_DECODE_ERROR) { initconfig_set_error(config, "decoding error"); From f029aff329b0d791c90b4c12e373c89b653cc227 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:23:20 +0200 Subject: [PATCH 07/16] Update comment --- Objects/unicodeobject.c | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 9529826438735a1..bfaba45634d805f 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -5496,8 +5496,9 @@ PyUnicode_DecodeUTF8Stateful(const char *s, // // On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. // -// On decoding error (if surrogateescape is zero), return -2. If wlen is -// non-NULL, write the start of the illegal byte sequence into *wlen. +// On decoding error (if surrogateescape is zero), return +// _Py_CODEC_DECODE_ERROR. If wlen is non-NULL, write the start of the illegal +// byte sequence into *wlen. // // Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if 'errors' error handler is not // supported. From 2fa4c3ac38a7536442b3d641c78e5307e3675e90 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:31:08 +0200 Subject: [PATCH 08/16] Update tests --- Lib/test/test_codecs.py | 32 +++++++++++++++--------------- Modules/_testinternalcapi.c | 9 ++++----- Python/fileutils.c | 39 ++++++++++++++++++------------------- 3 files changed, 39 insertions(+), 41 deletions(-) diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 78d1cb929d87b12..af267a0c9d3e606 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4071,13 +4071,13 @@ class LocaleCodecTest(unittest.TestCase): BYTES_STRINGS = (b"blatin1:\xa7\xe9", b"b255:\xff") SURROGATES = "\uDC80\uDCFF" - def encode_locale(self, text): + def encode_locale_surrogateescape(self, text): # Test Py_EncodeLocale(): use the "surrogateescape" error handler return _testlimitedcapi.encode_locale(text) def encode_locale_ex(self, text, errors="strict"): - # Test _Py_EncodeLocaleEx() - return _testinternalcapi.EncodeLocaleEx(text, 0, errors) + # Test _Py_EncodeLocale() + return _testinternalcapi.encode_locale(text, 0, errors) def check_encode_strings(self, errors): for text in self.STRINGS: @@ -4095,7 +4095,7 @@ def check_encode_strings(self, errors): if errors == "surrogateescape": with self.assertRaises(ValueError) as cm: - self.encode_locale(text) + self.encode_locale_surrogateescape(text) errmsg = f"Py_EncodeLocale failed: error_pos={error_pos}" self.assertEqual(str(cm.exception), errmsg) @@ -4105,7 +4105,7 @@ def check_encode_strings(self, errors): self.assertEqual(str(cm.exception), errmsg) else: if errors in ("strict", "surrogateescape"): - encoded = self.encode_locale(text) + encoded = self.encode_locale_surrogateescape(text) self.assertEqual(encoded, expected) encoded = self.encode_locale_ex(text, errors) @@ -4134,12 +4134,12 @@ def test_encode_unsupported_error_handler(self): self.encode_locale_ex('', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') - def decode_locale_ex(self, encoded, errors="strict"): - # Test _Py_DecodeLocaleEx() - return _testinternalcapi.DecodeLocaleEx(encoded, 0, errors) + def decode_locale(self, encoded, errors="strict"): + # Test _Py_DecodeLocale() + return _testinternalcapi.decode_locale(encoded, 0, errors) - def decode_locale(self, encoded): - # Test DecodeLocale(): use the "surrogateescape" error handler + def decode_locale_surrogateescape(self, encoded): + # Test Py_DecodeLocale(): use the "surrogateescape" error handler return _testlimitedcapi.decode_locale(encoded) def check_decode_strings(self, errors): @@ -4180,20 +4180,20 @@ def check_decode_strings(self, errors): if errors == "surrogateescape": with self.assertRaises(ValueError) as cm: - self.decode_locale(encoded) + self.decode_locale_surrogateescape(encoded) errmsg = f"Py_DecodeLocale failed: error_pos={error_pos}" self.assertEqual(str(cm.exception), errmsg) with self.assertRaises(RuntimeError) as cm: - self.decode_locale_ex(encoded, errors) + self.decode_locale(encoded, errors) errmsg = f"decode error: pos={error_pos}" self.assertEqual(str(cm.exception), errmsg) else: if errors == ("strict", "surrogateescape"): - decoded = self.decode_locale(encoded) + decoded = self.decode_locale_surrogateescape(encoded) self.assertEqual(decoded, expected) - decoded = self.decode_locale_ex(encoded, errors) + decoded = self.decode_locale(encoded, errors) self.assertEqual(decoded, expected) def test_decode_strict(self): @@ -4204,7 +4204,7 @@ def test_decode_surrogateescape(self): def test_decode_surrogatepass(self): try: - self.decode_locale_ex(b'', 'surrogatepass') + self.decode_locale(b'', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} decoder doesn't support " @@ -4216,7 +4216,7 @@ def test_decode_surrogatepass(self): def test_decode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.decode_locale_ex(b'', 'backslashreplace') + self.decode_locale(b'', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 4cb8f44ab4f7fdf..0d9c686c589aca8 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1048,7 +1048,7 @@ get_getpath_codeobject(PyObject *self, PyObject *Py_UNUSED(args)) { // Test _Py_EncodeLocale() static PyObject * -encode_locale_ex(PyObject *self, PyObject *args) +encode_locale(PyObject *self, PyObject *args) { PyObject *unicode; int current_locale = 0; @@ -1071,7 +1071,6 @@ encode_locale_ex(PyObject *self, PyObject *args) size_t error_pos = error_pos_canary; const size_t output_length_canary = (size_t)-456; size_t output_length = output_length_canary; - const char *reason_canary = (const char*)0x123456; int ret = _Py_EncodeLocale(wstr, &str, &output_length, &error_pos, current_locale, error_handler); @@ -1116,7 +1115,7 @@ encode_locale_ex(PyObject *self, PyObject *args) // Test _Py_DecodeLocale() static PyObject * -decode_locale_ex(PyObject *self, PyObject *args) +decode_locale(PyObject *self, PyObject *args) { char *str; int current_locale = 0; @@ -3358,8 +3357,8 @@ static PyMethodDef module_functions[] = { {"test_bytes_find", test_bytes_find, METH_NOARGS}, {"normalize_path", normalize_path, METH_O, NULL}, {"get_getpath_codeobject", get_getpath_codeobject, METH_NOARGS, NULL}, - {"EncodeLocaleEx", encode_locale_ex, METH_VARARGS}, - {"DecodeLocaleEx", decode_locale_ex, METH_VARARGS}, + {"encode_locale", encode_locale, METH_VARARGS}, + {"decode_locale", decode_locale, METH_VARARGS}, {"set_eval_frame_default", set_eval_frame_default, METH_NOARGS, NULL}, {"set_eval_frame_interp", set_eval_frame_interp, METH_VARARGS, NULL}, {"set_eval_frame_record", set_eval_frame_record, METH_O, NULL}, diff --git a/Python/fileutils.c b/Python/fileutils.c index bf2b71060221966..b3c8de07186a050 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -728,25 +728,24 @@ _Py_DecodeLocale(const char* arg, wchar_t **wstr, size_t *wlen, } -/* Decode a byte string from the locale encoding with the - surrogateescape error handler: undecodable bytes are decoded as characters - in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate - character, escape the bytes using the surrogateescape error handler instead - of decoding them. - - Return a pointer to a newly allocated wide character string, use - PyMem_RawFree() to free the memory. If size is not NULL, write the number of - wide characters excluding the null character into *size - - Return NULL on decoding error or memory allocation error. If *size* is not - NULL, *size is set to (size_t)-1 on memory error or set to (size_t)-2 on - decoding error. - - Decoding errors should never happen, unless there is a bug in the C - library. - - Use the Py_EncodeLocale() function to encode the character string back to a - byte string. */ +// Decode a byte string from the locale encoding with the +// surrogateescape error handler: undecodable bytes are decoded as characters +// in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate +// character, escape the bytes using the surrogateescape error handler instead +// of decoding them. +// +// Return a pointer to a newly allocated wide character string, use +// PyMem_RawFree() to free the memory. If wlen is not NULL, write the number of +// wide characters excluding the null character into *wlen. +// +// Return NULL on decoding error or memory allocation error. If wlen is not +// NULL, *wlen is set to (size_t)_Py_CODEC_MEMORY_ERROR on memory error or set +// to (size_t)_Py_CODEC_DECODE_ERROR on decoding error. +// +// Decoding errors should never happen, unless there is a bug in the C library. +// +// Use the Py_EncodeLocale() function to encode the character string back to a +// byte string. wchar_t* Py_DecodeLocale(const char* arg, size_t *wlen) { @@ -754,7 +753,7 @@ Py_DecodeLocale(const char* arg, size_t *wlen) int res = _Py_DecodeLocale(arg, &wstr, wlen, 0, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { - assert(res == -1 || res == -2); + assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_DECODE_ERROR); if (wlen != NULL) { *wlen = (size_t)res; } From 0caf0ce3a0ddd58e7735f6286001a35b83b6acd0 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:33:51 +0200 Subject: [PATCH 09/16] indent --- Modules/_testinternalcapi.c | 6 +++--- Objects/unicodeobject.c | 13 ++++++------- Python/fileutils.c | 18 ++++++++---------- Python/initconfig.c | 5 ++--- 4 files changed, 19 insertions(+), 23 deletions(-) diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 0d9c686c589aca8..96285902bc8835a 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1072,8 +1072,8 @@ encode_locale(PyObject *self, PyObject *args) const size_t output_length_canary = (size_t)-456; size_t output_length = output_length_canary; int ret = _Py_EncodeLocale(wstr, - &str, &output_length, &error_pos, - current_locale, error_handler); + &str, &output_length, &error_pos, + current_locale, error_handler); PyMem_Free(wstr); switch(ret) { @@ -1132,7 +1132,7 @@ decode_locale(PyObject *self, PyObject *args) const size_t wlen_canary = (size_t)-123; size_t wlen = wlen_canary; int ret = _Py_DecodeLocale(str, &wstr, &wlen, - current_locale, error_handler); + current_locale, error_handler); switch(ret) { case 0: diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index bfaba45634d805f..ec9287d560d5a0e 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3767,7 +3767,7 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, size_t str_len; size_t error_pos; int res = _Py_EncodeLocale(wstr, &str, &str_len, &error_pos, - current_locale, error_handler); + current_locale, error_handler); PyMem_Free(wstr); if (res != 0) { @@ -3985,8 +3985,7 @@ unicode_decode_locale(const char *str, Py_ssize_t len, wchar_t *wstr; size_t wlen; - int res = _Py_DecodeLocale(str, &wstr, &wlen, - current_locale, errors); + int res = _Py_DecodeLocale(str, &wstr, &wlen, current_locale, errors); if (res != 0) { if (res == _Py_CODEC_DECODE_ERROR) { PyObject *exc; @@ -5504,7 +5503,7 @@ PyUnicode_DecodeUTF8Stateful(const char *s, // supported. int _Py_DecodeUTF8(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, - _Py_error_handler errors) + _Py_error_handler errors) { assert(s != NULL); assert(wstr != NULL); @@ -5610,7 +5609,7 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, { wchar_t *wstr; int res = _Py_DecodeUTF8(arg, arglen, &wstr, wlen, - _Py_ERROR_SURROGATEESCAPE); + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { /* _Py_DecodeUTF8() must support _Py_ERROR_SURROGATEESCAPE */ assert(res != _Py_CODEC_UNSUPPORTED_ERROR_HANDLER); @@ -5638,7 +5637,7 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, */ int _Py_EncodeUTF8(const wchar_t *text, char **str, size_t *output_length, - size_t *error_pos, int raw_malloc, _Py_error_handler errors) + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { assert(str != NULL); assert(output_length != NULL); @@ -15234,7 +15233,7 @@ encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) size_t output_length; int res = _Py_EncodeUTF8(wstr, str, &output_length, - NULL, 1, _Py_ERROR_STRICT); + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { PyErr_Format(PyExc_RuntimeError, "cannot encode %s", name); return -1; diff --git a/Python/fileutils.c b/Python/fileutils.c index b3c8de07186a050..e8bda50aba24c7f 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -627,8 +627,7 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, - errors); + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); #else return decode_current_locale(arg, wstr, wlen, errors); #endif @@ -690,7 +689,7 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, // arg and wstr must not be NULL. int _Py_DecodeLocale(const char* arg, wchar_t **wstr, size_t *wlen, - int current_locale, _Py_error_handler errors) + int current_locale, _Py_error_handler errors) { assert(arg != NULL); assert(wstr != NULL); @@ -751,7 +750,7 @@ Py_DecodeLocale(const char* arg, size_t *wlen) { wchar_t *wstr; int res = _Py_DecodeLocale(arg, &wstr, wlen, 0, - _Py_ERROR_SURROGATEESCAPE); + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_DECODE_ERROR); if (wlen != NULL) { @@ -901,7 +900,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE return _Py_EncodeUTF8(text, str, output_length, - error_pos, raw_malloc, errors); + error_pos, raw_malloc, errors); #else return encode_current_locale(text, str, output_length, error_pos, raw_malloc, errors); @@ -910,8 +909,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, #ifdef _Py_FORCE_UTF8_FS_ENCODING return _Py_EncodeUTF8(text, str, output_length, - error_pos, - raw_malloc, errors); + error_pos, raw_malloc, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); #ifdef MS_WINDOWS @@ -919,7 +917,7 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, #endif if (use_utf8) { return _Py_EncodeUTF8(text, str, output_length, - error_pos, raw_malloc, errors); + error_pos, raw_malloc, errors); } #ifdef USE_FORCE_ASCII @@ -1044,8 +1042,8 @@ _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) int _Py_EncodeLocale(const wchar_t *text, char **str, size_t *output_length, - size_t *error_pos, int current_locale, - _Py_error_handler errors) + size_t *error_pos, int current_locale, + _Py_error_handler errors) { return encode_locale_impl(text, str, output_length, error_pos, 1, current_locale, errors); diff --git a/Python/initconfig.c b/Python/initconfig.c index 7a5ff699414c7ea..363739a1a1b43c9 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4155,7 +4155,7 @@ wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) char *utf8; size_t utf8_len; int res = _Py_EncodeUTF8(wstr, &utf8, &utf8_len, - NULL, 1, _Py_ERROR_STRICT); + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { initconfig_set_error(config, "encoding error"); return NULL; @@ -4325,8 +4325,7 @@ utf8_to_wstr(PyInitConfig *config, const char *str) { wchar_t *wstr; size_t wlen; - int res = _Py_DecodeUTF8(str, strlen(str), &wstr, &wlen, - _Py_ERROR_STRICT); + int res = _Py_DecodeUTF8(str, strlen(str), &wstr, &wlen, _Py_ERROR_STRICT); if (res == _Py_CODEC_DECODE_ERROR) { initconfig_set_error(config, "decoding error"); return NULL; From 7e0dfeb8461cc9aa28a7c231b62d9d3b88f475d2 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:37:39 +0200 Subject: [PATCH 10/16] Mark _Py_EncodeLocaleRaw() as static --- Include/internal/pycore_fileutils.h | 4 ---- Python/fileutils.c | 7 +++---- 2 files changed, 3 insertions(+), 8 deletions(-) diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index 5419c24cf059db5..a765eb5fe2d3228 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -40,10 +40,6 @@ PyAPI_FUNC(int) _Py_EncodeLocale( int current_locale, _Py_error_handler errors); -extern char* _Py_EncodeLocaleRaw( - const wchar_t *text, - size_t *error_pos); - extern PyObject* _Py_device_encoding(int); #if defined(MS_WINDOWS) || defined(__APPLE__) diff --git a/Python/fileutils.c b/Python/fileutils.c index e8bda50aba24c7f..c89eea63a24adf5 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -683,7 +683,7 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, // Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3) if the 'errors' error // handler is not supported: other than "strict" and "surrogateescape". // -// Use the Py_EncodeLocaleEx() function to encode the character string back to +// Use the _Py_EncodeLocale() function to encode the character string back to // a byte string. // // arg and wstr must not be NULL. @@ -749,8 +749,7 @@ wchar_t* Py_DecodeLocale(const char* arg, size_t *wlen) { wchar_t *wstr; - int res = _Py_DecodeLocale(arg, &wstr, wlen, 0, - _Py_ERROR_SURROGATEESCAPE); + int res = _Py_DecodeLocale(arg, &wstr, wlen, 0, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_DECODE_ERROR); if (wlen != NULL) { @@ -1033,7 +1032,7 @@ Py_EncodeLocale(const wchar_t *text, size_t *error_pos) /* Similar to Py_EncodeLocale(), but result must be freed by PyMem_RawFree() instead of PyMem_Free(). */ -char* +static char* _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) { return encode_locale(text, error_pos, 1, 0); From 23c1ff3bbfbb33c0c0ac731887702ef6ae81e556 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 18:50:19 +0200 Subject: [PATCH 11/16] Self review --- Doc/c-api/sys.rst | 43 ++++---- Lib/test/test_codecs.py | 26 ++--- Modules/_testinternalcapi.c | 7 +- Modules/_testlimitedcapi/codec.c | 29 ++++-- Objects/unicodeobject.c | 32 +++--- Python/fileutils.c | 169 +++++++++++++++---------------- 6 files changed, 150 insertions(+), 156 deletions(-) diff --git a/Doc/c-api/sys.rst b/Doc/c-api/sys.rst index 0a446f86e22eaf6..c2eeecdc313b658 100644 --- a/Doc/c-api/sys.rst +++ b/Doc/c-api/sys.rst @@ -152,28 +152,29 @@ Operating System Utilities ` and so that the LC_CTYPE locale is properly configured: see the :c:func:`Py_PreInitialize` function. - Decode a byte string from the :term:`filesystem encoding and error handler`. - If the error handler is :ref:`surrogateescape error handler - `, undecodable bytes are decoded as characters in range - U+DC80..U+DCFF; and if a byte sequence can be decoded as a surrogate - character, the bytes are escaped using the surrogateescape error handler - instead of decoding them. + Decode a byte string from the :term:`filesystem encoding ` with the :ref:`surrogateescape error handler + ` error handler. + + Undecodable bytes are decoded as characters in range U+DC80..U+DCFF. If a + byte sequence can be decoded as a surrogate character, escape the bytes + using the surrogateescape error handler instead of decoding them. Return a pointer to a newly allocated wide character string, use :c:func:`PyMem_RawFree` to free the memory. If size is not ``NULL``, write the number of wide characters excluding the null character into ``*size`` - Return ``NULL`` on decoding error or memory allocation error. If *size* is - not ``NULL``, ``*size`` is set to ``(size_t)-1`` on memory error or set to - ``(size_t)-2`` on decoding error. + On memory allocation failure, set *\*size* to ``(size_t)-1`` and return + ``NULL``. + + On decode error, set *\*size* to ``(size_t)-2`` and return ``NULL``. + Decoding errors should never happen, unless there is a bug in the C + library. The :term:`filesystem encoding and error handler` are selected by :c:func:`PyConfig_Read`: see :c:member:`~PyConfig.filesystem_encoding` and :c:member:`~PyConfig.filesystem_errors` members of :c:type:`PyConfig`. - Decoding errors should never happen, unless there is a bug in the C - library. - Use the :c:func:`Py_EncodeLocale` function to encode the character string back to a byte string. @@ -195,17 +196,19 @@ Operating System Utilities .. c:function:: char* Py_EncodeLocale(const wchar_t *text, size_t *error_pos) - Encode a wide character string to the :term:`filesystem encoding and error - handler`. If the error handler is :ref:`surrogateescape error handler - `, surrogate characters in the range U+DC80..U+DCFF are - converted to bytes 0x80..0xFF. + Encode a wide character string to the :term:`filesystem encoding ` with the :ref:`surrogateescape error handler + `. Surrogate characters in the range U+DC80..U+DCFF are + encoded to bytes 0x80..0xFF. Return a pointer to a newly allocated byte string, use :c:func:`PyMem_Free` - to free the memory. Return ``NULL`` on encoding error or memory allocation - error. + to free the memory. + + On memory allocation failure, set *\*error_pos* to ``(size_t)-1`` and return + ``NULL``. - If error_pos is not ``NULL``, ``*error_pos`` is set to ``(size_t)-1`` on - success, or set to the index of the invalid character on encoding error. + On encoding error, set *\*error_pos* to the index of the first invalid + character and return ``NULL``. The :term:`filesystem encoding and error handler` are selected by :c:func:`PyConfig_Read`: see :c:member:`~PyConfig.filesystem_encoding` and diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index af267a0c9d3e606..3da19c93e58a6b1 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4072,11 +4072,12 @@ class LocaleCodecTest(unittest.TestCase): SURROGATES = "\uDC80\uDCFF" def encode_locale_surrogateescape(self, text): - # Test Py_EncodeLocale(): use the "surrogateescape" error handler + # Test public Py_EncodeLocale() C API: + # use the "surrogateescape" error handler return _testlimitedcapi.encode_locale(text) - def encode_locale_ex(self, text, errors="strict"): - # Test _Py_EncodeLocale() + def encode_locale(self, text, errors="strict"): + # Test the internal _Py_EncodeLocale() C API return _testinternalcapi.encode_locale(text, 0, errors) def check_encode_strings(self, errors): @@ -4094,13 +4095,13 @@ def check_encode_strings(self, errors): self.fail("failed to compute error_pos") if errors == "surrogateescape": - with self.assertRaises(ValueError) as cm: + with self.assertRaises(RuntimeError) as cm: self.encode_locale_surrogateescape(text) - errmsg = f"Py_EncodeLocale failed: error_pos={error_pos}" + errmsg = f"encode error: pos={error_pos}" self.assertEqual(str(cm.exception), errmsg) with self.assertRaises(RuntimeError) as cm: - self.encode_locale_ex(text, errors) + self.encode_locale(text, errors) errmsg = f"encode error: pos={error_pos}" self.assertEqual(str(cm.exception), errmsg) else: @@ -4108,7 +4109,7 @@ def check_encode_strings(self, errors): encoded = self.encode_locale_surrogateescape(text) self.assertEqual(encoded, expected) - encoded = self.encode_locale_ex(text, errors) + encoded = self.encode_locale(text, errors) self.assertEqual(encoded, expected) def test_encode_strict(self): @@ -4119,7 +4120,7 @@ def test_encode_surrogateescape(self): def test_encode_surrogatepass(self): try: - self.encode_locale_ex('', 'surrogatepass') + self.encode_locale('', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} encoder doesn't support " @@ -4131,15 +4132,16 @@ def test_encode_surrogatepass(self): def test_encode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.encode_locale_ex('', 'backslashreplace') + self.encode_locale('', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') def decode_locale(self, encoded, errors="strict"): - # Test _Py_DecodeLocale() + # Test the internal _Py_DecodeLocale() C API return _testinternalcapi.decode_locale(encoded, 0, errors) def decode_locale_surrogateescape(self, encoded): - # Test Py_DecodeLocale(): use the "surrogateescape" error handler + # Test the public Py_DecodeLocale() C API: + # use the "surrogateescape" error handler return _testlimitedcapi.decode_locale(encoded) def check_decode_strings(self, errors): @@ -4179,7 +4181,7 @@ def check_decode_strings(self, errors): self.fail("failed to compute error_pos") if errors == "surrogateescape": - with self.assertRaises(ValueError) as cm: + with self.assertRaises(RuntimeError) as cm: self.decode_locale_surrogateescape(encoded) errmsg = f"Py_DecodeLocale failed: error_pos={error_pos}" self.assertEqual(str(cm.exception), errmsg) diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 96285902bc8835a..0d81e5613350e9c 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1103,10 +1103,7 @@ encode_locale(PyObject *self, PyObject *args) PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: - assert(str == NULL); - assert(output_length == 0); - assert(error_pos == 0); - PyErr_SetString(PyExc_ValueError, "unknown error code"); + PyErr_SetString(PyExc_SystemError, "unknown error code"); break; } return res; @@ -1157,7 +1154,7 @@ decode_locale(PyObject *self, PyObject *args) PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: - PyErr_SetString(PyExc_ValueError, "unknown error code"); + PyErr_SetString(PyExc_SystemError, "unknown error code"); break; } return res; diff --git a/Modules/_testlimitedcapi/codec.c b/Modules/_testlimitedcapi/codec.c index bc2531fb01f00f8..38febf70456dd19 100644 --- a/Modules/_testlimitedcapi/codec.c +++ b/Modules/_testlimitedcapi/codec.c @@ -25,25 +25,28 @@ decode_locale(PyObject *Py_UNUSED(module), PyObject *arg) return NULL; } - size_t wstr_len = (size_t)-123; - wchar_t *wstr = Py_DecodeLocale(str, &wstr_len); + const size_t size_canary = (size_t)-123; + size_t size = size_canary; + wchar_t *wstr = Py_DecodeLocale(str, &size); if (str == NULL) { - if (wstr_len == (size_t)-1) { + if (size == (size_t)-1) { PyErr_NoMemory(); } - else if (wstr_len == (size_t)-2) { - PyErr_SetString(PyExc_ValueError, "decode error"); + else if (size == (size_t)-2) { + PyErr_SetString(PyExc_RuntimeError, "decode error"); } else { PyErr_Format(PyExc_SystemError, "unknown Py_DecodeLocale() return value: %zd", - (Py_ssize_t)wstr_len); + (Py_ssize_t)size); } return NULL; } + assert(wstr != NULL); + assert(size != size_canary); - PyObject *result = PyUnicode_FromWideChar(wstr, wstr_len); + PyObject *result = PyUnicode_FromWideChar(wstr, size); PyMem_RawFree(wstr); return result; } @@ -69,10 +72,14 @@ encode_locale(PyObject *Py_UNUSED(module), PyObject *arg) PyMem_Free(wstr); if (str == NULL) { - assert(error_pos != error_pos_canary); - return PyErr_Format(PyExc_ValueError, - "Py_EncodeLocale failed: error_pos=%zd", - error_pos); + if (error_pos == (size_t)-1) { + return PyErr_NoMemory(); + } + else { + assert(error_pos != error_pos_canary); + return PyErr_Format(PyExc_RuntimeError, + "encode error: pos=%zd", error_pos); + } } assert(error_pos == error_pos_canary); diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index ec9287d560d5a0e..f9cfc84c3ab8979 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3773,12 +3773,12 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, if (res != 0) { if (res == _Py_CODEC_ENCODE_ERROR) { PyObject *exc; - assert(error_pos <= (PY_SSIZE_T_MAX - 1)); + assert(error_pos <= (size_t)(PY_SSIZE_T_MAX - 1)); exc = PyObject_CallFunction(PyExc_UnicodeEncodeError, "sOnns", "locale", unicode, (Py_ssize_t)error_pos, (Py_ssize_t)(error_pos+1), - "encoding error"); + "encode error"); if (exc != NULL) { PyCodec_StrictErrors(exc); Py_DECREF(exc); @@ -3989,12 +3989,12 @@ unicode_decode_locale(const char *str, Py_ssize_t len, if (res != 0) { if (res == _Py_CODEC_DECODE_ERROR) { PyObject *exc; - assert(wlen <= (PY_SSIZE_T_MAX - 1)); + assert(wlen <= (size_t)(PY_SSIZE_T_MAX - 1)); exc = PyObject_CallFunction(PyExc_UnicodeDecodeError, "sy#nns", "locale", str, len, (Py_ssize_t)wlen, (Py_ssize_t)(wlen + 1), - "decoding error"); + "decode error"); if (exc != NULL) { PyCodec_StrictErrors(exc); Py_DECREF(exc); @@ -5495,24 +5495,20 @@ PyUnicode_DecodeUTF8Stateful(const char *s, // // On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. // -// On decoding error (if surrogateescape is zero), return -// _Py_CODEC_DECODE_ERROR. If wlen is non-NULL, write the start of the illegal -// byte sequence into *wlen. +// On decoding error (if errors is "strict"), return _Py_CODEC_DECODE_ERROR. +// If wlen is non-NULL, write the start of the illegal byte sequence into +// *wlen. // -// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if 'errors' error handler is not +// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if errors error handler is not // supported. int _Py_DecodeUTF8(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, _Py_error_handler errors) { + assert(0 <= size); assert(s != NULL); assert(wstr != NULL); - const char *orig_s = s; - const char *e; - wchar_t *unicode; - Py_ssize_t outpos; - int surrogateescape = 0; int surrogatepass = 0; switch (errors) @@ -5531,18 +5527,18 @@ _Py_DecodeUTF8(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, /* Note: size will always be longer than the resulting Unicode character count */ - if (PY_SSIZE_T_MAX / (Py_ssize_t)sizeof(wchar_t) - 1 < size) { + if ((size_t)PY_SSIZE_T_MAX / sizeof(wchar_t) - 1 < (size_t)size) { return _Py_CODEC_MEMORY_ERROR; } - - unicode = PyMem_RawMalloc((size + 1) * sizeof(wchar_t)); + wchar_t *unicode = PyMem_RawMalloc((size + 1) * sizeof(wchar_t)); if (!unicode) { return _Py_CODEC_MEMORY_ERROR; } /* Unpack UTF-8 encoded data */ - e = s + size; - outpos = 0; + const char *orig_s = s; + const char *e = s + size; + Py_ssize_t outpos = 0; while (s < e) { Py_UCS4 ch; #if SIZEOF_WCHAR_T == 4 diff --git a/Python/fileutils.c b/Python/fileutils.c index c89eea63a24adf5..bdf23bc8cac0cd1 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -52,9 +52,9 @@ int _Py_open_cloexec_works = -1; #endif // wcstombs() error -static const size_t ENCODE_ERROR = ((size_t)-1); +static const size_t ENCODE_ERROR = (size_t)-1; // mbstowcs() and mbrtowc() errors -static const size_t DECODE_ERROR = ((size_t)-1); +static const size_t DECODE_ERROR = (size_t)-1; #ifdef HAVE_MBRTOWC static const size_t INCOMPLETE_CHARACTER = (size_t)-2; #endif @@ -437,9 +437,9 @@ _Py_ResetForceASCII(void) #if !defined(HAVE_MBRTOWC) || defined(USE_FORCE_ASCII) -// Decode a bytes string from ASCII. If 'errors' is "surrogateescape", handle -// non-ASCII with "surrogateescape" error handler and so the output is -// non-ASCII. +// Decode a bytes string from the ASCII encoding. If 'errors' is +// "surrogateescape", decode non-ASCII bytes with "surrogateescape" error +// handler and so the output is non-ASCII. // // On success, set *wstr and *wlen (if set) and return 0. // On error, return a negative number: _Py_CODEC_MEMORY_ERROR, @@ -498,7 +498,7 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, } #endif /* !HAVE_MBRTOWC */ -// Decode a bytes string from the current locale. +// Decode a bytes string from the current locale encoding. // On success, set *wstr and *wlen (if set) and return 0. // On error, return a negative number: _Py_CODEC_MEMORY_ERROR, // _Py_CODEC_DECODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. @@ -510,15 +510,6 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, assert(arg != NULL); assert(wstr != NULL); - wchar_t *res; - size_t argsize; - size_t count; -#ifdef HAVE_MBRTOWC - unsigned char *in; - wchar_t *out; - mbstate_t mbs; -#endif - int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { // Only support "strict" and "surrogateescape" @@ -530,10 +521,11 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, * mbstowcs which does not count the characters that * would result from conversion. Use an upper bound. */ - argsize = strlen(arg); + size_t argsize = strlen(arg); #else - argsize = _Py_mbstowcs(NULL, arg, 0); + size_t argsize = _Py_mbstowcs(NULL, arg, 0); #endif + wchar_t *res; if (argsize != DECODE_ERROR) { if (argsize > PY_SSIZE_T_MAX / sizeof(wchar_t) - 1) { return _Py_CODEC_MEMORY_ERROR; @@ -543,8 +535,11 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, return _Py_CODEC_MEMORY_ERROR; } - count = _Py_mbstowcs(res, arg, argsize + 1); + // +1 to write also the trailing NUL character + size_t count = _Py_mbstowcs(res, arg, argsize + 1); if (count != DECODE_ERROR) { + // Success + assert(count == argsize); *wstr = res; if (wlen != NULL) { *wlen = count; @@ -569,8 +564,9 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, return _Py_CODEC_MEMORY_ERROR; } - in = (unsigned char*)arg; - out = res; + unsigned char *in = (unsigned char*)arg; + wchar_t *out = res; + mbstate_t mbs; memset(&mbs, 0, sizeof mbs); while (argsize) { size_t converted = _Py_mbrtowc(out, (char*)in, argsize, &mbs); @@ -602,6 +598,7 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, } if (wlen != NULL) { *wlen = out - res; + assert(res[*wlen] == 0); } *wstr = res; return 0; @@ -662,11 +659,13 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, // Decode a byte string from the locale encoding. // -// Use the strict error handler if 'surrogateescape' is zero. Use the -// surrogateescape error handler if 'surrogateescape' is non-zero: undecodable -// bytes are decoded as characters in range U+DC80..U+DCFF. If a byte sequence -// can be decoded as a surrogate character, escape the bytes using the -// surrogateescape error handler instead of decoding them. +// Supported error handlers are _Py_ERROR_STRICT and _Py_ERROR_SURROGATEESCAPE. +// The UTF-8 decoder also supports _Py_ERROR_SURROGATEPASS. +// +// If errors is _Py_ERROR_SURROGATEESCAPE, undecodable bytes are decoded as +// characters in range U+DC80..U+DCFF. If a byte sequence can be decoded as a +// surrogate character, escape the bytes using the surrogateescape error +// handler instead of decoding them. // // On success, return 0 and write the newly allocated wide character string into // *wstr (use PyMem_RawFree() to free the memory). If wlen is not NULL, write @@ -674,13 +673,13 @@ decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, // // On error, return a negative number. // -// On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). +// On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. // -// On decoding error, return _Py_CODEC_DECODE_ERROR (-2). If wlen is not NULL, +// On decoding error, return _Py_CODEC_DECODE_ERROR. If wlen is not NULL, // write the start of invalid byte sequence in the input string into *wlen. If // reason is not NULL, write the decoding error message into *reason. // -// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3) if the 'errors' error +// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if the 'errors' error // handler is not supported: other than "strict" and "surrogateescape". // // Use the _Py_EncodeLocale() function to encode the character string back to @@ -720,7 +719,9 @@ _Py_DecodeLocale(const char* arg, wchar_t **wstr, size_t *wlen, // Success assert(*wstr != NULL); #ifdef Py_DEBUG - assert(wlen == NULL || *wlen != wlen_canary); + if (wlen != NULL) { + assert(*wlen == wcslen(*wstr)); + } #endif } return res; @@ -728,32 +729,31 @@ _Py_DecodeLocale(const char* arg, wchar_t **wstr, size_t *wlen, // Decode a byte string from the locale encoding with the -// surrogateescape error handler: undecodable bytes are decoded as characters +// surrogateescape error handler. Undecodable bytes are decoded as characters // in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate // character, escape the bytes using the surrogateescape error handler instead // of decoding them. // // Return a pointer to a newly allocated wide character string, use -// PyMem_RawFree() to free the memory. If wlen is not NULL, write the number of -// wide characters excluding the null character into *wlen. +// PyMem_RawFree() to free the memory. If size is not NULL, write the number of +// wide characters excluding the null character into *size. // -// Return NULL on decoding error or memory allocation error. If wlen is not -// NULL, *wlen is set to (size_t)_Py_CODEC_MEMORY_ERROR on memory error or set -// to (size_t)_Py_CODEC_DECODE_ERROR on decoding error. +// On memory allocation failure, set *size to (size_t)-1 and return NULL. // +// On decode error, set *size to (size_t)-2 and return NULL. // Decoding errors should never happen, unless there is a bug in the C library. // // Use the Py_EncodeLocale() function to encode the character string back to a // byte string. wchar_t* -Py_DecodeLocale(const char* arg, size_t *wlen) +Py_DecodeLocale(const char* arg, size_t *size) { wchar_t *wstr; - int res = _Py_DecodeLocale(arg, &wstr, wlen, 0, _Py_ERROR_SURROGATEESCAPE); + int res = _Py_DecodeLocale(arg, &wstr, size, 0, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_DECODE_ERROR); - if (wlen != NULL) { - *wlen = (size_t)res; + if (size != NULL) { + *size = (size_t)res; } return NULL; } @@ -761,7 +761,7 @@ Py_DecodeLocale(const char* arg, size_t *wlen) } -// If bytes is NULL, return the length in 'bytes' of the encoded string. +// If bytes is NULL, compute the length in 'bytes' of the encoded string. // Otherwise, encode the wide string into 'bytes', decrements 'size', and // return 0 on success. // On encoding error, set 'error_pos' (if set) and return ENCODE_ERROR. @@ -848,23 +848,18 @@ encode_current_locale(const wchar_t *text, char **str, size_t *output_length, } // First, compute the output length - char *result = NULL; const size_t len = wcslen(text); // Sanity check, it cannot happen in practice assert(len != ENCODE_ERROR); size_t size = encode_current_locale_impl(text, len, surrogateescape, NULL, 0, error_pos); if (size == ENCODE_ERROR) { - goto encode_error; + return _Py_CODEC_ENCODE_ERROR; } *output_length = size; - if (raw_malloc) { - result = PyMem_RawMalloc(size + 1); - } - else { - result = PyMem_Malloc(size + 1); - } + char *result; + result = raw_malloc ? PyMem_RawMalloc(size + 1) : PyMem_Malloc(size + 1); if (result == NULL) { return _Py_CODEC_MEMORY_ERROR; } @@ -873,21 +868,18 @@ encode_current_locale(const wchar_t *text, char **str, size_t *output_length, size = encode_current_locale_impl(text, len, surrogateescape, result, size, error_pos); if (size == ENCODE_ERROR) { - goto encode_error; + if (raw_malloc) { + PyMem_RawFree(result); + } + else { + PyMem_Free(result); + } + return _Py_CODEC_ENCODE_ERROR; } assert(size == 0); *str = result; return 0; - -encode_error: - if (raw_malloc) { - PyMem_RawFree(result); - } - else { - PyMem_Free(result); - } - return _Py_CODEC_ENCODE_ERROR; } @@ -944,16 +936,17 @@ encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, // of PyMem_Malloc(). // * current_locale: if non-zero, use the current LC_CTYPE, otherwise use // Python filesystem encoding. -// * errors: error handler like "strict" or "surrogateescape". +// * errors: supported error handlers are _Py_ERROR_STRICT and +// _Py_ERROR_SURROGATEESCAPE. The UTF-8 encoder also supports +// _Py_ERROR_SURROGATEPASS. // // Set *str to a newly allocated decoded string and return 0 on success. // Return a negative result on error: // -// * _Py_CODEC_MEMORY_ERROR (-1): memory allocation failure -// * _Py_CODEC_ENCODE_ERROR (-2): encoding error, set *error_pos. -// * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER (-3): the 'errors' error handler -// is not supported. Most functions only support "strict" and -// "surrogateescape". The UTF-8 encoder also supports "surrogatepass". +// * _Py_CODEC_MEMORY_ERROR: memory allocation failure +// * _Py_CODEC_ENCODE_ERROR: encoding error, set *error_pos. +// * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: the 'errors' error handler +// is not supported. // // text, str and output_length must not be NULL. static int @@ -965,11 +958,6 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, assert(str != NULL); assert(output_length != NULL); -#ifdef Py_DEBUG - const size_t output_length_canary = (size_t)-2; - *output_length = output_length_canary; -#endif - int res = encode_locale_inner(text, str, output_length, error_pos, raw_malloc, current_locale, errors); @@ -977,9 +965,7 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, // Error *str = NULL; *output_length = 0; - if (res == _Py_CODEC_ENCODE_ERROR) { - } - else { + if (res != _Py_CODEC_ENCODE_ERROR) { if (error_pos) { *error_pos = 0; } @@ -987,9 +973,7 @@ encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, } else { assert(*str != NULL); -#ifdef Py_DEBUG - assert(*output_length != output_length_canary); -#endif + assert(*output_length == strlen(*str)); // *error_pos is left unchanged } return res; @@ -1000,29 +984,34 @@ encode_locale(const wchar_t *text, size_t *error_pos, int raw_malloc, int current_locale) { char *str; - size_t output_length; - int res = encode_locale_impl(text, &str, &output_length, + size_t unused_output_length; + int res = encode_locale_impl(text, &str, &unused_output_length, error_pos, raw_malloc, current_locale, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { + assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_ENCODE_ERROR); + if (res == _Py_CODEC_MEMORY_ERROR && error_pos != NULL) { + *error_pos = (size_t)-1; + } return NULL; } - assert(strlen(str) == output_length); return str; } -/* Encode a wide character string to the locale encoding with the - surrogateescape error handler: surrogate characters in the range - U+DC80..U+DCFF are converted to bytes 0x80..0xFF. - - Return a pointer to a newly allocated byte string, use PyMem_Free() to free - the memory. Return NULL on encoding or memory allocation error. - - If error_pos is not NULL, *error_pos is set to (size_t)-1 on success, or set - to the index of the invalid character on encoding error. - - Use the Py_DecodeLocale() function to decode the bytes string back to a wide - character string. */ +// Encode a wide character string to the locale encoding with the +// surrogateescape error handler. Surrogate characters in the range +// U+DC80..U+DCFF are encoded to bytes 0x80..0xFF. +// +// Return a pointer to a newly allocated byte string, use PyMem_Free() to free +// the memory. Return NULL on encoding or memory allocation error. +// +// On memory allocation failure, set *error_pos to (size_t)-1 and return NULL. +// +// On encoding error, set *error_pos to the index of the first invalid +// character and return NULL. +// +// Use the Py_DecodeLocale() function to decode the bytes string back to a wide +// character string. char* Py_EncodeLocale(const wchar_t *text, size_t *error_pos) { From 56d285bf4561157c08309d9e2e0817a30771eb83 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 20:32:40 +0200 Subject: [PATCH 12/16] Remove redundant assertion --- Objects/unicodeobject.c | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index f9cfc84c3ab8979..157e005c3c0b600 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -15227,8 +15227,8 @@ encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) { assert(str != NULL); - size_t output_length; - int res = _Py_EncodeUTF8(wstr, str, &output_length, + size_t unused_output_length; + int res = _Py_EncodeUTF8(wstr, str, &unused_output_length, NULL, 1, _Py_ERROR_STRICT); if (res == -2) { PyErr_Format(PyExc_RuntimeError, "cannot encode %s", name); @@ -15238,7 +15238,6 @@ encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) PyErr_NoMemory(); return -1; } - assert(strlen(*str) == output_length); return 0; } From 9bbafbbd4b9cac2831d0ca8d4dbc9d7daffe899b Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 21:06:16 +0200 Subject: [PATCH 13/16] Test embedded NUL byte/character --- Lib/test/test_codecs.py | 13 +++++++++++-- Modules/_testinternalcapi.c | 10 +++++++--- Modules/_testlimitedcapi/codec.c | 8 ++++++-- 3 files changed, 24 insertions(+), 7 deletions(-) diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 3da19c93e58a6b1..eabae1f04ed98e8 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4067,7 +4067,8 @@ class LocaleCodecTest(unittest.TestCase): STRINGS = ("ascii", "ulatin1:\xa7\xe9", "u255:\xff", "UCS:\xe9\u20ac\U0010ffff", - "surrogates:\uDC80\uDCFF") + "surrogates:\uDC80\uDCFF", + "embed\0char") BYTES_STRINGS = (b"blatin1:\xa7\xe9", b"b255:\xff") SURROGATES = "\uDC80\uDCFF" @@ -4085,6 +4086,10 @@ def check_encode_strings(self, errors): with self.subTest(text=text): try: expected = text.encode(self.ENCODING, errors) + if b"\0" in expected: + # Py_EncodeLocale() and _Py_EncodeLocale() + # truncate the input string at the first NUL character + expected = expected.partition(b'\0')[0] except UnicodeEncodeError: for error_pos in range(len(text)): try: @@ -4169,8 +4174,12 @@ def check_decode_strings(self, errors): with self.subTest(encoded=encoded): try: expected = encoded.decode(self.ENCODING, errors) + if "\0" in expected: + # Py_DecodeLocale() and _Py_DecodeLocale() truncate + # the input string at the first NUL byte + expected = expected.partition('\0')[0] except UnicodeDecodeError: - for error_pos in range(len(text) - 1, -1, -1): + for error_pos in range(len(encoded) - 1, -1, -1): try: encoded[:error_pos].decode(self.ENCODING, errors) except UnicodeDecodeError: diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 0d81e5613350e9c..e1f87eb3b41c9f8 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1059,7 +1059,9 @@ encode_locale(PyObject *self, PyObject *args) return NULL; } - wchar_t *wstr = PyUnicode_AsWideCharString(unicode, NULL); + // Accept embedded null characters + Py_ssize_t unused_wlen; + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, &unused_wlen); if (wstr == NULL) { return NULL; } @@ -1115,11 +1117,13 @@ static PyObject * decode_locale(PyObject *self, PyObject *args) { char *str; + Py_ssize_t unused_len; int current_locale = 0; PyObject *res = NULL; const char *errors = NULL; - - if (!PyArg_ParseTuple(args, "y|is", &str, ¤t_locale, &errors)) { + // Accept embedded null bytes in str + if (!PyArg_ParseTuple(args, "y#|is", + &str, &unused_len, ¤t_locale, &errors)) { return NULL; } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); diff --git a/Modules/_testlimitedcapi/codec.c b/Modules/_testlimitedcapi/codec.c index 38febf70456dd19..61e5d4708c71d68 100644 --- a/Modules/_testlimitedcapi/codec.c +++ b/Modules/_testlimitedcapi/codec.c @@ -21,7 +21,9 @@ static PyObject * decode_locale(PyObject *Py_UNUSED(module), PyObject *arg) { const char *str; - if (PyArg_Parse(arg, "y", &str) < 0) { + Py_ssize_t unused_len; + // Accept embedded null bytes + if (PyArg_Parse(arg, "y#", &str, &unused_len) < 0) { return NULL; } @@ -61,7 +63,9 @@ encode_locale(PyObject *Py_UNUSED(module), PyObject *arg) return NULL; } - wchar_t *wstr = PyUnicode_AsWideCharString(unicode, NULL); + // Accept embedded null characters + Py_ssize_t unused_wlen; + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, &unused_wlen); if (wstr == NULL) { return NULL; } From 8bb54eba3206d51ad98ee28eda21e71d9cc40522 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 21:08:05 +0200 Subject: [PATCH 14/16] Update comments --- Lib/test/test_codecs.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index eabae1f04ed98e8..206f9bf8d0cf5c9 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4061,7 +4061,11 @@ def test_pickle(self): @unittest.skipIf(_testinternalcapi is None, 'need _testinternalcapi module') class LocaleCodecTest(unittest.TestCase): """ - Test indirectly _Py_DecodeUTF8Ex() and _Py_EncodeUTF8Ex(). + Test public Py_EncodeLocale() and Py_DecodeLocale() C API. + + Test internal _Py_EncodeLocale() and _Py_DecodeLocale() C API. + + Test indirectly _Py_DecodeUTF8() and _Py_EncodeUTF8(). """ ENCODING = sys.getfilesystemencoding() STRINGS = ("ascii", "ulatin1:\xa7\xe9", @@ -4078,7 +4082,7 @@ def encode_locale_surrogateescape(self, text): return _testlimitedcapi.encode_locale(text) def encode_locale(self, text, errors="strict"): - # Test the internal _Py_EncodeLocale() C API + # Test internal _Py_EncodeLocale() C API return _testinternalcapi.encode_locale(text, 0, errors) def check_encode_strings(self, errors): @@ -4141,7 +4145,7 @@ def test_encode_unsupported_error_handler(self): self.assertEqual(str(cm.exception), 'unsupported error handler') def decode_locale(self, encoded, errors="strict"): - # Test the internal _Py_DecodeLocale() C API + # Test internal _Py_DecodeLocale() C API return _testinternalcapi.decode_locale(encoded, 0, errors) def decode_locale_surrogateescape(self, encoded): From bbd3b91de5c220508a9562defae5dcb8baa51144 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 21:15:15 +0200 Subject: [PATCH 15/16] Remove unused variable --- Lib/test/test_codecs.py | 1 - 1 file changed, 1 deletion(-) diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 206f9bf8d0cf5c9..2715ad6d7b3f8a3 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4074,7 +4074,6 @@ class LocaleCodecTest(unittest.TestCase): "surrogates:\uDC80\uDCFF", "embed\0char") BYTES_STRINGS = (b"blatin1:\xa7\xe9", b"b255:\xff") - SURROGATES = "\uDC80\uDCFF" def encode_locale_surrogateescape(self, text): # Test public Py_EncodeLocale() C API: From 210231410d3eb7729d4d6136dffb2c96d3a55927 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 21:24:44 +0200 Subject: [PATCH 16/16] Fix the doc/comments --- Doc/c-api/sys.rst | 4 ++-- Python/fileutils.c | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/Doc/c-api/sys.rst b/Doc/c-api/sys.rst index c2eeecdc313b658..2f59e8498b88a20 100644 --- a/Doc/c-api/sys.rst +++ b/Doc/c-api/sys.rst @@ -154,7 +154,7 @@ Operating System Utilities Decode a byte string from the :term:`filesystem encoding ` with the :ref:`surrogateescape error handler - ` error handler. + `. Undecodable bytes are decoded as characters in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate character, escape the bytes @@ -207,7 +207,7 @@ Operating System Utilities On memory allocation failure, set *\*error_pos* to ``(size_t)-1`` and return ``NULL``. - On encoding error, set *\*error_pos* to the index of the first invalid + On encoding error, set *\*error_pos* to the index of the first unencodable character and return ``NULL``. The :term:`filesystem encoding and error handler` are selected by diff --git a/Python/fileutils.c b/Python/fileutils.c index bdf23bc8cac0cd1..8ed88047b35e4ff 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -1007,7 +1007,7 @@ encode_locale(const wchar_t *text, size_t *error_pos, // // On memory allocation failure, set *error_pos to (size_t)-1 and return NULL. // -// On encoding error, set *error_pos to the index of the first invalid +// On encoding error, set *error_pos to the index of the first unencodable // character and return NULL. // // Use the Py_DecodeLocale() function to decode the bytes string back to a wide