diff --git a/Doc/c-api/sys.rst b/Doc/c-api/sys.rst index 0a446f86e22eaf..2f59e8498b88a2 100644 --- a/Doc/c-api/sys.rst +++ b/Doc/c-api/sys.rst @@ -152,28 +152,29 @@ Operating System Utilities ` and so that the LC_CTYPE locale is properly configured: see the :c:func:`Py_PreInitialize` function. - Decode a byte string from the :term:`filesystem encoding and error handler`. - If the error handler is :ref:`surrogateescape error handler - `, undecodable bytes are decoded as characters in range - U+DC80..U+DCFF; and if a byte sequence can be decoded as a surrogate - character, the bytes are escaped using the surrogateescape error handler - instead of decoding them. + Decode a byte string from the :term:`filesystem encoding ` with the :ref:`surrogateescape error handler + `. + + Undecodable bytes are decoded as characters in range U+DC80..U+DCFF. If a + byte sequence can be decoded as a surrogate character, escape the bytes + using the surrogateescape error handler instead of decoding them. Return a pointer to a newly allocated wide character string, use :c:func:`PyMem_RawFree` to free the memory. If size is not ``NULL``, write the number of wide characters excluding the null character into ``*size`` - Return ``NULL`` on decoding error or memory allocation error. If *size* is - not ``NULL``, ``*size`` is set to ``(size_t)-1`` on memory error or set to - ``(size_t)-2`` on decoding error. + On memory allocation failure, set *\*size* to ``(size_t)-1`` and return + ``NULL``. + + On decode error, set *\*size* to ``(size_t)-2`` and return ``NULL``. + Decoding errors should never happen, unless there is a bug in the C + library. The :term:`filesystem encoding and error handler` are selected by :c:func:`PyConfig_Read`: see :c:member:`~PyConfig.filesystem_encoding` and :c:member:`~PyConfig.filesystem_errors` members of :c:type:`PyConfig`. - Decoding errors should never happen, unless there is a bug in the C - library. - Use the :c:func:`Py_EncodeLocale` function to encode the character string back to a byte string. @@ -195,17 +196,19 @@ Operating System Utilities .. c:function:: char* Py_EncodeLocale(const wchar_t *text, size_t *error_pos) - Encode a wide character string to the :term:`filesystem encoding and error - handler`. If the error handler is :ref:`surrogateescape error handler - `, surrogate characters in the range U+DC80..U+DCFF are - converted to bytes 0x80..0xFF. + Encode a wide character string to the :term:`filesystem encoding ` with the :ref:`surrogateescape error handler + `. Surrogate characters in the range U+DC80..U+DCFF are + encoded to bytes 0x80..0xFF. Return a pointer to a newly allocated byte string, use :c:func:`PyMem_Free` - to free the memory. Return ``NULL`` on encoding error or memory allocation - error. + to free the memory. + + On memory allocation failure, set *\*error_pos* to ``(size_t)-1`` and return + ``NULL``. - If error_pos is not ``NULL``, ``*error_pos`` is set to ``(size_t)-1`` on - success, or set to the index of the invalid character on encoding error. + On encoding error, set *\*error_pos* to the index of the first unencodable + character and return ``NULL``. The :term:`filesystem encoding and error handler` are selected by :c:func:`PyConfig_Read`: see :c:member:`~PyConfig.filesystem_encoding` and diff --git a/Include/internal/pycore_fileutils.h b/Include/internal/pycore_fileutils.h index 128790823aa879..a765eb5fe2d322 100644 --- a/Include/internal/pycore_fileutils.h +++ b/Include/internal/pycore_fileutils.h @@ -24,27 +24,22 @@ extern "C" { PyAPI_FUNC(_Py_error_handler) _Py_GetErrorHandler(const char *errors); // Export for '_testinternalcapi' shared extension -PyAPI_FUNC(int) _Py_DecodeLocaleEx( +PyAPI_FUNC(int) _Py_DecodeLocale( const char *arg, wchar_t **wstr, size_t *wlen, - const char **reason, int current_locale, _Py_error_handler errors); // Export for '_testinternalcapi' shared extension -PyAPI_FUNC(int) _Py_EncodeLocaleEx( +PyAPI_FUNC(int) _Py_EncodeLocale( const wchar_t *text, char **str, + size_t *output_length, size_t *error_pos, - const char **reason, int current_locale, _Py_error_handler errors); -extern char* _Py_EncodeLocaleRaw( - const wchar_t *text, - size_t *error_pos); - extern PyObject* _Py_device_encoding(int); #if defined(MS_WINDOWS) || defined(__APPLE__) @@ -190,19 +185,23 @@ extern int _Py_open_osfhandle(void *handle, int flags); ? _PyStatus_ERR("cannot decode " NAME) \ : _PyStatus_NO_MEMORY() -extern int _Py_DecodeUTF8Ex( +#define _Py_CODEC_MEMORY_ERROR -1 +#define _Py_CODEC_DECODE_ERROR -2 +#define _Py_CODEC_ENCODE_ERROR -2 +#define _Py_CODEC_UNSUPPORTED_ERROR_HANDLER -3 + +extern int _Py_DecodeUTF8( const char *arg, Py_ssize_t arglen, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors); -extern int _Py_EncodeUTF8Ex( +extern int _Py_EncodeUTF8( const wchar_t *text, char **str, + size_t *output_length, size_t *error_pos, - const char **reason, int raw_malloc, _Py_error_handler errors); diff --git a/Lib/test/test_codecs.py b/Lib/test/test_codecs.py index 1916ee5507e972..2715ad6d7b3f8a 100644 --- a/Lib/test/test_codecs.py +++ b/Lib/test/test_codecs.py @@ -4057,34 +4057,67 @@ def test_pickle(self): pickle.dumps(sr, proto) +@unittest.skipIf(_testlimitedcapi is None, 'need _testlimitedcapi module') @unittest.skipIf(_testinternalcapi is None, 'need _testinternalcapi module') class LocaleCodecTest(unittest.TestCase): """ - Test indirectly _Py_DecodeUTF8Ex() and _Py_EncodeUTF8Ex(). + Test public Py_EncodeLocale() and Py_DecodeLocale() C API. + + Test internal _Py_EncodeLocale() and _Py_DecodeLocale() C API. + + Test indirectly _Py_DecodeUTF8() and _Py_EncodeUTF8(). """ ENCODING = sys.getfilesystemencoding() STRINGS = ("ascii", "ulatin1:\xa7\xe9", "u255:\xff", "UCS:\xe9\u20ac\U0010ffff", - "surrogates:\uDC80\uDCFF") + "surrogates:\uDC80\uDCFF", + "embed\0char") BYTES_STRINGS = (b"blatin1:\xa7\xe9", b"b255:\xff") - SURROGATES = "\uDC80\uDCFF" - def encode(self, text, errors="strict"): - return _testinternalcapi.EncodeLocaleEx(text, 0, errors) + def encode_locale_surrogateescape(self, text): + # Test public Py_EncodeLocale() C API: + # use the "surrogateescape" error handler + return _testlimitedcapi.encode_locale(text) + + def encode_locale(self, text, errors="strict"): + # Test internal _Py_EncodeLocale() C API + return _testinternalcapi.encode_locale(text, 0, errors) def check_encode_strings(self, errors): for text in self.STRINGS: with self.subTest(text=text): try: expected = text.encode(self.ENCODING, errors) + if b"\0" in expected: + # Py_EncodeLocale() and _Py_EncodeLocale() + # truncate the input string at the first NUL character + expected = expected.partition(b'\0')[0] except UnicodeEncodeError: + for error_pos in range(len(text)): + try: + text[error_pos].encode(self.ENCODING, errors) + except UnicodeEncodeError: + break + else: + self.fail("failed to compute error_pos") + + if errors == "surrogateescape": + with self.assertRaises(RuntimeError) as cm: + self.encode_locale_surrogateescape(text) + errmsg = f"encode error: pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) + with self.assertRaises(RuntimeError) as cm: - self.encode(text, errors) - errmsg = str(cm.exception) - self.assertRegex(errmsg, r"encode error: pos=[0-9]+, reason=") + self.encode_locale(text, errors) + errmsg = f"encode error: pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) else: - encoded = self.encode(text, errors) + if errors in ("strict", "surrogateescape"): + encoded = self.encode_locale_surrogateescape(text) + self.assertEqual(encoded, expected) + + encoded = self.encode_locale(text, errors) self.assertEqual(encoded, expected) def test_encode_strict(self): @@ -4095,7 +4128,7 @@ def test_encode_surrogateescape(self): def test_encode_surrogatepass(self): try: - self.encode('', 'surrogatepass') + self.encode_locale('', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} encoder doesn't support " @@ -4107,11 +4140,17 @@ def test_encode_surrogatepass(self): def test_encode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.encode('', 'backslashreplace') + self.encode_locale('', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') - def decode(self, encoded, errors="strict"): - return _testinternalcapi.DecodeLocaleEx(encoded, 0, errors) + def decode_locale(self, encoded, errors="strict"): + # Test internal _Py_DecodeLocale() C API + return _testinternalcapi.decode_locale(encoded, 0, errors) + + def decode_locale_surrogateescape(self, encoded): + # Test the public Py_DecodeLocale() C API: + # use the "surrogateescape" error handler + return _testlimitedcapi.decode_locale(encoded) def check_decode_strings(self, errors): is_utf8 = (self.ENCODING == "utf-8") @@ -4138,13 +4177,37 @@ def check_decode_strings(self, errors): with self.subTest(encoded=encoded): try: expected = encoded.decode(self.ENCODING, errors) + if "\0" in expected: + # Py_DecodeLocale() and _Py_DecodeLocale() truncate + # the input string at the first NUL byte + expected = expected.partition('\0')[0] except UnicodeDecodeError: + for error_pos in range(len(encoded) - 1, -1, -1): + try: + encoded[:error_pos].decode(self.ENCODING, errors) + except UnicodeDecodeError: + pass + else: + break + else: + self.fail("failed to compute error_pos") + + if errors == "surrogateescape": + with self.assertRaises(RuntimeError) as cm: + self.decode_locale_surrogateescape(encoded) + errmsg = f"Py_DecodeLocale failed: error_pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) + with self.assertRaises(RuntimeError) as cm: - self.decode(encoded, errors) - errmsg = str(cm.exception) - self.assertStartsWith(errmsg, "decode error: ") + self.decode_locale(encoded, errors) + errmsg = f"decode error: pos={error_pos}" + self.assertEqual(str(cm.exception), errmsg) else: - decoded = self.decode(encoded, errors) + if errors == ("strict", "surrogateescape"): + decoded = self.decode_locale_surrogateescape(encoded) + self.assertEqual(decoded, expected) + + decoded = self.decode_locale(encoded, errors) self.assertEqual(decoded, expected) def test_decode_strict(self): @@ -4155,7 +4218,7 @@ def test_decode_surrogateescape(self): def test_decode_surrogatepass(self): try: - self.decode(b'', 'surrogatepass') + self.decode_locale(b'', 'surrogatepass') except ValueError as exc: if str(exc) == 'unsupported error handler': self.skipTest(f"{self.ENCODING!r} decoder doesn't support " @@ -4167,7 +4230,7 @@ def test_decode_surrogatepass(self): def test_decode_unsupported_error_handler(self): with self.assertRaises(ValueError) as cm: - self.decode(b'', 'backslashreplace') + self.decode_locale(b'', 'backslashreplace') self.assertEqual(str(cm.exception), 'unsupported error handler') diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index a2ce266eb35a65..e1f87eb3b41c9f 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -1046,92 +1046,119 @@ get_getpath_codeobject(PyObject *self, PyObject *Py_UNUSED(args)) { } +// Test _Py_EncodeLocale() static PyObject * -encode_locale_ex(PyObject *self, PyObject *args) +encode_locale(PyObject *self, PyObject *args) { PyObject *unicode; int current_locale = 0; - wchar_t *wstr; PyObject *res = NULL; const char *errors = NULL; if (!PyArg_ParseTuple(args, "U|is", &unicode, ¤t_locale, &errors)) { return NULL; } - wstr = PyUnicode_AsWideCharString(unicode, NULL); + + // Accept embedded null characters + Py_ssize_t unused_wlen; + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, &unused_wlen); if (wstr == NULL) { return NULL; } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); - char *str = NULL; - size_t error_pos; - const char *reason = NULL; - int ret = _Py_EncodeLocaleEx(wstr, - &str, &error_pos, &reason, - current_locale, error_handler); + const char *str_canary = (const char*)0x1234; + char *str = (char*)str_canary; + size_t error_pos_canary = (size_t)-123; + size_t error_pos = error_pos_canary; + const size_t output_length_canary = (size_t)-456; + size_t output_length = output_length_canary; + int ret = _Py_EncodeLocale(wstr, + &str, &output_length, &error_pos, + current_locale, error_handler); PyMem_Free(wstr); switch(ret) { case 0: - res = PyBytes_FromString(str); + assert(str != NULL && str != str_canary); + assert(output_length != output_length_canary); + assert(error_pos == error_pos_canary); + res = PyBytes_FromStringAndSize(str, output_length); PyMem_RawFree(str); break; - case -1: + case _Py_CODEC_MEMORY_ERROR: + assert(str == NULL); + assert(output_length == 0); + assert(error_pos == 0); PyErr_NoMemory(); break; - case -2: - PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu, reason=%s", - error_pos, reason); + case _Py_CODEC_ENCODE_ERROR: + assert(str == NULL); + assert(output_length == 0); + assert(error_pos != error_pos_canary); + PyErr_Format(PyExc_RuntimeError, "encode error: pos=%zu", error_pos); break; - case -3: + case _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: + assert(str == NULL); + assert(output_length == 0); + assert(error_pos == 0); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: - PyErr_SetString(PyExc_ValueError, "unknown error code"); + PyErr_SetString(PyExc_SystemError, "unknown error code"); break; } return res; } +// Test _Py_DecodeLocale() static PyObject * -decode_locale_ex(PyObject *self, PyObject *args) +decode_locale(PyObject *self, PyObject *args) { char *str; + Py_ssize_t unused_len; int current_locale = 0; PyObject *res = NULL; const char *errors = NULL; - - if (!PyArg_ParseTuple(args, "y|is", &str, ¤t_locale, &errors)) { + // Accept embedded null bytes in str + if (!PyArg_ParseTuple(args, "y#|is", + &str, &unused_len, ¤t_locale, &errors)) { return NULL; } _Py_error_handler error_handler = _Py_GetErrorHandler(errors); - wchar_t *wstr = NULL; - size_t wlen = 0; - const char *reason = NULL; - int ret = _Py_DecodeLocaleEx(str, - &wstr, &wlen, &reason, - current_locale, error_handler); + const wchar_t *wstr_canary = (const wchar_t*)0x12345; + wchar_t *wstr = (wchar_t*)wstr_canary; + const size_t wlen_canary = (size_t)-123; + size_t wlen = wlen_canary; + int ret = _Py_DecodeLocale(str, &wstr, &wlen, + current_locale, error_handler); switch(ret) { case 0: + assert(wstr != NULL && wstr != wstr_canary); + assert(wlen != wlen_canary); res = PyUnicode_FromWideChar(wstr, wlen); PyMem_RawFree(wstr); break; - case -1: + case _Py_CODEC_MEMORY_ERROR: + assert(wstr == NULL); + assert(wlen == 0); PyErr_NoMemory(); break; - case -2: - PyErr_Format(PyExc_RuntimeError, "decode error: pos=%zu, reason=%s", - wlen, reason); + case _Py_CODEC_DECODE_ERROR: + assert(wstr == NULL); + assert(wlen != wlen_canary); + PyErr_Format(PyExc_RuntimeError, "decode error: pos=%zu", wlen); break; - case -3: + case _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: + assert(wstr == NULL); + assert(wlen == 0); PyErr_SetString(PyExc_ValueError, "unsupported error handler"); break; default: - PyErr_SetString(PyExc_ValueError, "unknown error code"); + PyErr_SetString(PyExc_SystemError, "unknown error code"); break; } return res; @@ -3331,8 +3358,8 @@ static PyMethodDef module_functions[] = { {"test_bytes_find", test_bytes_find, METH_NOARGS}, {"normalize_path", normalize_path, METH_O, NULL}, {"get_getpath_codeobject", get_getpath_codeobject, METH_NOARGS, NULL}, - {"EncodeLocaleEx", encode_locale_ex, METH_VARARGS}, - {"DecodeLocaleEx", decode_locale_ex, METH_VARARGS}, + {"encode_locale", encode_locale, METH_VARARGS}, + {"decode_locale", decode_locale, METH_VARARGS}, {"set_eval_frame_default", set_eval_frame_default, METH_NOARGS, NULL}, {"set_eval_frame_interp", set_eval_frame_interp, METH_VARARGS, NULL}, {"set_eval_frame_record", set_eval_frame_record, METH_O, NULL}, diff --git a/Modules/_testlimitedcapi/codec.c b/Modules/_testlimitedcapi/codec.c index 44eecf99f678bb..61e5d4708c71d6 100644 --- a/Modules/_testlimitedcapi/codec.c +++ b/Modules/_testlimitedcapi/codec.c @@ -2,8 +2,8 @@ #ifdef Py_GIL_DISABLED # define Py_TARGET_ABI3T 0x030f0000 #else - // Need limited C API version 3.5 for PyCodec_NameReplaceErrors() -# define Py_LIMITED_API 0x03050000 + // Need limited C API version 3.13 for PyMem_RawFree() +# define Py_LIMITED_API 0x030d0000 #endif #include "parts.h" @@ -15,16 +15,93 @@ codec_namereplace_errors(PyObject *Py_UNUSED(module), PyObject *exc) return PyCodec_NameReplaceErrors(exc); } + +// Test Py_DecodeLocale() +static PyObject * +decode_locale(PyObject *Py_UNUSED(module), PyObject *arg) +{ + const char *str; + Py_ssize_t unused_len; + // Accept embedded null bytes + if (PyArg_Parse(arg, "y#", &str, &unused_len) < 0) { + return NULL; + } + + const size_t size_canary = (size_t)-123; + size_t size = size_canary; + wchar_t *wstr = Py_DecodeLocale(str, &size); + + if (str == NULL) { + if (size == (size_t)-1) { + PyErr_NoMemory(); + } + else if (size == (size_t)-2) { + PyErr_SetString(PyExc_RuntimeError, "decode error"); + } + else { + PyErr_Format(PyExc_SystemError, + "unknown Py_DecodeLocale() return value: %zd", + (Py_ssize_t)size); + } + return NULL; + } + assert(wstr != NULL); + assert(size != size_canary); + + PyObject *result = PyUnicode_FromWideChar(wstr, size); + PyMem_RawFree(wstr); + return result; +} + + +// Test Py_EncodeLocale() +static PyObject * +encode_locale(PyObject *Py_UNUSED(module), PyObject *arg) +{ + PyObject *unicode; + if (PyArg_Parse(arg, "U", &unicode) < 0) { + return NULL; + } + + // Accept embedded null characters + Py_ssize_t unused_wlen; + wchar_t *wstr = PyUnicode_AsWideCharString(unicode, &unused_wlen); + if (wstr == NULL) { + return NULL; + } + + const size_t error_pos_canary = (size_t)-123; + size_t error_pos = error_pos_canary; + char *str = Py_EncodeLocale(wstr, &error_pos); + PyMem_Free(wstr); + + if (str == NULL) { + if (error_pos == (size_t)-1) { + return PyErr_NoMemory(); + } + else { + assert(error_pos != error_pos_canary); + return PyErr_Format(PyExc_RuntimeError, + "encode error: pos=%zd", error_pos); + } + } + assert(error_pos == error_pos_canary); + + PyObject *result = PyBytes_FromString(str); + PyMem_Free(str); + return result; +} + + static PyMethodDef test_methods[] = { {"codec_namereplace_errors", codec_namereplace_errors, METH_O}, + {"decode_locale", decode_locale, METH_O}, + {"encode_locale", encode_locale, METH_O}, {NULL}, }; int _PyTestLimitedCAPI_Init_Codec(PyObject *module) { - if (PyModule_AddFunctions(module, test_methods) < 0) { - return -1; - } - return 0; + return PyModule_AddFunctions(module, test_methods); } diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index e630dbb77f8725..157e005c3c0b60 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -3764,35 +3764,37 @@ unicode_encode_locale(PyObject *unicode, _Py_error_handler error_handler, } char *str; + size_t str_len; size_t error_pos; - const char *reason; - int res = _Py_EncodeLocaleEx(wstr, &str, &error_pos, &reason, - current_locale, error_handler); + int res = _Py_EncodeLocale(wstr, &str, &str_len, &error_pos, + current_locale, error_handler); PyMem_Free(wstr); if (res != 0) { - if (res == -2) { + if (res == _Py_CODEC_ENCODE_ERROR) { PyObject *exc; + assert(error_pos <= (size_t)(PY_SSIZE_T_MAX - 1)); exc = PyObject_CallFunction(PyExc_UnicodeEncodeError, "sOnns", "locale", unicode, (Py_ssize_t)error_pos, (Py_ssize_t)(error_pos+1), - reason); + "encode error"); if (exc != NULL) { PyCodec_StrictErrors(exc); Py_DECREF(exc); } } - else if (res == -3) { + else if (res == _Py_CODEC_UNSUPPORTED_ERROR_HANDLER) { PyErr_SetString(PyExc_ValueError, "unsupported error handler"); } else { + assert(res == _Py_CODEC_MEMORY_ERROR); PyErr_NoMemory(); } return NULL; } - PyObject *bytes = PyBytes_FromString(str); + PyObject *bytes = PyBytes_FromStringAndSize(str, str_len); PyMem_RawFree(str); return bytes; } @@ -3983,26 +3985,26 @@ unicode_decode_locale(const char *str, Py_ssize_t len, wchar_t *wstr; size_t wlen; - const char *reason; - int res = _Py_DecodeLocaleEx(str, &wstr, &wlen, &reason, - current_locale, errors); + int res = _Py_DecodeLocale(str, &wstr, &wlen, current_locale, errors); if (res != 0) { - if (res == -2) { + if (res == _Py_CODEC_DECODE_ERROR) { PyObject *exc; + assert(wlen <= (size_t)(PY_SSIZE_T_MAX - 1)); exc = PyObject_CallFunction(PyExc_UnicodeDecodeError, "sy#nns", "locale", str, len, (Py_ssize_t)wlen, (Py_ssize_t)(wlen + 1), - reason); + "decode error"); if (exc != NULL) { PyCodec_StrictErrors(exc); Py_DECREF(exc); } } - else if (res == -3) { + else if (res == _Py_CODEC_UNSUPPORTED_ERROR_HANDLER) { PyErr_SetString(PyExc_ValueError, "unsupported error handler"); } else { + assert(res == _Py_CODEC_MEMORY_ERROR); PyErr_NoMemory(); } return NULL; @@ -5481,26 +5483,31 @@ PyUnicode_DecodeUTF8Stateful(const char *s, } -/* UTF-8 decoder: use surrogateescape error handler if 'surrogateescape' is - non-zero, use strict error handler otherwise. - - On success, write a pointer to a newly allocated wide character string into - *wstr (use PyMem_RawFree() to free the memory) and write the output length - (in number of wchar_t units) into *wlen (if wlen is set). - - On memory allocation failure, return -1. - - On decoding error (if surrogateescape is zero), return -2. If wlen is - non-NULL, write the start of the illegal byte sequence into *wlen. If reason - is not NULL, write the decoding error message into *reason. */ +// UTF-8 decoder. +// +// Supported error handlers: "strict", "surrogateescape" and "surrogatepass". +// +// On success, write a pointer to a newly allocated wide character string into +// *wstr (use PyMem_RawFree() to free the memory), write the output length +// (in number of wchar_t units) into *wlen (if wlen is set), and return 0. +// +// On error, return a negative number. +// +// On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. +// +// On decoding error (if errors is "strict"), return _Py_CODEC_DECODE_ERROR. +// If wlen is non-NULL, write the start of the illegal byte sequence into +// *wlen. +// +// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if errors error handler is not +// supported. int -_Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors) +_Py_DecodeUTF8(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, + _Py_error_handler errors) { - const char *orig_s = s; - const char *e; - wchar_t *unicode; - Py_ssize_t outpos; + assert(0 <= size); + assert(s != NULL); + assert(wstr != NULL); int surrogateescape = 0; int surrogatepass = 0; @@ -5515,23 +5522,23 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, surrogatepass = 1; break; default: - return -3; + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } /* Note: size will always be longer than the resulting Unicode character count */ - if (PY_SSIZE_T_MAX / (Py_ssize_t)sizeof(wchar_t) - 1 < size) { - return -1; + if ((size_t)PY_SSIZE_T_MAX / sizeof(wchar_t) - 1 < (size_t)size) { + return _Py_CODEC_MEMORY_ERROR; } - - unicode = PyMem_RawMalloc((size + 1) * sizeof(wchar_t)); + wchar_t *unicode = PyMem_RawMalloc((size + 1) * sizeof(wchar_t)); if (!unicode) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } /* Unpack UTF-8 encoded data */ - e = s + size; - outpos = 0; + const char *orig_s = s; + const char *e = s + size; + Py_ssize_t outpos = 0; while (s < e) { Py_UCS4 ch; #if SIZEOF_WCHAR_T == 4 @@ -5571,24 +5578,10 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, } else { PyMem_RawFree(unicode ); - if (reason != NULL) { - switch (ch) { - case 0: - *reason = "unexpected end of data"; - break; - case 1: - *reason = "invalid start byte"; - break; - /* 2, 3, 4 */ - default: - *reason = "invalid continuation byte"; - break; - } - } if (wlen != NULL) { *wlen = s - orig_s; } - return -2; + return _Py_CODEC_DECODE_ERROR; } } } @@ -5602,17 +5595,20 @@ _Py_DecodeUTF8Ex(const char *s, Py_ssize_t size, wchar_t **wstr, size_t *wlen, } +// Decode from UTF-8 with the "surrogateescape" error handler. +// On success, set *wlen and return a newly allocated string. +// On error, set *wlen to the error (_Py_CODEC_MEMORY_ERROR or +// _Py_CODEC_DECODE_ERROR) and the return NULL wchar_t* _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, size_t *wlen) { wchar_t *wstr; - int res = _Py_DecodeUTF8Ex(arg, arglen, - &wstr, wlen, - NULL, _Py_ERROR_SURROGATEESCAPE); + int res = _Py_DecodeUTF8(arg, arglen, &wstr, wlen, + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { - /* _Py_DecodeUTF8Ex() must support _Py_ERROR_SURROGATEESCAPE */ - assert(res != -3); + /* _Py_DecodeUTF8() must support _Py_ERROR_SURROGATEESCAPE */ + assert(res != _Py_CODEC_UNSUPPORTED_ERROR_HANDLER); if (wlen) { *wlen = (size_t)res; } @@ -5627,19 +5623,24 @@ _Py_DecodeUTF8_surrogateescape(const char *arg, Py_ssize_t arglen, On success, return 0 and write the newly allocated character string (use PyMem_Free() to free the memory) into *str. - On encoding failure, return -2 and write the position of the invalid - surrogate character into *error_pos (if error_pos is set) and the decoding - error message into *reason (if reason is set). + On encoding failure, return _Py_CODEC_ENCODE_ERROR (-2) and write the + position of the invalid surrogate character into *error_pos (if error_pos is + set). - On memory allocation failure, return -1. */ + On memory allocation failure, return _Py_CODEC_MEMORY_ERROR (-1). + + str and output_length must not be NULL +*/ int -_Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, - const char **reason, int raw_malloc, _Py_error_handler errors) +_Py_EncodeUTF8(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { - const Py_ssize_t max_char_size = 4; - Py_ssize_t len = wcslen(text); + assert(str != NULL); + assert(output_length != NULL); - assert(len >= 0); + // U+10ffff encoded to UTF-8 takes 4 bytes + const size_t max_char_size = 4; + size_t len = wcslen(text); int surrogateescape = 0; int surrogatepass = 0; @@ -5654,11 +5655,11 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, surrogatepass = 1; break; default: - return -3; + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } - if (len > PY_SSIZE_T_MAX / max_char_size - 1) { - return -1; + if (len > (size_t)PY_SSIZE_T_MAX / max_char_size - 1) { + return _Py_CODEC_MEMORY_ERROR; } char *bytes; if (raw_malloc) { @@ -5668,12 +5669,11 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, bytes = PyMem_Malloc((len + 1) * max_char_size); } if (bytes == NULL) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } char *p = bytes; - Py_ssize_t i; - for (i = 0; i < len; ) { + for (size_t i = 0; i < len; ) { Py_ssize_t ch_pos = i; Py_UCS4 ch = text[i]; i++; @@ -5689,7 +5689,6 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, if (ch < 0x80) { /* Encode ASCII */ *p++ = (char) ch; - } else if (ch < 0x0800) { /* Encode Latin-1 */ @@ -5702,16 +5701,13 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, if (error_pos != NULL) { *error_pos = (size_t)ch_pos; } - if (reason != NULL) { - *reason = "encoding error"; - } if (raw_malloc) { PyMem_RawFree(bytes); } else { PyMem_Free(bytes); } - return -2; + return _Py_CODEC_ENCODE_ERROR; } *p++ = (char)(ch & 0xff); } @@ -5729,29 +5725,29 @@ _Py_EncodeUTF8Ex(const wchar_t *text, char **str, size_t *error_pos, *p++ = (char)(0x80 | (ch & 0x3f)); } } - *p++ = '\0'; + *p++ = '\0'; // trailing NUL byte size_t final_size = (p - bytes); - char *bytes2; + assert(final_size >= 1); + *output_length = final_size - 1; // -1 for the trailing NUL byte + + char *result; if (raw_malloc) { - bytes2 = PyMem_RawRealloc(bytes, final_size); + result = PyMem_RawRealloc(bytes, final_size); } else { - bytes2 = PyMem_Realloc(bytes, final_size); + result = PyMem_Realloc(bytes, final_size); } - if (bytes2 == NULL) { - if (error_pos != NULL) { - *error_pos = (size_t)-1; - } + if (result == NULL) { if (raw_malloc) { PyMem_RawFree(bytes); } else { PyMem_Free(bytes); } - return -1; + return _Py_CODEC_MEMORY_ERROR; } - *str = bytes2; + *str = result; return 0; } @@ -15229,8 +15225,11 @@ unicode_iter(PyObject *seq) static int encode_wstr_utf8(wchar_t *wstr, char **str, const char *name) { - int res; - res = _Py_EncodeUTF8Ex(wstr, str, NULL, NULL, 1, _Py_ERROR_STRICT); + assert(str != NULL); + + size_t unused_output_length; + int res = _Py_EncodeUTF8(wstr, str, &unused_output_length, + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { PyErr_Format(PyExc_RuntimeError, "cannot encode %s", name); return -1; diff --git a/Python/fileutils.c b/Python/fileutils.c index 404fec83385e97..8ed88047b35e4f 100644 --- a/Python/fileutils.c +++ b/Python/fileutils.c @@ -51,13 +51,19 @@ extern int winerror_to_errno(int); int _Py_open_cloexec_works = -1; #endif +// wcstombs() error +static const size_t ENCODE_ERROR = (size_t)-1; // mbstowcs() and mbrtowc() errors -static const size_t DECODE_ERROR = ((size_t)-1); +static const size_t DECODE_ERROR = (size_t)-1; #ifdef HAVE_MBRTOWC static const size_t INCOMPLETE_CHARACTER = (size_t)-2; #endif +// Get the error handler as 'int surrogateescape'. +// Set '*surrogateescape' and return 0 on success. +// Return -1 if the error handler is not supported: other than "strict" and +// "surrogateescape". static int get_surrogateescape(_Py_error_handler errors, int *surrogateescape) { @@ -227,8 +233,7 @@ extern int _Py_normalize_encoding(const char *, char *, size_t, int); Values of force_ascii: 1: the workaround is used: Py_EncodeLocale() uses - encode_ascii_surrogateescape() and Py_DecodeLocale() uses - decode_ascii() + encode_ascii() and Py_DecodeLocale() uses decode_ascii() 0: the workaround is not used: Py_EncodeLocale() uses wcstombs() and Py_DecodeLocale() uses mbstowcs() -1: unknown, need to call check_force_ascii() to get the value @@ -353,36 +358,41 @@ _Py_ResetForceASCII(void) } +// Encode a wide string to the ASCII encoding. If errors is +// _Py_ERROR_SURROGATEESCAPE, handle non-ASCII character using the +// surrogateescape error handler. In that case, the output is non-ASCII. +// +// Set *str to a newly allocated string on success. +// Return a negative number on error: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_ENCODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// +// str and output_length must not be NULL static int -encode_ascii(const wchar_t *text, char **str, - size_t *error_pos, const char **reason, - int raw_malloc, _Py_error_handler errors) +encode_ascii(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, _Py_error_handler errors) { - char *result = NULL, *out; - size_t len, i; - wchar_t ch; + assert(str != NULL); + assert(output_length != NULL); int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } - len = wcslen(text); + size_t len = wcslen(text); + *output_length = len; /* +1 for NULL byte */ - if (raw_malloc) { - result = PyMem_RawMalloc(len + 1); - } - else { - result = PyMem_Malloc(len + 1); - } + char *result; + result = raw_malloc ? PyMem_RawMalloc(len + 1) : PyMem_Malloc(len + 1); if (result == NULL) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } - out = result; - for (i=0; i PY_SSIZE_T_MAX / sizeof(wchar_t)) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } res = PyMem_RawMalloc(argsize * sizeof(wchar_t)); if (!res) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } out = res; for (in = (unsigned char*)arg; *in; in++) { unsigned char ch = *in; - if (ch < 128) { + if (ch <= 127) { *out++ = ch; } else { @@ -462,10 +483,7 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, if (wlen) { *wlen = in - (unsigned char*)arg; } - if (reason) { - *reason = "decoding error"; - } - return -2; + return _Py_CODEC_DECODE_ERROR; } *out++ = 0xdc00 + ch; } @@ -480,22 +498,22 @@ decode_ascii(const char *arg, wchar_t **wstr, size_t *wlen, } #endif /* !HAVE_MBRTOWC */ +// Decode a bytes string from the current locale encoding. +// On success, set *wstr and *wlen (if set) and return 0. +// On error, return a negative number: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_DECODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// On decode, set *wlen (if set). static int decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, - const char **reason, _Py_error_handler errors) + _Py_error_handler errors) { - wchar_t *res; - size_t argsize; - size_t count; -#ifdef HAVE_MBRTOWC - unsigned char *in; - wchar_t *out; - mbstate_t mbs; -#endif + assert(arg != NULL); + assert(wstr != NULL); int surrogateescape; if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; } #ifdef HAVE_BROKEN_MBSTOWCS @@ -503,21 +521,25 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, * mbstowcs which does not count the characters that * would result from conversion. Use an upper bound. */ - argsize = strlen(arg); + size_t argsize = strlen(arg); #else - argsize = _Py_mbstowcs(NULL, arg, 0); + size_t argsize = _Py_mbstowcs(NULL, arg, 0); #endif + wchar_t *res; if (argsize != DECODE_ERROR) { if (argsize > PY_SSIZE_T_MAX / sizeof(wchar_t) - 1) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } res = (wchar_t *)PyMem_RawMalloc((argsize + 1) * sizeof(wchar_t)); if (!res) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } - count = _Py_mbstowcs(res, arg, argsize + 1); + // +1 to write also the trailing NUL character + size_t count = _Py_mbstowcs(res, arg, argsize + 1); if (count != DECODE_ERROR) { + // Success + assert(count == argsize); *wstr = res; if (wlen != NULL) { *wlen = count; @@ -535,15 +557,16 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, actual output could use less memory. */ argsize = strlen(arg) + 1; if (argsize > PY_SSIZE_T_MAX / sizeof(wchar_t)) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } res = (wchar_t*)PyMem_RawMalloc(argsize * sizeof(wchar_t)); if (!res) { - return -1; + return _Py_CODEC_MEMORY_ERROR; } - in = (unsigned char*)arg; - out = res; + unsigned char *in = (unsigned char*)arg; + wchar_t *out = res; + mbstate_t mbs; memset(&mbs, 0, sizeof mbs); while (argsize) { size_t converted = _Py_mbrtowc(out, (char*)in, argsize, &mbs); @@ -575,6 +598,7 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, } if (wlen != NULL) { *wlen = out - res; + assert(res[*wlen] == 0); } *wstr = res; return 0; @@ -584,66 +608,37 @@ decode_current_locale(const char* arg, wchar_t **wstr, size_t *wlen, if (wlen) { *wlen = in - (unsigned char*)arg; } - if (reason) { - *reason = "decoding error"; - } - return -2; + return _Py_CODEC_DECODE_ERROR; #else /* HAVE_MBRTOWC */ /* Cannot use C locale for escaping; manually escape as if charset is ASCII (i.e. escape all bytes > 128. This will still roundtrip correctly in the locale's charset, which must be an ASCII superset. */ - return decode_ascii(arg, wstr, wlen, reason, errors); + return decode_ascii(arg, wstr, wlen, errors); #endif /* HAVE_MBRTOWC */ } -/* Decode a byte string from the locale encoding. - - Use the strict error handler if 'surrogateescape' is zero. Use the - surrogateescape error handler if 'surrogateescape' is non-zero: undecodable - bytes are decoded as characters in range U+DC80..U+DCFF. If a byte sequence - can be decoded as a surrogate character, escape the bytes using the - surrogateescape error handler instead of decoding them. - - On success, return 0 and write the newly allocated wide character string into - *wstr (use PyMem_RawFree() to free the memory). If wlen is not NULL, write - the number of wide characters excluding the null character into *wlen. - - On memory allocation failure, return -1. - - On decoding error, return -2. If wlen is not NULL, write the start of - invalid byte sequence in the input string into *wlen. If reason is not NULL, - write the decoding error message into *reason. - - Return -3 if the error handler 'errors' is not supported. - - Use the Py_EncodeLocaleEx() function to encode the character string back to - a byte string. */ -int -_Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, - const char **reason, +static int +decode_locale_impl(const char* arg, wchar_t **wstr, size_t *wlen, int current_locale, _Py_error_handler errors) { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, reason, - errors); + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); #else - return decode_current_locale(arg, wstr, wlen, reason, errors); + return decode_current_locale(arg, wstr, wlen, errors); #endif } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, reason, - errors); + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); #ifdef MS_WINDOWS use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_DecodeUTF8Ex(arg, strlen(arg), wstr, wlen, reason, - errors); + return _Py_DecodeUTF8(arg, strlen(arg), wstr, wlen, errors); } #ifdef USE_FORCE_ASCII @@ -653,45 +648,112 @@ _Py_DecodeLocaleEx(const char* arg, wchar_t **wstr, size_t *wlen, if (force_ascii) { /* force ASCII encoding to workaround mbstowcs() issue */ - return decode_ascii(arg, wstr, wlen, reason, errors); + return decode_ascii(arg, wstr, wlen, errors); } #endif - return decode_current_locale(arg, wstr, wlen, reason, errors); + return decode_current_locale(arg, wstr, wlen, errors); #endif /* !_Py_FORCE_UTF8_FS_ENCODING */ } -/* Decode a byte string from the locale encoding with the - surrogateescape error handler: undecodable bytes are decoded as characters - in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate - character, escape the bytes using the surrogateescape error handler instead - of decoding them. +// Decode a byte string from the locale encoding. +// +// Supported error handlers are _Py_ERROR_STRICT and _Py_ERROR_SURROGATEESCAPE. +// The UTF-8 decoder also supports _Py_ERROR_SURROGATEPASS. +// +// If errors is _Py_ERROR_SURROGATEESCAPE, undecodable bytes are decoded as +// characters in range U+DC80..U+DCFF. If a byte sequence can be decoded as a +// surrogate character, escape the bytes using the surrogateescape error +// handler instead of decoding them. +// +// On success, return 0 and write the newly allocated wide character string into +// *wstr (use PyMem_RawFree() to free the memory). If wlen is not NULL, write +// the number of wide characters excluding the null character into *wlen. +// +// On error, return a negative number. +// +// On memory allocation failure, return _Py_CODEC_MEMORY_ERROR. +// +// On decoding error, return _Py_CODEC_DECODE_ERROR. If wlen is not NULL, +// write the start of invalid byte sequence in the input string into *wlen. If +// reason is not NULL, write the decoding error message into *reason. +// +// Return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER if the 'errors' error +// handler is not supported: other than "strict" and "surrogateescape". +// +// Use the _Py_EncodeLocale() function to encode the character string back to +// a byte string. +// +// arg and wstr must not be NULL. +int +_Py_DecodeLocale(const char* arg, wchar_t **wstr, size_t *wlen, + int current_locale, _Py_error_handler errors) +{ + assert(arg != NULL); + assert(wstr != NULL); - Return a pointer to a newly allocated wide character string, use - PyMem_RawFree() to free the memory. If size is not NULL, write the number of - wide characters excluding the null character into *size +#ifdef Py_DEBUG + size_t wlen_canary = (size_t)-2; + if (wlen) { + *wlen = wlen_canary; + } +#endif - Return NULL on decoding error or memory allocation error. If *size* is not - NULL, *size is set to (size_t)-1 on memory error or set to (size_t)-2 on - decoding error. + int res = decode_locale_impl(arg, wstr, wlen, current_locale, errors); + if (res < 0) { + // Error + *wstr = NULL; + if (res == _Py_CODEC_DECODE_ERROR) { +#ifdef Py_DEBUG + assert(wlen == NULL || *wlen != wlen_canary); +#endif + } + else { + if (wlen) { + *wlen = 0; + } + } + } + else { + // Success + assert(*wstr != NULL); +#ifdef Py_DEBUG + if (wlen != NULL) { + assert(*wlen == wcslen(*wstr)); + } +#endif + } + return res; +} - Decoding errors should never happen, unless there is a bug in the C - library. - Use the Py_EncodeLocale() function to encode the character string back to a - byte string. */ +// Decode a byte string from the locale encoding with the +// surrogateescape error handler. Undecodable bytes are decoded as characters +// in range U+DC80..U+DCFF. If a byte sequence can be decoded as a surrogate +// character, escape the bytes using the surrogateescape error handler instead +// of decoding them. +// +// Return a pointer to a newly allocated wide character string, use +// PyMem_RawFree() to free the memory. If size is not NULL, write the number of +// wide characters excluding the null character into *size. +// +// On memory allocation failure, set *size to (size_t)-1 and return NULL. +// +// On decode error, set *size to (size_t)-2 and return NULL. +// Decoding errors should never happen, unless there is a bug in the C library. +// +// Use the Py_EncodeLocale() function to encode the character string back to a +// byte string. wchar_t* -Py_DecodeLocale(const char* arg, size_t *wlen) +Py_DecodeLocale(const char* arg, size_t *size) { wchar_t *wstr; - int res = _Py_DecodeLocaleEx(arg, &wstr, wlen, - NULL, 0, - _Py_ERROR_SURROGATEESCAPE); + int res = _Py_DecodeLocale(arg, &wstr, size, 0, _Py_ERROR_SURROGATEESCAPE); if (res != 0) { - assert(res != -3); - if (wlen != NULL) { - *wlen = (size_t)res; + assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_DECODE_ERROR); + if (size != NULL) { + *size = (size_t)res; } return NULL; } @@ -699,143 +761,154 @@ Py_DecodeLocale(const char* arg, size_t *wlen) } -static int -encode_current_locale(const wchar_t *text, char **str, - size_t *error_pos, const char **reason, - int raw_malloc, _Py_error_handler errors) +// If bytes is NULL, compute the length in 'bytes' of the encoded string. +// Otherwise, encode the wide string into 'bytes', decrements 'size', and +// return 0 on success. +// On encoding error, set 'error_pos' (if set) and return ENCODE_ERROR. +static size_t +encode_current_locale_impl(const wchar_t *text, const size_t len, + int surrogateescape, + char *bytes, size_t size, + size_t *error_pos) { - const size_t len = wcslen(text); - char *result = NULL, *bytes = NULL; - size_t i, size, converted; - wchar_t c, buf[2]; - - int surrogateescape; - if (get_surrogateescape(errors, &surrogateescape) < 0) { - return -3; - } - - /* The function works in two steps: - 1. compute the length of the output buffer in bytes (size) - 2. outputs the bytes */ - size = 0; + wchar_t buf[2]; + // The second character is always the NUL character buf[1] = 0; - while (1) { - for (i=0; i < len; i++) { - c = text[i]; - if (c >= 0xdc80 && c <= 0xdcff) { - if (!surrogateescape) { - goto encode_error; - } - /* UTF-8b surrogate */ - if (bytes != NULL) { - *bytes++ = c - 0xdc00; - size--; - } - else { - size++; + + for (size_t i=0; i < len; i++) { + wchar_t c = text[i]; + if (c >= 0xdc80 && c <= 0xdcff) { + if (!surrogateescape) { + if (error_pos != NULL) { + *error_pos = i; } - continue; + return ENCODE_ERROR; + } + /* UTF-8b surrogate */ + if (bytes != NULL) { + *bytes++ = c - 0xdc00; + size--; } else { - buf[0] = c; - if (bytes != NULL) { - converted = wcstombs(bytes, buf, size); - } - else { - converted = wcstombs(NULL, buf, 0); - } - if (converted == DECODE_ERROR) { - goto encode_error; - } - if (bytes != NULL) { - bytes += converted; - size -= converted; - } - else { - size += converted; - } + size++; } } - if (result != NULL) { - *bytes = '\0'; - break; + else { + // Encode a single character using wcstombs() + buf[0] = c; + size_t converted; + if (bytes != NULL) { + converted = wcstombs(bytes, buf, size); + } + else { + converted = wcstombs(NULL, buf, 0); + } + if (converted == ENCODE_ERROR) { + if (error_pos != NULL) { + *error_pos = i; + } + return ENCODE_ERROR; + } + if (bytes != NULL) { + bytes += converted; + size -= converted; + } + else { + size += converted; + } } + } + if (bytes) { + *bytes = '\0'; + } + // Sanity check, it cannot happen in practice + assert(size != ENCODE_ERROR); + return size; +} - size += 1; /* nul byte at the end */ + +// Encode a wide string to the current locale encoding. +// Set *str to a newly allocated string on success. +// Return a negative number on error: _Py_CODEC_MEMORY_ERROR, +// _Py_CODEC_ENCODE_ERROR or _Py_CODEC_UNSUPPORTED_ERROR_HANDLER. +// +// str and output_length must not be NULL +static int +encode_current_locale(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + _Py_error_handler errors) +{ + assert(str != NULL); + assert(output_length != NULL); + + int surrogateescape; + if (get_surrogateescape(errors, &surrogateescape) < 0) { + // Only support "strict" and "surrogateescape" + return _Py_CODEC_UNSUPPORTED_ERROR_HANDLER; + } + + // First, compute the output length + const size_t len = wcslen(text); + // Sanity check, it cannot happen in practice + assert(len != ENCODE_ERROR); + size_t size = encode_current_locale_impl(text, len, surrogateescape, + NULL, 0, error_pos); + if (size == ENCODE_ERROR) { + return _Py_CODEC_ENCODE_ERROR; + } + + *output_length = size; + char *result; + result = raw_malloc ? PyMem_RawMalloc(size + 1) : PyMem_Malloc(size + 1); + if (result == NULL) { + return _Py_CODEC_MEMORY_ERROR; + } + + // Second, encode characters + size = encode_current_locale_impl(text, len, surrogateescape, + result, size, error_pos); + if (size == ENCODE_ERROR) { if (raw_malloc) { - result = PyMem_RawMalloc(size); + PyMem_RawFree(result); } else { - result = PyMem_Malloc(size); - } - if (result == NULL) { - return -1; + PyMem_Free(result); } - bytes = result; + return _Py_CODEC_ENCODE_ERROR; } + assert(size == 0); + *str = result; return 0; - -encode_error: - if (raw_malloc) { - PyMem_RawFree(result); - } - else { - PyMem_Free(result); - } - if (error_pos != NULL) { - *error_pos = i; - } - if (reason) { - *reason = "encoding error"; - } - return -2; } -/* Encode a string to the locale encoding. - - Parameters: - - * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead - of PyMem_Malloc(). - * current_locale: if non-zero, use the current LC_CTYPE, otherwise use - Python filesystem encoding. - * errors: error handler like "strict" or "surrogateescape". - - Return value: - - 0: success, *str is set to a newly allocated decoded string. - -1: memory allocation failure - -2: encoding error, set *error_pos and *reason (if set). - -3: the error handler 'errors' is not supported. - */ static int -encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, - const char **reason, - int raw_malloc, int current_locale, _Py_error_handler errors) +encode_locale_inner(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + int current_locale, _Py_error_handler errors) { if (current_locale) { #ifdef _Py_FORCE_UTF8_LOCALE - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8(text, str, output_length, + error_pos, raw_malloc, errors); #else - return encode_current_locale(text, str, error_pos, reason, - raw_malloc, errors); + return encode_current_locale(text, str, output_length, + error_pos, raw_malloc, errors); #endif } #ifdef _Py_FORCE_UTF8_FS_ENCODING - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8(text, str, output_length, + error_pos, raw_malloc, errors); #else int use_utf8 = (_PyRuntime.preconfig.utf8_mode >= 1); #ifdef MS_WINDOWS use_utf8 |= (_PyRuntime.preconfig.legacy_windows_fs_encoding == 0); #endif if (use_utf8) { - return _Py_EncodeUTF8Ex(text, str, error_pos, reason, - raw_malloc, errors); + return _Py_EncodeUTF8(text, str, output_length, + error_pos, raw_malloc, errors); } #ifdef USE_FORCE_ASCII @@ -844,45 +917,101 @@ encode_locale_ex(const wchar_t *text, char **str, size_t *error_pos, } if (force_ascii) { - return encode_ascii(text, str, error_pos, reason, - raw_malloc, errors); + return encode_ascii(text, str, output_length, + error_pos, raw_malloc, errors); } #endif - return encode_current_locale(text, str, error_pos, reason, - raw_malloc, errors); + return encode_current_locale(text, str, output_length, + error_pos, raw_malloc, errors); #endif /* _Py_FORCE_UTF8_FS_ENCODING */ } + +// Encode a wide string to the locale encoding. +// +// Parameters: +// +// * raw_malloc: if non-zero, allocate memory using PyMem_RawMalloc() instead +// of PyMem_Malloc(). +// * current_locale: if non-zero, use the current LC_CTYPE, otherwise use +// Python filesystem encoding. +// * errors: supported error handlers are _Py_ERROR_STRICT and +// _Py_ERROR_SURROGATEESCAPE. The UTF-8 encoder also supports +// _Py_ERROR_SURROGATEPASS. +// +// Set *str to a newly allocated decoded string and return 0 on success. +// Return a negative result on error: +// +// * _Py_CODEC_MEMORY_ERROR: memory allocation failure +// * _Py_CODEC_ENCODE_ERROR: encoding error, set *error_pos. +// * _Py_CODEC_UNSUPPORTED_ERROR_HANDLER: the 'errors' error handler +// is not supported. +// +// text, str and output_length must not be NULL. +static int +encode_locale_impl(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int raw_malloc, + int current_locale, _Py_error_handler errors) +{ + assert(text != NULL); + assert(str != NULL); + assert(output_length != NULL); + + int res = encode_locale_inner(text, str, output_length, + error_pos, raw_malloc, + current_locale, errors); + if (res < 0) { + // Error + *str = NULL; + *output_length = 0; + if (res != _Py_CODEC_ENCODE_ERROR) { + if (error_pos) { + *error_pos = 0; + } + } + } + else { + assert(*str != NULL); + assert(*output_length == strlen(*str)); + // *error_pos is left unchanged + } + return res; +} + static char* encode_locale(const wchar_t *text, size_t *error_pos, int raw_malloc, int current_locale) { char *str; - int res = encode_locale_ex(text, &str, error_pos, NULL, - raw_malloc, current_locale, - _Py_ERROR_SURROGATEESCAPE); - if (res != -2 && error_pos) { - *error_pos = (size_t)-1; - } + size_t unused_output_length; + int res = encode_locale_impl(text, &str, &unused_output_length, + error_pos, raw_malloc, current_locale, + _Py_ERROR_SURROGATEESCAPE); if (res != 0) { + assert(res == _Py_CODEC_MEMORY_ERROR || res == _Py_CODEC_ENCODE_ERROR); + if (res == _Py_CODEC_MEMORY_ERROR && error_pos != NULL) { + *error_pos = (size_t)-1; + } return NULL; } return str; } -/* Encode a wide character string to the locale encoding with the - surrogateescape error handler: surrogate characters in the range - U+DC80..U+DCFF are converted to bytes 0x80..0xFF. - - Return a pointer to a newly allocated byte string, use PyMem_Free() to free - the memory. Return NULL on encoding or memory allocation error. - - If error_pos is not NULL, *error_pos is set to (size_t)-1 on success, or set - to the index of the invalid character on encoding error. - - Use the Py_DecodeLocale() function to decode the bytes string back to a wide - character string. */ +// Encode a wide character string to the locale encoding with the +// surrogateescape error handler. Surrogate characters in the range +// U+DC80..U+DCFF are encoded to bytes 0x80..0xFF. +// +// Return a pointer to a newly allocated byte string, use PyMem_Free() to free +// the memory. Return NULL on encoding or memory allocation error. +// +// On memory allocation failure, set *error_pos to (size_t)-1 and return NULL. +// +// On encoding error, set *error_pos to the index of the first unencodable +// character and return NULL. +// +// Use the Py_DecodeLocale() function to decode the bytes string back to a wide +// character string. char* Py_EncodeLocale(const wchar_t *text, size_t *error_pos) { @@ -892,7 +1021,7 @@ Py_EncodeLocale(const wchar_t *text, size_t *error_pos) /* Similar to Py_EncodeLocale(), but result must be freed by PyMem_RawFree() instead of PyMem_Free(). */ -char* +static char* _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) { return encode_locale(text, error_pos, 1, 0); @@ -900,12 +1029,12 @@ _Py_EncodeLocaleRaw(const wchar_t *text, size_t *error_pos) int -_Py_EncodeLocaleEx(const wchar_t *text, char **str, - size_t *error_pos, const char **reason, - int current_locale, _Py_error_handler errors) +_Py_EncodeLocale(const wchar_t *text, char **str, size_t *output_length, + size_t *error_pos, int current_locale, + _Py_error_handler errors) { - return encode_locale_ex(text, str, error_pos, reason, 1, - current_locale, errors); + return encode_locale_impl(text, str, output_length, + error_pos, 1, current_locale, errors); } @@ -944,9 +1073,8 @@ _Py_GetLocaleEncoding(void) } wchar_t *wstr; - int res = decode_current_locale(encoding, &wstr, NULL, - NULL, _Py_ERROR_SURROGATEESCAPE); - if (res < 0) { + if (decode_current_locale(encoding, &wstr, NULL, + _Py_ERROR_SURROGATEESCAPE) < 0) { return NULL; } return wstr; diff --git a/Python/initconfig.c b/Python/initconfig.c index 178e6f992fa11f..363739a1a1b43c 100644 --- a/Python/initconfig.c +++ b/Python/initconfig.c @@ -4153,7 +4153,9 @@ static char* wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) { char *utf8; - int res = _Py_EncodeUTF8Ex(wstr, &utf8, NULL, NULL, 1, _Py_ERROR_STRICT); + size_t utf8_len; + int res = _Py_EncodeUTF8(wstr, &utf8, &utf8_len, + NULL, 1, _Py_ERROR_STRICT); if (res == -2) { initconfig_set_error(config, "encoding error"); return NULL; @@ -4164,7 +4166,7 @@ wstr_to_utf8(PyInitConfig *config, wchar_t *wstr) } // Copy to use the malloc() memory allocator - size_t size = strlen(utf8) + 1; + size_t size = utf8_len + 1; char *str = malloc(size); if (str == NULL) { PyMem_RawFree(utf8); @@ -4323,12 +4325,13 @@ utf8_to_wstr(PyInitConfig *config, const char *str) { wchar_t *wstr; size_t wlen; - int res = _Py_DecodeUTF8Ex(str, strlen(str), &wstr, &wlen, NULL, _Py_ERROR_STRICT); - if (res == -2) { + int res = _Py_DecodeUTF8(str, strlen(str), &wstr, &wlen, _Py_ERROR_STRICT); + if (res == _Py_CODEC_DECODE_ERROR) { initconfig_set_error(config, "decoding error"); return NULL; } if (res < 0) { + assert(res == _Py_CODEC_MEMORY_ERROR); config->status = _PyStatus_NO_MEMORY(); return NULL; }