From 8e0d8a207f0de8e4ad2db7edf9ac29a489bd70c6 Mon Sep 17 00:00:00 2001 From: Cody Maloney Date: Mon, 5 Oct 2026 17:30:16 -0700 Subject: [PATCH] gh-87613: Use AC vectorcall for bytes and bytearray Speeds up common construction of both objects. Enables optimizing construction off uniquely referenced temporary objects. Previously the argument tuples / dictionaries required by the call protocol resulted in arguments never being unique temporaries. With vectorcall they can be. Avoid copying by adopting the existing allocation when possible. Benchmark Run on my linux dev machine: PGO+LTO, clang 23, --with-tail-call-interp (note: AI Written, but I think sufficiently representative) ```python """bytes() and bytearray() construction. temp/ passes a unique temporary (adopted, no copy); named/ binds the same source to a name first so it is always copied. """ import pyperf runner = pyperf.Runner() runner.timeit('call/bytes()', 'bytes()') runner.timeit('call/bytes(int)', 'bytes(16)') runner.timeit('call/bytes(bytes)', "bytes(b'abcdefghabcdefgh')") runner.timeit("call/bytes(str, 'ascii')", "bytes('abcdefgh', 'ascii')") runner.timeit('call/bytearray()', 'bytearray()') runner.timeit('call/bytearray(int)', 'bytearray(16)') runner.timeit('call/bytearray(bytes)', "bytearray(b'abcdefghabcdefgh')") runner.timeit("call/bytearray(str, 'ascii')", "bytearray('abcdefgh', 'ascii')") for size_name, size in (('16B', 16), ('4KiB', 4096), ('1MiB', 1 << 20)): # Slicing in the statement makes a fresh unique temporary. setup = f"b = b'x' * {size}; ba = bytearray(b)" runner.timeit(f'temp/bytes(bytearray) {size_name}', 'bytes(ba[1:])', setup) runner.timeit(f'named/bytes(bytearray) {size_name}', 'x = ba[1:]; bytes(x); del x', setup) runner.timeit(f'temp/bytearray(bytes) {size_name}', 'bytearray(b[1:])', setup) runner.timeit(f'named/bytearray(bytes) {size_name}', 'x = b[1:]; bytearray(x); del x', setup) runner.timeit(f'temp/bytearray(bytearray) {size_name}', 'bytearray(ba[1:])', setup) runner.timeit(f'named/bytearray(bytearray) {size_name}', 'x = ba[1:]; bytearray(x); del x', setup) ``` | Benchmark | main | patch | |---------------------------------|:-------:|:----------------------:| | temp/bytearray(bytearray) 1MiB | 515 us | 14.9 us: 34.53x faster | | temp/bytearray(bytes) 1MiB | 508 us | 14.8 us: 34.27x faster | | temp/bytes(bytearray) 1MiB | 507 us | 14.9 us: 34.05x faster | | temp/bytearray(bytes) 4KiB | 144 ns | 79.5 ns: 1.81x faster | | temp/bytearray(bytearray) 4KiB | 155 ns | 94.7 ns: 1.63x faster | | temp/bytearray(bytes) 16B | 56.4 ns | 36.1 ns: 1.56x faster | | temp/bytes(bytearray) 4KiB | 150 ns | 96.4 ns: 1.55x faster | | temp/bytearray(bytearray) 16B | 64.1 ns | 47.3 ns: 1.35x faster | | call/bytes() | 14.7 ns | 11.1 ns: 1.32x faster | | call/bytes(int) | 31.5 ns | 24.3 ns: 1.29x faster | | call/bytes(str, 'ascii') | 36.9 ns | 28.7 ns: 1.28x faster | | call/bytes(bytes) | 34.3 ns | 27.3 ns: 1.26x faster | | temp/bytes(bytearray) 16B | 59.4 ns | 49.5 ns: 1.20x faster | | call/bytearray(str, 'ascii') | 45.6 ns | 39.1 ns: 1.17x faster | | call/bytearray(bytes) | 44.3 ns | 38.9 ns: 1.14x faster | | call/bytearray(int) | 41.7 ns | 37.3 ns: 1.12x faster | | named/bytearray(bytes) 16B | 59.6 ns | 54.4 ns: 1.10x faster | | named/bytes(bytearray) 16B | 61.3 ns | 56.2 ns: 1.09x faster | | named/bytearray(bytearray) 16B | 65.3 ns | 60.9 ns: 1.07x faster | | named/bytes(bytearray) 4KiB | 154 ns | 148 ns: 1.04x faster | | call/bytearray() | 22.1 ns | 21.4 ns: 1.03x faster | | named/bytearray(bytearray) 4KiB | 156 ns | 151 ns: 1.03x faster | | named/bytearray(bytes) 4KiB | 146 ns | 143 ns: 1.02x faster | | Geometric mean | (ref) | 1.77x faster | Benchmark hidden because not significant (3): named/bytes(bytearray) 1MiB, named/bytearray(bytes) 1MiB, named/bytearray(bytearray) 1MiB --- Lib/test/test_bytes.py | 18 ++++ ...6-10-05-13-19-49.gh-issue-87613.JLTGBy.rst | 2 + Objects/bytearrayobject.c | 31 +++++- Objects/bytesobject.c | 9 +- Objects/clinic/bytearrayobject.c.h | 95 +++++++++++++++++-- Objects/clinic/bytesobject.c.h | 71 +++++++++++++- 6 files changed, 213 insertions(+), 13 deletions(-) create mode 100644 Misc/NEWS.d/next/Core_and_Builtins/2026-10-05-13-19-49.gh-issue-87613.JLTGBy.rst diff --git a/Lib/test/test_bytes.py b/Lib/test/test_bytes.py index 96190a7f5821709..ed5d43014953b8d 100644 --- a/Lib/test/test_bytes.py +++ b/Lib/test/test_bytes.py @@ -16,6 +16,7 @@ import sys import tempfile import textwrap +lazy import tracemalloc import threading import unittest from _codecs import _unregister_error as _codecs_unregister_error @@ -1809,6 +1810,23 @@ def hashing_codec(name): self.assertEqual(taken, b'Hello') self.assertEqual(hash(taken), hash(b'Hello')) + @support.cpython_only + @support.bigmemtest(size=support._1G, memuse=1) + def test_unique_temporary_take_memory(self, size): + # Optimization: When passed a unique temporary adopt it rather than + # allocate + copy + deallocate. + def peak(func): + tracemalloc.start() + try: + func() + return tracemalloc.get_traced_memory()[1] + finally: + tracemalloc.stop() + # Copy would be 2.0, 1.5 gives space for other allocations. + self.assertLess(peak(lambda: bytearray(bytes(size))), size * 1.5) + self.assertLess(peak(lambda: bytearray(bytearray(size))), size * 1.5) + self.assertLess(peak(lambda: bytes(bytearray(size))), size * 1.5) + def test_take_bytes_reentrant_resize(self): # gh-153570: n.__index__() can resize the bytearray, so take_bytes() # must re-read the size afterwards. It cached the size before the diff --git a/Misc/NEWS.d/next/Core_and_Builtins/2026-10-05-13-19-49.gh-issue-87613.JLTGBy.rst b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-05-13-19-49.gh-issue-87613.JLTGBy.rst new file mode 100644 index 000000000000000..830e8ef1dc23b28 --- /dev/null +++ b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-05-13-19-49.gh-issue-87613.JLTGBy.rst @@ -0,0 +1,2 @@ +Speed up :class:`bytes` and :class:`bytearray` construction by adding +:ref:`vectorcall` support. diff --git a/Objects/bytearrayobject.c b/Objects/bytearrayobject.c index 496d04a1704d46d..cc9fbce5313eb0c 100644 --- a/Objects/bytearrayobject.c +++ b/Objects/bytearrayobject.c @@ -977,6 +977,7 @@ bytearray_new(PyTypeObject *type, PyObject *args, PyObject *kwds) } /*[clinic input] +@vectorcall bytearray.__init__ source as arg: object = NULL @@ -988,7 +989,7 @@ bytearray.__init__ static int bytearray___init___impl(PyByteArrayObject *self, PyObject *arg, const char *encoding, const char *errors) -/*[clinic end generated code: output=4ce1304649c2f8b3 input=1141a7122eefd7b9]*/ +/*[clinic end generated code: output=4ce1304649c2f8b3 input=24ddb84055f432dd]*/ { Py_ssize_t count; PyObject *it; @@ -1082,6 +1083,33 @@ bytearray___init___impl(PyByteArrayObject *self, PyObject *arg, } } + /* optimization: Adopt bytes rather than copying. */ + if (PyBytes_CheckExact(arg) + && PyUnstable_Object_IsUniqueReferencedTemporary(arg) + && _PyBytes_GET_CACHED_HASH((PyBytesObject *)arg) == -1) + { + /* Reinit before taking a reference: it asserts the bytes is + uniquely referenced. */ + self->ob_bytes_object = arg; + bytearray_reinit_from_bytes(self, PyBytes_GET_SIZE(arg)); + Py_INCREF(arg); + return 0; + } + + /* optimization: Adopt bytearray rather than copying. */ + if (PyByteArray_CheckExact(arg) + && PyUnstable_Object_IsUniqueReferencedTemporary(arg)) + { + PyObject *bytes = bytearray_take_bytes_impl((PyByteArrayObject *)arg, + Py_None); + if (bytes == NULL) { + return -1; + } + self->ob_bytes_object = bytes; + bytearray_reinit_from_bytes(self, PyBytes_GET_SIZE(bytes)); + return 0; + } + /* Use the buffer API */ if (PyObject_CheckBuffer(arg)) { Py_ssize_t size; @@ -3027,6 +3055,7 @@ PyTypeObject PyByteArray_Type = { PyType_GenericAlloc, /* tp_alloc */ bytearray_new, /* tp_new */ PyObject_Free, /* tp_free */ + .tp_vectorcall = bytearray_vectorcall, .tp_version_tag = _Py_TYPE_VERSION_BYTEARRAY, }; diff --git a/Objects/bytesobject.c b/Objects/bytesobject.c index 1917055d25198f1..7cda6c06fb5c675 100644 --- a/Objects/bytesobject.c +++ b/Objects/bytesobject.c @@ -2918,6 +2918,7 @@ static PyObject * bytes_subtype_new(PyTypeObject *, PyObject *); /*[clinic input] +@vectorcall @classmethod bytes.__new__ as bytes_new @@ -2930,7 +2931,7 @@ bytes.__new__ as bytes_new static PyObject * bytes_new_impl(PyTypeObject *type, PyObject *x, const char *encoding, const char *errors) -/*[clinic end generated code: output=1e0c471be311a425 input=f0a966d19b7262b4]*/ +/*[clinic end generated code: output=1e0c471be311a425 input=b0248d22e221a095]*/ { PyObject *bytes; PyObject *func; @@ -3002,6 +3003,11 @@ bytes_new_impl(PyTypeObject *type, PyObject *x, const char *encoding, bytes = _PyBytes_FromSizeZero(size); } } + /* Adopt unique temporary bytearray rather than copying. */ + else if (PyByteArray_CheckExact(x) + && PyUnstable_Object_IsUniqueReferencedTemporary(x)) { + bytes = PyObject_CallMethodNoArgs(x, &_Py_ID(take_bytes)); + } else { bytes = PyBytes_FromObject(x); } @@ -3316,6 +3322,7 @@ PyTypeObject PyBytes_Type = { bytes_type_alloc, /* tp_alloc */ bytes_new, /* tp_new */ PyObject_Free, /* tp_free */ + .tp_vectorcall = bytes_vectorcall, .tp_version_tag = _Py_TYPE_VERSION_BYTES, ._tp_iteritem = bytes_iteritem, }; diff --git a/Objects/clinic/bytearrayobject.c.h b/Objects/clinic/bytearrayobject.c.h index b074851f1fbf78f..9c8ba841200f306 100644 --- a/Objects/clinic/bytearrayobject.c.h +++ b/Objects/clinic/bytearrayobject.c.h @@ -4,11 +4,11 @@ preserve #if defined(Py_BUILD_CORE) && !defined(Py_BUILD_CORE_MODULE) # include "pycore_gc.h" // PyGC_Head -# include "pycore_runtime.h" // _Py_ID() #endif #include "pycore_abstract.h" // _PyNumber_Index() #include "pycore_critical_section.h"// Py_BEGIN_CRITICAL_SECTION() #include "pycore_modsupport.h" // _PyArg_UnpackKeywords() +#include "pycore_runtime.h" // _Py_SINGLETON() #include "pycore_unicodeobject.h" // _PyUnicode_AsUTF8NoNUL() static int @@ -16,7 +16,8 @@ bytearray___init___impl(PyByteArrayObject *self, PyObject *arg, const char *encoding, const char *errors); static int -bytearray___init__(PyObject *self, PyObject *args, PyObject *kwargs) +bytearray___init___helper(PyObject *self, PyObject *const *args, + Py_ssize_t nargs, Py_ssize_t nkw, PyObject *kwargs, PyObject *kwnames) { int return_value = -1; #if defined(Py_BUILD_CORE) && !defined(Py_BUILD_CORE_MODULE) @@ -48,13 +49,12 @@ bytearray___init__(PyObject *self, PyObject *args, PyObject *kwargs) #undef KWTUPLE PyObject *argsbuf[3]; PyObject * const *fastargs; - Py_ssize_t nargs = PyTuple_GET_SIZE(args); - Py_ssize_t noptargs = nargs + (kwargs ? PyDict_GET_SIZE(kwargs) : 0) - 0; + Py_ssize_t noptargs = nargs + nkw - 0; PyObject *arg = NULL; const char *encoding = NULL; const char *errors = NULL; - fastargs = _PyArg_UnpackKeywords(_PyTuple_CAST(args)->ob_item, nargs, kwargs, NULL, &_parser, + fastargs = _PyArg_UnpackKeywords(args, nargs, kwargs, kwnames, &_parser, /*minpos*/ 0, /*maxpos*/ 3, /*minkw*/ 0, /*varpos*/ 0, argsbuf); if (!fastargs) { goto exit; @@ -96,6 +96,89 @@ bytearray___init__(PyObject *self, PyObject *args, PyObject *kwargs) return return_value; } +static int +bytearray___init__(PyObject *self, PyObject *args, PyObject *kwargs) +{ + return bytearray___init___helper(self, _PyTuple_CAST(args)->ob_item, + PyTuple_GET_SIZE(args), + kwargs ? PyDict_GET_SIZE(kwargs) : 0, + kwargs, NULL); +} + +static PyObject * +bytearray_vectorcall(PyObject *type, PyObject *const *args, + size_t nargsf, PyObject *kwnames) +{ + PyObject *return_value = NULL; + Py_ssize_t nargs = PyVectorcall_NARGS(nargsf); + PyObject *self; + int _result; + PyObject *arg = NULL; + const char *encoding = NULL; + const char *errors = NULL; + + assert(Py_Is(_PyType_CAST(type), &PyByteArray_Type)); + /* Make sure the type object is immutable: the generated + * vectorcall doesn't deal e.g. with users reassigning __init__. */ + assert(PyType_HasFeature(_PyType_CAST(type), Py_TPFLAGS_IMMUTABLETYPE)); + if (kwnames != NULL || nargs > 3) { + self = _PyType_CAST(type)->tp_new(_PyType_CAST(type), + (PyObject *)&_Py_SINGLETON(tuple_empty), NULL); + if (self == NULL) { + return NULL; + } + _result = bytearray___init___helper(self, args, nargs, + kwnames ? PyTuple_GET_SIZE(kwnames) : 0, + NULL, kwnames); + if (_result != 0) { + Py_DECREF(self); + return NULL; + } + return self; + } + if (nargs < 1) { + goto skip_optional; + } + arg = args[0]; + if (nargs < 2) { + goto skip_optional; + } + if (!PyUnicode_Check(args[1])) { + _PyArg_BadArgument("bytearray", "argument 'encoding'", "str", args[1]); + goto exit; + } + encoding = _PyUnicode_AsUTF8NoNUL(args[1]); + if (encoding == NULL) { + goto exit; + } + if (nargs < 3) { + goto skip_optional; + } + if (!PyUnicode_Check(args[2])) { + _PyArg_BadArgument("bytearray", "argument 'errors'", "str", args[2]); + goto exit; + } + errors = _PyUnicode_AsUTF8NoNUL(args[2]); + if (errors == NULL) { + goto exit; + } +skip_optional: + self = _PyType_CAST(type)->tp_new(_PyType_CAST(type), + (PyObject *)&_Py_SINGLETON(tuple_empty), NULL); + if (self == NULL) { + goto exit; + } + _result = bytearray___init___impl((PyByteArrayObject *)self, arg, encoding, errors); + if (_result != 0) { + Py_DECREF(self); + goto exit; + } + return_value = self; + +exit: + return return_value; +} + PyDoc_STRVAR(bytearray_find__doc__, "find($self, sub[, start[, end]], /)\n" "--\n" @@ -1863,4 +1946,4 @@ bytearray_sizeof(PyObject *self, PyObject *Py_UNUSED(ignored)) { return bytearray_sizeof_impl((PyByteArrayObject *)self); } -/*[clinic end generated code: output=16fb40de6aefa5f8 input=a9049054013a1b77]*/ +/*[clinic end generated code: output=b67d1f1e84c7ed0f input=a9049054013a1b77]*/ diff --git a/Objects/clinic/bytesobject.c.h b/Objects/clinic/bytesobject.c.h index 56e8d089334a7a7..2b2850b5933cf9b 100644 --- a/Objects/clinic/bytesobject.c.h +++ b/Objects/clinic/bytesobject.c.h @@ -1357,7 +1357,8 @@ bytes_new_impl(PyTypeObject *type, PyObject *x, const char *encoding, const char *errors); static PyObject * -bytes_new(PyTypeObject *type, PyObject *args, PyObject *kwargs) +bytes_new_helper(PyTypeObject *type, PyObject *const *args, + Py_ssize_t nargs, Py_ssize_t nkw, PyObject *kwargs, PyObject *kwnames) { PyObject *return_value = NULL; #if defined(Py_BUILD_CORE) && !defined(Py_BUILD_CORE_MODULE) @@ -1389,13 +1390,12 @@ bytes_new(PyTypeObject *type, PyObject *args, PyObject *kwargs) #undef KWTUPLE PyObject *argsbuf[3]; PyObject * const *fastargs; - Py_ssize_t nargs = PyTuple_GET_SIZE(args); - Py_ssize_t noptargs = nargs + (kwargs ? PyDict_GET_SIZE(kwargs) : 0) - 0; + Py_ssize_t noptargs = nargs + nkw - 0; PyObject *x = NULL; const char *encoding = NULL; const char *errors = NULL; - fastargs = _PyArg_UnpackKeywords(_PyTuple_CAST(args)->ob_item, nargs, kwargs, NULL, &_parser, + fastargs = _PyArg_UnpackKeywords(args, nargs, kwargs, kwnames, &_parser, /*minpos*/ 0, /*maxpos*/ 3, /*minkw*/ 0, /*varpos*/ 0, argsbuf); if (!fastargs) { goto exit; @@ -1436,4 +1436,65 @@ bytes_new(PyTypeObject *type, PyObject *args, PyObject *kwargs) exit: return return_value; } -/*[clinic end generated code: output=2de57d656cc2f041 input=a9049054013a1b77]*/ + +static PyObject * +bytes_new(PyTypeObject *type, PyObject *args, PyObject *kwargs) +{ + return bytes_new_helper(type, _PyTuple_CAST(args)->ob_item, + PyTuple_GET_SIZE(args), + kwargs ? PyDict_GET_SIZE(kwargs) : 0, + kwargs, NULL); +} + +static PyObject * +bytes_vectorcall(PyObject *type, PyObject *const *args, + size_t nargsf, PyObject *kwnames) +{ + PyObject *return_value = NULL; + Py_ssize_t nargs = PyVectorcall_NARGS(nargsf); + PyObject *x = NULL; + const char *encoding = NULL; + const char *errors = NULL; + + assert(Py_Is(_PyType_CAST(type), &PyBytes_Type)); + /* Make sure the type object is immutable: the generated + * vectorcall doesn't deal e.g. with users reassigning __init__. */ + assert(PyType_HasFeature(_PyType_CAST(type), Py_TPFLAGS_IMMUTABLETYPE)); + if (kwnames != NULL || nargs > 3) { + return bytes_new_helper(_PyType_CAST(type), args, nargs, + kwnames ? PyTuple_GET_SIZE(kwnames) : 0, + NULL, kwnames); + } + if (nargs < 1) { + goto skip_optional; + } + x = args[0]; + if (nargs < 2) { + goto skip_optional; + } + if (!PyUnicode_Check(args[1])) { + _PyArg_BadArgument("bytes", "argument 'encoding'", "str", args[1]); + goto exit; + } + encoding = _PyUnicode_AsUTF8NoNUL(args[1]); + if (encoding == NULL) { + goto exit; + } + if (nargs < 3) { + goto skip_optional; + } + if (!PyUnicode_Check(args[2])) { + _PyArg_BadArgument("bytes", "argument 'errors'", "str", args[2]); + goto exit; + } + errors = _PyUnicode_AsUTF8NoNUL(args[2]); + if (errors == NULL) { + goto exit; + } +skip_optional: + return_value = bytes_new_impl(_PyType_CAST(type), x, encoding, errors); + +exit: + return return_value; +} +/*[clinic end generated code: output=9ea978e8e97e33ae input=a9049054013a1b77]*/