From 815b22cf0f3fb058bf90ce3433935900483683be Mon Sep 17 00:00:00 2001 From: Pieter Eendebak Date: Thu, 8 Oct 2026 20:45:44 +0200 Subject: [PATCH 1/3] gh-159037: Speed up the pre-pass of str.join() Compute the maximum character and whether memcpy() can be used from the OR of the item kinds and the AND of their ASCII flags after the pre-pass, instead of per item. Count one separator per item and subtract the extra one after the loop. Co-Authored-By: Claude Fable 5.1 --- Objects/unicodeobject.c | 44 ++++++++++++++++++++++++----------------- 1 file changed, 26 insertions(+), 18 deletions(-) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index ded13cc37f69f17..1df1e9b6a4513f2 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -10457,10 +10457,10 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq PyObject *item; Py_ssize_t sz, i, res_offset; Py_UCS4 maxchar; - Py_UCS4 item_maxchar; + /* kinds is the bitwise OR of the PyUnicode_KIND() values (1, 2, 4). */ + unsigned int kinds = 0, all_ascii = 1; int use_memcpy; unsigned char *res_data = NULL, *sep_data = NULL; - PyObject *last_obj; int kind = 0; /* If empty sequence, return u"". */ @@ -10469,14 +10469,12 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq } /* If singleton sequence with an exact Unicode, return that. */ - last_obj = NULL; if (seqlen == 1) { if (PyUnicode_CheckExact(items[0])) { res = items[0]; return Py_NewRef(res); } seplen = 0; - maxchar = 0; } else { /* Set up sep and seplen */ @@ -10486,7 +10484,6 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq if (!sep) goto onError; seplen = 1; - maxchar = 32; } else { if (!PyUnicode_Check(separator)) { @@ -10498,12 +10495,12 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq } sep = separator; seplen = PyUnicode_GET_LENGTH(separator); - maxchar = PyUnicode_MAX_CHAR_VALUE(separator); /* inc refcount to keep this code path symmetric with the above case of a blank separator */ Py_INCREF(sep); } - last_obj = sep; + kinds = PyUnicode_KIND(sep); + all_ascii = PyUnicode_IS_ASCII(sep); } /* There are at least two things to join, or else we have a subclass @@ -10527,23 +10524,34 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq i, Py_TYPE(item)->tp_name); goto onError; } - add_sz = PyUnicode_GET_LENGTH(item); - item_maxchar = PyUnicode_MAX_CHAR_VALUE(item); - maxchar = Py_MAX(maxchar, item_maxchar); - if (i != 0) { - add_sz += seplen; - } + /* Count one separator per item; the extra one is removed below. */ + add_sz = (size_t)PyUnicode_GET_LENGTH(item) + (size_t)seplen; + kinds |= PyUnicode_KIND(item); + all_ascii &= PyUnicode_IS_ASCII(item); if (add_sz > (size_t)(PY_SSIZE_T_MAX - sz)) { PyErr_SetString(PyExc_OverflowError, "join() result is too long for a Python string"); goto onError; } sz += add_sz; - if (use_memcpy && last_obj != NULL) { - if (PyUnicode_KIND(last_obj) != PyUnicode_KIND(item)) - use_memcpy = 0; - } - last_obj = item; + } + sz -= seplen; + /* memcpy() can be used if the separator and all items have the same + kind: the result then has that kind too. */ + if (kinds & (kinds - 1)) { + use_memcpy = 0; + } + if (all_ascii) { + maxchar = 0x7f; + } + else if (kinds & PyUnicode_4BYTE_KIND) { + maxchar = MAX_UNICODE; + } + else if (kinds & PyUnicode_2BYTE_KIND) { + maxchar = 0xffff; + } + else { + maxchar = 0xff; } res = PyUnicode_New(sz, maxchar); From f3a2c156082f2f859c4804e0c4356e2153f30d4b Mon Sep 17 00:00:00 2001 From: Pieter Eendebak Date: Thu, 8 Oct 2026 20:45:44 +0200 Subject: [PATCH 2/3] gh-159037: Add str.join() tests for separators and items of mixed kinds Co-Authored-By: Claude Fable 5.1 --- Lib/test/test_str.py | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/Lib/test/test_str.py b/Lib/test/test_str.py index bef0e62fb42bccc..3aa01e541d1d74a 100644 --- a/Lib/test/test_str.py +++ b/Lib/test/test_str.py @@ -566,6 +566,18 @@ def __str__(self): return self.sval self.checkequalnofix('a b c d', ' ', 'join', ['a', 'b', 'c', 'd']) self.checkequalnofix('abcd', '', 'join', ('a', 'b', 'c', 'd')) self.checkequalnofix('w x y z', ' ', 'join', string_tests.Sequence('wxyz')) + # mixed kinds of the separator and the items + self.checkequalnofix('a\xe9b', '\xe9', 'join', ['a', 'b']) + self.checkequalnofix('a\u20acb', '\u20ac', 'join', ['a', 'b']) + self.checkequalnofix('a\xe9', '', 'join', ['a', '\xe9']) + self.checkequalnofix('\u20ac\xe9', '', 'join', ['\u20ac', '\xe9']) + self.checkequalnofix('\xe9-\u20ac-\U0001f600', '-', 'join', + ['\xe9', '\u20ac', '\U0001f600']) + self.checkequalnofix('\U0001f600-a', '-', 'join', ['\U0001f600', 'a']) + # a single item which is a str subclass is copied + class S(str): pass + for s in 'a', '\xe9', '\u20ac', '\U0001f600': + self.checkequalnofix(s, '-', 'join', [S(s)]) self.checkraises(TypeError, ' ', 'join', ['1', '2', MyWrapper('foo')]) self.checkraises(TypeError, ' ', 'join', ['1', '2', '3', bytes()]) self.checkraises(TypeError, ' ', 'join', [1, 2, 3]) From 866211dda4b96b6cc0224b57a2b0836637b55d33 Mon Sep 17 00:00:00 2001 From: "blurb-it[bot]" <43283697+blurb-it[bot]@users.noreply.github.com> Date: Thu, 8 Oct 2026 19:03:21 +0000 Subject: [PATCH 3/3] =?UTF-8?q?=F0=9F=93=9C=F0=9F=A4=96=20Added=20by=20blu?= =?UTF-8?q?rb=5Fit.?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../2026-10-08-19-03-14.gh-issue-159037.nbQ5wk.rst | 1 + 1 file changed, 1 insertion(+) create mode 100644 Misc/NEWS.d/next/Core_and_Builtins/2026-10-08-19-03-14.gh-issue-159037.nbQ5wk.rst diff --git a/Misc/NEWS.d/next/Core_and_Builtins/2026-10-08-19-03-14.gh-issue-159037.nbQ5wk.rst b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-08-19-03-14.gh-issue-159037.nbQ5wk.rst new file mode 100644 index 000000000000000..a27cc2782d0b7c0 --- /dev/null +++ b/Misc/NEWS.d/next/Core_and_Builtins/2026-10-08-19-03-14.gh-issue-159037.nbQ5wk.rst @@ -0,0 +1 @@ +Improve performance of :meth:`str.join`.