Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions Lib/test/test_str.py
Original file line number Diff line number Diff line change
Expand Up @@ -566,6 +566,18 @@ def __str__(self): return self.sval
self.checkequalnofix('a b c d', ' ', 'join', ['a', 'b', 'c', 'd'])
self.checkequalnofix('abcd', '', 'join', ('a', 'b', 'c', 'd'))
self.checkequalnofix('w x y z', ' ', 'join', string_tests.Sequence('wxyz'))
# mixed kinds of the separator and the items
self.checkequalnofix('a\xe9b', '\xe9', 'join', ['a', 'b'])
self.checkequalnofix('a\u20acb', '\u20ac', 'join', ['a', 'b'])
self.checkequalnofix('a\xe9', '', 'join', ['a', '\xe9'])
self.checkequalnofix('\u20ac\xe9', '', 'join', ['\u20ac', '\xe9'])
self.checkequalnofix('\xe9-\u20ac-\U0001f600', '-', 'join',
['\xe9', '\u20ac', '\U0001f600'])
self.checkequalnofix('\U0001f600-a', '-', 'join', ['\U0001f600', 'a'])
# a single item which is a str subclass is copied
class S(str): pass
for s in 'a', '\xe9', '\u20ac', '\U0001f600':
self.checkequalnofix(s, '-', 'join', [S(s)])
self.checkraises(TypeError, ' ', 'join', ['1', '2', MyWrapper('foo')])
self.checkraises(TypeError, ' ', 'join', ['1', '2', '3', bytes()])
self.checkraises(TypeError, ' ', 'join', [1, 2, 3])
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1 @@
Improve performance of :meth:`str.join`.
44 changes: 26 additions & 18 deletions Objects/unicodeobject.c
Original file line number Diff line number Diff line change
Expand Up @@ -10457,10 +10457,10 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq
PyObject *item;
Py_ssize_t sz, i, res_offset;
Py_UCS4 maxchar;
Py_UCS4 item_maxchar;
/* kinds is the bitwise OR of the PyUnicode_KIND() values (1, 2, 4). */
unsigned int kinds = 0, all_ascii = 1;
int use_memcpy;
unsigned char *res_data = NULL, *sep_data = NULL;
PyObject *last_obj;
int kind = 0;

/* If empty sequence, return u"". */
Expand All @@ -10469,14 +10469,12 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq
}

/* If singleton sequence with an exact Unicode, return that. */
last_obj = NULL;
if (seqlen == 1) {
if (PyUnicode_CheckExact(items[0])) {
res = items[0];
return Py_NewRef(res);
}
seplen = 0;
maxchar = 0;
}
else {
/* Set up sep and seplen */
Expand All @@ -10486,7 +10484,6 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq
if (!sep)
goto onError;
seplen = 1;
maxchar = 32;
}
else {
if (!PyUnicode_Check(separator)) {
Expand All @@ -10498,12 +10495,12 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq
}
sep = separator;
seplen = PyUnicode_GET_LENGTH(separator);
maxchar = PyUnicode_MAX_CHAR_VALUE(separator);
/* inc refcount to keep this code path symmetric with the
above case of a blank separator */
Py_INCREF(sep);
}
last_obj = sep;
kinds = PyUnicode_KIND(sep);
all_ascii = PyUnicode_IS_ASCII(sep);
}

/* There are at least two things to join, or else we have a subclass
Expand All @@ -10527,23 +10524,34 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq
i, Py_TYPE(item)->tp_name);
goto onError;
}
add_sz = PyUnicode_GET_LENGTH(item);
item_maxchar = PyUnicode_MAX_CHAR_VALUE(item);
maxchar = Py_MAX(maxchar, item_maxchar);
if (i != 0) {
add_sz += seplen;
}
/* Count one separator per item; the extra one is removed below. */
add_sz = (size_t)PyUnicode_GET_LENGTH(item) + (size_t)seplen;
kinds |= PyUnicode_KIND(item);
all_ascii &= PyUnicode_IS_ASCII(item);
if (add_sz > (size_t)(PY_SSIZE_T_MAX - sz)) {
PyErr_SetString(PyExc_OverflowError,
"join() result is too long for a Python string");
goto onError;
}
sz += add_sz;
if (use_memcpy && last_obj != NULL) {
if (PyUnicode_KIND(last_obj) != PyUnicode_KIND(item))
use_memcpy = 0;
}
last_obj = item;
}
sz -= seplen;
/* memcpy() can be used if the separator and all items have the same
kind: the result then has that kind too. */
if (kinds & (kinds - 1)) {
use_memcpy = 0;
}
if (all_ascii) {
maxchar = 0x7f;
}
else if (kinds & PyUnicode_4BYTE_KIND) {
maxchar = MAX_UNICODE;
}
else if (kinds & PyUnicode_2BYTE_KIND) {
maxchar = 0xffff;
}
else {
maxchar = 0xff;
}

res = PyUnicode_New(sz, maxchar);
Expand Down
Loading