From d43b1b5d2ab1b5cd5fd1ceecae2fb42d930ab0d3 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Sun, 4 Oct 2026 01:56:05 +0200 Subject: [PATCH] gh-158451: Optimize PyUnicode_Join() Move tests on the separator outside the loop. Add a loop version for empty separator. --- Objects/unicodeobject.c | 90 ++++++++++++++++++++++++++--------------- 1 file changed, 57 insertions(+), 33 deletions(-) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index e630dbb77f87251..c6ffc937a7ad9f3 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -1835,8 +1835,7 @@ PyUnicode_Resize(PyObject **p_unicode, Py_ssize_t length) static PyObject* get_latin1_char(Py_UCS1 ch) { - PyObject *o = LATIN1(ch); - return o; + return LATIN1(ch); } static PyObject* @@ -10489,9 +10488,7 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq /* Set up sep and seplen */ if (separator == NULL) { /* fall back to a blank space separator */ - sep = PyUnicode_FromOrdinal(' '); - if (!sep) - goto onError; + sep = get_latin1_char(' '); seplen = 1; maxchar = 32; } @@ -10562,51 +10559,78 @@ _PyUnicode_JoinArray(PyObject *separator, PyObject *const *items, Py_ssize_t seq use_memcpy = 0; #else if (use_memcpy) { - res_data = PyUnicode_1BYTE_DATA(res); + res_data = PyUnicode_DATA(res); kind = PyUnicode_KIND(res); if (seplen != 0) - sep_data = PyUnicode_1BYTE_DATA(sep); + sep_data = PyUnicode_DATA(sep); } #endif if (use_memcpy) { - for (i = 0; i < seqlen; ++i) { - Py_ssize_t itemlen; - item = items[i]; - - /* Copy item, and maybe the separator. */ - if (i && seplen != 0) { - memcpy(res_data, - sep_data, - kind * seplen); - res_data += kind * seplen; - } - - itemlen = PyUnicode_GET_LENGTH(item); + if (seplen != 0) { + item = items[0]; + Py_ssize_t itemlen = PyUnicode_GET_LENGTH(item); if (itemlen != 0) { - memcpy(res_data, - PyUnicode_DATA(item), - kind * itemlen); + memcpy(res_data, PyUnicode_DATA(item), kind * itemlen); res_data += kind * itemlen; } + + for (i = 1; i < seqlen; ++i) { + /* Copy item, and maybe the separator. */ + memcpy(res_data, sep_data, kind * seplen); + res_data += kind * seplen; + + item = items[i]; + itemlen = PyUnicode_GET_LENGTH(item); + if (itemlen != 0) { + memcpy(res_data, PyUnicode_DATA(item), kind * itemlen); + res_data += kind * itemlen; + } + } + } + else { + for (i = 0; i < seqlen; ++i) { + item = items[i]; + Py_ssize_t itemlen = PyUnicode_GET_LENGTH(item); + if (itemlen != 0) { + memcpy(res_data, PyUnicode_DATA(item), kind * itemlen); + res_data += kind * itemlen; + } + } } assert(res_data == PyUnicode_1BYTE_DATA(res) + kind * PyUnicode_GET_LENGTH(res)); } else { - for (i = 0, res_offset = 0; i < seqlen; ++i) { - Py_ssize_t itemlen; - item = items[i]; + if (seplen != 0) { + res_offset = 0; + item = items[0]; + Py_ssize_t itemlen = PyUnicode_GET_LENGTH(item); + if (itemlen != 0) { + _PyUnicode_FastCopyCharacters(res, res_offset, item, 0, itemlen); + res_offset += itemlen; + } - /* Copy item, and maybe the separator. */ - if (i && seplen != 0) { + for (i = 1; i < seqlen; ++i) { + /* Copy item, and maybe the separator. */ _PyUnicode_FastCopyCharacters(res, res_offset, sep, 0, seplen); res_offset += seplen; - } - itemlen = PyUnicode_GET_LENGTH(item); - if (itemlen != 0) { - _PyUnicode_FastCopyCharacters(res, res_offset, item, 0, itemlen); - res_offset += itemlen; + item = items[i]; + itemlen = PyUnicode_GET_LENGTH(item); + if (itemlen != 0) { + _PyUnicode_FastCopyCharacters(res, res_offset, item, 0, itemlen); + res_offset += itemlen; + } + } + } + else { + for (i = 0, res_offset = 0; i < seqlen; ++i) { + item = items[i]; + Py_ssize_t itemlen = PyUnicode_GET_LENGTH(item); + if (itemlen != 0) { + _PyUnicode_FastCopyCharacters(res, res_offset, item, 0, itemlen); + res_offset += itemlen; + } } } assert(res_offset == PyUnicode_GET_LENGTH(res));