From 33239e5cdf28cb07e0e4ed7c6f3e7657b13b5154 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Wed, 30 Sep 2026 03:14:12 +0200 Subject: [PATCH] gh-158451: Add _PyUnicodeWriter_WriteLatin1String() test Fix _PyUnicodeWriter_WriteLatin1String() when the writer buffer kind is UCS-2 or UCS-4. --- Lib/test/test_capi/test_unicode.py | 18 ++++ ...-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst | 2 + Modules/_testcapi/unicode.c | 27 ++++++ Objects/unicode_writer.c | 91 +++++++------------ 4 files changed, 81 insertions(+), 57 deletions(-) create mode 100644 Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst diff --git a/Lib/test/test_capi/test_unicode.py b/Lib/test/test_capi/test_unicode.py index 36ecca48230ddd3..c511ffc0fda13c8 100644 --- a/Lib/test/test_capi/test_unicode.py +++ b/Lib/test/test_capi/test_unicode.py @@ -1970,6 +1970,24 @@ def test_ascii(self): writer.write_ascii(b"Python! ", 6) self.assertEqual(writer.finish(), "Hello Python") + def test_write_latin1(self): + # Test _PyUnicodeWriter_WriteLatin1String() + writer = self.create_writer(0) + # Start with ASCII buffer + writer.write_latin1(b"abc IGNORED", 3) + writer.write_latin1(b"IGNORED", 0) + # Change buffer kind to UCS-1 + writer.write_latin1(b"\xe9", 1) + # Change buffer kind to UCS-2 + writer.write_str('[\u20ac]') + writer.write_latin1(b"def\xa0", 4) + # Change buffer kind to UCS-4 + writer.write_str('[\U0010ffff]') + writer.write_latin1(b"ghi\xff.", 5) + writer.write_latin1(b"IGNORED", 0) + self.assertEqual(writer.finish(), + "abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.") + def test_invalid_utf8(self): writer = self.create_writer(0) with self.assertRaises(UnicodeDecodeError): diff --git a/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst b/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst new file mode 100644 index 000000000000000..a16dd024287076f --- /dev/null +++ b/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst @@ -0,0 +1,2 @@ +Fix :c:func:`!_PyUnicodeWriter_WriteLatin1String` when the writer buffer +kind is UCS-2 or UCS-4. Patch by Victor Stinner. diff --git a/Modules/_testcapi/unicode.c b/Modules/_testcapi/unicode.c index c62813f4f3768d7..ba9b205e07ef8bb 100644 --- a/Modules/_testcapi/unicode.c +++ b/Modules/_testcapi/unicode.c @@ -685,6 +685,32 @@ writer_write_substring(PyObject *self_raw, PyObject *args) } +static PyObject* +writer_write_latin1(PyObject *self_raw, PyObject *args) +{ + WriterObject *self = (WriterObject *)self_raw; + if (writer_check(self) < 0) { + return NULL; + } + + const char *str; + Py_ssize_t bsize, size; + if (!PyArg_ParseTuple(args, "z#n", &str, &bsize, &size)) { + return NULL; + } + + _PyUnicodeWriter *writer = (_PyUnicodeWriter*)self->writer; +_Py_COMP_DIAG_PUSH +_Py_COMP_DIAG_IGNORE_DEPR_DECLS + if (_PyUnicodeWriter_WriteLatin1String(writer, str, size) < 0) { + return NULL; + } +_Py_COMP_DIAG_POP + + Py_RETURN_NONE; +} + + static PyObject* writer_decodeutf8stateful(PyObject *self_raw, PyObject *args) { @@ -778,6 +804,7 @@ static PyMethodDef writer_methods[] = { {"write_str", _PyCFunction_CAST(writer_write_str), METH_O}, {"write_repr", _PyCFunction_CAST(writer_write_repr), METH_O}, {"write_substring", _PyCFunction_CAST(writer_write_substring), METH_VARARGS}, + {"write_latin1", _PyCFunction_CAST(writer_write_latin1), METH_VARARGS}, {"decodeutf8stateful", _PyCFunction_CAST(writer_decodeutf8stateful), METH_VARARGS}, {"get_pointer", _PyCFunction_CAST(writer_get_pointer), METH_VARARGS}, {"get_buffer", _PyCFunction_CAST(writer_get_buffer), METH_VARARGS}, diff --git a/Objects/unicode_writer.c b/Objects/unicode_writer.c index 8637be921e24546..0949e45d51cbad7 100644 --- a/Objects/unicode_writer.c +++ b/Objects/unicode_writer.c @@ -62,59 +62,6 @@ OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE. #include "stringlib/undef.h" -/* Copy an ASCII or latin1 char* string into a Python Unicode string. - - WARNING: The function doesn't copy the terminating null character and - doesn't check the maximum character (may write a latin1 character in an - ASCII string). */ -static void -unicode_write_cstr(PyObject *unicode, Py_ssize_t index, - const char *str, Py_ssize_t len) -{ - int kind = PyUnicode_KIND(unicode); - const void *data = PyUnicode_DATA(unicode); - const char *end = str + len; - - assert(index + len <= PyUnicode_GET_LENGTH(unicode)); - switch (kind) { - case PyUnicode_1BYTE_KIND: { -#ifdef Py_DEBUG - if (PyUnicode_IS_ASCII(unicode)) { - Py_UCS4 maxchar = ucs1lib_find_max_char( - (const Py_UCS1*)str, - (const Py_UCS1*)str + len); - assert(maxchar < 128); - } -#endif - memcpy((char *) data + index, str, len); - break; - } - case PyUnicode_2BYTE_KIND: { - Py_UCS2 *start = (Py_UCS2 *)data + index; - Py_UCS2 *ucs2 = start; - - for (; str < end; ++ucs2, ++str) - *ucs2 = (Py_UCS2)*str; - - assert((ucs2 - start) <= PyUnicode_GET_LENGTH(unicode)); - break; - } - case PyUnicode_4BYTE_KIND: { - Py_UCS4 *start = (Py_UCS4 *)data + index; - Py_UCS4 *ucs4 = start; - - for (; str < end; ++ucs4, ++str) - *ucs4 = (Py_UCS4)*str; - - assert((ucs4 - start) <= PyUnicode_GET_LENGTH(unicode)); - break; - } - default: - Py_UNREACHABLE(); - } -} - - void _PyUnicodeWriter_Init(_PyUnicodeWriter *writer) { @@ -550,13 +497,43 @@ int _PyUnicodeWriter_WriteLatin1String(_PyUnicodeWriter *writer, const char *str, Py_ssize_t len) { - Py_UCS4 maxchar; + if (len == 0) { + return 0; + } - maxchar = ucs1lib_find_max_char((const Py_UCS1*)str, (const Py_UCS1*)str + len); - if (_PyUnicodeWriter_Prepare(writer, len, maxchar) == -1) + const Py_UCS1 *ucs1 = (const Py_UCS1 *)str; + Py_UCS4 maxchar = ucs1lib_find_max_char(ucs1, ucs1 + len); + if (_PyUnicodeWriter_Prepare(writer, len, maxchar) < 0) { return -1; + } assert(_PyUnicodeWriter_CanWrite(writer)); - unicode_write_cstr(writer->buffer, writer->pos, str, len); + + Py_ssize_t index = writer->pos; + switch (writer->kind) { + case PyUnicode_1BYTE_KIND: { + memcpy((Py_UCS1 *)writer->data + index, ucs1, len); + break; + } + case PyUnicode_2BYTE_KIND: { + Py_UCS2 *ucs2 = (Py_UCS2 *)writer->data + index; + const Py_UCS1 *end = ucs1 + len; + for (; ucs1 < end; ++ucs2, ++ucs1) { + *ucs2 = (Py_UCS2)*ucs1; + } + break; + } + case PyUnicode_4BYTE_KIND: { + Py_UCS4 *ucs4 = (Py_UCS4 *)writer->data + index; + const Py_UCS1 *end = ucs1 + len; + for (; ucs1 < end; ++ucs4, ++ucs1) { + *ucs4 = (Py_UCS4)*ucs1; + } + break; + } + default: + Py_UNREACHABLE(); + } + writer->pos += len; return 0; }