Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 18 additions & 0 deletions Lib/test/test_capi/test_unicode.py
Original file line number Diff line number Diff line change
Expand Up @@ -1970,6 +1970,24 @@ def test_ascii(self):
writer.write_ascii(b"Python! <truncated>", 6)
self.assertEqual(writer.finish(), "Hello Python")

def test_write_latin1(self):
# Test _PyUnicodeWriter_WriteLatin1String()
writer = self.create_writer(0)
# Start with ASCII buffer
writer.write_latin1(b"abc IGNORED", 3)
writer.write_latin1(b"IGNORED", 0)
# Change buffer kind to UCS-1
writer.write_latin1(b"\xe9", 1)
# Change buffer kind to UCS-2
writer.write_str('[\u20ac]')
writer.write_latin1(b"def\xa0", 4)
# Change buffer kind to UCS-4
writer.write_str('[\U0010ffff]')
writer.write_latin1(b"ghi\xff.", 5)
writer.write_latin1(b"IGNORED", 0)
self.assertEqual(writer.finish(),
"abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.")

def test_invalid_utf8(self):
writer = self.create_writer(0)
with self.assertRaises(UnicodeDecodeError):
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,2 @@
Fix :c:func:`!_PyUnicodeWriter_WriteLatin1String` when the writer buffer
kind is UCS-2 or UCS-4. Patch by Victor Stinner.
27 changes: 27 additions & 0 deletions Modules/_testcapi/unicode.c
Original file line number Diff line number Diff line change
Expand Up @@ -685,6 +685,32 @@ writer_write_substring(PyObject *self_raw, PyObject *args)
}


static PyObject*
writer_write_latin1(PyObject *self_raw, PyObject *args)
{
WriterObject *self = (WriterObject *)self_raw;
if (writer_check(self) < 0) {
return NULL;
}

const char *str;
Py_ssize_t bsize, size;
if (!PyArg_ParseTuple(args, "z#n", &str, &bsize, &size)) {
return NULL;
}

_PyUnicodeWriter *writer = (_PyUnicodeWriter*)self->writer;
_Py_COMP_DIAG_PUSH
_Py_COMP_DIAG_IGNORE_DEPR_DECLS
if (_PyUnicodeWriter_WriteLatin1String(writer, str, size) < 0) {
return NULL;
}
_Py_COMP_DIAG_POP

Py_RETURN_NONE;
}


static PyObject*
writer_decodeutf8stateful(PyObject *self_raw, PyObject *args)
{
Expand Down Expand Up @@ -778,6 +804,7 @@ static PyMethodDef writer_methods[] = {
{"write_str", _PyCFunction_CAST(writer_write_str), METH_O},
{"write_repr", _PyCFunction_CAST(writer_write_repr), METH_O},
{"write_substring", _PyCFunction_CAST(writer_write_substring), METH_VARARGS},
{"write_latin1", _PyCFunction_CAST(writer_write_latin1), METH_VARARGS},
{"decodeutf8stateful", _PyCFunction_CAST(writer_decodeutf8stateful), METH_VARARGS},
{"get_pointer", _PyCFunction_CAST(writer_get_pointer), METH_VARARGS},
{"get_buffer", _PyCFunction_CAST(writer_get_buffer), METH_VARARGS},
Expand Down
91 changes: 34 additions & 57 deletions Objects/unicode_writer.c
Original file line number Diff line number Diff line change
Expand Up @@ -62,59 +62,6 @@ OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
#include "stringlib/undef.h"


/* Copy an ASCII or latin1 char* string into a Python Unicode string.

WARNING: The function doesn't copy the terminating null character and
doesn't check the maximum character (may write a latin1 character in an
ASCII string). */
static void
unicode_write_cstr(PyObject *unicode, Py_ssize_t index,
const char *str, Py_ssize_t len)
{
int kind = PyUnicode_KIND(unicode);
const void *data = PyUnicode_DATA(unicode);
const char *end = str + len;

assert(index + len <= PyUnicode_GET_LENGTH(unicode));
switch (kind) {
case PyUnicode_1BYTE_KIND: {
#ifdef Py_DEBUG
if (PyUnicode_IS_ASCII(unicode)) {
Py_UCS4 maxchar = ucs1lib_find_max_char(
(const Py_UCS1*)str,
(const Py_UCS1*)str + len);
assert(maxchar < 128);
}
#endif
memcpy((char *) data + index, str, len);
break;
}
case PyUnicode_2BYTE_KIND: {
Py_UCS2 *start = (Py_UCS2 *)data + index;
Py_UCS2 *ucs2 = start;

for (; str < end; ++ucs2, ++str)
*ucs2 = (Py_UCS2)*str;

assert((ucs2 - start) <= PyUnicode_GET_LENGTH(unicode));
break;
}
case PyUnicode_4BYTE_KIND: {
Py_UCS4 *start = (Py_UCS4 *)data + index;
Py_UCS4 *ucs4 = start;

for (; str < end; ++ucs4, ++str)
*ucs4 = (Py_UCS4)*str;

assert((ucs4 - start) <= PyUnicode_GET_LENGTH(unicode));
break;
}
default:
Py_UNREACHABLE();
}
}


void
_PyUnicodeWriter_Init(_PyUnicodeWriter *writer)
{
Expand Down Expand Up @@ -550,13 +497,43 @@ int
_PyUnicodeWriter_WriteLatin1String(_PyUnicodeWriter *writer,
const char *str, Py_ssize_t len)
{
Py_UCS4 maxchar;
if (len == 0) {
return 0;
}

maxchar = ucs1lib_find_max_char((const Py_UCS1*)str, (const Py_UCS1*)str + len);
if (_PyUnicodeWriter_Prepare(writer, len, maxchar) == -1)
const Py_UCS1 *ucs1 = (const Py_UCS1 *)str;
Py_UCS4 maxchar = ucs1lib_find_max_char(ucs1, ucs1 + len);
if (_PyUnicodeWriter_Prepare(writer, len, maxchar) < 0) {
return -1;
}
assert(_PyUnicodeWriter_CanWrite(writer));
unicode_write_cstr(writer->buffer, writer->pos, str, len);

Py_ssize_t index = writer->pos;
switch (writer->kind) {
case PyUnicode_1BYTE_KIND: {
memcpy((Py_UCS1 *)writer->data + index, ucs1, len);
break;
}
case PyUnicode_2BYTE_KIND: {
Py_UCS2 *ucs2 = (Py_UCS2 *)writer->data + index;
const Py_UCS1 *end = ucs1 + len;
for (; ucs1 < end; ++ucs2, ++ucs1) {
*ucs2 = (Py_UCS2)*ucs1;
}
break;
}
case PyUnicode_4BYTE_KIND: {
Py_UCS4 *ucs4 = (Py_UCS4 *)writer->data + index;
const Py_UCS1 *end = ucs1 + len;
for (; ucs1 < end; ++ucs4, ++ucs1) {
*ucs4 = (Py_UCS4)*ucs1;
}
break;
}
default:
Py_UNREACHABLE();
}

writer->pos += len;
return 0;
}
Expand Down
Loading