Skip to content

Commit 26d3f5d

Browse files
authored
gh-158451: Add _PyUnicodeWriter_WriteLatin1String() test (#158464)
Fix _PyUnicodeWriter_WriteLatin1String() when the writer buffer kind is UCS-2 or UCS-4.
1 parent 00307b0 commit 26d3f5d

4 files changed

Lines changed: 81 additions & 57 deletions

File tree

‎Lib/test/test_capi/test_unicode.py‎

Lines changed: 18 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1970,6 +1970,24 @@ def test_ascii(self):
19701970
writer.write_ascii(b"Python! <truncated>", 6)
19711971
self.assertEqual(writer.finish(), "Hello Python")
19721972

1973+
def test_write_latin1(self):
1974+
# Test _PyUnicodeWriter_WriteLatin1String()
1975+
writer = self.create_writer(0)
1976+
# Start with ASCII buffer
1977+
writer.write_latin1(b"abc IGNORED", 3)
1978+
writer.write_latin1(b"IGNORED", 0)
1979+
# Change buffer kind to UCS-1
1980+
writer.write_latin1(b"\xe9", 1)
1981+
# Change buffer kind to UCS-2
1982+
writer.write_str('[\u20ac]')
1983+
writer.write_latin1(b"def\xa0", 4)
1984+
# Change buffer kind to UCS-4
1985+
writer.write_str('[\U0010ffff]')
1986+
writer.write_latin1(b"ghi\xff.", 5)
1987+
writer.write_latin1(b"IGNORED", 0)
1988+
self.assertEqual(writer.finish(),
1989+
"abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.")
1990+
19731991
def test_invalid_utf8(self):
19741992
writer = self.create_writer(0)
19751993
with self.assertRaises(UnicodeDecodeError):
Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,2 @@
1+
Fix :c:func:`!_PyUnicodeWriter_WriteLatin1String` when the writer buffer
2+
kind is UCS-2 or UCS-4. Patch by Victor Stinner.

‎Modules/_testcapi/unicode.c‎

Lines changed: 27 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -685,6 +685,32 @@ writer_write_substring(PyObject *self_raw, PyObject *args)
685685
}
686686

687687

688+
static PyObject*
689+
writer_write_latin1(PyObject *self_raw, PyObject *args)
690+
{
691+
WriterObject *self = (WriterObject *)self_raw;
692+
if (writer_check(self) < 0) {
693+
return NULL;
694+
}
695+
696+
const char *str;
697+
Py_ssize_t bsize, size;
698+
if (!PyArg_ParseTuple(args, "z#n", &str, &bsize, &size)) {
699+
return NULL;
700+
}
701+
702+
_PyUnicodeWriter *writer = (_PyUnicodeWriter*)self->writer;
703+
_Py_COMP_DIAG_PUSH
704+
_Py_COMP_DIAG_IGNORE_DEPR_DECLS
705+
if (_PyUnicodeWriter_WriteLatin1String(writer, str, size) < 0) {
706+
return NULL;
707+
}
708+
_Py_COMP_DIAG_POP
709+
710+
Py_RETURN_NONE;
711+
}
712+
713+
688714
static PyObject*
689715
writer_decodeutf8stateful(PyObject *self_raw, PyObject *args)
690716
{
@@ -778,6 +804,7 @@ static PyMethodDef writer_methods[] = {
778804
{"write_str", _PyCFunction_CAST(writer_write_str), METH_O},
779805
{"write_repr", _PyCFunction_CAST(writer_write_repr), METH_O},
780806
{"write_substring", _PyCFunction_CAST(writer_write_substring), METH_VARARGS},
807+
{"write_latin1", _PyCFunction_CAST(writer_write_latin1), METH_VARARGS},
781808
{"decodeutf8stateful", _PyCFunction_CAST(writer_decodeutf8stateful), METH_VARARGS},
782809
{"get_pointer", _PyCFunction_CAST(writer_get_pointer), METH_VARARGS},
783810
{"get_buffer", _PyCFunction_CAST(writer_get_buffer), METH_VARARGS},

‎Objects/unicode_writer.c‎

Lines changed: 34 additions & 57 deletions
Original file line numberDiff line numberDiff line change
@@ -62,59 +62,6 @@ OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS SOFTWARE.
6262
#include "stringlib/undef.h"
6363

6464

65-
/* Copy an ASCII or latin1 char* string into a Python Unicode string.
66-
67-
WARNING: The function doesn't copy the terminating null character and
68-
doesn't check the maximum character (may write a latin1 character in an
69-
ASCII string). */
70-
static void
71-
unicode_write_cstr(PyObject *unicode, Py_ssize_t index,
72-
const char *str, Py_ssize_t len)
73-
{
74-
int kind = PyUnicode_KIND(unicode);
75-
const void *data = PyUnicode_DATA(unicode);
76-
const char *end = str + len;
77-
78-
assert(index + len <= PyUnicode_GET_LENGTH(unicode));
79-
switch (kind) {
80-
case PyUnicode_1BYTE_KIND: {
81-
#ifdef Py_DEBUG
82-
if (PyUnicode_IS_ASCII(unicode)) {
83-
Py_UCS4 maxchar = ucs1lib_find_max_char(
84-
(const Py_UCS1*)str,
85-
(const Py_UCS1*)str + len);
86-
assert(maxchar < 128);
87-
}
88-
#endif
89-
memcpy((char *) data + index, str, len);
90-
break;
91-
}
92-
case PyUnicode_2BYTE_KIND: {
93-
Py_UCS2 *start = (Py_UCS2 *)data + index;
94-
Py_UCS2 *ucs2 = start;
95-
96-
for (; str < end; ++ucs2, ++str)
97-
*ucs2 = (Py_UCS2)*str;
98-
99-
assert((ucs2 - start) <= PyUnicode_GET_LENGTH(unicode));
100-
break;
101-
}
102-
case PyUnicode_4BYTE_KIND: {
103-
Py_UCS4 *start = (Py_UCS4 *)data + index;
104-
Py_UCS4 *ucs4 = start;
105-
106-
for (; str < end; ++ucs4, ++str)
107-
*ucs4 = (Py_UCS4)*str;
108-
109-
assert((ucs4 - start) <= PyUnicode_GET_LENGTH(unicode));
110-
break;
111-
}
112-
default:
113-
Py_UNREACHABLE();
114-
}
115-
}
116-
117-
11865
void
11966
_PyUnicodeWriter_Init(_PyUnicodeWriter *writer)
12067
{
@@ -550,13 +497,43 @@ int
550497
_PyUnicodeWriter_WriteLatin1String(_PyUnicodeWriter *writer,
551498
const char *str, Py_ssize_t len)
552499
{
553-
Py_UCS4 maxchar;
500+
if (len == 0) {
501+
return 0;
502+
}
554503

555-
maxchar = ucs1lib_find_max_char((const Py_UCS1*)str, (const Py_UCS1*)str + len);
556-
if (_PyUnicodeWriter_Prepare(writer, len, maxchar) == -1)
504+
const Py_UCS1 *ucs1 = (const Py_UCS1 *)str;
505+
Py_UCS4 maxchar = ucs1lib_find_max_char(ucs1, ucs1 + len);
506+
if (_PyUnicodeWriter_Prepare(writer, len, maxchar) < 0) {
557507
return -1;
508+
}
558509
assert(_PyUnicodeWriter_CanWrite(writer));
559-
unicode_write_cstr(writer->buffer, writer->pos, str, len);
510+
511+
Py_ssize_t index = writer->pos;
512+
switch (writer->kind) {
513+
case PyUnicode_1BYTE_KIND: {
514+
memcpy((Py_UCS1 *)writer->data + index, ucs1, len);
515+
break;
516+
}
517+
case PyUnicode_2BYTE_KIND: {
518+
Py_UCS2 *ucs2 = (Py_UCS2 *)writer->data + index;
519+
const Py_UCS1 *end = ucs1 + len;
520+
for (; ucs1 < end; ++ucs2, ++ucs1) {
521+
*ucs2 = (Py_UCS2)*ucs1;
522+
}
523+
break;
524+
}
525+
case PyUnicode_4BYTE_KIND: {
526+
Py_UCS4 *ucs4 = (Py_UCS4 *)writer->data + index;
527+
const Py_UCS1 *end = ucs1 + len;
528+
for (; ucs1 < end; ++ucs4, ++ucs1) {
529+
*ucs4 = (Py_UCS4)*ucs1;
530+
}
531+
break;
532+
}
533+
default:
534+
Py_UNREACHABLE();
535+
}
536+
560537
writer->pos += len;
561538
return 0;
562539
}

0 commit comments

Comments
 (0)