diff --git a/Doc/c-api/unicode.rst b/Doc/c-api/unicode.rst index 3b635fa7fa37441..dcae88b2c1cd9f9 100644 --- a/Doc/c-api/unicode.rst +++ b/Doc/c-api/unicode.rst @@ -425,6 +425,10 @@ APIs: the UCS1 range, it will be transformed into UCS1 (:c:macro:`PyUnicode_1BYTE_KIND`). + All characters must be in range [U+0000; U+10ffff]. If *kind* is + :c:macro:`PyUnicode_4BYTE_KIND` and the string contains invalid characters, + the behavior is undefined. + .. versionadded:: 3.3 @@ -1896,10 +1900,13 @@ object. .. c:function:: int PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *writer, const Py_UCS4 *str, Py_ssize_t size) - Writer the UCS4 string *str* into *writer*. + Write the UCS4 string *str* into *writer*. *size* is a number of UCS4 characters. + All characters must be in range [U+0000; U+10ffff]. If the string contains + invalid characters, the behavior is undefined. + On success, return ``0``. On error, set an exception, leave the writer unchanged, and return ``-1``. diff --git a/Lib/test/test_capi/test_unicode.py b/Lib/test/test_capi/test_unicode.py index 032b910a280083b..8c59c0b4e58b118 100644 --- a/Lib/test/test_capi/test_unicode.py +++ b/Lib/test/test_capi/test_unicode.py @@ -19,6 +19,11 @@ from _testcapi import PY_SSIZE_T_MIN, PY_SSIZE_T_MAX, SIZEOF_WCHAR_T +MAX_UNICODE = _testinternalcapi._Py_MAX_UNICODE +# The first invalid character after MAX_UNICODE +INVALID_CHAR = MAX_UNICODE + 1 +# Maximum invalid character which fits into 32-bit Py_UCS4 +MAX_INVALID_CHAR = 0xFFFF_FFFF NULL = None class Str(str): @@ -41,6 +46,12 @@ def __str__(self): SSTATE_INTERNED_IMMORTAL_STATIC = 3 +def assert_invalid_string(testcase, text): + # Check that a Unicode string contains invalid characters: + # not in range [U+0000; U+10ffff] + testcase.assertRaises(SystemError, list, text) + + class CAPITest(unittest.TestCase): def _test_check(self, check, *, exact): @@ -73,14 +84,14 @@ def test_new(self): self.assertEqual(new(0, maxchar), '') self.assertEqual(new(5, maxchar), chr(maxchar)*5) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX, maxchar) - self.assertEqual(new(0, 0x110000), '') + self.assertEqual(new(0, INVALID_CHAR), '') self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x4f60) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x4f60) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x1f600) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x1f600) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4, 0x1f600) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4+1, 0x1f600) - self.assertRaises(SystemError, new, 5, 0x110000) + self.assertRaises(SystemError, new, 5, INVALID_CHAR) self.assertRaises(SystemError, new, -1, 0) self.assertRaises(SystemError, new, PY_SSIZE_T_MIN, 0) @@ -115,7 +126,7 @@ def test_fill(self): s = strings[0] self.assertRaises(IndexError, fill, s, -1, 0, 0x78) self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78) - self.assertRaises(ValueError, fill, s, 0, 0, 0x110000) + self.assertRaises(ValueError, fill, s, 0, 0, INVALID_CHAR) self.assertRaises(SystemError, fill, b'abc', 0, 0, 0x78) self.assertRaises(SystemError, fill, [], 0, 0, 0x78) # CRASHES fill(s, 0, NULL, 0, 0) @@ -129,7 +140,7 @@ def _test_writechar(self, writechar, *, check): '\U0001f600\U0001f601\U0001f602' ] # one character for every kind + out of range code - chars = [0x78, 0xa9, 0x20ac, 0x1f638, 0x110000] + chars = [0x78, 0xa9, 0x20ac, 0x1f638, INVALID_CHAR] for i, s in enumerate(strings): for j, c in enumerate(chars): if j <= i: @@ -301,7 +312,22 @@ def test_fromkindanddata(self): self.assertRaises(ValueError, fromkindanddata, 1, NULL, -1) self.assertRaises(ValueError, fromkindanddata, 1, NULL, PY_SSIZE_T_MIN) # CRASHES fromkindanddata(1, NULL, 1) - # CRASHES fromkindanddata(4, b'\xff\xff\xff\xff') + + # Test invalid UCS-4 string. Create an invalid string in release mode, + # or raise SystemError in debug mode. + for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR): + with self.subTest(invalid_char=invalid_char): + # Test single character + ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder) + self.assertRaises(SystemError, fromkindanddata, 4, ucs4_char) + + # Test multiple characters + s = 'valid'.encode(enc4) + ucs4_char + if support.Py_DEBUG: + self.assertRaises(SystemError, fromkindanddata, 4, s) + else: + result = fromkindanddata(4, s) + assert_invalid_string(self, result) def test_substring(self): """Test PyUnicode_Substring()""" @@ -446,7 +472,7 @@ def check_format(expected, format, *args): check_format('\U0010ffff', b'%c', c_int(0x10ffff)) with self.assertRaises(OverflowError): - PyUnicode_FromFormat(b'%c', c_int(0x110000)) + PyUnicode_FromFormat(b'%c', c_int(INVALID_CHAR)) # Issue #18183 check_format('\U00010000\U00100000', b'%c%c', c_int(0x10000), c_int(0x100000)) @@ -1015,7 +1041,7 @@ def test_fromordinal(self): self.assertEqual(fromordinal(0x20ac), '\u20ac') self.assertEqual(fromordinal(0x1f600), '\U0001f600') - self.assertRaises(ValueError, fromordinal, 0x110000) + self.assertRaises(ValueError, fromordinal, INVALID_CHAR) self.assertRaises(ValueError, fromordinal, -1) def test_asutf8(self): @@ -1367,8 +1393,8 @@ def test_findchar(self): self.assertEqual(unicode_findchar(str, ord(ch), 0, len(str), -1), i) str = "!>_ *maxchar) { *maxchar = ch; if (*maxchar > MAX_UNICODE) { - PyErr_Format(PyExc_ValueError, - "character U+%x is not in range [U+0000; U+%x]", - ch, MAX_UNICODE); + unicode_invalid_character(PyExc_ValueError, ch); return -1; } } @@ -1834,17 +1842,15 @@ get_latin1_char(Py_UCS1 ch) static PyObject* unicode_char(Py_UCS4 ch) { - PyObject *unicode; - - assert(ch <= MAX_UNICODE); - if (ch < 256) { return get_latin1_char(ch); } - unicode = PyUnicode_New(1, ch); - if (unicode == NULL) + // Raise SystemError if the character is not in range [U+0000; MAX_UNICODE] + PyObject *unicode = PyUnicode_New(1, ch); + if (unicode == NULL) { return NULL; + } assert(PyUnicode_KIND(unicode) != PyUnicode_1BYTE_KIND); if (PyUnicode_KIND(unicode) == PyUnicode_2BYTE_KIND) { @@ -2224,15 +2230,36 @@ static PyObject* _PyUnicode_FromUCS4(const Py_UCS4 *u, Py_ssize_t size) { PyObject *res; - Py_UCS4 max_char; if (size == 0) _Py_RETURN_UNICODE_EMPTY(); assert(size > 0); - if (size == 1) + + if (size == 1) { + // Raise SystemError if the character is invalid return unicode_char(u[0]); + } + +#ifdef Py_DEBUG + // Check for invalid characters in debug mode + Py_UCS4 max_char = 127; + for (Py_ssize_t i = 0; i < size; i++) { + Py_UCS4 ch = u[i]; + if (ch > max_char) { + if (ch > MAX_UNICODE) { + unicode_invalid_character(PyExc_SystemError, ch); + return NULL; + } + max_char = ch; + } + } +#else + // gh-158445: Return MAX_UNICODE even if the string contains invalid + // characters. Checking for invalid characters would make the function + // slower whereas it's unlikely in practice. + Py_UCS4 max_char = ucs4lib_find_max_char(u, u + size); +#endif - max_char = ucs4lib_find_max_char(u, u + size); res = PyUnicode_New(size, max_char); if (!res) return NULL; @@ -2266,9 +2293,27 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer, return 0; } - Py_UCS4 max_char = ucs4lib_find_max_char(str, str + size); +#ifdef Py_DEBUG + // Check for invalid characters in debug mode + Py_UCS4 maxchar = 127; + for (Py_ssize_t i = 0; i < size; i++) { + Py_UCS4 ch = str[i]; + if (ch > maxchar) { + if (ch > MAX_UNICODE) { + unicode_invalid_character(PyExc_SystemError, ch); + return -1; + } + maxchar = ch; + } + } +#else + // gh-158445: Return MAX_UNICODE even if the string contains invalid + // characters. Checking for invalid characters would make the function + // slower whereas it's unlikely in practice. + Py_UCS4 maxchar = ucs4lib_find_max_char(str, str + size); +#endif - if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) { + if (_PyUnicodeWriter_Prepare(writer, size, maxchar) < 0) { return -1; } assert(_PyUnicodeWriter_CanWrite(writer));