Skip to content

Commit 4a8e756

Browse files
committed
gh-158445: Detect invalid UCS4 characters in debug mode
In debug mode, PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND) and PyUnicodeWriter_WriteUCS4() now raise a SystemError if a character is not in range [U+0000; U+10ffff], instead of creating an invalid str object. In release mode, PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND) raises a SystemError if the size is 1. * In debug mode, unicode_char() no longer fails with an assertion error if the character is invalid. Instead, raise SystemError. * Add tests on invalid UCS-4 characters. * Add _testinternalcapi._Py_MAX_UNICODE. * Add unicode_invalid_character() helper function.
1 parent dc0b1f8 commit 4a8e756

4 files changed

Lines changed: 135 additions & 31 deletions

File tree

‎Doc/c-api/unicode.rst‎

Lines changed: 8 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -425,6 +425,10 @@ APIs:
425425
the UCS1 range, it will be transformed into UCS1
426426
(:c:macro:`PyUnicode_1BYTE_KIND`).
427427
428+
All characters must be in range [U+0000; U+10ffff]. If *kind* is
429+
:c:macro:`PyUnicode_4BYTE_KIND` and the string contains invalid characters,
430+
the behavior is undefined.
431+
428432
.. versionadded:: 3.3
429433
430434
@@ -1896,10 +1900,13 @@ object.
18961900
18971901
.. c:function:: int PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *writer, const Py_UCS4 *str, Py_ssize_t size)
18981902
1899-
Writer the UCS4 string *str* into *writer*.
1903+
Write the UCS4 string *str* into *writer*.
19001904
19011905
*size* is a number of UCS4 characters.
19021906
1907+
All characters must be in range [U+0000; U+10ffff]. If the string contains
1908+
invalid characters, the behavior is undefined.
1909+
19031910
On success, return ``0``.
19041911
On error, set an exception, leave the writer unchanged, and return ``-1``.
19051912

‎Lib/test/test_capi/test_unicode.py‎

Lines changed: 64 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -19,6 +19,11 @@
1919
from _testcapi import PY_SSIZE_T_MIN, PY_SSIZE_T_MAX, SIZEOF_WCHAR_T
2020

2121

22+
MAX_UNICODE = _testinternalcapi._Py_MAX_UNICODE
23+
# The first invalid character after MAX_UNICODE
24+
INVALID_CHAR = MAX_UNICODE + 1
25+
# Maximum invalid character which fits into 32-bit Py_UCS4
26+
MAX_INVALID_CHAR = 0xFFFF_FFFF
2227
NULL = None
2328

2429
class Str(str):
@@ -41,6 +46,12 @@ def __str__(self):
4146
SSTATE_INTERNED_IMMORTAL_STATIC = 3
4247

4348

49+
def assert_invalid_string(testcase, text):
50+
# Check that a Unicode string contains invalid characters:
51+
# not in range [U+0000; U+10ffff]
52+
testcase.assertRaises(SystemError, list, text)
53+
54+
4455
class CAPITest(unittest.TestCase):
4556

4657
def _test_check(self, check, *, exact):
@@ -73,14 +84,14 @@ def test_new(self):
7384
self.assertEqual(new(0, maxchar), '')
7485
self.assertEqual(new(5, maxchar), chr(maxchar)*5)
7586
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX, maxchar)
76-
self.assertEqual(new(0, 0x110000), '')
87+
self.assertEqual(new(0, INVALID_CHAR), '')
7788
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x4f60)
7889
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x4f60)
7990
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x1f600)
8091
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x1f600)
8192
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4, 0x1f600)
8293
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4+1, 0x1f600)
83-
self.assertRaises(SystemError, new, 5, 0x110000)
94+
self.assertRaises(SystemError, new, 5, INVALID_CHAR)
8495
self.assertRaises(SystemError, new, -1, 0)
8596
self.assertRaises(SystemError, new, PY_SSIZE_T_MIN, 0)
8697

@@ -115,7 +126,7 @@ def test_fill(self):
115126
s = strings[0]
116127
self.assertRaises(IndexError, fill, s, -1, 0, 0x78)
117128
self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78)
118-
self.assertRaises(ValueError, fill, s, 0, 0, 0x110000)
129+
self.assertRaises(ValueError, fill, s, 0, 0, INVALID_CHAR)
119130
self.assertRaises(SystemError, fill, b'abc', 0, 0, 0x78)
120131
self.assertRaises(SystemError, fill, [], 0, 0, 0x78)
121132
# CRASHES fill(s, 0, NULL, 0, 0)
@@ -129,7 +140,7 @@ def _test_writechar(self, writechar, *, check):
129140
'\U0001f600\U0001f601\U0001f602'
130141
]
131142
# one character for every kind + out of range code
132-
chars = [0x78, 0xa9, 0x20ac, 0x1f638, 0x110000]
143+
chars = [0x78, 0xa9, 0x20ac, 0x1f638, INVALID_CHAR]
133144
for i, s in enumerate(strings):
134145
for j, c in enumerate(chars):
135146
if j <= i:
@@ -301,7 +312,22 @@ def test_fromkindanddata(self):
301312
self.assertRaises(ValueError, fromkindanddata, 1, NULL, -1)
302313
self.assertRaises(ValueError, fromkindanddata, 1, NULL, PY_SSIZE_T_MIN)
303314
# CRASHES fromkindanddata(1, NULL, 1)
304-
# CRASHES fromkindanddata(4, b'\xff\xff\xff\xff')
315+
316+
# Test invalid UCS-4 string. Create an invalid string in release mode,
317+
# or raise SystemError in debug mode.
318+
for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
319+
with self.subTest(invalid_char=invalid_char):
320+
# Test single character
321+
ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
322+
self.assertRaises(SystemError, fromkindanddata, 4, ucs4_char)
323+
324+
# Test multiple characters
325+
s = 'valid'.encode(enc4) + ucs4_char
326+
if support.Py_DEBUG:
327+
self.assertRaises(SystemError, fromkindanddata, 4, s)
328+
else:
329+
result = fromkindanddata(4, s)
330+
assert_invalid_string(self, result)
305331

306332
def test_substring(self):
307333
"""Test PyUnicode_Substring()"""
@@ -446,7 +472,7 @@ def check_format(expected, format, *args):
446472
check_format('\U0010ffff',
447473
b'%c', c_int(0x10ffff))
448474
with self.assertRaises(OverflowError):
449-
PyUnicode_FromFormat(b'%c', c_int(0x110000))
475+
PyUnicode_FromFormat(b'%c', c_int(INVALID_CHAR))
450476
# Issue #18183
451477
check_format('\U00010000\U00100000',
452478
b'%c%c', c_int(0x10000), c_int(0x100000))
@@ -1015,7 +1041,7 @@ def test_fromordinal(self):
10151041
self.assertEqual(fromordinal(0x20ac), '\u20ac')
10161042
self.assertEqual(fromordinal(0x1f600), '\U0001f600')
10171043

1018-
self.assertRaises(ValueError, fromordinal, 0x110000)
1044+
self.assertRaises(ValueError, fromordinal, INVALID_CHAR)
10191045
self.assertRaises(ValueError, fromordinal, -1)
10201046

10211047
def test_asutf8(self):
@@ -1367,8 +1393,8 @@ def test_findchar(self):
13671393
self.assertEqual(unicode_findchar(str, ord(ch), 0, len(str), -1), i)
13681394

13691395
str = "!>_<!"
1370-
self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), 1), -1)
1371-
self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), -1), -1)
1396+
self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), 1), -1)
1397+
self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), -1), -1)
13721398
# start < end
13731399
self.assertEqual(unicode_findchar(str, ord('!'), 1, len(str)+1, 1), 4)
13741400
self.assertEqual(unicode_findchar(str, ord('!'), 1, PY_SSIZE_T_MAX, 1), 4)
@@ -1757,7 +1783,7 @@ def test_max_char_value(self):
17571783
self.assertEqual(max_char_value('ascii'), 0x7f)
17581784
self.assertEqual(max_char_value('latin1:\xe9'), 0xff)
17591785
self.assertEqual(max_char_value('bmp:\u20ac'), 0xffff)
1760-
self.assertEqual(max_char_value('\U0010ffff'), 0x10_ffff)
1786+
self.assertEqual(max_char_value(chr(0x10_0000)), 0x10_ffff)
17611787

17621788
# CRASHES max_char_value(NULL)
17631789

@@ -1935,8 +1961,8 @@ def test_write_char(self):
19351961
writer.write_char(ord('$'))
19361962
writer.write_char(0x20ac)
19371963
writer.write_char(0x10_ffff)
1938-
self.assertRaises(ValueError, writer.write_char, 0x11_0000)
1939-
self.assertRaises(ValueError, writer.write_char, 0xFFFF_FFFF)
1964+
self.assertRaises(ValueError, writer.write_char, INVALID_CHAR)
1965+
self.assertRaises(ValueError, writer.write_char, MAX_INVALID_CHAR)
19401966
self.assertEqual(writer.finish(),
19411967
"\0$\u20AC\U0010FFFF")
19421968

@@ -2106,15 +2132,37 @@ def test_ucs4(self):
21062132
writer.write_ucs4("pair\uD83D\uDC0D".encode(encoding, 'surrogatepass'))
21072133
writer.write_char(ord("-"))
21082134
writer.write_ucs4("null[\0]".encode(encoding), 7)
2109-
invalid = (b'\x00\x00\x11\x00' if sys.byteorder == 'little' else
2110-
b'\x00\x11\x00\x00')
2111-
# CRASHES writer.write_ucs4("invalid".encode(encoding) + invalid)
21122135
writer.write_ucs4(NULL, 0)
21132136
# CRASHES writer.write_ucs4(NULL, 1)
21142137
self.assertEqual(writer.finish(),
21152138
"lone\udc80-pair\ud83d\udc0d-null[\x00]")
21162139

2117-
# invalid size
2140+
# Invalid UCS-4 characters. Create an invalid string in release mode,
2141+
# or raise SystemError in debug mode.
2142+
writer = self.create_writer(0)
2143+
for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
2144+
with self.subTest(invalid_char=invalid_char):
2145+
# Test single character
2146+
ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
2147+
if support.Py_DEBUG:
2148+
self.assertRaises(SystemError, writer.write_ucs4, ucs4_char)
2149+
else:
2150+
writer.write_ucs4(ucs4_char)
2151+
2152+
# Test multiple characters
2153+
s = 'valid'.encode(encoding) + ucs4_char
2154+
if support.Py_DEBUG:
2155+
self.assertRaises(SystemError, writer.write_ucs4, s)
2156+
else:
2157+
writer.write_ucs4(s)
2158+
2159+
if support.Py_DEBUG:
2160+
self.assertEqual(writer.finish(), '')
2161+
else:
2162+
result = writer.finish()
2163+
assert_invalid_string(self, result)
2164+
2165+
# Invalid size
21182166
writer = self.create_writer(0)
21192167
with self.assertRaises(ValueError):
21202168
writer.write_ucs4("text".encode(encoding), -1)

‎Modules/_testinternalcapi.c‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -3544,6 +3544,10 @@ module_exec(PyObject *module)
35443544
}
35453545
PyModule_AddObject(module, "SelfInterruptingContextManager", (PyObject *)&SelfInterruptingContextManager_Type);
35463546

3547+
if (PyModule_AddIntMacro(module, _Py_MAX_UNICODE) < 0) {
3548+
return 1;
3549+
}
3550+
35473551
return 0;
35483552
}
35493553

‎Objects/unicodeobject.c‎

Lines changed: 59 additions & 14 deletions
Original file line numberDiff line numberDiff line change
@@ -1592,6 +1592,16 @@ PyUnicode_CopyCharacters(PyObject *to, Py_ssize_t to_start,
15921592
return how_many;
15931593
}
15941594

1595+
1596+
static void
1597+
unicode_invalid_character(PyObject *exc, Py_UCS4 ch)
1598+
{
1599+
PyErr_Format(exc,
1600+
"character U+%x is not in range [U+0000; U+%x]",
1601+
ch, MAX_UNICODE);
1602+
}
1603+
1604+
15951605
/* Find the maximum code point and count the number of surrogate pairs so a
15961606
correct string length can be computed before converting a string to UCS4.
15971607
This function counts single surrogates as a character and not as a pair.
@@ -1627,9 +1637,7 @@ find_maxchar_surrogates(const wchar_t *begin, const wchar_t *end,
16271637
if (ch > *maxchar) {
16281638
*maxchar = ch;
16291639
if (*maxchar > MAX_UNICODE) {
1630-
PyErr_Format(PyExc_ValueError,
1631-
"character U+%x is not in range [U+0000; U+%x]",
1632-
ch, MAX_UNICODE);
1640+
unicode_invalid_character(PyExc_ValueError, ch);
16331641
return -1;
16341642
}
16351643
}
@@ -1834,17 +1842,15 @@ get_latin1_char(Py_UCS1 ch)
18341842
static PyObject*
18351843
unicode_char(Py_UCS4 ch)
18361844
{
1837-
PyObject *unicode;
1838-
1839-
assert(ch <= MAX_UNICODE);
1840-
18411845
if (ch < 256) {
18421846
return get_latin1_char(ch);
18431847
}
18441848

1845-
unicode = PyUnicode_New(1, ch);
1846-
if (unicode == NULL)
1849+
// Raise SystemError if the character is not in range [U+0000; MAX_UNICODE]
1850+
PyObject *unicode = PyUnicode_New(1, ch);
1851+
if (unicode == NULL) {
18471852
return NULL;
1853+
}
18481854

18491855
assert(PyUnicode_KIND(unicode) != PyUnicode_1BYTE_KIND);
18501856
if (PyUnicode_KIND(unicode) == PyUnicode_2BYTE_KIND) {
@@ -2224,15 +2230,36 @@ static PyObject*
22242230
_PyUnicode_FromUCS4(const Py_UCS4 *u, Py_ssize_t size)
22252231
{
22262232
PyObject *res;
2227-
Py_UCS4 max_char;
22282233

22292234
if (size == 0)
22302235
_Py_RETURN_UNICODE_EMPTY();
22312236
assert(size > 0);
2232-
if (size == 1)
2237+
2238+
if (size == 1) {
2239+
// Raise SystemError if the character is invalid
22332240
return unicode_char(u[0]);
2241+
}
2242+
2243+
#ifdef Py_DEBUG
2244+
// Check for invalid characters in debug mode
2245+
Py_UCS4 max_char = 127;
2246+
for (Py_ssize_t i = 0; i < size; i++) {
2247+
Py_UCS4 ch = u[i];
2248+
if (ch > max_char) {
2249+
if (ch > MAX_UNICODE) {
2250+
unicode_invalid_character(PyExc_SystemError, ch);
2251+
return NULL;
2252+
}
2253+
max_char = ch;
2254+
}
2255+
}
2256+
#else
2257+
// gh-158445: Return MAX_UNICODE even if the string contains invalid
2258+
// characters. Checking for invalid characters would make the function
2259+
// slower whereas it's unlikely in practice.
2260+
Py_UCS4 max_char = ucs4lib_find_max_char(u, u + size);
2261+
#endif
22342262

2235-
max_char = ucs4lib_find_max_char(u, u + size);
22362263
res = PyUnicode_New(size, max_char);
22372264
if (!res)
22382265
return NULL;
@@ -2266,9 +2293,27 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer,
22662293
return 0;
22672294
}
22682295

2269-
Py_UCS4 max_char = ucs4lib_find_max_char(str, str + size);
2296+
#ifdef Py_DEBUG
2297+
// Check for invalid characters in debug mode
2298+
Py_UCS4 maxchar = 127;
2299+
for (Py_ssize_t i = 0; i < size; i++) {
2300+
Py_UCS4 ch = str[i];
2301+
if (ch > maxchar) {
2302+
if (ch > MAX_UNICODE) {
2303+
unicode_invalid_character(PyExc_SystemError, ch);
2304+
return -1;
2305+
}
2306+
maxchar = ch;
2307+
}
2308+
}
2309+
#else
2310+
// gh-158445: Return MAX_UNICODE even if the string contains invalid
2311+
// characters. Checking for invalid characters would make the function
2312+
// slower whereas it's unlikely in practice.
2313+
Py_UCS4 maxchar = ucs4lib_find_max_char(str, str + size);
2314+
#endif
22702315

2271-
if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) {
2316+
if (_PyUnicodeWriter_Prepare(writer, size, maxchar) < 0) {
22722317
return -1;
22732318
}
22742319
assert(_PyUnicodeWriter_CanWrite(writer));

0 commit comments

Comments
 (0)