Skip to content

Commit 7eada7c

Browse files
authored
gh-158451: Fix python-gdb.py for invalid Unicode strings (#158475)
python-gdb.py is now able to format an invalid Unicode string: render invalid characters as "\Uhhhhhhhh" instead of raising an exception. The PyUnicodeWriter API fills a UCS-4 buffer with 0xff byte pattern to detect usage of uninitialized characters. It produces invalid characters '\Uffffffff'. PyBytesObjectPtr now iterates on bytes (list of integers), instead of creating a temporary Unicode string. Add more bytes tests in test_gdb pretty printer. Remove the unused 'ch2' variable.
1 parent 77c0675 commit 7eada7c

3 files changed

Lines changed: 49 additions & 38 deletions

File tree

‎Lib/test/test_gdb/test_pretty_print.py‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -103,13 +103,16 @@ def test_bytes(self):
103103
self.assertGdbRepr(b'And now for something hopefully the same')
104104
self.assertGdbRepr(b'string with embedded NUL here \0 and then some more text')
105105
self.assertGdbRepr(b'this is a tab:\t'
106+
b' this is a slash:\\'
106107
b' this is a slash-N:\n'
107108
b' this is a slash-R:\r'
108109
)
110+
# Test double quotes (")
111+
self.assertGdbRepr(b"it's a quote")
109112

110113
self.assertGdbRepr(b'this is byte 255:\xff and byte 128:\x80')
111114

112-
self.assertGdbRepr(bytes([b for b in range(255)]))
115+
self.assertGdbRepr(bytes([b for b in range(256)]))
113116

114117
@support.requires_resource('cpu')
115118
def test_strings(self):
Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,3 @@
1+
``python-gdb.py`` is now able to format an invalid Unicode string: render
2+
invalid characters as ``\Uhhhhhhhh`` instead of raising an exception. Patch by
3+
Victor Stinner.

‎Tools/gdb/libpython.py‎

Lines changed: 42 additions & 37 deletions
Original file line numberDiff line numberDiff line change
@@ -46,6 +46,9 @@
4646
import sys
4747

4848

49+
MAX_UNICODE = 0x10_ffff
50+
51+
4952
# Look up the gdb.Type for some standard types:
5053
# Those need to be refreshed as types (pointer sizes) may change when
5154
# gdb loads different executables
@@ -1378,29 +1381,33 @@ def write_repr(self, out, visited):
13781381
class PyBytesObjectPtr(PyObjectPtr):
13791382
_typename = 'PyBytesObject'
13801383

1381-
def __str__(self):
1384+
def get_bytes(self):
13821385
field_ob_size = self.field('ob_size')
13831386
field_ob_sval = self.field('ob_sval')
13841387
char_ptr = field_ob_sval.address.cast(_type_unsigned_char_ptr())
1385-
return ''.join([chr(char_ptr[i]) for i in safe_range(field_ob_size)])
1388+
return [char_ptr[i] for i in safe_range(field_ob_size)]
1389+
1390+
def __str__(self):
1391+
as_bytes = self.get_bytes()
1392+
return ''.join([chr(byte) for byte in as_bytes])
13861393

13871394
def proxyval(self, visited):
13881395
return str(self)
13891396

13901397
def write_repr(self, out, visited):
13911398
# Write this out as a Python bytes literal, i.e. with a "b" prefix
13921399

1393-
# Get a PyStringObject* within the Python gdb process:
1394-
proxy = self.proxyval(visited)
1400+
as_bytes = self.get_bytes()
13951401

13961402
# Transliteration of Python's Objects/bytesobject.c:PyBytes_Repr
13971403
# to Python code:
13981404
quote = "'"
1399-
if "'" in proxy and not '"' in proxy:
1405+
if ord("'") in as_bytes and ord('"') not in as_bytes:
14001406
quote = '"'
14011407
out.write('b')
14021408
out.write(quote)
1403-
for byte in proxy:
1409+
for value in as_bytes:
1410+
byte = chr(value)
14041411
if byte == quote or byte == '\\':
14051412
out.write('\\')
14061413
out.write(byte)
@@ -1410,10 +1417,10 @@ def write_repr(self, out, visited):
14101417
out.write('\\n')
14111418
elif byte == '\r':
14121419
out.write('\\r')
1413-
elif byte < ' ' or ord(byte) >= 0x7f:
1420+
elif value < ord(' ') or value >= 0x7f:
14141421
out.write('\\x')
1415-
out.write(hexdigits[(ord(byte) & 0xf0) >> 4])
1416-
out.write(hexdigits[ord(byte) & 0xf])
1422+
out.write(hexdigits[(value & 0xf0) >> 4])
1423+
out.write(hexdigits[value & 0xf])
14171424
else:
14181425
out.write(byte)
14191426
out.write(quote)
@@ -1466,10 +1473,17 @@ def _unichr_is_printable(char):
14661473
return unicodedata.category(char) not in ("C", "Z")
14671474

14681475

1476+
def safe_chr(i):
1477+
if i <= MAX_UNICODE:
1478+
return chr(i)
1479+
else:
1480+
return f'\\U{i:08x}'
1481+
1482+
14691483
class PyUnicodeObjectPtr(PyObjectPtr):
14701484
_typename = 'PyUnicodeObject'
14711485

1472-
def proxyval(self, visited):
1486+
def get_code_points(self):
14731487
compact = self.field('_base')
14741488
ascii = compact['_base']
14751489
state = ascii['state']
@@ -1491,11 +1505,13 @@ def proxyval(self, visited):
14911505

14921506
# Gather a list of ints from the code point array; these are either
14931507
# UCS-1, UCS-2 or UCS-4 code points:
1494-
code_points = [int(field_str[i]) for i in safe_range(field_length)]
1508+
return [int(field_str[i]) for i in safe_range(field_length)]
14951509

1510+
def proxyval(self, visited):
1511+
code_points = self.get_code_points()
14961512
# Convert the int code points to unicode characters, and generate a
14971513
# local unicode instance.
1498-
result = ''.join(map(chr, code_points))
1514+
result = ''.join(map(safe_chr, code_points))
14991515
return result
15001516

15011517
def write_repr(self, out, visited):
@@ -1506,20 +1522,18 @@ def write_repr(self, out, visited):
15061522
encoding = gdb.host_charset()
15071523

15081524
# Get a PyUnicodeObject* within the Python gdb process:
1509-
proxy = self.proxyval(visited)
1525+
code_points = self.get_code_points()
15101526

15111527
# Transliteration of Python's Object/unicodeobject.c:unicode_repr
15121528
# to Python:
1513-
if "'" in proxy and '"' not in proxy:
1529+
if ord("'") in code_points and ord('"') not in code_points:
15141530
quote = '"'
15151531
else:
15161532
quote = "'"
15171533
out.write(quote)
15181534

1519-
i = 0
1520-
while i < len(proxy):
1521-
ch = proxy[i]
1522-
i += 1
1535+
for code_point in code_points:
1536+
ch = safe_chr(code_point)
15231537

15241538
# Escape quotes and backslashes
15251539
if ch == quote or ch == '\\':
@@ -1535,24 +1549,24 @@ def write_repr(self, out, visited):
15351549
out.write('\\r')
15361550

15371551
# Map non-printable US ASCII to '\xhh' */
1538-
elif ch < ' ' or ord(ch) == 0x7F:
1552+
elif ch < ' ' or code_point == 0x7F:
15391553
out.write('\\x')
1540-
out.write(hexdigits[(ord(ch) >> 4) & 0x000F])
1541-
out.write(hexdigits[ord(ch) & 0x000F])
1554+
out.write(hexdigits[(code_point >> 4) & 0x000F])
1555+
out.write(hexdigits[code_point & 0x000F])
15421556

15431557
# Copy ASCII characters as-is
1544-
elif ord(ch) < 0x7F:
1558+
elif code_point < 0x7F:
15451559
out.write(ch)
15461560

15471561
# Non-ASCII characters
15481562
else:
1549-
ucs = ch
1550-
ch2 = None
1551-
1552-
printable = ucs.isprintable()
1563+
if code_point <= MAX_UNICODE:
1564+
printable = ch.isprintable()
1565+
else:
1566+
printable = False
15531567
if printable:
15541568
try:
1555-
ucs.encode(encoding)
1569+
ch.encode(encoding)
15561570
# LookupError or ValueError if the host charset is unknown
15571571
# or invalid.
15581572
except (UnicodeEncodeError, LookupError, ValueError):
@@ -1561,14 +1575,7 @@ def write_repr(self, out, visited):
15611575
# Map Unicode whitespace and control characters
15621576
# (categories Z* and C* except ASCII space)
15631577
if not printable:
1564-
if ch2 is not None:
1565-
# Match Python's representation of non-printable
1566-
# wide characters.
1567-
code = (ord(ch) & 0x03FF) << 10
1568-
code |= ord(ch2) & 0x03FF
1569-
code += 0x00010000
1570-
else:
1571-
code = ord(ucs)
1578+
code = code_point
15721579

15731580
# Map 8-bit characters to '\\xhh'
15741581
if code <= 0xff:
@@ -1596,8 +1603,6 @@ def write_repr(self, out, visited):
15961603
else:
15971604
# Copy characters as-is
15981605
out.write(ch)
1599-
if ch2 is not None:
1600-
out.write(ch2)
16011606

16021607
out.write(quote)
16031608

0 commit comments

Comments
 (0)