Skip to content

Commit 13b1a4b

Browse files
[3.14] gh-158451: Fix python-gdb.py for invalid Unicode strings (GH-158475) (#158495)
gh-158451: Fix python-gdb.py for invalid Unicode strings (GH-158475) python-gdb.py is now able to format an invalid Unicode string: render invalid characters as "\Uhhhhhhhh" instead of raising an exception. The PyUnicodeWriter API fills a UCS-4 buffer with 0xff byte pattern to detect usage of uninitialized characters. It produces invalid characters '\Uffffffff'. PyBytesObjectPtr now iterates on bytes (list of integers), instead of creating a temporary Unicode string. Add more bytes tests in test_gdb pretty printer. Remove the unused 'ch2' variable. (cherry picked from commit 7eada7c) Co-authored-by: Victor Stinner <vstinner@python.org>
1 parent 4443885 commit 13b1a4b

3 files changed

Lines changed: 49 additions & 38 deletions

File tree

‎Lib/test/test_gdb/test_pretty_print.py‎

Lines changed: 4 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -96,13 +96,16 @@ def test_bytes(self):
9696
self.assertGdbRepr(b'And now for something hopefully the same')
9797
self.assertGdbRepr(b'string with embedded NUL here \0 and then some more text')
9898
self.assertGdbRepr(b'this is a tab:\t'
99+
b' this is a slash:\\'
99100
b' this is a slash-N:\n'
100101
b' this is a slash-R:\r'
101102
)
103+
# Test double quotes (")
104+
self.assertGdbRepr(b"it's a quote")
102105

103106
self.assertGdbRepr(b'this is byte 255:\xff and byte 128:\x80')
104107

105-
self.assertGdbRepr(bytes([b for b in range(255)]))
108+
self.assertGdbRepr(bytes([b for b in range(256)]))
106109

107110
@support.requires_resource('cpu')
108111
def test_strings(self):
Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,3 @@
1+
``python-gdb.py`` is now able to format an invalid Unicode string: render
2+
invalid characters as ``\Uhhhhhhhh`` instead of raising an exception. Patch by
3+
Victor Stinner.

‎Tools/gdb/libpython.py‎

Lines changed: 42 additions & 37 deletions
Original file line numberDiff line numberDiff line change
@@ -46,6 +46,9 @@
4646
import sys
4747

4848

49+
MAX_UNICODE = 0x10_ffff
50+
51+
4952
# Look up the gdb.Type for some standard types:
5053
# Those need to be refreshed as types (pointer sizes) may change when
5154
# gdb loads different executables
@@ -1367,29 +1370,33 @@ def write_repr(self, out, visited):
13671370
class PyBytesObjectPtr(PyObjectPtr):
13681371
_typename = 'PyBytesObject'
13691372

1370-
def __str__(self):
1373+
def get_bytes(self):
13711374
field_ob_size = self.field('ob_size')
13721375
field_ob_sval = self.field('ob_sval')
13731376
char_ptr = field_ob_sval.address.cast(_type_unsigned_char_ptr())
1374-
return ''.join([chr(char_ptr[i]) for i in safe_range(field_ob_size)])
1377+
return [char_ptr[i] for i in safe_range(field_ob_size)]
1378+
1379+
def __str__(self):
1380+
as_bytes = self.get_bytes()
1381+
return ''.join([chr(byte) for byte in as_bytes])
13751382

13761383
def proxyval(self, visited):
13771384
return str(self)
13781385

13791386
def write_repr(self, out, visited):
13801387
# Write this out as a Python bytes literal, i.e. with a "b" prefix
13811388

1382-
# Get a PyStringObject* within the Python gdb process:
1383-
proxy = self.proxyval(visited)
1389+
as_bytes = self.get_bytes()
13841390

13851391
# Transliteration of Python's Objects/bytesobject.c:PyBytes_Repr
13861392
# to Python code:
13871393
quote = "'"
1388-
if "'" in proxy and not '"' in proxy:
1394+
if ord("'") in as_bytes and ord('"') not in as_bytes:
13891395
quote = '"'
13901396
out.write('b')
13911397
out.write(quote)
1392-
for byte in proxy:
1398+
for value in as_bytes:
1399+
byte = chr(value)
13931400
if byte == quote or byte == '\\':
13941401
out.write('\\')
13951402
out.write(byte)
@@ -1399,10 +1406,10 @@ def write_repr(self, out, visited):
13991406
out.write('\\n')
14001407
elif byte == '\r':
14011408
out.write('\\r')
1402-
elif byte < ' ' or ord(byte) >= 0x7f:
1409+
elif value < ord(' ') or value >= 0x7f:
14031410
out.write('\\x')
1404-
out.write(hexdigits[(ord(byte) & 0xf0) >> 4])
1405-
out.write(hexdigits[ord(byte) & 0xf])
1411+
out.write(hexdigits[(value & 0xf0) >> 4])
1412+
out.write(hexdigits[value & 0xf])
14061413
else:
14071414
out.write(byte)
14081415
out.write(quote)
@@ -1455,10 +1462,17 @@ def _unichr_is_printable(char):
14551462
return unicodedata.category(char) not in ("C", "Z")
14561463

14571464

1465+
def safe_chr(i):
1466+
if i <= MAX_UNICODE:
1467+
return chr(i)
1468+
else:
1469+
return f'\\U{i:08x}'
1470+
1471+
14581472
class PyUnicodeObjectPtr(PyObjectPtr):
14591473
_typename = 'PyUnicodeObject'
14601474

1461-
def proxyval(self, visited):
1475+
def get_code_points(self):
14621476
compact = self.field('_base')
14631477
ascii = compact['_base']
14641478
state = ascii['state']
@@ -1480,11 +1494,13 @@ def proxyval(self, visited):
14801494

14811495
# Gather a list of ints from the code point array; these are either
14821496
# UCS-1, UCS-2 or UCS-4 code points:
1483-
code_points = [int(field_str[i]) for i in safe_range(field_length)]
1497+
return [int(field_str[i]) for i in safe_range(field_length)]
14841498

1499+
def proxyval(self, visited):
1500+
code_points = self.get_code_points()
14851501
# Convert the int code points to unicode characters, and generate a
14861502
# local unicode instance.
1487-
result = ''.join(map(chr, code_points))
1503+
result = ''.join(map(safe_chr, code_points))
14881504
return result
14891505

14901506
def write_repr(self, out, visited):
@@ -1495,20 +1511,18 @@ def write_repr(self, out, visited):
14951511
encoding = gdb.host_charset()
14961512

14971513
# Get a PyUnicodeObject* within the Python gdb process:
1498-
proxy = self.proxyval(visited)
1514+
code_points = self.get_code_points()
14991515

15001516
# Transliteration of Python's Object/unicodeobject.c:unicode_repr
15011517
# to Python:
1502-
if "'" in proxy and '"' not in proxy:
1518+
if ord("'") in code_points and ord('"') not in code_points:
15031519
quote = '"'
15041520
else:
15051521
quote = "'"
15061522
out.write(quote)
15071523

1508-
i = 0
1509-
while i < len(proxy):
1510-
ch = proxy[i]
1511-
i += 1
1524+
for code_point in code_points:
1525+
ch = safe_chr(code_point)
15121526

15131527
# Escape quotes and backslashes
15141528
if ch == quote or ch == '\\':
@@ -1524,24 +1538,24 @@ def write_repr(self, out, visited):
15241538
out.write('\\r')
15251539

15261540
# Map non-printable US ASCII to '\xhh' */
1527-
elif ch < ' ' or ord(ch) == 0x7F:
1541+
elif ch < ' ' or code_point == 0x7F:
15281542
out.write('\\x')
1529-
out.write(hexdigits[(ord(ch) >> 4) & 0x000F])
1530-
out.write(hexdigits[ord(ch) & 0x000F])
1543+
out.write(hexdigits[(code_point >> 4) & 0x000F])
1544+
out.write(hexdigits[code_point & 0x000F])
15311545

15321546
# Copy ASCII characters as-is
1533-
elif ord(ch) < 0x7F:
1547+
elif code_point < 0x7F:
15341548
out.write(ch)
15351549

15361550
# Non-ASCII characters
15371551
else:
1538-
ucs = ch
1539-
ch2 = None
1540-
1541-
printable = ucs.isprintable()
1552+
if code_point <= MAX_UNICODE:
1553+
printable = ch.isprintable()
1554+
else:
1555+
printable = False
15421556
if printable:
15431557
try:
1544-
ucs.encode(encoding)
1558+
ch.encode(encoding)
15451559
# LookupError or ValueError if the host charset is unknown
15461560
# or invalid.
15471561
except (UnicodeEncodeError, LookupError, ValueError):
@@ -1550,14 +1564,7 @@ def write_repr(self, out, visited):
15501564
# Map Unicode whitespace and control characters
15511565
# (categories Z* and C* except ASCII space)
15521566
if not printable:
1553-
if ch2 is not None:
1554-
# Match Python's representation of non-printable
1555-
# wide characters.
1556-
code = (ord(ch) & 0x03FF) << 10
1557-
code |= ord(ch2) & 0x03FF
1558-
code += 0x00010000
1559-
else:
1560-
code = ord(ucs)
1567+
code = code_point
15611568

15621569
# Map 8-bit characters to '\\xhh'
15631570
if code <= 0xff:
@@ -1585,8 +1592,6 @@ def write_repr(self, out, visited):
15851592
else:
15861593
# Copy characters as-is
15871594
out.write(ch)
1588-
if ch2 is not None:
1589-
out.write(ch2)
15901595

15911596
out.write(quote)
15921597

0 commit comments

Comments
 (0)