https://github.com/python/cpython/commit/7eada7c2c63189fbe753cf5e84b77baf4099a4a1
commit: 7eada7c2c63189fbe753cf5e84b77baf4099a4a1
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T14:12:59+02:00
summary:
gh-158451: Fix python-gdb.py for invalid Unicode strings (#158475)
python-gdb.py is now able to format an invalid Unicode string:
render invalid characters as "\Uhhhhhhhh" instead of raising an
exception.
The PyUnicodeWriter API fills a UCS-4 buffer with 0xff byte pattern
to detect usage of uninitialized characters. It produces invalid
characters '\Uffffffff'.
PyBytesObjectPtr now iterates on bytes (list of integers), instead of
creating a temporary Unicode string. Add more bytes tests in test_gdb
pretty printer.
Remove the unused 'ch2' variable.
files:
A Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
M Lib/test/test_gdb/test_pretty_print.py
M Tools/gdb/libpython.py
diff --git a/Lib/test/test_gdb/test_pretty_print.py
b/Lib/test/test_gdb/test_pretty_print.py
index edb79ef86d31db5..ea047a12b802e88 100644
--- a/Lib/test/test_gdb/test_pretty_print.py
+++ b/Lib/test/test_gdb/test_pretty_print.py
@@ -103,13 +103,16 @@ def test_bytes(self):
self.assertGdbRepr(b'And now for something hopefully the same')
self.assertGdbRepr(b'string with embedded NUL here \0 and then some
more text')
self.assertGdbRepr(b'this is a tab:\t'
+ b' this is a slash:\\'
b' this is a slash-N:\n'
b' this is a slash-R:\r'
)
+ # Test double quotes (")
+ self.assertGdbRepr(b"it's a quote")
self.assertGdbRepr(b'this is byte 255:\xff and byte 128:\x80')
- self.assertGdbRepr(bytes([b for b in range(255)]))
+ self.assertGdbRepr(bytes([b for b in range(256)]))
@support.requires_resource('cpu')
def test_strings(self):
diff --git
a/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
b/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
new file mode 100644
index 000000000000000..2d84d6f0a1393f2
--- /dev/null
+++
b/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
@@ -0,0 +1,3 @@
+``python-gdb.py`` is now able to format an invalid Unicode string: render
+invalid characters as ``\Uhhhhhhhh`` instead of raising an exception. Patch by
+Victor Stinner.
diff --git a/Tools/gdb/libpython.py b/Tools/gdb/libpython.py
index 422e4f605920a0c..a7f89442dd21dba 100755
--- a/Tools/gdb/libpython.py
+++ b/Tools/gdb/libpython.py
@@ -46,6 +46,9 @@
import sys
+MAX_UNICODE = 0x10_ffff
+
+
# Look up the gdb.Type for some standard types:
# Those need to be refreshed as types (pointer sizes) may change when
# gdb loads different executables
@@ -1378,11 +1381,15 @@ def write_repr(self, out, visited):
class PyBytesObjectPtr(PyObjectPtr):
_typename = 'PyBytesObject'
- def __str__(self):
+ def get_bytes(self):
field_ob_size = self.field('ob_size')
field_ob_sval = self.field('ob_sval')
char_ptr = field_ob_sval.address.cast(_type_unsigned_char_ptr())
- return ''.join([chr(char_ptr[i]) for i in safe_range(field_ob_size)])
+ return [char_ptr[i] for i in safe_range(field_ob_size)]
+
+ def __str__(self):
+ as_bytes = self.get_bytes()
+ return ''.join([chr(byte) for byte in as_bytes])
def proxyval(self, visited):
return str(self)
@@ -1390,17 +1397,17 @@ def proxyval(self, visited):
def write_repr(self, out, visited):
# Write this out as a Python bytes literal, i.e. with a "b" prefix
- # Get a PyStringObject* within the Python gdb process:
- proxy = self.proxyval(visited)
+ as_bytes = self.get_bytes()
# Transliteration of Python's Objects/bytesobject.c:PyBytes_Repr
# to Python code:
quote = "'"
- if "'" in proxy and not '"' in proxy:
+ if ord("'") in as_bytes and ord('"') not in as_bytes:
quote = '"'
out.write('b')
out.write(quote)
- for byte in proxy:
+ for value in as_bytes:
+ byte = chr(value)
if byte == quote or byte == '\\':
out.write('\\')
out.write(byte)
@@ -1410,10 +1417,10 @@ def write_repr(self, out, visited):
out.write('\\n')
elif byte == '\r':
out.write('\\r')
- elif byte < ' ' or ord(byte) >= 0x7f:
+ elif value < ord(' ') or value >= 0x7f:
out.write('\\x')
- out.write(hexdigits[(ord(byte) & 0xf0) >> 4])
- out.write(hexdigits[ord(byte) & 0xf])
+ out.write(hexdigits[(value & 0xf0) >> 4])
+ out.write(hexdigits[value & 0xf])
else:
out.write(byte)
out.write(quote)
@@ -1466,10 +1473,17 @@ def _unichr_is_printable(char):
return unicodedata.category(char) not in ("C", "Z")
+def safe_chr(i):
+ if i <= MAX_UNICODE:
+ return chr(i)
+ else:
+ return f'\\U{i:08x}'
+
+
class PyUnicodeObjectPtr(PyObjectPtr):
_typename = 'PyUnicodeObject'
- def proxyval(self, visited):
+ def get_code_points(self):
compact = self.field('_base')
ascii = compact['_base']
state = ascii['state']
@@ -1491,11 +1505,13 @@ def proxyval(self, visited):
# Gather a list of ints from the code point array; these are either
# UCS-1, UCS-2 or UCS-4 code points:
- code_points = [int(field_str[i]) for i in safe_range(field_length)]
+ return [int(field_str[i]) for i in safe_range(field_length)]
+ def proxyval(self, visited):
+ code_points = self.get_code_points()
# Convert the int code points to unicode characters, and generate a
# local unicode instance.
- result = ''.join(map(chr, code_points))
+ result = ''.join(map(safe_chr, code_points))
return result
def write_repr(self, out, visited):
@@ -1506,20 +1522,18 @@ def write_repr(self, out, visited):
encoding = gdb.host_charset()
# Get a PyUnicodeObject* within the Python gdb process:
- proxy = self.proxyval(visited)
+ code_points = self.get_code_points()
# Transliteration of Python's Object/unicodeobject.c:unicode_repr
# to Python:
- if "'" in proxy and '"' not in proxy:
+ if ord("'") in code_points and ord('"') not in code_points:
quote = '"'
else:
quote = "'"
out.write(quote)
- i = 0
- while i < len(proxy):
- ch = proxy[i]
- i += 1
+ for code_point in code_points:
+ ch = safe_chr(code_point)
# Escape quotes and backslashes
if ch == quote or ch == '\\':
@@ -1535,24 +1549,24 @@ def write_repr(self, out, visited):
out.write('\\r')
# Map non-printable US ASCII to '\xhh' */
- elif ch < ' ' or ord(ch) == 0x7F:
+ elif ch < ' ' or code_point == 0x7F:
out.write('\\x')
- out.write(hexdigits[(ord(ch) >> 4) & 0x000F])
- out.write(hexdigits[ord(ch) & 0x000F])
+ out.write(hexdigits[(code_point >> 4) & 0x000F])
+ out.write(hexdigits[code_point & 0x000F])
# Copy ASCII characters as-is
- elif ord(ch) < 0x7F:
+ elif code_point < 0x7F:
out.write(ch)
# Non-ASCII characters
else:
- ucs = ch
- ch2 = None
-
- printable = ucs.isprintable()
+ if code_point <= MAX_UNICODE:
+ printable = ch.isprintable()
+ else:
+ printable = False
if printable:
try:
- ucs.encode(encoding)
+ ch.encode(encoding)
# LookupError or ValueError if the host charset is unknown
# or invalid.
except (UnicodeEncodeError, LookupError, ValueError):
@@ -1561,14 +1575,7 @@ def write_repr(self, out, visited):
# Map Unicode whitespace and control characters
# (categories Z* and C* except ASCII space)
if not printable:
- if ch2 is not None:
- # Match Python's representation of non-printable
- # wide characters.
- code = (ord(ch) & 0x03FF) << 10
- code |= ord(ch2) & 0x03FF
- code += 0x00010000
- else:
- code = ord(ucs)
+ code = code_point
# Map 8-bit characters to '\\xhh'
if code <= 0xff:
@@ -1596,8 +1603,6 @@ def write_repr(self, out, visited):
else:
# Copy characters as-is
out.write(ch)
- if ch2 is not None:
- out.write(ch2)
out.write(quote)
_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]