https://github.com/python/cpython/commit/13b1a4bed3a4e9641a810ddc553d6628da7f4d7b
commit: 13b1a4bed3a4e9641a810ddc553d6628da7f4d7b
branch: 3.14
author: Miss Islington (bot) <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T12:41:03Z
summary:

[3.14] gh-158451: Fix python-gdb.py for invalid Unicode strings (GH-158475) 
(#158495)

gh-158451: Fix python-gdb.py for invalid Unicode strings (GH-158475)

python-gdb.py is now able to format an invalid Unicode string:
render invalid characters as "\Uhhhhhhhh" instead of raising an
exception.

The PyUnicodeWriter API fills a UCS-4 buffer with 0xff byte pattern
to detect usage of uninitialized characters. It produces invalid
characters '\Uffffffff'.

PyBytesObjectPtr now iterates on bytes (list of integers), instead of
creating a temporary Unicode string. Add more bytes tests in test_gdb
pretty printer.

Remove the unused 'ch2' variable.
(cherry picked from commit 7eada7c2c63189fbe753cf5e84b77baf4099a4a1)

Co-authored-by: Victor Stinner <[email protected]>

files:
A Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
M Lib/test/test_gdb/test_pretty_print.py
M Tools/gdb/libpython.py

diff --git a/Lib/test/test_gdb/test_pretty_print.py 
b/Lib/test/test_gdb/test_pretty_print.py
index a258d7c0e60c7e1..563379c63b68902 100644
--- a/Lib/test/test_gdb/test_pretty_print.py
+++ b/Lib/test/test_gdb/test_pretty_print.py
@@ -96,13 +96,16 @@ def test_bytes(self):
         self.assertGdbRepr(b'And now for something hopefully the same')
         self.assertGdbRepr(b'string with embedded NUL here \0 and then some 
more text')
         self.assertGdbRepr(b'this is a tab:\t'
+                           b' this is a slash:\\'
                            b' this is a slash-N:\n'
                            b' this is a slash-R:\r'
                            )
+        # Test double quotes (")
+        self.assertGdbRepr(b"it's a quote")
 
         self.assertGdbRepr(b'this is byte 255:\xff and byte 128:\x80')
 
-        self.assertGdbRepr(bytes([b for b in range(255)]))
+        self.assertGdbRepr(bytes([b for b in range(256)]))
 
     @support.requires_resource('cpu')
     def test_strings(self):
diff --git 
a/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst 
b/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
new file mode 100644
index 000000000000000..2d84d6f0a1393f2
--- /dev/null
+++ 
b/Misc/NEWS.d/next/Tools-Demos/2026-09-30-06-55-37.gh-issue-158451.pZE56z.rst
@@ -0,0 +1,3 @@
+``python-gdb.py`` is now able to format an invalid Unicode string: render
+invalid characters as ``\Uhhhhhhhh`` instead of raising an exception. Patch by
+Victor Stinner.
diff --git a/Tools/gdb/libpython.py b/Tools/gdb/libpython.py
index f765d96d9cb7a97..78f15d0c8f93045 100755
--- a/Tools/gdb/libpython.py
+++ b/Tools/gdb/libpython.py
@@ -46,6 +46,9 @@
 import sys
 
 
+MAX_UNICODE = 0x10_ffff
+
+
 # Look up the gdb.Type for some standard types:
 # Those need to be refreshed as types (pointer sizes) may change when
 # gdb loads different executables
@@ -1367,11 +1370,15 @@ def write_repr(self, out, visited):
 class PyBytesObjectPtr(PyObjectPtr):
     _typename = 'PyBytesObject'
 
-    def __str__(self):
+    def get_bytes(self):
         field_ob_size = self.field('ob_size')
         field_ob_sval = self.field('ob_sval')
         char_ptr = field_ob_sval.address.cast(_type_unsigned_char_ptr())
-        return ''.join([chr(char_ptr[i]) for i in safe_range(field_ob_size)])
+        return [char_ptr[i] for i in safe_range(field_ob_size)]
+
+    def __str__(self):
+        as_bytes = self.get_bytes()
+        return ''.join([chr(byte) for byte in as_bytes])
 
     def proxyval(self, visited):
         return str(self)
@@ -1379,17 +1386,17 @@ def proxyval(self, visited):
     def write_repr(self, out, visited):
         # Write this out as a Python bytes literal, i.e. with a "b" prefix
 
-        # Get a PyStringObject* within the Python gdb process:
-        proxy = self.proxyval(visited)
+        as_bytes = self.get_bytes()
 
         # Transliteration of Python's Objects/bytesobject.c:PyBytes_Repr
         # to Python code:
         quote = "'"
-        if "'" in proxy and not '"' in proxy:
+        if ord("'") in as_bytes and ord('"') not in as_bytes:
             quote = '"'
         out.write('b')
         out.write(quote)
-        for byte in proxy:
+        for value in as_bytes:
+            byte = chr(value)
             if byte == quote or byte == '\\':
                 out.write('\\')
                 out.write(byte)
@@ -1399,10 +1406,10 @@ def write_repr(self, out, visited):
                 out.write('\\n')
             elif byte == '\r':
                 out.write('\\r')
-            elif byte < ' ' or ord(byte) >= 0x7f:
+            elif value < ord(' ') or value >= 0x7f:
                 out.write('\\x')
-                out.write(hexdigits[(ord(byte) & 0xf0) >> 4])
-                out.write(hexdigits[ord(byte) & 0xf])
+                out.write(hexdigits[(value & 0xf0) >> 4])
+                out.write(hexdigits[value & 0xf])
             else:
                 out.write(byte)
         out.write(quote)
@@ -1455,10 +1462,17 @@ def _unichr_is_printable(char):
     return unicodedata.category(char) not in ("C", "Z")
 
 
+def safe_chr(i):
+    if i <= MAX_UNICODE:
+        return chr(i)
+    else:
+        return f'\\U{i:08x}'
+
+
 class PyUnicodeObjectPtr(PyObjectPtr):
     _typename = 'PyUnicodeObject'
 
-    def proxyval(self, visited):
+    def get_code_points(self):
         compact = self.field('_base')
         ascii = compact['_base']
         state = ascii['state']
@@ -1480,11 +1494,13 @@ def proxyval(self, visited):
 
         # Gather a list of ints from the code point array; these are either
         # UCS-1, UCS-2 or UCS-4 code points:
-        code_points = [int(field_str[i]) for i in safe_range(field_length)]
+        return [int(field_str[i]) for i in safe_range(field_length)]
 
+    def proxyval(self, visited):
+        code_points = self.get_code_points()
         # Convert the int code points to unicode characters, and generate a
         # local unicode instance.
-        result = ''.join(map(chr, code_points))
+        result = ''.join(map(safe_chr, code_points))
         return result
 
     def write_repr(self, out, visited):
@@ -1495,20 +1511,18 @@ def write_repr(self, out, visited):
         encoding = gdb.host_charset()
 
         # Get a PyUnicodeObject* within the Python gdb process:
-        proxy = self.proxyval(visited)
+        code_points = self.get_code_points()
 
         # Transliteration of Python's Object/unicodeobject.c:unicode_repr
         # to Python:
-        if "'" in proxy and '"' not in proxy:
+        if ord("'") in code_points and ord('"') not in code_points:
             quote = '"'
         else:
             quote = "'"
         out.write(quote)
 
-        i = 0
-        while i < len(proxy):
-            ch = proxy[i]
-            i += 1
+        for code_point in code_points:
+            ch = safe_chr(code_point)
 
             # Escape quotes and backslashes
             if ch == quote or ch == '\\':
@@ -1524,24 +1538,24 @@ def write_repr(self, out, visited):
                 out.write('\\r')
 
             # Map non-printable US ASCII to '\xhh' */
-            elif ch < ' ' or ord(ch) == 0x7F:
+            elif ch < ' ' or code_point == 0x7F:
                 out.write('\\x')
-                out.write(hexdigits[(ord(ch) >> 4) & 0x000F])
-                out.write(hexdigits[ord(ch) & 0x000F])
+                out.write(hexdigits[(code_point >> 4) & 0x000F])
+                out.write(hexdigits[code_point & 0x000F])
 
             # Copy ASCII characters as-is
-            elif ord(ch) < 0x7F:
+            elif code_point < 0x7F:
                 out.write(ch)
 
             # Non-ASCII characters
             else:
-                ucs = ch
-                ch2 = None
-
-                printable = ucs.isprintable()
+                if code_point <= MAX_UNICODE:
+                    printable = ch.isprintable()
+                else:
+                    printable = False
                 if printable:
                     try:
-                        ucs.encode(encoding)
+                        ch.encode(encoding)
                     # LookupError or ValueError if the host charset is unknown
                     # or invalid.
                     except (UnicodeEncodeError, LookupError, ValueError):
@@ -1550,14 +1564,7 @@ def write_repr(self, out, visited):
                 # Map Unicode whitespace and control characters
                 # (categories Z* and C* except ASCII space)
                 if not printable:
-                    if ch2 is not None:
-                        # Match Python's representation of non-printable
-                        # wide characters.
-                        code = (ord(ch) & 0x03FF) << 10
-                        code |= ord(ch2) & 0x03FF
-                        code += 0x00010000
-                    else:
-                        code = ord(ucs)
+                    code = code_point
 
                     # Map 8-bit characters to '\\xhh'
                     if code <= 0xff:
@@ -1585,8 +1592,6 @@ def write_repr(self, out, visited):
                 else:
                     # Copy characters as-is
                     out.write(ch)
-                    if ch2 is not None:
-                        out.write(ch2)
 
         out.write(quote)
 

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to