https://github.com/python/cpython/commit/a14a4a6bc9a6e1b930b1a4ba3678cb9ab551c006
commit: a14a4a6bc9a6e1b930b1a4ba3678cb9ab551c006
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T17:19:51+02:00
summary:

gh-158445: Detect invalid UCS4 characters in debug mode (#158502)

In debug mode, PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND) and
PyUnicodeWriter_WriteUCS4() now raise a SystemError if a character is
not in range [U+0000; U+10ffff], instead of creating an invalid str
object.

In release mode, PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND)
raises a SystemError if the size is 1.

* In debug mode, unicode_char() no longer fails with an assertion
  error if the character is invalid. Instead, raise SystemError.
* Add tests on invalid UCS-4 characters.
* Add _testinternalcapi._Py_MAX_UNICODE.
* Add unicode_invalid_character() helper function.

files:
M Doc/c-api/unicode.rst
M Lib/test/test_capi/test_unicode.py
M Modules/_testinternalcapi.c
M Objects/unicodeobject.c

diff --git a/Doc/c-api/unicode.rst b/Doc/c-api/unicode.rst
index 5f918b447950d9..fcdcd26ff70cb0 100644
--- a/Doc/c-api/unicode.rst
+++ b/Doc/c-api/unicode.rst
@@ -425,6 +425,10 @@ APIs:
    the UCS1 range, it will be transformed into UCS1
    (:c:macro:`PyUnicode_1BYTE_KIND`).
 
+   All characters must be in range [U+0000; U+10ffff]. If *kind* is
+   :c:macro:`PyUnicode_4BYTE_KIND` and the string contains invalid characters,
+   the behavior is undefined.
+
    .. versionadded:: 3.3
 
 
@@ -1896,10 +1900,13 @@ object.
 
 .. c:function:: int PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *writer, const 
Py_UCS4 *str, Py_ssize_t size)
 
-   Writer the UCS4 string *str* into *writer*.
+   Write the UCS4 string *str* into *writer*.
 
    *size* is a number of UCS4 characters.
 
+   All characters must be in range [U+0000; U+10ffff]. If the string contains
+   invalid characters, the behavior is undefined.
+
    On success, return ``0``.
    On error, set an exception, leave the writer unchanged, and return ``-1``.
 
diff --git a/Lib/test/test_capi/test_unicode.py 
b/Lib/test/test_capi/test_unicode.py
index 9c91d6eb512d2a..c5926dce397409 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -19,6 +19,11 @@
 from _testcapi import PY_SSIZE_T_MIN, PY_SSIZE_T_MAX, SIZEOF_WCHAR_T
 
 
+MAX_UNICODE = _testinternalcapi._Py_MAX_UNICODE
+# The first invalid character after MAX_UNICODE
+INVALID_CHAR = MAX_UNICODE + 1
+# Maximum invalid character which fits into 32-bit Py_UCS4
+MAX_INVALID_CHAR = 0xFFFF_FFFF
 NULL = None
 
 class Str(str):
@@ -41,6 +46,12 @@ def __str__(self):
 SSTATE_INTERNED_IMMORTAL_STATIC = 3
 
 
+def assert_invalid_string(testcase, text):
+    # Check that a Unicode string contains invalid characters:
+    # not in range [U+0000; U+10ffff]
+    testcase.assertRaises(SystemError, list, text)
+
+
 class CAPITest(unittest.TestCase):
 
     def _test_check(self, check, *, exact):
@@ -73,14 +84,14 @@ def test_new(self):
             self.assertEqual(new(0, maxchar), '')
             self.assertEqual(new(5, maxchar), chr(maxchar)*5)
             self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX, maxchar)
-        self.assertEqual(new(0, 0x110000), '')
+        self.assertEqual(new(0, INVALID_CHAR), '')
         self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x4f60)
         self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x4f60)
         self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x1f600)
         self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x1f600)
         self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4, 0x1f600)
         self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4+1, 0x1f600)
-        self.assertRaises(SystemError, new, 5, 0x110000)
+        self.assertRaises(SystemError, new, 5, INVALID_CHAR)
         self.assertRaises(SystemError, new, -1, 0)
         self.assertRaises(SystemError, new, PY_SSIZE_T_MIN, 0)
 
@@ -115,7 +126,7 @@ def test_fill(self):
         s = strings[0]
         self.assertRaises(IndexError, fill, s, -1, 0, 0x78)
         self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78)
-        self.assertRaises(ValueError, fill, s, 0, 0, 0x110000)
+        self.assertRaises(ValueError, fill, s, 0, 0, INVALID_CHAR)
         self.assertRaises(SystemError, fill, b'abc', 0, 0, 0x78)
         self.assertRaises(SystemError, fill, [], 0, 0, 0x78)
         # CRASHES fill(s, 0, NULL, 0, 0)
@@ -129,7 +140,7 @@ def _test_writechar(self, writechar, *, check):
             '\U0001f600\U0001f601\U0001f602'
         ]
         # one character for every kind + out of range code
-        chars = [0x78, 0xa9, 0x20ac, 0x1f638, 0x110000]
+        chars = [0x78, 0xa9, 0x20ac, 0x1f638, INVALID_CHAR]
         for i, s in enumerate(strings):
             for j, c in enumerate(chars):
                 if j <= i:
@@ -301,7 +312,22 @@ def test_fromkindanddata(self):
         self.assertRaises(ValueError, fromkindanddata, 1, NULL, -1)
         self.assertRaises(ValueError, fromkindanddata, 1, NULL, PY_SSIZE_T_MIN)
         # CRASHES fromkindanddata(1, NULL, 1)
-        # CRASHES fromkindanddata(4, b'\xff\xff\xff\xff')
+
+        # Test invalid UCS-4 string. Create an invalid string in release mode,
+        # or raise SystemError in debug mode.
+        for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
+            with self.subTest(invalid_char=invalid_char):
+                # Test single character
+                ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
+                self.assertRaises(SystemError, fromkindanddata, 4, ucs4_char)
+
+                # Test multiple characters
+                s = 'valid'.encode(enc4) + ucs4_char
+                if support.Py_DEBUG:
+                    self.assertRaises(SystemError, fromkindanddata, 4, s)
+                else:
+                    result = fromkindanddata(4, s)
+                    assert_invalid_string(self, result)
 
     def test_substring(self):
         """Test PyUnicode_Substring()"""
@@ -446,7 +472,7 @@ def check_format(expected, format, *args):
         check_format('\U0010ffff',
                      b'%c', c_int(0x10ffff))
         with self.assertRaises(OverflowError):
-            PyUnicode_FromFormat(b'%c', c_int(0x110000))
+            PyUnicode_FromFormat(b'%c', c_int(INVALID_CHAR))
         # Issue #18183
         check_format('\U00010000\U00100000',
                      b'%c%c', c_int(0x10000), c_int(0x100000))
@@ -1023,7 +1049,7 @@ def test_fromordinal(self):
         self.assertEqual(fromordinal(0x20ac), '\u20ac')
         self.assertEqual(fromordinal(0x1f600), '\U0001f600')
 
-        self.assertRaises(ValueError, fromordinal, 0x110000)
+        self.assertRaises(ValueError, fromordinal, INVALID_CHAR)
         self.assertRaises(ValueError, fromordinal, -1)
 
     def test_asutf8(self):
@@ -1375,8 +1401,8 @@ def test_findchar(self):
                 self.assertEqual(unicode_findchar(str, ord(ch), 0, len(str), 
-1), i)
 
         str = "!>_<!"
-        self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), 1), -1)
-        self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), -1), -1)
+        self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), 1), 
-1)
+        self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), -1), 
-1)
         # start < end
         self.assertEqual(unicode_findchar(str, ord('!'), 1, len(str)+1, 1), 4)
         self.assertEqual(unicode_findchar(str, ord('!'), 1, PY_SSIZE_T_MAX, 
1), 4)
@@ -1765,7 +1791,7 @@ def test_max_char_value(self):
         self.assertEqual(max_char_value('ascii'), 0x7f)
         self.assertEqual(max_char_value('latin1:\xe9'), 0xff)
         self.assertEqual(max_char_value('bmp:\u20ac'), 0xffff)
-        self.assertEqual(max_char_value('\U0010ffff'), 0x10_ffff)
+        self.assertEqual(max_char_value(chr(0x10_0000)), 0x10_ffff)
 
         # CRASHES max_char_value(NULL)
 
@@ -1943,8 +1969,8 @@ def test_write_char(self):
         writer.write_char(ord('$'))
         writer.write_char(0x20ac)
         writer.write_char(0x10_ffff)
-        self.assertRaises(ValueError, writer.write_char, 0x11_0000)
-        self.assertRaises(ValueError, writer.write_char, 0xFFFF_FFFF)
+        self.assertRaises(ValueError, writer.write_char, INVALID_CHAR)
+        self.assertRaises(ValueError, writer.write_char, MAX_INVALID_CHAR)
         self.assertEqual(writer.finish(),
                          "\0$\u20AC\U0010FFFF")
 
@@ -2137,15 +2163,37 @@ def test_ucs4(self):
         writer.write_ucs4("pair\uD83D\uDC0D".encode(encoding, 'surrogatepass'))
         writer.write_char(ord("-"))
         writer.write_ucs4("null[\0]".encode(encoding), 7)
-        invalid = (b'\x00\x00\x11\x00' if sys.byteorder == 'little' else
-                   b'\x00\x11\x00\x00')
-        # CRASHES writer.write_ucs4("invalid".encode(encoding) + invalid)
         writer.write_ucs4(NULL, 0)
         # CRASHES writer.write_ucs4(NULL, 1)
         self.assertEqual(writer.finish(),
                          "lone\udc80-pair\ud83d\udc0d-null[\x00]")
 
-        # invalid size
+        # Invalid UCS-4 characters. Create an invalid string in release mode,
+        # or raise SystemError in debug mode.
+        writer = self.create_writer(0)
+        for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
+            with self.subTest(invalid_char=invalid_char):
+                # Test single character
+                ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
+                if support.Py_DEBUG:
+                    self.assertRaises(SystemError, writer.write_ucs4, 
ucs4_char)
+                else:
+                    writer.write_ucs4(ucs4_char)
+
+                # Test multiple characters
+                s = 'valid'.encode(encoding) + ucs4_char
+                if support.Py_DEBUG:
+                    self.assertRaises(SystemError, writer.write_ucs4, s)
+                else:
+                    writer.write_ucs4(s)
+
+        if support.Py_DEBUG:
+            self.assertEqual(writer.finish(), '')
+        else:
+            result = writer.finish()
+            assert_invalid_string(self, result)
+
+        # Invalid size
         writer = self.create_writer(0)
         with self.assertRaises(ValueError):
             writer.write_ucs4("text".encode(encoding), -1)
diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c
index 049b990d65be5f..a2ce266eb35a65 100644
--- a/Modules/_testinternalcapi.c
+++ b/Modules/_testinternalcapi.c
@@ -3544,6 +3544,10 @@ module_exec(PyObject *module)
     }
     PyModule_AddObject(module, "SelfInterruptingContextManager", (PyObject 
*)&SelfInterruptingContextManager_Type);
 
+    if (PyModule_AddIntMacro(module, _Py_MAX_UNICODE) < 0) {
+        return 1;
+    }
+
     return 0;
 }
 
diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c
index 68ce13afd9bf37..e630dbb77f8725 100644
--- a/Objects/unicodeobject.c
+++ b/Objects/unicodeobject.c
@@ -1592,6 +1592,16 @@ PyUnicode_CopyCharacters(PyObject *to, Py_ssize_t 
to_start,
     return how_many;
 }
 
+
+static void
+unicode_invalid_character(PyObject *exc, Py_UCS4 ch)
+{
+    PyErr_Format(exc,
+                 "character U+%x is not in range [U+0000; U+%x]",
+                 ch, MAX_UNICODE);
+}
+
+
 /* Find the maximum code point and count the number of surrogate pairs so a
    correct string length can be computed before converting a string to UCS4.
    This function counts single surrogates as a character and not as a pair.
@@ -1627,9 +1637,7 @@ find_maxchar_surrogates(const wchar_t *begin, const 
wchar_t *end,
         if (ch > *maxchar) {
             *maxchar = ch;
             if (*maxchar > MAX_UNICODE) {
-                PyErr_Format(PyExc_ValueError,
-                             "character U+%x is not in range [U+0000; U+%x]",
-                             ch, MAX_UNICODE);
+                unicode_invalid_character(PyExc_ValueError, ch);
                 return -1;
             }
         }
@@ -1834,17 +1842,15 @@ get_latin1_char(Py_UCS1 ch)
 static PyObject*
 unicode_char(Py_UCS4 ch)
 {
-    PyObject *unicode;
-
-    assert(ch <= MAX_UNICODE);
-
     if (ch < 256) {
         return get_latin1_char(ch);
     }
 
-    unicode = PyUnicode_New(1, ch);
-    if (unicode == NULL)
+    // Raise SystemError if the character is not in range [U+0000; MAX_UNICODE]
+    PyObject *unicode = PyUnicode_New(1, ch);
+    if (unicode == NULL) {
         return NULL;
+    }
 
     assert(PyUnicode_KIND(unicode) != PyUnicode_1BYTE_KIND);
     if (PyUnicode_KIND(unicode) == PyUnicode_2BYTE_KIND) {
@@ -2224,15 +2230,36 @@ static PyObject*
 _PyUnicode_FromUCS4(const Py_UCS4 *u, Py_ssize_t size)
 {
     PyObject *res;
-    Py_UCS4 max_char;
 
     if (size == 0)
         _Py_RETURN_UNICODE_EMPTY();
     assert(size > 0);
-    if (size == 1)
+
+    if (size == 1) {
+        // Raise SystemError if the character is invalid
         return unicode_char(u[0]);
+    }
+
+#ifdef Py_DEBUG
+    // Check for invalid characters in debug mode
+    Py_UCS4 max_char = 127;
+    for (Py_ssize_t i = 0; i < size; i++) {
+        Py_UCS4 ch = u[i];
+        if (ch > max_char) {
+            if (ch > MAX_UNICODE) {
+                unicode_invalid_character(PyExc_SystemError, ch);
+                return NULL;
+            }
+            max_char = ch;
+        }
+    }
+#else
+    // gh-158445: Return MAX_UNICODE even if the string contains invalid
+    // characters. Checking for invalid characters would make the function
+    // slower whereas it's unlikely in practice.
+    Py_UCS4 max_char = ucs4lib_find_max_char(u, u + size);
+#endif
 
-    max_char = ucs4lib_find_max_char(u, u + size);
     res = PyUnicode_New(size, max_char);
     if (!res)
         return NULL;
@@ -2266,9 +2293,27 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer,
         return 0;
     }
 
-    Py_UCS4 max_char = ucs4lib_find_max_char(str, str + size);
+#ifdef Py_DEBUG
+    // Check for invalid characters in debug mode
+    Py_UCS4 maxchar = 127;
+    for (Py_ssize_t i = 0; i < size; i++) {
+        Py_UCS4 ch = str[i];
+        if (ch > maxchar) {
+            if (ch > MAX_UNICODE) {
+                unicode_invalid_character(PyExc_SystemError, ch);
+                return -1;
+            }
+            maxchar = ch;
+        }
+    }
+#else
+    // gh-158445: Return MAX_UNICODE even if the string contains invalid
+    // characters. Checking for invalid characters would make the function
+    // slower whereas it's unlikely in practice.
+    Py_UCS4 maxchar = ucs4lib_find_max_char(str, str + size);
+#endif
 
-    if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) {
+    if (_PyUnicodeWriter_Prepare(writer, size, maxchar) < 0) {
         return -1;
     }
     assert(_PyUnicodeWriter_CanWrite(writer));

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to