https://github.com/python/cpython/commit/a14a4a6bc9a6e1b930b1a4ba3678cb9ab551c006
commit: a14a4a6bc9a6e1b930b1a4ba3678cb9ab551c006
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T17:19:51+02:00
summary:
gh-158445: Detect invalid UCS4 characters in debug mode (#158502)
In debug mode, PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND) and
PyUnicodeWriter_WriteUCS4() now raise a SystemError if a character is
not in range [U+0000; U+10ffff], instead of creating an invalid str
object.
In release mode, PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND)
raises a SystemError if the size is 1.
* In debug mode, unicode_char() no longer fails with an assertion
error if the character is invalid. Instead, raise SystemError.
* Add tests on invalid UCS-4 characters.
* Add _testinternalcapi._Py_MAX_UNICODE.
* Add unicode_invalid_character() helper function.
files:
M Doc/c-api/unicode.rst
M Lib/test/test_capi/test_unicode.py
M Modules/_testinternalcapi.c
M Objects/unicodeobject.c
diff --git a/Doc/c-api/unicode.rst b/Doc/c-api/unicode.rst
index 5f918b447950d9..fcdcd26ff70cb0 100644
--- a/Doc/c-api/unicode.rst
+++ b/Doc/c-api/unicode.rst
@@ -425,6 +425,10 @@ APIs:
the UCS1 range, it will be transformed into UCS1
(:c:macro:`PyUnicode_1BYTE_KIND`).
+ All characters must be in range [U+0000; U+10ffff]. If *kind* is
+ :c:macro:`PyUnicode_4BYTE_KIND` and the string contains invalid characters,
+ the behavior is undefined.
+
.. versionadded:: 3.3
@@ -1896,10 +1900,13 @@ object.
.. c:function:: int PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *writer, const
Py_UCS4 *str, Py_ssize_t size)
- Writer the UCS4 string *str* into *writer*.
+ Write the UCS4 string *str* into *writer*.
*size* is a number of UCS4 characters.
+ All characters must be in range [U+0000; U+10ffff]. If the string contains
+ invalid characters, the behavior is undefined.
+
On success, return ``0``.
On error, set an exception, leave the writer unchanged, and return ``-1``.
diff --git a/Lib/test/test_capi/test_unicode.py
b/Lib/test/test_capi/test_unicode.py
index 9c91d6eb512d2a..c5926dce397409 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -19,6 +19,11 @@
from _testcapi import PY_SSIZE_T_MIN, PY_SSIZE_T_MAX, SIZEOF_WCHAR_T
+MAX_UNICODE = _testinternalcapi._Py_MAX_UNICODE
+# The first invalid character after MAX_UNICODE
+INVALID_CHAR = MAX_UNICODE + 1
+# Maximum invalid character which fits into 32-bit Py_UCS4
+MAX_INVALID_CHAR = 0xFFFF_FFFF
NULL = None
class Str(str):
@@ -41,6 +46,12 @@ def __str__(self):
SSTATE_INTERNED_IMMORTAL_STATIC = 3
+def assert_invalid_string(testcase, text):
+ # Check that a Unicode string contains invalid characters:
+ # not in range [U+0000; U+10ffff]
+ testcase.assertRaises(SystemError, list, text)
+
+
class CAPITest(unittest.TestCase):
def _test_check(self, check, *, exact):
@@ -73,14 +84,14 @@ def test_new(self):
self.assertEqual(new(0, maxchar), '')
self.assertEqual(new(5, maxchar), chr(maxchar)*5)
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX, maxchar)
- self.assertEqual(new(0, 0x110000), '')
+ self.assertEqual(new(0, INVALID_CHAR), '')
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x4f60)
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x4f60)
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x1f600)
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x1f600)
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4, 0x1f600)
self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4+1, 0x1f600)
- self.assertRaises(SystemError, new, 5, 0x110000)
+ self.assertRaises(SystemError, new, 5, INVALID_CHAR)
self.assertRaises(SystemError, new, -1, 0)
self.assertRaises(SystemError, new, PY_SSIZE_T_MIN, 0)
@@ -115,7 +126,7 @@ def test_fill(self):
s = strings[0]
self.assertRaises(IndexError, fill, s, -1, 0, 0x78)
self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78)
- self.assertRaises(ValueError, fill, s, 0, 0, 0x110000)
+ self.assertRaises(ValueError, fill, s, 0, 0, INVALID_CHAR)
self.assertRaises(SystemError, fill, b'abc', 0, 0, 0x78)
self.assertRaises(SystemError, fill, [], 0, 0, 0x78)
# CRASHES fill(s, 0, NULL, 0, 0)
@@ -129,7 +140,7 @@ def _test_writechar(self, writechar, *, check):
'\U0001f600\U0001f601\U0001f602'
]
# one character for every kind + out of range code
- chars = [0x78, 0xa9, 0x20ac, 0x1f638, 0x110000]
+ chars = [0x78, 0xa9, 0x20ac, 0x1f638, INVALID_CHAR]
for i, s in enumerate(strings):
for j, c in enumerate(chars):
if j <= i:
@@ -301,7 +312,22 @@ def test_fromkindanddata(self):
self.assertRaises(ValueError, fromkindanddata, 1, NULL, -1)
self.assertRaises(ValueError, fromkindanddata, 1, NULL, PY_SSIZE_T_MIN)
# CRASHES fromkindanddata(1, NULL, 1)
- # CRASHES fromkindanddata(4, b'\xff\xff\xff\xff')
+
+ # Test invalid UCS-4 string. Create an invalid string in release mode,
+ # or raise SystemError in debug mode.
+ for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
+ with self.subTest(invalid_char=invalid_char):
+ # Test single character
+ ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
+ self.assertRaises(SystemError, fromkindanddata, 4, ucs4_char)
+
+ # Test multiple characters
+ s = 'valid'.encode(enc4) + ucs4_char
+ if support.Py_DEBUG:
+ self.assertRaises(SystemError, fromkindanddata, 4, s)
+ else:
+ result = fromkindanddata(4, s)
+ assert_invalid_string(self, result)
def test_substring(self):
"""Test PyUnicode_Substring()"""
@@ -446,7 +472,7 @@ def check_format(expected, format, *args):
check_format('\U0010ffff',
b'%c', c_int(0x10ffff))
with self.assertRaises(OverflowError):
- PyUnicode_FromFormat(b'%c', c_int(0x110000))
+ PyUnicode_FromFormat(b'%c', c_int(INVALID_CHAR))
# Issue #18183
check_format('\U00010000\U00100000',
b'%c%c', c_int(0x10000), c_int(0x100000))
@@ -1023,7 +1049,7 @@ def test_fromordinal(self):
self.assertEqual(fromordinal(0x20ac), '\u20ac')
self.assertEqual(fromordinal(0x1f600), '\U0001f600')
- self.assertRaises(ValueError, fromordinal, 0x110000)
+ self.assertRaises(ValueError, fromordinal, INVALID_CHAR)
self.assertRaises(ValueError, fromordinal, -1)
def test_asutf8(self):
@@ -1375,8 +1401,8 @@ def test_findchar(self):
self.assertEqual(unicode_findchar(str, ord(ch), 0, len(str),
-1), i)
str = "!>_<!"
- self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), 1), -1)
- self.assertEqual(unicode_findchar(str, 0x110000, 0, len(str), -1), -1)
+ self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), 1),
-1)
+ self.assertEqual(unicode_findchar(str, INVALID_CHAR, 0, len(str), -1),
-1)
# start < end
self.assertEqual(unicode_findchar(str, ord('!'), 1, len(str)+1, 1), 4)
self.assertEqual(unicode_findchar(str, ord('!'), 1, PY_SSIZE_T_MAX,
1), 4)
@@ -1765,7 +1791,7 @@ def test_max_char_value(self):
self.assertEqual(max_char_value('ascii'), 0x7f)
self.assertEqual(max_char_value('latin1:\xe9'), 0xff)
self.assertEqual(max_char_value('bmp:\u20ac'), 0xffff)
- self.assertEqual(max_char_value('\U0010ffff'), 0x10_ffff)
+ self.assertEqual(max_char_value(chr(0x10_0000)), 0x10_ffff)
# CRASHES max_char_value(NULL)
@@ -1943,8 +1969,8 @@ def test_write_char(self):
writer.write_char(ord('$'))
writer.write_char(0x20ac)
writer.write_char(0x10_ffff)
- self.assertRaises(ValueError, writer.write_char, 0x11_0000)
- self.assertRaises(ValueError, writer.write_char, 0xFFFF_FFFF)
+ self.assertRaises(ValueError, writer.write_char, INVALID_CHAR)
+ self.assertRaises(ValueError, writer.write_char, MAX_INVALID_CHAR)
self.assertEqual(writer.finish(),
"\0$\u20AC\U0010FFFF")
@@ -2137,15 +2163,37 @@ def test_ucs4(self):
writer.write_ucs4("pair\uD83D\uDC0D".encode(encoding, 'surrogatepass'))
writer.write_char(ord("-"))
writer.write_ucs4("null[\0]".encode(encoding), 7)
- invalid = (b'\x00\x00\x11\x00' if sys.byteorder == 'little' else
- b'\x00\x11\x00\x00')
- # CRASHES writer.write_ucs4("invalid".encode(encoding) + invalid)
writer.write_ucs4(NULL, 0)
# CRASHES writer.write_ucs4(NULL, 1)
self.assertEqual(writer.finish(),
"lone\udc80-pair\ud83d\udc0d-null[\x00]")
- # invalid size
+ # Invalid UCS-4 characters. Create an invalid string in release mode,
+ # or raise SystemError in debug mode.
+ writer = self.create_writer(0)
+ for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR):
+ with self.subTest(invalid_char=invalid_char):
+ # Test single character
+ ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder)
+ if support.Py_DEBUG:
+ self.assertRaises(SystemError, writer.write_ucs4,
ucs4_char)
+ else:
+ writer.write_ucs4(ucs4_char)
+
+ # Test multiple characters
+ s = 'valid'.encode(encoding) + ucs4_char
+ if support.Py_DEBUG:
+ self.assertRaises(SystemError, writer.write_ucs4, s)
+ else:
+ writer.write_ucs4(s)
+
+ if support.Py_DEBUG:
+ self.assertEqual(writer.finish(), '')
+ else:
+ result = writer.finish()
+ assert_invalid_string(self, result)
+
+ # Invalid size
writer = self.create_writer(0)
with self.assertRaises(ValueError):
writer.write_ucs4("text".encode(encoding), -1)
diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c
index 049b990d65be5f..a2ce266eb35a65 100644
--- a/Modules/_testinternalcapi.c
+++ b/Modules/_testinternalcapi.c
@@ -3544,6 +3544,10 @@ module_exec(PyObject *module)
}
PyModule_AddObject(module, "SelfInterruptingContextManager", (PyObject
*)&SelfInterruptingContextManager_Type);
+ if (PyModule_AddIntMacro(module, _Py_MAX_UNICODE) < 0) {
+ return 1;
+ }
+
return 0;
}
diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c
index 68ce13afd9bf37..e630dbb77f8725 100644
--- a/Objects/unicodeobject.c
+++ b/Objects/unicodeobject.c
@@ -1592,6 +1592,16 @@ PyUnicode_CopyCharacters(PyObject *to, Py_ssize_t
to_start,
return how_many;
}
+
+static void
+unicode_invalid_character(PyObject *exc, Py_UCS4 ch)
+{
+ PyErr_Format(exc,
+ "character U+%x is not in range [U+0000; U+%x]",
+ ch, MAX_UNICODE);
+}
+
+
/* Find the maximum code point and count the number of surrogate pairs so a
correct string length can be computed before converting a string to UCS4.
This function counts single surrogates as a character and not as a pair.
@@ -1627,9 +1637,7 @@ find_maxchar_surrogates(const wchar_t *begin, const
wchar_t *end,
if (ch > *maxchar) {
*maxchar = ch;
if (*maxchar > MAX_UNICODE) {
- PyErr_Format(PyExc_ValueError,
- "character U+%x is not in range [U+0000; U+%x]",
- ch, MAX_UNICODE);
+ unicode_invalid_character(PyExc_ValueError, ch);
return -1;
}
}
@@ -1834,17 +1842,15 @@ get_latin1_char(Py_UCS1 ch)
static PyObject*
unicode_char(Py_UCS4 ch)
{
- PyObject *unicode;
-
- assert(ch <= MAX_UNICODE);
-
if (ch < 256) {
return get_latin1_char(ch);
}
- unicode = PyUnicode_New(1, ch);
- if (unicode == NULL)
+ // Raise SystemError if the character is not in range [U+0000; MAX_UNICODE]
+ PyObject *unicode = PyUnicode_New(1, ch);
+ if (unicode == NULL) {
return NULL;
+ }
assert(PyUnicode_KIND(unicode) != PyUnicode_1BYTE_KIND);
if (PyUnicode_KIND(unicode) == PyUnicode_2BYTE_KIND) {
@@ -2224,15 +2230,36 @@ static PyObject*
_PyUnicode_FromUCS4(const Py_UCS4 *u, Py_ssize_t size)
{
PyObject *res;
- Py_UCS4 max_char;
if (size == 0)
_Py_RETURN_UNICODE_EMPTY();
assert(size > 0);
- if (size == 1)
+
+ if (size == 1) {
+ // Raise SystemError if the character is invalid
return unicode_char(u[0]);
+ }
+
+#ifdef Py_DEBUG
+ // Check for invalid characters in debug mode
+ Py_UCS4 max_char = 127;
+ for (Py_ssize_t i = 0; i < size; i++) {
+ Py_UCS4 ch = u[i];
+ if (ch > max_char) {
+ if (ch > MAX_UNICODE) {
+ unicode_invalid_character(PyExc_SystemError, ch);
+ return NULL;
+ }
+ max_char = ch;
+ }
+ }
+#else
+ // gh-158445: Return MAX_UNICODE even if the string contains invalid
+ // characters. Checking for invalid characters would make the function
+ // slower whereas it's unlikely in practice.
+ Py_UCS4 max_char = ucs4lib_find_max_char(u, u + size);
+#endif
- max_char = ucs4lib_find_max_char(u, u + size);
res = PyUnicode_New(size, max_char);
if (!res)
return NULL;
@@ -2266,9 +2293,27 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer,
return 0;
}
- Py_UCS4 max_char = ucs4lib_find_max_char(str, str + size);
+#ifdef Py_DEBUG
+ // Check for invalid characters in debug mode
+ Py_UCS4 maxchar = 127;
+ for (Py_ssize_t i = 0; i < size; i++) {
+ Py_UCS4 ch = str[i];
+ if (ch > maxchar) {
+ if (ch > MAX_UNICODE) {
+ unicode_invalid_character(PyExc_SystemError, ch);
+ return -1;
+ }
+ maxchar = ch;
+ }
+ }
+#else
+ // gh-158445: Return MAX_UNICODE even if the string contains invalid
+ // characters. Checking for invalid characters would make the function
+ // slower whereas it's unlikely in practice.
+ Py_UCS4 maxchar = ucs4lib_find_max_char(str, str + size);
+#endif
- if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) {
+ if (_PyUnicodeWriter_Prepare(writer, size, maxchar) < 0) {
return -1;
}
assert(_PyUnicodeWriter_CanWrite(writer));
_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]