https://github.com/python/cpython/commit/84b0669f81c0cf9b0b1accdbf02cf06747bccb04
commit: 84b0669f81c0cf9b0b1accdbf02cf06747bccb04
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-10-05T07:16:37Z
summary:

gh-157710: Soft deprecate C API modifying str objects (#157711)

Soft deprecate PyUnicode_New(), PyUnicode_CopyCharacters(),
PyUnicode_Fill(), PyUnicode_Resize(), PyUnicode_WRITE() and
PyUnicode_WriteChar() functions. Use the PyUnicodeWriter API instead.

* Add tests on PyUnicode_New() and PyUnicode_Resize(). Check that the
  result is either a mutable string or the empty string singleton.
* Check that PyUnicode_Fill(), PyUnicode_CopyCharacters() and
  PyUnicode_WriteChar() fail to modify a string with 2 references.

Co-authored-by: Peter Bierma <[email protected]>

files:
A Misc/NEWS.d/next/C_API/2026-09-17-23-18-12.gh-issue-157710.1xKQAH.rst
M Doc/c-api/unicode.rst
M Doc/whatsnew/3.16.rst
M Lib/test/test_capi/test_unicode.py
M Modules/_testcapi/unicode.c
M Modules/_testlimitedcapi/unicode.c

diff --git a/Doc/c-api/unicode.rst b/Doc/c-api/unicode.rst
index fcdcd26ff70cb0e..584732b05c811ff 100644
--- a/Doc/c-api/unicode.rst
+++ b/Doc/c-api/unicode.rst
@@ -168,8 +168,14 @@ access to internal read-only data of Unicode objects:
    The function performs no checks for any of its requirements,
    and is intended for usage in loops.
 
+   While :class:`str` objects are usually immutable in Python, this special C 
API allows
+   mutating a fresh :class:`str` object if the string has not been "used" yet.
+
    .. versionadded:: 3.3
 
+   .. soft-deprecated:: next
+      Use the :c:type:`PyUnicodeWriter` API instead.
+
 
 .. c:function:: Py_UCS4 PyUnicode_READ(int kind, void *data, Py_ssize_t index)
 
@@ -407,9 +413,15 @@ APIs:
    using the :c:type:`PyUnicodeWriter` API, or one of the ``PyUnicode_From*``
    functions below.
 
+   While :class:`str` objects are usually immutable in Python, this special C 
API
+   returns a :class:`str` object that can be mutated, except if *size* is 
zero, in which
+   case it returns the immutable empty string constant.
 
    .. versionadded:: 3.3
 
+   .. soft-deprecated:: next
+      Use the :c:type:`PyUnicodeWriter` API instead.
+
 
 .. c:function:: PyObject* PyUnicode_FromKindAndData(int kind, const void 
*buffer, \
                                                     Py_ssize_t size)
@@ -754,11 +766,16 @@ APIs:
    possible.  Returns ``-1`` and sets an exception on error, otherwise returns
    the number of copied characters.
 
-   The string must not have been “used” yet.
+   While :class:`str` objects are usually immutable in Python, this special C 
API allows
+   mutating a fresh :class:`str` object if the string has not been "used" yet.
+
    See :c:func:`PyUnicode_New` for details.
 
    .. versionadded:: 3.3
 
+   .. soft-deprecated:: next
+      Use the :c:type:`PyUnicodeWriter` API instead.
+
 
 .. c:function:: int PyUnicode_Resize(PyObject **unicode, Py_ssize_t length);
 
@@ -774,6 +791,14 @@ APIs:
    The function doesn't check string content, the result may not be a
    string in canonical representation.
 
+   While :class:`str` objects are usually immutable in Python, this special C 
API
+   can resize a :class:`str` object in-place if the string has not been "used" 
yet.
+   It returns a :class:`str` object which can be mutated, except if *size* is 
zero, in
+   which case it returns the immutable empty string constant.
+
+   .. soft-deprecated:: next
+      Use the :c:type:`PyUnicodeWriter` API instead.
+
 
 .. c:function:: Py_ssize_t PyUnicode_Fill(PyObject *unicode, Py_ssize_t start, 
\
                         Py_ssize_t length, Py_UCS4 fill_char)
@@ -784,14 +809,19 @@ APIs:
    Fail if *fill_char* is bigger than the string maximum character, or if the
    string has more than 1 reference.
 
-   The string must not have been “used” yet.
-   See :c:func:`PyUnicode_New` for details.
-
    Return the number of written characters, or return ``-1`` and raise an
    exception on error.
 
+   While :class:`str` objects are usually immutable in Python, this special C 
API allows
+   mutating a fresh :class:`str` object if the string has not been "used" yet.
+
+   See :c:func:`PyUnicode_New` for details.
+
    .. versionadded:: 3.3
 
+   .. soft-deprecated:: next
+      Use the :c:type:`PyUnicodeWriter` API instead.
+
 
 .. c:function:: int PyUnicode_WriteChar(PyObject *unicode, Py_ssize_t index, \
                                         Py_UCS4 character)
@@ -804,11 +834,16 @@ APIs:
    See :c:func:`PyUnicode_WRITE` for a version that skips these checks,
    making them your responsibility.
 
-   The string must not have been “used” yet.
+   While :class:`str` objects are usually immutable in Python, this special C 
API allows
+   mutating a fresh :class:`str` object if the string has not been "used" yet.
+
    See :c:func:`PyUnicode_New` for details.
 
    .. versionadded:: 3.3
 
+   .. soft-deprecated:: next
+      Use the :c:type:`PyUnicodeWriter` API instead.
+
 
 .. c:function:: Py_UCS4 PyUnicode_ReadChar(PyObject *unicode, Py_ssize_t index)
 
diff --git a/Doc/whatsnew/3.16.rst b/Doc/whatsnew/3.16.rst
index 16e03e9ef065321..65db9bd6bf70ba0 100644
--- a/Doc/whatsnew/3.16.rst
+++ b/Doc/whatsnew/3.16.rst
@@ -1224,6 +1224,13 @@ Deprecated C APIs
   :c:func:`PyModule_GetFilenameObject` instead is still recommended.
   (Contributed by Victor Stinner in :gh:`154757`.)
 
+* Soft deprecate functions modifying Unicode strings:
+  :c:func:`PyUnicode_New`, :c:func:`PyUnicode_CopyCharacters`,
+  :c:func:`PyUnicode_Fill`, :c:func:`PyUnicode_Resize`,
+  :c:func:`PyUnicode_WRITE` and :c:func:`PyUnicode_WriteChar`.
+  Use the safer :c:type:`PyUnicodeWriter` API instead.
+  (Contributed by Victor Stinner in :gh:`157710`.)
+
 .. Add C API deprecations above alphabetically, not here at the end.
 
 .. include:: ../deprecations/c-api-pending-removal-in-3.18.rst
diff --git a/Lib/test/test_capi/test_unicode.py 
b/Lib/test/test_capi/test_unicode.py
index d43318770a4e62c..9237809e7f0dcca 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -25,6 +25,7 @@
 # Maximum invalid character which fits into 32-bit Py_UCS4
 MAX_INVALID_CHAR = 0xFFFF_FFFF
 NULL = None
+USED_STR_ERROR = 'Cannot modify a string currently used'
 
 class Str(str):
     pass
@@ -76,9 +77,27 @@ def test_checkexact(self):
         # Test PyUnicode_CheckExact()
         self._test_check(_testlimitedcapi.unicode_checkexact, exact=True)
 
+    def assert_is_mutable(self, result, refcnt):
+        # Check that result is a "mutable" Unicode string
+        self.assertEqual(refcnt, 1)
+        self.assertFalse(sys._is_immortal(result))
+
+    def assert_is_empty_singleton(self, result):
+        # Check that result is the empty string singleton
+        self.assertEqual(result, '')
+        self.assertTrue(sys._is_immortal(result))
+
     def test_new(self):
         """Test PyUnicode_New()"""
-        new = _testcapi.unicode_new
+        _unicode_new = _testcapi.unicode_new
+
+        def new(size, maxchar):
+            result = _unicode_new(size, maxchar)
+            if size != 0:
+                self.assert_is_mutable(result, sys.getrefcount(result))
+            else:
+                self.assert_is_empty_singleton(result)
+            return result
 
         for maxchar in 0, 0x61, 0xa1, 0x4f60, 0x1f600, 0x10ffff:
             self.assertEqual(new(0, maxchar), '')
@@ -123,6 +142,10 @@ def test_fill(self):
                         self.assertEqual(fill(to, start, length, fill_char),
                                         (expected, filled))
 
+        # A string with 2 references cannot be modified
+        with self.assertRaisesRegex(SystemError, USED_STR_ERROR):
+            fill('abc', 0, 3, ord('x'), incref=True)
+
         s = strings[0]
         self.assertRaises(IndexError, fill, s, -1, 0, 0x78)
         self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78)
@@ -162,7 +185,12 @@ def _test_writechar(self, writechar, *, check):
 
     def test_writechar(self):
         """Test PyUnicode_WriteChar()"""
-        self._test_writechar(_testlimitedcapi.unicode_writechar, check=True)
+        writechar = _testlimitedcapi.unicode_writechar
+        self._test_writechar(writechar, check=True)
+
+        # A string with 2 references cannot be modified
+        with self.assertRaisesRegex(SystemError, USED_STR_ERROR):
+            writechar('abc', 1, ord('x'), incref=True)
 
     def test_write_macro(self):
         """Test PyUnicode_WRITE()"""
@@ -187,18 +215,16 @@ def resize(s, length, new=True, compute_hash=False):
                 self.assertFalse(is_new_obj)
             elif length == 0:
                 # Get the empty Unicode string
-                self.assertEqual(result, '')
-                self.assertTrue(sys._is_immortal(result))
+                self.assert_is_empty_singleton(result)
                 self.assertTrue(is_new_obj)
             elif (not new) or compute_hash:
                 # Get a fresh copy
-                self.assertEqual(refcnt, 1)
+                self.assert_is_mutable(result, refcnt)
                 self.assertTrue(is_new_obj)
-                self.assertFalse(sys._is_immortal(result))
             else:
                 # In-size replace can return the same address, or not.
                 # So 'is_new_obj' cannot be tested.
-                self.assertFalse(sys._is_immortal(result))
+                self.assert_is_mutable(result, refcnt)
 
             return result
 
@@ -1793,6 +1819,11 @@ def test_copycharacters(self):
         self.assertRaises(SystemError, unicode_copycharacters, s, 0, s, 0, 
PY_SSIZE_T_MIN)
         self.assertRaises(SystemError, unicode_copycharacters, s, 0, b'', 0, 0)
         self.assertRaises(SystemError, unicode_copycharacters, s, 0, [], 0, 0)
+
+        # A string with 2 references cannot be modified
+        with self.assertRaisesRegex(SystemError, USED_STR_ERROR):
+            unicode_copycharacters('abc', 0, 'abc', 0, 1, incref=True)
+
         # CRASHES unicode_copycharacters(s, 0, NULL, 0, 0)
         # TODO: Test PyUnicode_CopyCharacters() with non-unicode and
         # non-modifiable unicode as "to".
diff --git 
a/Misc/NEWS.d/next/C_API/2026-09-17-23-18-12.gh-issue-157710.1xKQAH.rst 
b/Misc/NEWS.d/next/C_API/2026-09-17-23-18-12.gh-issue-157710.1xKQAH.rst
new file mode 100644
index 000000000000000..4b203e430e94d87
--- /dev/null
+++ b/Misc/NEWS.d/next/C_API/2026-09-17-23-18-12.gh-issue-157710.1xKQAH.rst
@@ -0,0 +1,5 @@
+Soft deprecate functions modifying Unicode strings:
+:c:func:`PyUnicode_New`, :c:func:`PyUnicode_CopyCharacters`,
+:c:func:`PyUnicode_Fill`, :c:func:`PyUnicode_Resize`,
+:c:func:`PyUnicode_WRITE` and :c:func:`PyUnicode_WriteChar`. Use
+the :c:type:`PyUnicodeWriter` API instead. Patch by Victor Stinner.
diff --git a/Modules/_testcapi/unicode.c b/Modules/_testcapi/unicode.c
index 5138fbab1f4ed64..ec3e9a623519f6e 100644
--- a/Modules/_testcapi/unicode.c
+++ b/Modules/_testcapi/unicode.c
@@ -59,13 +59,20 @@ unicode_copy(PyObject *unicode)
 
 /* Test PyUnicode_Fill() */
 static PyObject *
-unicode_fill(PyObject *self, PyObject *args)
+unicode_fill(PyObject *self, PyObject *args, PyObject *kwargs)
 {
+    static char *kwlist[] = {"to", "start", "length", "fill_char",
+                             "incref", NULL};
     PyObject *to, *to_copy;
     Py_ssize_t start, length, filled;
     unsigned int fill_char;
+    int incref = 0;
 
-    if (!PyArg_ParseTuple(args, "OnnI", &to, &start, &length, &fill_char)) {
+    if (!PyArg_ParseTupleAndKeywords(args, kwargs,
+                                     "OnnI|p", kwlist,
+                                     &to, &start, &length,
+                                     &fill_char, &incref))
+    {
         return NULL;
     }
 
@@ -74,7 +81,14 @@ unicode_fill(PyObject *self, PyObject *args)
         return NULL;
     }
 
+    if (incref) {
+        Py_INCREF(to_copy);
+    }
     filled = PyUnicode_Fill(to_copy, start, length, (Py_UCS4)fill_char);
+    if (incref) {
+        Py_DECREF(to_copy);
+    }
+
     if (filled == -1 && PyErr_Occurred()) {
         Py_DECREF(to_copy);
         return NULL;
@@ -190,13 +204,18 @@ unicode_asutf8(PyObject *self, PyObject *args)
 
 /* Test PyUnicode_CopyCharacters() */
 static PyObject *
-unicode_copycharacters(PyObject *self, PyObject *args)
+unicode_copycharacters(PyObject *self, PyObject *args, PyObject *kwargs)
 {
+    static char *kwlist[] = {"to", "to_start", "from", "from_start",
+                             "howmany", "incref", NULL};
     PyObject *from, *to, *to_copy;
     Py_ssize_t from_start, to_start, how_many, copied;
+    int incref = 0;
 
-    if (!PyArg_ParseTuple(args, "UnOnn", &to, &to_start,
-                          &from, &from_start, &how_many)) {
+    if (!PyArg_ParseTupleAndKeywords(args, kwargs,
+                                     "UnOnn|p", kwlist,
+                                     &to, &to_start, &from, &from_start,
+                                     &how_many, &incref)) {
         return NULL;
     }
 
@@ -210,8 +229,15 @@ unicode_copycharacters(PyObject *self, PyObject *args)
         return NULL;
     }
 
+    if (incref) {
+        Py_INCREF(to_copy);
+    }
     copied = PyUnicode_CopyCharacters(to_copy, to_start, from,
                                       from_start, how_many);
+    if (incref) {
+        Py_DECREF(to_copy);
+    }
+
     if (copied == -1 && PyErr_Occurred()) {
         Py_DECREF(to_copy);
         return NULL;
@@ -856,12 +882,12 @@ static PyType_Spec Writer_spec = {
 
 static PyMethodDef TestMethods[] = {
     {"unicode_new",              unicode_new,                    METH_VARARGS},
-    {"unicode_fill",             unicode_fill,                   METH_VARARGS},
+    {"unicode_fill",             _PyCFunction_CAST(unicode_fill), METH_VARARGS 
| METH_KEYWORDS},
     {"unicode_fromkindanddata",  unicode_fromkindanddata,        METH_VARARGS},
     {"unicode_asucs4",           unicode_asucs4,                 METH_VARARGS},
     {"unicode_asucs4copy",       unicode_asucs4copy,             METH_VARARGS},
     {"unicode_asutf8",           unicode_asutf8,                 METH_VARARGS},
-    {"unicode_copycharacters",   unicode_copycharacters,         METH_VARARGS},
+    {"unicode_copycharacters",   _PyCFunction_CAST(unicode_copycharacters), 
METH_VARARGS | METH_KEYWORDS},
     {"unicode_GET_CACHED_HASH",  unicode_GET_CACHED_HASH,        METH_O},
     {"test_py_identifier",       test_py_identifier,             METH_NOARGS},
     {"corrupt_unicode",          corrupt_unicode,                METH_VARARGS},
diff --git a/Modules/_testlimitedcapi/unicode.c 
b/Modules/_testlimitedcapi/unicode.c
index 53fff1916ac1ab6..3d88cbb9fb304d3 100644
--- a/Modules/_testlimitedcapi/unicode.c
+++ b/Modules/_testlimitedcapi/unicode.c
@@ -140,14 +140,19 @@ unicode_copy(PyObject *unicode)
 
 /* Test PyUnicode_WriteChar() */
 static PyObject *
-unicode_writechar(PyObject *self, PyObject *args)
+unicode_writechar(PyObject *self, PyObject *args, PyObject *kwargs)
 {
+    static char *kwlist[] = {"to", "index", "character", "incref", NULL};
     PyObject *to, *to_copy;
     Py_ssize_t index;
     unsigned int character;
     int result;
+    int incref = 0;
 
-    if (!PyArg_ParseTuple(args, "OnI", &to, &index, &character)) {
+    if (!PyArg_ParseTupleAndKeywords(args, kwargs,
+                                     "OnI|p", kwlist,
+                                     &to, &index, &character, &incref))
+    {
         return NULL;
     }
 
@@ -156,7 +161,14 @@ unicode_writechar(PyObject *self, PyObject *args)
         return NULL;
     }
 
+    if (incref) {
+        Py_INCREF(to_copy);
+    }
     result = PyUnicode_WriteChar(to_copy, index, (Py_UCS4)character);
+    if (incref) {
+        Py_DECREF(to_copy);
+    }
+
     if (result == -1 && PyErr_Occurred()) {
         Py_DECREF(to_copy);
         return NULL;
@@ -1915,7 +1927,7 @@ static PyMethodDef TestMethods[] = {
      test_unicode_compare_with_ascii,                            METH_NOARGS},
     {"test_string_from_format",  test_string_from_format,        METH_NOARGS},
     {"test_widechar",            test_widechar,                  METH_NOARGS},
-    {"unicode_writechar",        unicode_writechar,              METH_VARARGS},
+    {"unicode_writechar",        _PyCFunction_CAST(unicode_writechar), 
METH_VARARGS | METH_KEYWORDS},
     {"unicode_resize",           unicode_resize,                 METH_VARARGS},
     {"unicode_resize_null",      unicode_resize_null,            METH_VARARGS},
     {"unicode_append",           unicode_append,                 METH_VARARGS},

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to