https://github.com/python/cpython/commit/26d3f5dd6a9c6699f430138db0e92d224f958747
commit: 26d3f5dd6a9c6699f430138db0e92d224f958747
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T03:53:38+02:00
summary:

gh-158451: Add _PyUnicodeWriter_WriteLatin1String() test (#158464)

Fix _PyUnicodeWriter_WriteLatin1String() when the writer buffer kind
is UCS-2 or UCS-4.

files:
A Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
M Lib/test/test_capi/test_unicode.py
M Modules/_testcapi/unicode.c
M Objects/unicode_writer.c

diff --git a/Lib/test/test_capi/test_unicode.py 
b/Lib/test/test_capi/test_unicode.py
index 36ecca48230ddd..c511ffc0fda13c 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -1970,6 +1970,24 @@ def test_ascii(self):
         writer.write_ascii(b"Python! <truncated>", 6)
         self.assertEqual(writer.finish(), "Hello Python")
 
+    def test_write_latin1(self):
+        # Test _PyUnicodeWriter_WriteLatin1String()
+        writer = self.create_writer(0)
+        # Start with ASCII buffer
+        writer.write_latin1(b"abc IGNORED", 3)
+        writer.write_latin1(b"IGNORED", 0)
+        # Change buffer kind to UCS-1
+        writer.write_latin1(b"\xe9", 1)
+        # Change buffer kind to UCS-2
+        writer.write_str('[\u20ac]')
+        writer.write_latin1(b"def\xa0", 4)
+        # Change buffer kind to UCS-4
+        writer.write_str('[\U0010ffff]')
+        writer.write_latin1(b"ghi\xff.", 5)
+        writer.write_latin1(b"IGNORED", 0)
+        self.assertEqual(writer.finish(),
+                         "abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.")
+
     def test_invalid_utf8(self):
         writer = self.create_writer(0)
         with self.assertRaises(UnicodeDecodeError):
diff --git 
a/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst 
b/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
new file mode 100644
index 00000000000000..a16dd024287076
--- /dev/null
+++ b/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
@@ -0,0 +1,2 @@
+Fix :c:func:`!_PyUnicodeWriter_WriteLatin1String` when the writer buffer
+kind is UCS-2 or UCS-4. Patch by Victor Stinner.
diff --git a/Modules/_testcapi/unicode.c b/Modules/_testcapi/unicode.c
index c62813f4f3768d..ba9b205e07ef8b 100644
--- a/Modules/_testcapi/unicode.c
+++ b/Modules/_testcapi/unicode.c
@@ -685,6 +685,32 @@ writer_write_substring(PyObject *self_raw, PyObject *args)
 }
 
 
+static PyObject*
+writer_write_latin1(PyObject *self_raw, PyObject *args)
+{
+    WriterObject *self = (WriterObject *)self_raw;
+    if (writer_check(self) < 0) {
+        return NULL;
+    }
+
+    const char *str;
+    Py_ssize_t bsize, size;
+    if (!PyArg_ParseTuple(args, "z#n", &str, &bsize, &size)) {
+        return NULL;
+    }
+
+    _PyUnicodeWriter *writer = (_PyUnicodeWriter*)self->writer;
+_Py_COMP_DIAG_PUSH
+_Py_COMP_DIAG_IGNORE_DEPR_DECLS
+    if (_PyUnicodeWriter_WriteLatin1String(writer, str, size) < 0) {
+        return NULL;
+    }
+_Py_COMP_DIAG_POP
+
+    Py_RETURN_NONE;
+}
+
+
 static PyObject*
 writer_decodeutf8stateful(PyObject *self_raw, PyObject *args)
 {
@@ -778,6 +804,7 @@ static PyMethodDef writer_methods[] = {
     {"write_str", _PyCFunction_CAST(writer_write_str), METH_O},
     {"write_repr", _PyCFunction_CAST(writer_write_repr), METH_O},
     {"write_substring", _PyCFunction_CAST(writer_write_substring), 
METH_VARARGS},
+    {"write_latin1", _PyCFunction_CAST(writer_write_latin1), METH_VARARGS},
     {"decodeutf8stateful", _PyCFunction_CAST(writer_decodeutf8stateful), 
METH_VARARGS},
     {"get_pointer", _PyCFunction_CAST(writer_get_pointer), METH_VARARGS},
     {"get_buffer", _PyCFunction_CAST(writer_get_buffer), METH_VARARGS},
diff --git a/Objects/unicode_writer.c b/Objects/unicode_writer.c
index 8637be921e2454..0949e45d51cbad 100644
--- a/Objects/unicode_writer.c
+++ b/Objects/unicode_writer.c
@@ -62,59 +62,6 @@ OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS 
SOFTWARE.
 #include "stringlib/undef.h"
 
 
-/* Copy an ASCII or latin1 char* string into a Python Unicode string.
-
-   WARNING: The function doesn't copy the terminating null character and
-   doesn't check the maximum character (may write a latin1 character in an
-   ASCII string). */
-static void
-unicode_write_cstr(PyObject *unicode, Py_ssize_t index,
-                   const char *str, Py_ssize_t len)
-{
-    int kind = PyUnicode_KIND(unicode);
-    const void *data = PyUnicode_DATA(unicode);
-    const char *end = str + len;
-
-    assert(index + len <= PyUnicode_GET_LENGTH(unicode));
-    switch (kind) {
-    case PyUnicode_1BYTE_KIND: {
-#ifdef Py_DEBUG
-        if (PyUnicode_IS_ASCII(unicode)) {
-            Py_UCS4 maxchar = ucs1lib_find_max_char(
-                (const Py_UCS1*)str,
-                (const Py_UCS1*)str + len);
-            assert(maxchar < 128);
-        }
-#endif
-        memcpy((char *) data + index, str, len);
-        break;
-    }
-    case PyUnicode_2BYTE_KIND: {
-        Py_UCS2 *start = (Py_UCS2 *)data + index;
-        Py_UCS2 *ucs2 = start;
-
-        for (; str < end; ++ucs2, ++str)
-            *ucs2 = (Py_UCS2)*str;
-
-        assert((ucs2 - start) <= PyUnicode_GET_LENGTH(unicode));
-        break;
-    }
-    case PyUnicode_4BYTE_KIND: {
-        Py_UCS4 *start = (Py_UCS4 *)data + index;
-        Py_UCS4 *ucs4 = start;
-
-        for (; str < end; ++ucs4, ++str)
-            *ucs4 = (Py_UCS4)*str;
-
-        assert((ucs4 - start) <= PyUnicode_GET_LENGTH(unicode));
-        break;
-    }
-    default:
-        Py_UNREACHABLE();
-    }
-}
-
-
 void
 _PyUnicodeWriter_Init(_PyUnicodeWriter *writer)
 {
@@ -550,13 +497,43 @@ int
 _PyUnicodeWriter_WriteLatin1String(_PyUnicodeWriter *writer,
                                    const char *str, Py_ssize_t len)
 {
-    Py_UCS4 maxchar;
+    if (len == 0) {
+        return 0;
+    }
 
-    maxchar = ucs1lib_find_max_char((const Py_UCS1*)str, (const Py_UCS1*)str + 
len);
-    if (_PyUnicodeWriter_Prepare(writer, len, maxchar) == -1)
+    const Py_UCS1 *ucs1 = (const Py_UCS1 *)str;
+    Py_UCS4 maxchar = ucs1lib_find_max_char(ucs1, ucs1 + len);
+    if (_PyUnicodeWriter_Prepare(writer, len, maxchar) < 0) {
         return -1;
+    }
     assert(_PyUnicodeWriter_CanWrite(writer));
-    unicode_write_cstr(writer->buffer, writer->pos, str, len);
+
+    Py_ssize_t index = writer->pos;
+    switch (writer->kind) {
+    case PyUnicode_1BYTE_KIND: {
+        memcpy((Py_UCS1 *)writer->data + index, ucs1, len);
+        break;
+    }
+    case PyUnicode_2BYTE_KIND: {
+        Py_UCS2 *ucs2 = (Py_UCS2 *)writer->data + index;
+        const Py_UCS1 *end = ucs1 + len;
+        for (; ucs1 < end; ++ucs2, ++ucs1) {
+            *ucs2 = (Py_UCS2)*ucs1;
+        }
+        break;
+    }
+    case PyUnicode_4BYTE_KIND: {
+        Py_UCS4 *ucs4 = (Py_UCS4 *)writer->data + index;
+        const Py_UCS1 *end = ucs1 + len;
+        for (; ucs1 < end; ++ucs4, ++ucs1) {
+            *ucs4 = (Py_UCS4)*ucs1;
+        }
+        break;
+    }
+    default:
+        Py_UNREACHABLE();
+    }
+
     writer->pos += len;
     return 0;
 }

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to