https://github.com/python/cpython/commit/26d3f5dd6a9c6699f430138db0e92d224f958747
commit: 26d3f5dd6a9c6699f430138db0e92d224f958747
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T03:53:38+02:00
summary:
gh-158451: Add _PyUnicodeWriter_WriteLatin1String() test (#158464)
Fix _PyUnicodeWriter_WriteLatin1String() when the writer buffer kind
is UCS-2 or UCS-4.
files:
A Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
M Lib/test/test_capi/test_unicode.py
M Modules/_testcapi/unicode.c
M Objects/unicode_writer.c
diff --git a/Lib/test/test_capi/test_unicode.py
b/Lib/test/test_capi/test_unicode.py
index 36ecca48230ddd..c511ffc0fda13c 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -1970,6 +1970,24 @@ def test_ascii(self):
writer.write_ascii(b"Python! <truncated>", 6)
self.assertEqual(writer.finish(), "Hello Python")
+ def test_write_latin1(self):
+ # Test _PyUnicodeWriter_WriteLatin1String()
+ writer = self.create_writer(0)
+ # Start with ASCII buffer
+ writer.write_latin1(b"abc IGNORED", 3)
+ writer.write_latin1(b"IGNORED", 0)
+ # Change buffer kind to UCS-1
+ writer.write_latin1(b"\xe9", 1)
+ # Change buffer kind to UCS-2
+ writer.write_str('[\u20ac]')
+ writer.write_latin1(b"def\xa0", 4)
+ # Change buffer kind to UCS-4
+ writer.write_str('[\U0010ffff]')
+ writer.write_latin1(b"ghi\xff.", 5)
+ writer.write_latin1(b"IGNORED", 0)
+ self.assertEqual(writer.finish(),
+ "abc\xe9[\u20ac]def\xa0[\U0010ffff]ghi\xff.")
+
def test_invalid_utf8(self):
writer = self.create_writer(0)
with self.assertRaises(UnicodeDecodeError):
diff --git
a/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
b/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
new file mode 100644
index 00000000000000..a16dd024287076
--- /dev/null
+++ b/Misc/NEWS.d/next/C_API/2026-09-30-03-19-34.gh-issue-158451.pUb2BJ.rst
@@ -0,0 +1,2 @@
+Fix :c:func:`!_PyUnicodeWriter_WriteLatin1String` when the writer buffer
+kind is UCS-2 or UCS-4. Patch by Victor Stinner.
diff --git a/Modules/_testcapi/unicode.c b/Modules/_testcapi/unicode.c
index c62813f4f3768d..ba9b205e07ef8b 100644
--- a/Modules/_testcapi/unicode.c
+++ b/Modules/_testcapi/unicode.c
@@ -685,6 +685,32 @@ writer_write_substring(PyObject *self_raw, PyObject *args)
}
+static PyObject*
+writer_write_latin1(PyObject *self_raw, PyObject *args)
+{
+ WriterObject *self = (WriterObject *)self_raw;
+ if (writer_check(self) < 0) {
+ return NULL;
+ }
+
+ const char *str;
+ Py_ssize_t bsize, size;
+ if (!PyArg_ParseTuple(args, "z#n", &str, &bsize, &size)) {
+ return NULL;
+ }
+
+ _PyUnicodeWriter *writer = (_PyUnicodeWriter*)self->writer;
+_Py_COMP_DIAG_PUSH
+_Py_COMP_DIAG_IGNORE_DEPR_DECLS
+ if (_PyUnicodeWriter_WriteLatin1String(writer, str, size) < 0) {
+ return NULL;
+ }
+_Py_COMP_DIAG_POP
+
+ Py_RETURN_NONE;
+}
+
+
static PyObject*
writer_decodeutf8stateful(PyObject *self_raw, PyObject *args)
{
@@ -778,6 +804,7 @@ static PyMethodDef writer_methods[] = {
{"write_str", _PyCFunction_CAST(writer_write_str), METH_O},
{"write_repr", _PyCFunction_CAST(writer_write_repr), METH_O},
{"write_substring", _PyCFunction_CAST(writer_write_substring),
METH_VARARGS},
+ {"write_latin1", _PyCFunction_CAST(writer_write_latin1), METH_VARARGS},
{"decodeutf8stateful", _PyCFunction_CAST(writer_decodeutf8stateful),
METH_VARARGS},
{"get_pointer", _PyCFunction_CAST(writer_get_pointer), METH_VARARGS},
{"get_buffer", _PyCFunction_CAST(writer_get_buffer), METH_VARARGS},
diff --git a/Objects/unicode_writer.c b/Objects/unicode_writer.c
index 8637be921e2454..0949e45d51cbad 100644
--- a/Objects/unicode_writer.c
+++ b/Objects/unicode_writer.c
@@ -62,59 +62,6 @@ OF OR IN CONNECTION WITH THE USE OR PERFORMANCE OF THIS
SOFTWARE.
#include "stringlib/undef.h"
-/* Copy an ASCII or latin1 char* string into a Python Unicode string.
-
- WARNING: The function doesn't copy the terminating null character and
- doesn't check the maximum character (may write a latin1 character in an
- ASCII string). */
-static void
-unicode_write_cstr(PyObject *unicode, Py_ssize_t index,
- const char *str, Py_ssize_t len)
-{
- int kind = PyUnicode_KIND(unicode);
- const void *data = PyUnicode_DATA(unicode);
- const char *end = str + len;
-
- assert(index + len <= PyUnicode_GET_LENGTH(unicode));
- switch (kind) {
- case PyUnicode_1BYTE_KIND: {
-#ifdef Py_DEBUG
- if (PyUnicode_IS_ASCII(unicode)) {
- Py_UCS4 maxchar = ucs1lib_find_max_char(
- (const Py_UCS1*)str,
- (const Py_UCS1*)str + len);
- assert(maxchar < 128);
- }
-#endif
- memcpy((char *) data + index, str, len);
- break;
- }
- case PyUnicode_2BYTE_KIND: {
- Py_UCS2 *start = (Py_UCS2 *)data + index;
- Py_UCS2 *ucs2 = start;
-
- for (; str < end; ++ucs2, ++str)
- *ucs2 = (Py_UCS2)*str;
-
- assert((ucs2 - start) <= PyUnicode_GET_LENGTH(unicode));
- break;
- }
- case PyUnicode_4BYTE_KIND: {
- Py_UCS4 *start = (Py_UCS4 *)data + index;
- Py_UCS4 *ucs4 = start;
-
- for (; str < end; ++ucs4, ++str)
- *ucs4 = (Py_UCS4)*str;
-
- assert((ucs4 - start) <= PyUnicode_GET_LENGTH(unicode));
- break;
- }
- default:
- Py_UNREACHABLE();
- }
-}
-
-
void
_PyUnicodeWriter_Init(_PyUnicodeWriter *writer)
{
@@ -550,13 +497,43 @@ int
_PyUnicodeWriter_WriteLatin1String(_PyUnicodeWriter *writer,
const char *str, Py_ssize_t len)
{
- Py_UCS4 maxchar;
+ if (len == 0) {
+ return 0;
+ }
- maxchar = ucs1lib_find_max_char((const Py_UCS1*)str, (const Py_UCS1*)str +
len);
- if (_PyUnicodeWriter_Prepare(writer, len, maxchar) == -1)
+ const Py_UCS1 *ucs1 = (const Py_UCS1 *)str;
+ Py_UCS4 maxchar = ucs1lib_find_max_char(ucs1, ucs1 + len);
+ if (_PyUnicodeWriter_Prepare(writer, len, maxchar) < 0) {
return -1;
+ }
assert(_PyUnicodeWriter_CanWrite(writer));
- unicode_write_cstr(writer->buffer, writer->pos, str, len);
+
+ Py_ssize_t index = writer->pos;
+ switch (writer->kind) {
+ case PyUnicode_1BYTE_KIND: {
+ memcpy((Py_UCS1 *)writer->data + index, ucs1, len);
+ break;
+ }
+ case PyUnicode_2BYTE_KIND: {
+ Py_UCS2 *ucs2 = (Py_UCS2 *)writer->data + index;
+ const Py_UCS1 *end = ucs1 + len;
+ for (; ucs1 < end; ++ucs2, ++ucs1) {
+ *ucs2 = (Py_UCS2)*ucs1;
+ }
+ break;
+ }
+ case PyUnicode_4BYTE_KIND: {
+ Py_UCS4 *ucs4 = (Py_UCS4 *)writer->data + index;
+ const Py_UCS1 *end = ucs1 + len;
+ for (; ucs1 < end; ++ucs4, ++ucs1) {
+ *ucs4 = (Py_UCS4)*ucs1;
+ }
+ break;
+ }
+ default:
+ Py_UNREACHABLE();
+ }
+
writer->pos += len;
return 0;
}
_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]