https://github.com/python/cpython/commit/7352b6aaa99ed68b055cb3c063657a2e3ba82190
commit: 7352b6aaa99ed68b055cb3c063657a2e3ba82190
branch: main
author: Vyron Vasileiadis <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-28T11:08:26+02:00
summary:
gh-158213: Don't return a str subclass from PyUnicodeWriter_Finish() (#158214)
Since GH-157861, every writer uses the read-only optimization of
_PyUnicodeWriter_WriteStr(), so the first write of a str subclass
instance into an empty writer kept that object as the buffer, and
PyUnicodeWriter_Finish() returned it. io.StringIO.getvalue() then
returned the written object itself, and the next write re-read it
through its __str__() method, which changed the contents and could make
read() read past the end of the buffer.
Only use the read-only optimization for exact str objects. A subclass
is copied into a new buffer, as before GH-157861.
files:
M Lib/test/test_capi/test_unicode.py
M Lib/test/test_io/test_memoryio.py
M Objects/unicode_writer.c
diff --git a/Lib/test/test_capi/test_unicode.py
b/Lib/test/test_capi/test_unicode.py
index f4bd961017b0ede..032b910a280083b 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -1910,6 +1910,18 @@ def test_create(self):
self.assertGreater(writer.get_buffer()[0], len(s))
self.assertEqual(writer.finish(), s)
+ def test_str_subclass(self):
+ # The read-only optimization must not return a str subclass
+ class MyStr(str):
+ def __str__(self):
+ return self
+
+ writer = self.create_writer(0)
+ writer.write_str(MyStr('abc'))
+ result = writer.finish()
+ self.assertEqual(result, 'abc')
+ self.assertIs(type(result), str)
+
def test_repr_null(self):
writer = self.create_writer(0)
writer.write_utf8(b'var=', -1)
diff --git a/Lib/test/test_io/test_memoryio.py
b/Lib/test/test_io/test_memoryio.py
index b378505aa8f7db2..b6f3aa93e7aa43f 100644
--- a/Lib/test/test_io/test_memoryio.py
+++ b/Lib/test/test_io/test_memoryio.py
@@ -1118,7 +1118,11 @@ def __str__(self):
s = MyStr("correct")
memio = self.ioclass()
memio.write(s)
- self.assertEqual(memio.getvalue(), "correct")
+ value = memio.getvalue()
+ self.assertEqual(value, "correct")
+ self.assertIs(type(value), str)
+ memio.write("!")
+ self.assertEqual(memio.getvalue(), "correct!")
# Also test the fast path where pos == string_size (STATE_ACCUMULATING)
memio2 = self.ioclass()
diff --git a/Objects/unicode_writer.c b/Objects/unicode_writer.c
index d6564ce84ed54eb..c1a2af4d9ac1fe9 100644
--- a/Objects/unicode_writer.c
+++ b/Objects/unicode_writer.c
@@ -313,7 +313,7 @@ _PyUnicodeWriter_WriteStr(_PyUnicodeWriter *writer,
PyObject *str)
Py_UCS4 maxchar = PyUnicode_MAX_CHAR_VALUE(str);
if (maxchar > writer->maxchar || len > writer->size - writer->pos) {
- if (writer->buffer == NULL) {
+ if (writer->buffer == NULL && PyUnicode_CheckExact(str)) {
assert(_PyUnicode_CheckConsistency(str, 1));
writer->readonly = 1;
writer->buffer = Py_NewRef(str);
_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]