https://github.com/python/cpython/commit/00307b0e69341ed3325012e6a83a810c397f3649
commit: 00307b0e69341ed3325012e6a83a810c397f3649
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T02:47:46+02:00
summary:
gh-158451: Add _PyUnicodeWriter_SetBuffer() helper function (#158452)
Replace _PyUnicodeWriter_Update() with _PyUnicodeWriter_SetBuffer().
Add _PyUnicodeWriter_SetReadOnly() helper function.
Add _PyUnicodeWriter_FinishWithSize() for PyUnicode_DecodeUTF7Stateful().
files:
M Include/internal/pycore_unicodeobject.h
M Objects/unicode_writer.c
M Objects/unicodeobject.c
diff --git a/Include/internal/pycore_unicodeobject.h
b/Include/internal/pycore_unicodeobject.h
index c091aa94371a75..1ee4fc4c4517ff 100644
--- a/Include/internal/pycore_unicodeobject.h
+++ b/Include/internal/pycore_unicodeobject.h
@@ -130,22 +130,39 @@ _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer)
#endif
static inline void
-_PyUnicodeWriter_Update(_PyUnicodeWriter *writer)
+_PyUnicodeWriter_SetBuffer(_PyUnicodeWriter *writer, PyObject *buffer)
{
- PyObject *buffer = writer->buffer;
- writer->maxchar = PyUnicode_MAX_CHAR_VALUE(buffer);
+ assert(writer->pos <= PyUnicode_GET_LENGTH(buffer));
+
+ // Py_DECREF() the previous buffer (if any)
+ Py_XSETREF(writer->buffer, buffer);
writer->data = PyUnicode_DATA(buffer);
writer->kind = PyUnicode_KIND(buffer);
+ writer->maxchar = PyUnicode_MAX_CHAR_VALUE(buffer);
+ writer->size = PyUnicode_GET_LENGTH(buffer);
+ writer->readonly = 0;
+}
- if (!writer->readonly) {
- writer->size = PyUnicode_GET_LENGTH(buffer);
- }
- else {
- /* Copy-on-write mode: set buffer size to 0 so
- * _PyUnicodeWriter_Prepare() will copy (and enlarge) the buffer on
- * next write. */
- writer->size = 0;
- }
+static inline void
+_PyUnicodeWriter_SetReadOnly(_PyUnicodeWriter *writer, PyObject *obj,
+ Py_ssize_t length)
+{
+ assert(writer->buffer == NULL);
+ assert(writer->pos == 0);
+ // Micro-optimization: pass length as a parameter, as it's usually known
+ // by the caller
+ assert(length == PyUnicode_GET_LENGTH(obj));
+
+ writer->buffer = obj;
+ writer->data = NULL;
+ /* Set kind and size to 0 to make sure that the next
+ * _PyUnicodeWriter_Prepare() call allocates a new buffer and copies
+ * characters. */
+ writer->kind = 0;
+ writer->maxchar = PyUnicode_MAX_CHAR_VALUE(obj);
+ writer->size = 0;
+ writer->pos = length;
+ writer->readonly = 1;
}
static inline int
@@ -156,11 +173,9 @@ _PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer,
Py_UCS4 ch)
// If the first write is a Latin1 character, use the singleton
// as a read-only object
PyObject *obj = _Py_LATIN1_CHR(ch);
- writer->readonly = 1;
- writer->buffer = obj; // Py_NewRef() is not need on immortal object
- _PyUnicodeWriter_Update(writer);
- assert(writer->pos == 0);
- writer->pos = 1;
+ // Py_NewRef() is not need on immortal object
+ _PyUnicodeWriter_SetReadOnly(writer, obj, 1);
+
// The next write will create a new buffer and copy the string
return 0;
}
@@ -176,6 +191,10 @@ _PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer,
Py_UCS4 ch)
return 0;
}
+extern PyObject* _PyUnicodeWriter_FinishWithSize(
+ _PyUnicodeWriter *writer,
+ Py_ssize_t size);
+
/* --- Unicode API -------------------------------------------------------- */
// Export for '_json' shared extension
diff --git a/Objects/unicode_writer.c b/Objects/unicode_writer.c
index 26deffa6baac63..8637be921e2454 100644
--- a/Objects/unicode_writer.c
+++ b/Objects/unicode_writer.c
@@ -178,8 +178,7 @@ _PyUnicodeWriter_InitWithBuffer(_PyUnicodeWriter *writer,
PyObject *buffer)
assert(PyUnstable_Object_IsUniquelyReferenced(buffer));
memset(writer, 0, sizeof(*writer));
- writer->buffer = buffer;
- _PyUnicodeWriter_Update(writer);
+ _PyUnicodeWriter_SetBuffer(writer, buffer);
writer->min_length = writer->size;
assert(_PyUnicodeWriter_CanWrite(writer));
}
@@ -204,16 +203,19 @@ _PyUnicodeWriter_PrepareInternal(_PyUnicodeWriter *writer,
maxchar = Py_MAX(maxchar, writer->min_char);
- PyObject *newbuffer;
+ PyObject *new_buffer;
if (writer->buffer == NULL) {
assert(!writer->readonly);
+
// Do not overallocate at the first allocation, but use min_length
- if (alloc < writer->min_length)
+ if (alloc < writer->min_length) {
alloc = writer->min_length;
+ }
- writer->buffer = PyUnicode_New(alloc, maxchar);
- if (writer->buffer == NULL)
+ new_buffer = PyUnicode_New(alloc, maxchar);
+ if (new_buffer == NULL) {
return -1;
+ }
}
else if (alloc > writer->size) {
// Do not overallocate at the first allocation, but use min_length
@@ -223,38 +225,42 @@ _PyUnicodeWriter_PrepareInternal(_PyUnicodeWriter *writer,
/* overallocate to limit the number of realloc() */
alloc += alloc / OVERALLOCATE_FACTOR;
}
- if (alloc < writer->min_length)
+ if (alloc < writer->min_length) {
alloc = writer->min_length;
+ }
if (maxchar > writer->maxchar || writer->readonly) {
/* resize + widen */
maxchar = Py_MAX(maxchar, writer->maxchar);
- newbuffer = PyUnicode_New(alloc, maxchar);
- if (newbuffer == NULL)
+ new_buffer = PyUnicode_New(alloc, maxchar);
+ if (new_buffer == NULL) {
return -1;
- _PyUnicode_FastCopyCharacters(newbuffer, 0,
+ }
+ _PyUnicode_FastCopyCharacters(new_buffer, 0,
writer->buffer, 0, writer->pos);
- writer->readonly = 0;
- Py_DECREF(writer->buffer);
- writer->buffer = newbuffer;
}
else {
- newbuffer = _PyUnicode_ResizeCompact(writer->buffer, alloc);
- if (newbuffer == NULL)
+ new_buffer = _PyUnicode_ResizeCompact(writer->buffer, alloc);
+ if (new_buffer == NULL) {
return -1;
- writer->buffer = newbuffer;
+ }
+ // Do not DECREF the old buffer
+ writer->buffer = NULL;
}
}
- else if (maxchar > writer->maxchar) {
+ else {
+ assert(maxchar > writer->maxchar);
assert(!writer->readonly);
- newbuffer = PyUnicode_New(writer->size, maxchar);
- if (newbuffer == NULL)
+
+ new_buffer = PyUnicode_New(writer->size, maxchar);
+ if (new_buffer == NULL) {
return -1;
- _PyUnicode_FastCopyCharacters(newbuffer, 0,
+ }
+ _PyUnicode_FastCopyCharacters(new_buffer, 0,
writer->buffer, 0, writer->pos);
- Py_SETREF(writer->buffer, newbuffer);
}
- _PyUnicodeWriter_Update(writer);
+
+ _PyUnicodeWriter_SetBuffer(writer, new_buffer);
return 0;
#undef OVERALLOCATE_FACTOR
@@ -316,11 +322,7 @@ _PyUnicodeWriter_WriteStr(_PyUnicodeWriter *writer,
PyObject *str)
if (maxchar > writer->maxchar || len > writer->size - writer->pos) {
if (writer->buffer == NULL && PyUnicode_CheckExact(str)) {
assert(_PyUnicode_CheckConsistency(str, 1));
- writer->readonly = 1;
- writer->buffer = Py_NewRef(str);
- _PyUnicodeWriter_Update(writer);
- writer->pos += len;
- // The next write will create a new buffer and copy the string
+ _PyUnicodeWriter_SetReadOnly(writer, Py_NewRef(str), len);
return 0;
}
if (_PyUnicodeWriter_PrepareInternal(writer, len, maxchar) == -1)
@@ -457,10 +459,7 @@ _PyUnicodeWriter_WriteASCIIString(_PyUnicodeWriter *writer,
if (str == NULL)
return -1;
- writer->readonly = 1;
- writer->buffer = str;
- _PyUnicodeWriter_Update(writer);
- writer->pos += len;
+ _PyUnicodeWriter_SetReadOnly(writer, str, len);
return 0;
}
@@ -639,6 +638,26 @@ _PyUnicodeWriter_Finish(_PyUnicodeWriter *writer)
}
+PyObject *
+_PyUnicodeWriter_FinishWithSize(_PyUnicodeWriter *writer, Py_ssize_t size)
+{
+ assert(0 <= size);
+ if (writer->buffer != NULL) {
+ assert(size <= writer->pos);
+ assert(size <= PyUnicode_GET_LENGTH(writer->buffer));
+ if (size < writer->pos) {
+ // Truncate the string: we may need to adjust the string kind
+ writer->recheck_maxchar = 1;
+ }
+ }
+ else {
+ assert(size == 0);
+ }
+ writer->pos = size;
+ return _PyUnicodeWriter_Finish(writer);
+}
+
+
PyObject*
PyUnicodeWriter_Finish(PyUnicodeWriter *writer)
{
diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c
index 893621f041c9ad..43b720e7cde218 100644
--- a/Objects/unicodeobject.c
+++ b/Objects/unicodeobject.c
@@ -4761,15 +4761,10 @@ PyUnicode_DecodeUTF7Stateful(const char *s,
if (consumed) {
if (inShift) {
*consumed = startinpos;
- if (writer.pos != shiftOutStart && writer.maxchar > 127) {
- PyObject *result = PyUnicode_FromKindAndData(
- writer.kind, writer.data, shiftOutStart);
- Py_XDECREF(errorHandler);
- Py_XDECREF(exc);
- _PyUnicodeWriter_Dealloc(&writer);
- return result;
- }
- writer.pos = shiftOutStart; /* back off output */
+
+ Py_XDECREF(errorHandler);
+ Py_XDECREF(exc);
+ return _PyUnicodeWriter_FinishWithSize(&writer, shiftOutStart);
}
else {
*consumed = s-starts;
_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]