https://github.com/python/cpython/commit/70e6f3cfe1da77859f1d497f6354b8a87048ed65
commit: 70e6f3cfe1da77859f1d497f6354b8a87048ed65
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-09-30T05:31:36+02:00
summary:
gh-158451: Convert _PyUnicodeWriter_Prepare() to static inline function
(#158466)
Add assertion to _PyUnicodeWriter_Prepare() to check that length is
not negative.
Add some assertions to check that _PyUnicodeWriter_Prepare() is
called with len >= 1.
files:
M Include/cpython/unicodeobject.h
M Include/internal/pycore_unicodeobject.h
M Lib/test/test_capi/test_unicode.py
M Objects/longobject.c
M Objects/unicode_formatter.c
M Objects/unicode_writer.c
M Objects/unicodeobject.c
diff --git a/Include/cpython/unicodeobject.h b/Include/cpython/unicodeobject.h
index 3da18a6ad35db3..54defdb2061aa9 100644
--- a/Include/cpython/unicodeobject.h
+++ b/Include/cpython/unicodeobject.h
@@ -557,25 +557,33 @@ typedef struct {
_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(void) _PyUnicodeWriter_Init(
_PyUnicodeWriter *writer);
-/* Prepare the buffer to write 'length' characters
- with the specified maximum character.
-
- Return 0 on success, raise an exception and return -1 on error. */
-#define _PyUnicodeWriter_Prepare(WRITER, LENGTH, MAXCHAR) \
- (((MAXCHAR) <= (WRITER)->maxchar \
- && (LENGTH) <= (WRITER)->size - (WRITER)->pos) \
- ? 0 \
- : (((LENGTH) == 0) \
- ? 0 \
- : _PyUnicodeWriter_PrepareInternal((WRITER), (LENGTH), (MAXCHAR))))
-
-/* Don't call this function directly, use the _PyUnicodeWriter_Prepare() macro
- instead. */
+// Don't call this function directly, use _PyUnicodeWriter_Prepare() instead.
_Py_DEPRECATED_EXTERNALLY(3.14) PyAPI_FUNC(int)
_PyUnicodeWriter_PrepareInternal(
_PyUnicodeWriter *writer,
Py_ssize_t length,
Py_UCS4 maxchar);
+// Prepare the buffer to write 'length' characters
+// with the specified maximum character.
+//
+// Return 0 on success. Set an exception and return -1 on error.
+_Py_DEPRECATED_EXTERNALLY(3.14) static inline int
+_PyUnicodeWriter_Prepare(_PyUnicodeWriter *writer,
+ Py_ssize_t length, Py_UCS4 maxchar)
+{
+ assert(0 <= length);
+ if (maxchar <= writer->maxchar && length <= (writer->size - writer->pos)) {
+ return 0;
+ }
+ if (length == 0) {
+ return 0;
+ }
+_Py_COMP_DIAG_PUSH
+_Py_COMP_DIAG_IGNORE_DEPR_DECLS
+ return _PyUnicodeWriter_PrepareInternal(writer, length, maxchar);
+_Py_COMP_DIAG_POP
+}
+
/* Prepare the buffer to have at least the kind KIND.
For example, kind=PyUnicode_2BYTE_KIND ensures that the writer will
support characters in range U+000-U+FFFF.
diff --git a/Include/internal/pycore_unicodeobject.h
b/Include/internal/pycore_unicodeobject.h
index 1ee4fc4c4517ff..3b930e857e3704 100644
--- a/Include/internal/pycore_unicodeobject.h
+++ b/Include/internal/pycore_unicodeobject.h
@@ -168,15 +168,13 @@ _PyUnicodeWriter_SetReadOnly(_PyUnicodeWriter *writer,
PyObject *obj,
static inline int
_PyUnicodeWriter_WriteCharInline(_PyUnicodeWriter *writer, Py_UCS4 ch)
{
- if (ch > writer->maxchar || 1 > writer->size - writer->pos) {
+ if (ch > writer->maxchar || 1 > (writer->size - writer->pos)) {
if (writer->buffer == NULL && ch <= 255) {
// If the first write is a Latin1 character, use the singleton
// as a read-only object
PyObject *obj = _Py_LATIN1_CHR(ch);
- // Py_NewRef() is not need on immortal object
+ // Py_NewRef() is not needed on immortal object
_PyUnicodeWriter_SetReadOnly(writer, obj, 1);
-
- // The next write will create a new buffer and copy the string
return 0;
}
diff --git a/Lib/test/test_capi/test_unicode.py
b/Lib/test/test_capi/test_unicode.py
index c511ffc0fda13c..6965b9dc11b57d 100644
--- a/Lib/test/test_capi/test_unicode.py
+++ b/Lib/test/test_capi/test_unicode.py
@@ -461,6 +461,14 @@ def check_format(expected, format, *args):
check_format('%abc',
b'%%%s', b'abc')
+ # test "%s" with empty string
+ check_format('x=',
+ b'x=%s', b'')
+ check_format('x=',
+ b'x=%0s', b'')
+ check_format('x=',
+ b'x=%.3s', b'')
+
# truncated string
check_format('abc',
b'%.3s', b'abcdef')
diff --git a/Objects/longobject.c b/Objects/longobject.c
index b9f00ca6fb471b..9577c8ef4d6ed0 100644
--- a/Objects/longobject.c
+++ b/Objects/longobject.c
@@ -2212,7 +2212,7 @@ long_to_decimal_string_internal(PyObject *aa,
}
}
if (writer) {
- if (_PyUnicodeWriter_Prepare(writer, strlen, '9') == -1) {
+ if (_PyUnicodeWriter_Prepare(writer, strlen, 127) == -1) {
Py_DECREF(scratch);
return -1;
}
@@ -2227,7 +2227,7 @@ long_to_decimal_string_internal(PyObject *aa,
}
}
else {
- str = PyUnicode_New(strlen, '9');
+ str = PyUnicode_New(strlen, 127);
if (str == NULL) {
Py_DECREF(scratch);
return -1;
@@ -2385,9 +2385,10 @@ long_format_binary(PyObject *aa, int base, int alternate,
/* 2 characters for prefix */
sz += 2;
}
+ assert(sz >= 1);
if (writer) {
- if (_PyUnicodeWriter_Prepare(writer, sz, 'x') == -1) {
+ if (_PyUnicodeWriter_Prepare(writer, sz, 127) == -1) {
return -1;
}
assert(_PyUnicodeWriter_CanWrite(writer));
diff --git a/Objects/unicode_formatter.c b/Objects/unicode_formatter.c
index 2b7681f7ff3ce9..8cbc774584fa68 100644
--- a/Objects/unicode_formatter.c
+++ b/Objects/unicode_formatter.c
@@ -1349,6 +1349,7 @@ format_long_internal(PyObject *value, const
InternalFormatSpec *format,
if (n_total == -1) {
goto done;
}
+ assert(n_total >= 1);
/* Allocate the memory. */
if (_PyUnicodeWriter_Prepare(writer, n_total, maxchar) == -1)
@@ -1503,6 +1504,7 @@ format_float_internal(PyObject *value,
if (n_total == -1) {
goto done;
}
+ assert(n_total >= 1);
/* Allocate the memory. */
if (_PyUnicodeWriter_Prepare(writer, n_total, maxchar) == -1)
@@ -1714,6 +1716,7 @@ format_complex_internal(PyObject *value,
/* Add 1 for the 'j', and optionally 2 for parens. */
calc_padding(n_re_total + n_im_total + 1 + add_parens * 2,
format->width, format->align, &lpad, &rpad, &total);
+ assert(total >= 1);
if (lpad || rpad)
maxchar = Py_MAX(maxchar, format->fill_char);
diff --git a/Objects/unicode_writer.c b/Objects/unicode_writer.c
index be3cfb2538b780..6d921ed24de298 100644
--- a/Objects/unicode_writer.c
+++ b/Objects/unicode_writer.c
@@ -138,9 +138,10 @@ _PyUnicodeWriter_PrepareInternal(_PyUnicodeWriter *writer,
assert(length >= 0);
assert(maxchar <= _Py_MAX_UNICODE);
- /* ensure that the _PyUnicodeWriter_Prepare macro was used */
- assert((maxchar > writer->maxchar && length >= 0)
- || length > 0);
+ // Check that _PyUnicodeWriter_Prepare() or _PyUnicodeWriter_PrepareKind()
+ // was used
+ assert(maxchar > writer->maxchar
+ || (length > (writer->size - writer->pos) && length >= 1));
if (length > PY_SSIZE_T_MAX - writer->pos) {
PyErr_NoMemory();
@@ -336,11 +337,12 @@ _PyUnicodeWriter_WriteSubstring(_PyUnicodeWriter *writer,
PyObject *str,
Py_ssize_t start, Py_ssize_t end)
{
assert(0 <= start);
- assert(end <= PyUnicode_GET_LENGTH(str));
assert(start <= end);
+ assert(end <= PyUnicode_GET_LENGTH(str));
- if (start == 0 && end == PyUnicode_GET_LENGTH(str))
+ if (start == 0 && end == PyUnicode_GET_LENGTH(str)) {
return _PyUnicodeWriter_WriteStr(writer, str);
+ }
Py_ssize_t len = end - start;
if (len == 0) {
diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c
index 43b720e7cde218..68ce13afd9bf37 100644
--- a/Objects/unicodeobject.c
+++ b/Objects/unicodeobject.c
@@ -2545,9 +2545,9 @@ unicode_fromformat_write_str(_PyUnicodeWriter *writer,
PyObject *str,
Py_UCS4 maxchar;
length = PyUnicode_GET_LENGTH(str);
- if ((precision == -1 || precision >= length)
- && width <= length)
+ if ((precision == -1 || precision >= length) && width <= length) {
return _PyUnicodeWriter_WriteStr(writer, str);
+ }
if (precision != -1)
length = Py_MIN(precision, length);
@@ -2837,7 +2837,7 @@ unicode_fromformat_arg(_PyUnicodeWriter *writer,
#undef SPRINT
#undef DO_SPRINTS
- assert(len >= 0);
+ assert(len >= 1);
int sign = (buffer[0] == '-');
len -= sign;
_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]