https://github.com/python/cpython/commit/eb30e3d9d4b523e3f8de2cd1485c9ae9a996ab03
commit: eb30e3d9d4b523e3f8de2cd1485c9ae9a996ab03
branch: main
author: Victor Stinner <[email protected]>
committer: vstinner <[email protected]>
date: 2026-10-01T12:25:38Z
summary:

Optimize PyFloat_Pack*() functions (#158483)

Use _Py_bswap16/32/64() functions to swap bytes. PyFloat_Pack4() and
PyFloat_Pack8() avoid temporary buffer copy in the native byte order.

files:
M Objects/floatobject.c

diff --git a/Objects/floatobject.c b/Objects/floatobject.c
index 1bb5311ac678a5..d3ae337236a693 100644
--- a/Objects/floatobject.c
+++ b/Objects/floatobject.c
@@ -5,6 +5,7 @@
 
 #include "Python.h"
 #include "pycore_abstract.h"      // _PyNumber_Index()
+#include "pycore_bitutils.h"      // _Py_bswap16()
 #include "pycore_dtoa.h"          // _Py_dg_dtoa()
 #include "pycore_floatobject.h"   // _PyFloat_FormatAdvancedWriter()
 #include "pycore_freelist.h"      // _Py_FREELIST_FREE(), _Py_FREELIST_POP()
@@ -1894,7 +1895,6 @@ _PyFloat_DebugMallocStats(FILE *out)
 int
 PyFloat_Pack2(double x, char *data, int le)
 {
-    unsigned char *p = (unsigned char *)data;
 #if _Py_HAVE_FLOAT16
     /* Conversion can change NaNs type or alter payload.  Here we
        just fallback to the generic code, instead of providing
@@ -1906,16 +1906,14 @@ PyFloat_Pack2(double x, char *data, int le)
             goto Overflow;
         }
 
-        unsigned char s[sizeof(_Float16)];
-
-        memcpy(s, &y, sizeof(_Float16));
         if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) {
-            p[1] = s[0];
-            p[0] = s[1];
+            uint16_t word;
+            memcpy(&word, &y, 2);
+            word = _Py_bswap16(word);
+            memcpy(data, &word, 2);
         }
         else {
-            p[0] = s[0];
-            p[1] = s[1];
+            memcpy(data, &y, sizeof(_Float16));
         }
         return 0;
     }
@@ -1923,8 +1921,7 @@ PyFloat_Pack2(double x, char *data, int le)
     unsigned char sign;
     int e;
     double f;
-    unsigned short bits;
-    int incr = 1;
+    uint16_t bits;
 
     if (x == 0.0) {
         sign = (copysign(1.0, x) == -1.0);
@@ -2004,18 +2001,10 @@ PyFloat_Pack2(double x, char *data, int le)
     bits |= (e << 10) | (sign << 15);
 
     /* Write out result. */
-    if (le) {
-        p += 1;
-        incr = -1;
+    if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) {
+        bits = _Py_bswap16(bits);
     }
-
-    /* First byte */
-    *p = (unsigned char)((bits >> 8) & 0xFF);
-    p += incr;
-
-    /* Second byte */
-    *p = (unsigned char)(bits & 0xFF);
-
+    memcpy(data, &bits, 2);
     return 0;
 
   Overflow:
@@ -2027,10 +2016,7 @@ PyFloat_Pack2(double x, char *data, int le)
 int
 PyFloat_Pack4(double x, char *data, int le)
 {
-    unsigned char *p = (unsigned char *)data;
     float y = (float)x;
-    int i, incr = 1;
-
     if (isinf(y) && !isinf(x)) {
         PyErr_SetString(PyExc_OverflowError,
                         "float too large to pack with f format");
@@ -2054,8 +2040,8 @@ PyFloat_Pack4(double x, char *data, int le)
         }
 #else
         uint32_t u32;
-
         memcpy(&u32, &y, 4);
+
         /* Workaround RISC-V: "If a NaN value is converted to a
          * different floating-point type, the result is the
          * canonical NaN of the new type".  The canonical NaN here
@@ -2075,17 +2061,14 @@ PyFloat_Pack4(double x, char *data, int le)
 #endif
     }
 
-    unsigned char s[sizeof(float)];
-    memcpy(s, &y, sizeof(float));
-
     if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) {
-        p += 3;
-        incr = -1;
+        uint32_t word;
+        memcpy(&word, &y, 4);
+        word = _Py_bswap32(word);
+        memcpy(data, &word, 4);
     }
-
-    for (i = 0; i < 4; i++) {
-        *p = s[i];
-        p += incr;
+    else {
+        memcpy(data, &y, sizeof(float));
     }
     return 0;
 }
@@ -2093,20 +2076,14 @@ PyFloat_Pack4(double x, char *data, int le)
 int
 PyFloat_Pack8(double x, char *data, int le)
 {
-    unsigned char *p = (unsigned char *)data;
-    unsigned char as_bytes[8];
-    memcpy(as_bytes, &x, 8);
-    const unsigned char *s = as_bytes;
-    int i, incr = 1;
-
     if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) {
-        p += 7;
-        incr = -1;
+        uint64_t word;
+        memcpy(&word, &x, 8);
+        word = _Py_bswap64(word);
+        memcpy(data, &word, 8);
     }
-
-    for (i = 0; i < 8; i++) {
-        *p = *s++;
-        p += incr;
+    else {
+        memcpy(data, &x, 8);
     }
     return 0;
 }

_______________________________________________
Python-checkins mailing list -- [email protected]
To unsubscribe send an email to [email protected]
https://mail.python.org/mailman3//lists/python-checkins.python.org
Member address: [email protected]

Reply via email to