From aa273f8bf520b3bd3ec36070400ca5a8b0a9ba02 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Fri, 25 Sep 2026 23:32:50 +0200 Subject: [PATCH 1/2] Optimize PyFloat_Pack*() functions PyFloat_Pack4() and PyFloat_Pack8() avoid temporary buffer in the native byte order. Use _Py_bswapXX() functions to reverse bytes. --- Objects/floatobject.c | 70 ++++++++++++++++--------------------------- 1 file changed, 26 insertions(+), 44 deletions(-) diff --git a/Objects/floatobject.c b/Objects/floatobject.c index 1bb5311ac678a5..8d18cbf5920b1b 100644 --- a/Objects/floatobject.c +++ b/Objects/floatobject.c @@ -5,6 +5,7 @@ #include "Python.h" #include "pycore_abstract.h" // _PyNumber_Index() +#include "pycore_bitutils.h" // _Py_bswap16() #include "pycore_dtoa.h" // _Py_dg_dtoa() #include "pycore_floatobject.h" // _PyFloat_FormatAdvancedWriter() #include "pycore_freelist.h" // _Py_FREELIST_FREE(), _Py_FREELIST_POP() @@ -1894,7 +1895,6 @@ _PyFloat_DebugMallocStats(FILE *out) int PyFloat_Pack2(double x, char *data, int le) { - unsigned char *p = (unsigned char *)data; #if _Py_HAVE_FLOAT16 /* Conversion can change NaNs type or alter payload. Here we just fallback to the generic code, instead of providing @@ -1906,16 +1906,14 @@ PyFloat_Pack2(double x, char *data, int le) goto Overflow; } - unsigned char s[sizeof(_Float16)]; - - memcpy(s, &y, sizeof(_Float16)); if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) { - p[1] = s[0]; - p[0] = s[1]; + uint16_t word; + memcpy(&word, &y, 2); + word = _Py_bswap16(word); // Swap bytes + memcpy(data, &word, 2); } else { - p[0] = s[0]; - p[1] = s[1]; + memcpy(data, &y, sizeof(_Float16)); } return 0; } @@ -1924,7 +1922,6 @@ PyFloat_Pack2(double x, char *data, int le) int e; double f; unsigned short bits; - int incr = 1; if (x == 0.0) { sign = (copysign(1.0, x) == -1.0); @@ -2004,18 +2001,15 @@ PyFloat_Pack2(double x, char *data, int le) bits |= (e << 10) | (sign << 15); /* Write out result. */ + unsigned char *p = (unsigned char *)data; if (le) { - p += 1; - incr = -1; + p[0] = (unsigned char)(bits & 0xFF); + p[1] = (unsigned char)((bits >> 8) & 0xFF); + } + else { + p[0] = (unsigned char)((bits >> 8) & 0xFF); + p[1] = (unsigned char)(bits & 0xFF); } - - /* First byte */ - *p = (unsigned char)((bits >> 8) & 0xFF); - p += incr; - - /* Second byte */ - *p = (unsigned char)(bits & 0xFF); - return 0; Overflow: @@ -2027,10 +2021,7 @@ PyFloat_Pack2(double x, char *data, int le) int PyFloat_Pack4(double x, char *data, int le) { - unsigned char *p = (unsigned char *)data; float y = (float)x; - int i, incr = 1; - if (isinf(y) && !isinf(x)) { PyErr_SetString(PyExc_OverflowError, "float too large to pack with f format"); @@ -2054,8 +2045,8 @@ PyFloat_Pack4(double x, char *data, int le) } #else uint32_t u32; - memcpy(&u32, &y, 4); + /* Workaround RISC-V: "If a NaN value is converted to a * different floating-point type, the result is the * canonical NaN of the new type". The canonical NaN here @@ -2075,17 +2066,14 @@ PyFloat_Pack4(double x, char *data, int le) #endif } - unsigned char s[sizeof(float)]; - memcpy(s, &y, sizeof(float)); - if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) { - p += 3; - incr = -1; + uint32_t word; + memcpy(&word, &y, 4); + word = _Py_bswap32(word); // Swap bytes + memcpy(data, &word, 4); } - - for (i = 0; i < 4; i++) { - *p = s[i]; - p += incr; + else { + memcpy(data, &y, sizeof(float)); } return 0; } @@ -2093,20 +2081,14 @@ PyFloat_Pack4(double x, char *data, int le) int PyFloat_Pack8(double x, char *data, int le) { - unsigned char *p = (unsigned char *)data; - unsigned char as_bytes[8]; - memcpy(as_bytes, &x, 8); - const unsigned char *s = as_bytes; - int i, incr = 1; - if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) { - p += 7; - incr = -1; + uint64_t word; + memcpy(&word, &x, 8); + word = _Py_bswap64(word); // Swap bytes + memcpy(data, &word, 8); } - - for (i = 0; i < 8; i++) { - *p = *s++; - p += incr; + else { + memcpy(data, &x, 8); } return 0; } From 89bb00023246724736a86f7f250374efcf83cf53 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Wed, 30 Sep 2026 14:29:56 +0200 Subject: [PATCH 2/2] Also use _Py_bswap16() at the end of PyFloat_Pack2() --- Objects/floatobject.c | 13 ++++--------- 1 file changed, 4 insertions(+), 9 deletions(-) diff --git a/Objects/floatobject.c b/Objects/floatobject.c index 8d18cbf5920b1b..2b8ba856f324a1 100644 --- a/Objects/floatobject.c +++ b/Objects/floatobject.c @@ -1921,7 +1921,7 @@ PyFloat_Pack2(double x, char *data, int le) unsigned char sign; int e; double f; - unsigned short bits; + uint16_t bits; if (x == 0.0) { sign = (copysign(1.0, x) == -1.0); @@ -2001,15 +2001,10 @@ PyFloat_Pack2(double x, char *data, int le) bits |= (e << 10) | (sign << 15); /* Write out result. */ - unsigned char *p = (unsigned char *)data; - if (le) { - p[0] = (unsigned char)(bits & 0xFF); - p[1] = (unsigned char)((bits >> 8) & 0xFF); - } - else { - p[0] = (unsigned char)((bits >> 8) & 0xFF); - p[1] = (unsigned char)(bits & 0xFF); + if ((_PY_FLOAT_LITTLE_ENDIAN && !le) || (_PY_FLOAT_BIG_ENDIAN && le)) { + bits = _Py_bswap16(bits); // Swap bytes } + memcpy(data, &bits, 2); return 0; Overflow: