From da5cb3361114b452488e77ff22db7d7194da5299 Mon Sep 17 00:00:00 2001 From: Victor Stinner Date: Tue, 29 Sep 2026 19:58:28 +0200 Subject: [PATCH] gh-158445: Reject invalid UCS4 in PyUnicode_FromKindAndData() PyUnicode_FromKindAndData(PyUnicode_4BYTE_KIND) and PyUnicodeWriter_WriteUCS4() now raise an exception if a character is not in the [U+0000; U+10ffff] range, instead of creating an invalid str object. * Add _testinternalcapi._Py_MAX_UNICODE. * Add unicode_invalid_character() helper function. --- Doc/c-api/unicode.rst | 2 +- Lib/test/test_capi/test_unicode.py | 54 +++++++++++++------ ...-09-29-20-13-43.gh-issue-158445.nbHlaw.rst | 4 ++ Modules/_testinternalcapi.c | 4 ++ Objects/unicodeobject.c | 48 ++++++++++++++--- 5 files changed, 88 insertions(+), 24 deletions(-) create mode 100644 Misc/NEWS.d/next/C_API/2026-09-29-20-13-43.gh-issue-158445.nbHlaw.rst diff --git a/Doc/c-api/unicode.rst b/Doc/c-api/unicode.rst index 3b635fa7fa3744..8b1c1e480d0990 100644 --- a/Doc/c-api/unicode.rst +++ b/Doc/c-api/unicode.rst @@ -1896,7 +1896,7 @@ object. .. c:function:: int PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *writer, const Py_UCS4 *str, Py_ssize_t size) - Writer the UCS4 string *str* into *writer*. + Write the UCS4 string *str* into *writer*. *size* is a number of UCS4 characters. diff --git a/Lib/test/test_capi/test_unicode.py b/Lib/test/test_capi/test_unicode.py index 032b910a280083..122d1f79e2b13b 100644 --- a/Lib/test/test_capi/test_unicode.py +++ b/Lib/test/test_capi/test_unicode.py @@ -19,6 +19,11 @@ from _testcapi import PY_SSIZE_T_MIN, PY_SSIZE_T_MAX, SIZEOF_WCHAR_T +MAX_UNICODE = _testinternalcapi._Py_MAX_UNICODE +# The first invalid character after MAX_UNICODE +INVALID_CHAR = MAX_UNICODE + 1 +# Maximum invalid character which fits into 32-bit Py_UCS4 +MAX_INVALID_CHAR = 0xFFFF_FFFF NULL = None class Str(str): @@ -73,14 +78,14 @@ def test_new(self): self.assertEqual(new(0, maxchar), '') self.assertEqual(new(5, maxchar), chr(maxchar)*5) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX, maxchar) - self.assertEqual(new(0, 0x110000), '') + self.assertEqual(new(0, INVALID_CHAR), '') self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x4f60) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x4f60) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2, 0x1f600) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//2+1, 0x1f600) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4, 0x1f600) self.assertRaises(MemoryError, new, PY_SSIZE_T_MAX//4+1, 0x1f600) - self.assertRaises(SystemError, new, 5, 0x110000) + self.assertRaises(SystemError, new, 5, INVALID_CHAR) self.assertRaises(SystemError, new, -1, 0) self.assertRaises(SystemError, new, PY_SSIZE_T_MIN, 0) @@ -115,7 +120,7 @@ def test_fill(self): s = strings[0] self.assertRaises(IndexError, fill, s, -1, 0, 0x78) self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78) - self.assertRaises(ValueError, fill, s, 0, 0, 0x110000) + self.assertRaises(ValueError, fill, s, 0, 0, INVALID_CHAR) self.assertRaises(SystemError, fill, b'abc', 0, 0, 0x78) self.assertRaises(SystemError, fill, [], 0, 0, 0x78) # CRASHES fill(s, 0, NULL, 0, 0) @@ -129,7 +134,7 @@ def _test_writechar(self, writechar, *, check): '\U0001f600\U0001f601\U0001f602' ] # one character for every kind + out of range code - chars = [0x78, 0xa9, 0x20ac, 0x1f638, 0x110000] + chars = [0x78, 0xa9, 0x20ac, 0x1f638, INVALID_CHAR] for i, s in enumerate(strings): for j, c in enumerate(chars): if j <= i: @@ -301,7 +306,17 @@ def test_fromkindanddata(self): self.assertRaises(ValueError, fromkindanddata, 1, NULL, -1) self.assertRaises(ValueError, fromkindanddata, 1, NULL, PY_SSIZE_T_MIN) # CRASHES fromkindanddata(1, NULL, 1) - # CRASHES fromkindanddata(4, b'\xff\xff\xff\xff') + + # Test invalid UCS-4 string + for invalid_char in (INVALID_CHAR, MAX_INVALID_CHAR): + with self.subTest(invalid_char=invalid_char): + # Test single character + ucs4_char = invalid_char.to_bytes(4, byteorder=sys.byteorder) + self.assertRaises(ValueError, fromkindanddata, 4, ucs4_char) + + # Test multiple characters + s = 'valid'.encode(enc4) + ucs4_char + self.assertRaises(ValueError, fromkindanddata, 4, s) def test_substring(self): """Test PyUnicode_Substring()""" @@ -446,7 +461,7 @@ def check_format(expected, format, *args): check_format('\U0010ffff', b'%c', c_int(0x10ffff)) with self.assertRaises(OverflowError): - PyUnicode_FromFormat(b'%c', c_int(0x110000)) + PyUnicode_FromFormat(b'%c', c_int(INVALID_CHAR)) # Issue #18183 check_format('\U00010000\U00100000', b'%c%c', c_int(0x10000), c_int(0x100000)) @@ -1015,7 +1030,7 @@ def test_fromordinal(self): self.assertEqual(fromordinal(0x20ac), '\u20ac') self.assertEqual(fromordinal(0x1f600), '\U0001f600') - self.assertRaises(ValueError, fromordinal, 0x110000) + self.assertRaises(ValueError, fromordinal, INVALID_CHAR) self.assertRaises(ValueError, fromordinal, -1) def test_asutf8(self): @@ -1367,8 +1382,8 @@ def test_findchar(self): self.assertEqual(unicode_findchar(str, ord(ch), 0, len(str), -1), i) str = "!>_` and :c:func:`PyUnicodeWriter_WriteUCS4` now raise +an exception if a character is not in the [U+0000; U+10ffff] range, instead of +creating an invalid str object. Patch by Victor Stinner. diff --git a/Modules/_testinternalcapi.c b/Modules/_testinternalcapi.c index 049b990d65be5f..a2ce266eb35a65 100644 --- a/Modules/_testinternalcapi.c +++ b/Modules/_testinternalcapi.c @@ -3544,6 +3544,10 @@ module_exec(PyObject *module) } PyModule_AddObject(module, "SelfInterruptingContextManager", (PyObject *)&SelfInterruptingContextManager_Type); + if (PyModule_AddIntMacro(module, _Py_MAX_UNICODE) < 0) { + return 1; + } + return 0; } diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 8446fdbfcb64a9..80764c74f49bc6 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -1592,6 +1592,14 @@ PyUnicode_CopyCharacters(PyObject *to, Py_ssize_t to_start, return how_many; } +static void +unicode_invalid_character(Py_UCS4 ch) +{ + PyErr_Format(PyExc_ValueError, + "character U+%x is not in range [U+0000; U+%x]", + ch, MAX_UNICODE); +} + /* Find the maximum code point and count the number of surrogate pairs so a correct string length can be computed before converting a string to UCS4. This function counts single surrogates as a character and not as a pair. @@ -1627,9 +1635,7 @@ find_maxchar_surrogates(const wchar_t *begin, const wchar_t *end, if (ch > *maxchar) { *maxchar = ch; if (*maxchar > MAX_UNICODE) { - PyErr_Format(PyExc_ValueError, - "character U+%x is not in range [U+0000; U+%x]", - ch, MAX_UNICODE); + unicode_invalid_character(ch); return -1; } } @@ -2224,15 +2230,29 @@ static PyObject* _PyUnicode_FromUCS4(const Py_UCS4 *u, Py_ssize_t size) { PyObject *res; - Py_UCS4 max_char; if (size == 0) _Py_RETURN_UNICODE_EMPTY(); assert(size > 0); - if (size == 1) + + // ucs4lib_find_max_char() cannot be used, it ignores limit greater + // than MAX_UNICODE + Py_UCS4 max_char = 127; + for (Py_ssize_t i = 0; i < size; i++) { + Py_UCS4 ch = u[i]; + if (ch > max_char) { + if (ch > MAX_UNICODE) { + unicode_invalid_character(ch); + return NULL; + } + max_char = ch; + } + } + + if (size == 1) { return unicode_char(u[0]); + } - max_char = ucs4lib_find_max_char(u, u + size); res = PyUnicode_New(size, max_char); if (!res) return NULL; @@ -2266,9 +2286,21 @@ PyUnicodeWriter_WriteUCS4(PyUnicodeWriter *pub_writer, return 0; } - Py_UCS4 max_char = ucs4lib_find_max_char(str, str + size); + // ucs4lib_find_max_char() cannot be used, it ignores limit greater + // than MAX_UNICODE + Py_UCS4 maxchar = 127; + for (Py_ssize_t i = 0; i < size; i++) { + Py_UCS4 ch = str[i]; + if (ch > maxchar) { + if (ch > MAX_UNICODE) { + unicode_invalid_character(ch); + return -1; + } + maxchar = ch; + } + } - if (_PyUnicodeWriter_Prepare(writer, size, max_char) < 0) { + if (_PyUnicodeWriter_Prepare(writer, size, maxchar) < 0) { return -1; } assert(_PyUnicodeWriter_CanWrite(writer));