From 6f4f3591eb8b729d80519e8dde7f1c605cc5d6c9 Mon Sep 17 00:00:00 2001 From: Petr Viktorin Date: Thu, 8 Oct 2026 18:25:48 +0200 Subject: [PATCH 1/6] Add Unicode checks - `_Py*UnicodeObject_CAST` macros assert they have the right bits for the shape - `_PyUnicode_IsModifiable` & `_PyUnicodeWriter_CanWrite` assert that UTF-8 storage hasn't been allocated yet --- Include/cpython/unicodeobject.h | 4 ++-- Include/internal/pycore_unicodeobject.h | 9 +++++++++ Objects/unicodeobject.c | 8 +++----- 3 files changed, 14 insertions(+), 7 deletions(-) diff --git a/Include/cpython/unicodeobject.h b/Include/cpython/unicodeobject.h index 54defdb2061aa96..0f13d6fd11ffde8 100644 --- a/Include/cpython/unicodeobject.h +++ b/Include/cpython/unicodeobject.h @@ -186,10 +186,10 @@ typedef struct { (assert(PyUnicode_Check(op)), \ _Py_CAST(PyASCIIObject*, (op))) #define _PyCompactUnicodeObject_CAST(op) \ - (assert(PyUnicode_Check(op)), \ + (assert(!_PyASCIIObject_CAST(op)->state.ascii || !_PyASCIIObject_CAST(op)->state.compact), \ _Py_CAST(PyCompactUnicodeObject*, (op))) #define _PyUnicodeObject_CAST(op) \ - (assert(PyUnicode_Check(op)), \ + (assert(!_PyASCIIObject_CAST(op)->state.compact), \ _Py_CAST(PyUnicodeObject*, (op))) diff --git a/Include/internal/pycore_unicodeobject.h b/Include/internal/pycore_unicodeobject.h index b8e24da693be08a..f00eb51b7499de0 100644 --- a/Include/internal/pycore_unicodeobject.h +++ b/Include/internal/pycore_unicodeobject.h @@ -11,6 +11,7 @@ extern "C" { #include "pycore_fileutils.h" // _Py_error_handler #include "pycore_ucnhash.h" // _PyUnicode_Name_CAPI #include "pycore_runtime.h" // _Py_LATIN1_CHR() +#include "pycore_pyatomic_ft_wrappers.h" // FT_ATOMIC_LOAD_PTR_ACQUIRE // Maximum code point of Unicode 6.0: 0x10ffff (1,114,111). @@ -108,6 +109,11 @@ _PyUnicode_EnsureUnicode(PyObject *obj) return 0; } +static inline char* _PyUnicode_UTF8(PyObject *op) +{ + return FT_ATOMIC_LOAD_PTR_ACQUIRE(_PyCompactUnicodeObject_CAST(op)->utf8); +} + #ifndef NDEBUG static inline int _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer) @@ -125,6 +131,9 @@ _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer) assert(PyUnstable_Unicode_GET_CACHED_HASH(buffer) == -1); assert(!PyUnicode_CHECK_INTERNED(buffer)); assert(!_Py_IsImmortal(buffer)); + assert(PyUnicode_IS_COMPACT_ASCII(buffer) + || _PyUnicode_UTF8(buffer) == NULL + || _PyUnicode_UTF8(buffer) == PyUnicode_DATA(buffer)); return 1; } #endif diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index ded13cc37f69f17..5c2b9cc35d07b53 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -114,11 +114,6 @@ NOTE: In the interpreter's initialization phase, some globals are currently # define _PyUnicode_CHECK(op) PyUnicode_Check(op) #endif -static inline char* _PyUnicode_UTF8(PyObject *op) -{ - return FT_ATOMIC_LOAD_PTR_ACQUIRE(_PyCompactUnicodeObject_CAST(op)->utf8); -} - static inline char* PyUnicode_UTF8(PyObject *op) { assert(_PyUnicode_CHECK(op)); @@ -1767,6 +1762,9 @@ _PyUnicode_IsModifiable(PyObject *unicode) return 0; if (PyUnicode_CHECK_INTERNED(unicode)) return 0; + if (_PyUnicode_HAS_UTF8_MEMORY(unicode)) { + return 0; + } #ifdef Py_DEBUG /* singleton refcount is greater than 1 */ assert(!unicode_is_singleton(unicode)); From c624e7a6a7014e1a8b975efb26dc42dc6e390907 Mon Sep 17 00:00:00 2001 From: Petr Viktorin Date: Fri, 9 Oct 2026 09:37:47 +0200 Subject: [PATCH 2/6] Apply batched suggestions from code review Co-authored-by: Victor Stinner --- Include/cpython/unicodeobject.h | 3 ++- Include/internal/pycore_unicodeobject.h | 3 ++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/Include/cpython/unicodeobject.h b/Include/cpython/unicodeobject.h index 0f13d6fd11ffde8..ce12dd16658c759 100644 --- a/Include/cpython/unicodeobject.h +++ b/Include/cpython/unicodeobject.h @@ -186,7 +186,8 @@ typedef struct { (assert(PyUnicode_Check(op)), \ _Py_CAST(PyASCIIObject*, (op))) #define _PyCompactUnicodeObject_CAST(op) \ - (assert(!_PyASCIIObject_CAST(op)->state.ascii || !_PyASCIIObject_CAST(op)->state.compact), \ + (assert(!(_PyASCIIObject_CAST(op)->state.ascii \ + && _PyASCIIObject_CAST(op)->state.compact)), \ _Py_CAST(PyCompactUnicodeObject*, (op))) #define _PyUnicodeObject_CAST(op) \ (assert(!_PyASCIIObject_CAST(op)->state.compact), \ diff --git a/Include/internal/pycore_unicodeobject.h b/Include/internal/pycore_unicodeobject.h index f00eb51b7499de0..adf35bf250f6208 100644 --- a/Include/internal/pycore_unicodeobject.h +++ b/Include/internal/pycore_unicodeobject.h @@ -109,7 +109,8 @@ _PyUnicode_EnsureUnicode(PyObject *obj) return 0; } -static inline char* _PyUnicode_UTF8(PyObject *op) +static inline char* +_PyUnicode_UTF8(PyObject *op) { return FT_ATOMIC_LOAD_PTR_ACQUIRE(_PyCompactUnicodeObject_CAST(op)->utf8); } From be2d004c3c8561114d89229c986a1fd414d99cb5 Mon Sep 17 00:00:00 2001 From: Petr Viktorin Date: Fri, 9 Oct 2026 09:40:09 +0200 Subject: [PATCH 3/6] Move _PyUnicode_HAS_UTF8_MEMORY to the header --- Include/internal/pycore_unicodeobject.h | 13 ++++++++++--- Objects/unicodeobject.c | 9 --------- 2 files changed, 10 insertions(+), 12 deletions(-) diff --git a/Include/internal/pycore_unicodeobject.h b/Include/internal/pycore_unicodeobject.h index adf35bf250f6208..6bd83fd3e9cd261 100644 --- a/Include/internal/pycore_unicodeobject.h +++ b/Include/internal/pycore_unicodeobject.h @@ -115,6 +115,15 @@ _PyUnicode_UTF8(PyObject *op) return FT_ATOMIC_LOAD_PTR_ACQUIRE(_PyCompactUnicodeObject_CAST(op)->utf8); } +/* true if the Unicode object has an allocated UTF-8 memory block + (not shared with other data) */ +static inline int _PyUnicode_HAS_UTF8_MEMORY(PyObject *op) +{ + return (!PyUnicode_IS_COMPACT_ASCII(op) + && _PyUnicode_UTF8(op) != NULL + && _PyUnicode_UTF8(op) != PyUnicode_DATA(op)); +} + #ifndef NDEBUG static inline int _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer) @@ -132,9 +141,7 @@ _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer) assert(PyUnstable_Unicode_GET_CACHED_HASH(buffer) == -1); assert(!PyUnicode_CHECK_INTERNED(buffer)); assert(!_Py_IsImmortal(buffer)); - assert(PyUnicode_IS_COMPACT_ASCII(buffer) - || _PyUnicode_UTF8(buffer) == NULL - || _PyUnicode_UTF8(buffer) == PyUnicode_DATA(buffer)); + assert(!_PyUnicode_HAS_UTF8_MEMORY(buffer)); return 1; } #endif diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 5c2b9cc35d07b53..6d15e1dd7443b71 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -170,15 +170,6 @@ static inline int _PyUnicode_SHARE_UTF8(PyObject *op) return (_PyUnicode_UTF8(op) == PyUnicode_DATA(op)); } -/* true if the Unicode object has an allocated UTF-8 memory block - (not shared with other data) */ -static inline int _PyUnicode_HAS_UTF8_MEMORY(PyObject *op) -{ - return (!PyUnicode_IS_COMPACT_ASCII(op) - && _PyUnicode_UTF8(op) != NULL - && _PyUnicode_UTF8(op) != PyUnicode_DATA(op)); -} - #define LATIN1 _Py_LATIN1_CHR From 739df74c4b93c509688b1f3e0345a0a2c13791b9 Mon Sep 17 00:00:00 2001 From: Petr Viktorin Date: Fri, 9 Oct 2026 11:24:23 +0200 Subject: [PATCH 4/6] Fix and test _PyUnicode_Dump --- Include/internal/pycore_unicodeobject.h | 4 +++ Lib/test/test_capi/test_unicode.py | 42 ++++++++++++++++++++++++- Objects/unicodeobject.c | 33 ++++++++++--------- 3 files changed, 61 insertions(+), 18 deletions(-) diff --git a/Include/internal/pycore_unicodeobject.h b/Include/internal/pycore_unicodeobject.h index 6bd83fd3e9cd261..261df2b11008181 100644 --- a/Include/internal/pycore_unicodeobject.h +++ b/Include/internal/pycore_unicodeobject.h @@ -458,6 +458,10 @@ extern int _PyUnicode_WideCharString_Opt_Converter(PyObject *, void *); // Export for test_peg_generator PyAPI_FUNC(Py_ssize_t) _PyUnicode_ScanIdentifier(PyObject *); +#ifdef Py_DEBUG +PyAPI_FUNC(void) _PyUnicode_Dump(PyObject *op); +#endif + /* --- Runtime lifecycle -------------------------------------------------- */ extern void _PyUnicode_InitState(PyInterpreterState *); diff --git a/Lib/test/test_capi/test_unicode.py b/Lib/test/test_capi/test_unicode.py index 9237809e7f0dcca..534c5c5ca2abb53 100644 --- a/Lib/test/test_capi/test_unicode.py +++ b/Lib/test/test_capi/test_unicode.py @@ -4,7 +4,7 @@ from test import support from test.support import import_helper from test.support import threading_helper -from test.support.script_helper import assert_python_failure +from test.support.script_helper import assert_python_failure, assert_python_ok from threading import Thread try: @@ -1966,6 +1966,46 @@ def copy(text): # CRASHES unicode_equal("abc", NULL) # CRASHES unicode_equal(NULL, "abc") + def test_pyunicode_dump(self): + try: + ctypes.pythonapi._PyUnicode_Dump + except AttributeError: + self.skipTest("_PyUnicode_Dump not available") + proc = assert_python_ok('-c', """ + import sys, ctypes + _PyUnicode_Dump = ctypes.pythonapi._PyUnicode_Dump + _PyUnicode_Dump.argtypes = [ctypes.py_object] + _PyUnicode_Dump.restype = None + PyUnicode_AsUTF8 = ctypes.pythonapi.PyUnicode_AsUTF8 + PyUnicode_AsUTF8.argtypes = [ctypes.py_object] + PyUnicode_AsUTF8.restype = ctypes.c_char_p + for s in ( + "ASCII parrot", "latin1 møøse", + "UCS2 half‐a‐bee", "UCS4 \N{RABBIT}" + ): + print(s, flush=True) + _PyUnicode_Dump(s) + PyUnicode_AsUTF8(s) + _PyUnicode_Dump(s) + sys.stdout.flush() + """.encode(), PYTHONIOENCODING='UTF-8') + self.assertRegex(proc.out.decode(), textwrap.dedent(r""" + \A + ASCII parrot\n + ascii: len=12, data=[^\n]*\n + ascii: len=12, data=[^\n]*\n + latin1 møøse\n + latin1: len=12, utf8=NULL \(0\), data=[^\n]*\n + latin1: len=12, utf8=[^\n]* \(14\), data=[^\n]*\n + UCS2 half‐a‐bee\n + UCS2: len=15, utf8=NULL \(0\), data=[^\n]*\n + UCS2: len=15, utf8=[^\n]* \(19\), data=[^\n]*\n + UCS4 \N{RABBIT}\n + UCS4: len=6, utf8=NULL \(0\), data=[^\n]*\n + UCS4: len=6, utf8=[^\n]* \(9\), data=[^\n]*\n + \Z + """).strip().replace('\n', '')) + class PyUnicodeWriterTest(unittest.TestCase): def create_writer(self, size): diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 6d15e1dd7443b71..5a71e729755a94f 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -1237,8 +1237,10 @@ const void *_PyUnicode_data(void *unicode_raw) { printf("compact %d\n", PyUnicode_IS_COMPACT(unicode)); printf("compact ascii %d\n", PyUnicode_IS_COMPACT_ASCII(unicode)); printf("ascii op %p\n", (void*)(_PyASCIIObject_CAST(unicode) + 1)); - printf("compact op %p\n", (void*)(_PyCompactUnicodeObject_CAST(unicode) + 1)); - printf("compact data %p\n", _PyUnicode_COMPACT_DATA(unicode)); + if (!PyUnicode_IS_COMPACT_ASCII(unicode)) { + printf("compact op %p\n", (void*)(_PyCompactUnicodeObject_CAST(unicode) + 1)); + printf("compact data %p\n", _PyUnicode_COMPACT_DATA(unicode)); + } return PyUnicode_DATA(unicode); } @@ -1246,25 +1248,22 @@ void _PyUnicode_Dump(PyObject *op) { PyASCIIObject *ascii = _PyASCIIObject_CAST(op); - PyCompactUnicodeObject *compact = _PyCompactUnicodeObject_CAST(op); - PyUnicodeObject *unicode = _PyUnicodeObject_CAST(op); - const void *data; + const void *data = PyUnicode_DATA(op); - if (ascii->state.compact) - { - if (ascii->state.ascii) - data = (ascii + 1); - else - data = (compact + 1); - } - else - data = unicode->data.any; - printf("%s: len=%zu, ", unicode_kind_name(op), ascii->length); + printf("%s: len=%zu", unicode_kind_name(op), ascii->length); if (!ascii->state.ascii) { - printf("utf8=%p (%zu)", (void *)compact->utf8, compact->utf8_length); + PyCompactUnicodeObject *compact = _PyCompactUnicodeObject_CAST(op); + if (compact->utf8 == NULL) { + printf(", utf8=NULL"); + } + else { + printf(", utf8=%p", (void *)compact->utf8); + } + printf(" (%zu)", compact->utf8_length); } - printf(", data=%p\n", data); + printf(", data=%p\n", PyUnicode_DATA(op)); + fflush(stdout); } #endif From 45020eedc41795dd471c683a6b3d9eea69c56278 Mon Sep 17 00:00:00 2001 From: Petr Viktorin Date: Fri, 9 Oct 2026 12:02:17 +0200 Subject: [PATCH 5/6] Remove unused variable --- Objects/unicodeobject.c | 1 - 1 file changed, 1 deletion(-) diff --git a/Objects/unicodeobject.c b/Objects/unicodeobject.c index 5a71e729755a94f..da1ba44dba81290 100644 --- a/Objects/unicodeobject.c +++ b/Objects/unicodeobject.c @@ -1248,7 +1248,6 @@ void _PyUnicode_Dump(PyObject *op) { PyASCIIObject *ascii = _PyASCIIObject_CAST(op); - const void *data = PyUnicode_DATA(op); printf("%s: len=%zu", unicode_kind_name(op), ascii->length); From 55fecc52f0bd5c3816dfe5b48f0395b66a8cd1b9 Mon Sep 17 00:00:00 2001 From: Petr Viktorin Date: Fri, 9 Oct 2026 13:55:11 +0200 Subject: [PATCH 6/6] Handle Windows newlines --- Lib/test/test_capi/test_unicode.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/Lib/test/test_capi/test_unicode.py b/Lib/test/test_capi/test_unicode.py index 534c5c5ca2abb53..43792f7a48e3e25 100644 --- a/Lib/test/test_capi/test_unicode.py +++ b/Lib/test/test_capi/test_unicode.py @@ -1989,7 +1989,8 @@ def test_pyunicode_dump(self): _PyUnicode_Dump(s) sys.stdout.flush() """.encode(), PYTHONIOENCODING='UTF-8') - self.assertRegex(proc.out.decode(), textwrap.dedent(r""" + stdout = proc.out.decode().replace('\r', '') + self.assertRegex(stdout, textwrap.dedent(r""" \A ASCII parrot\n ascii: len=12, data=[^\n]*\n