FazBrowse GitHub Viewer | Trending |
URL:
| Home
Tools: [Download Repo ZIP]   [Original HTTPS Page]

gh-157710: Add more asserts to internal PyUnicode operations by encukou · Pull Request #159033 · python/cpython · GitHub

Repository navigation

Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension .c  (1) .h  (2) .py  (1) All 3 file types selected
Viewed files
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Unified
Split
Hide whitespace
Diff view
Unified
Split
Hide whitespace
5 changes: 3 additions & 2 deletions Include/cpython/unicodeobject.h
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -186,10 +186,11 @@ typedef struct {
(assert(PyUnicode_Check(op)), \
_Py_CAST(PyASCIIObject*, (op)))
#define _PyCompactUnicodeObject_CAST(op) \
(assert(PyUnicode_Check(op)), \
(assert(!(_PyASCIIObject_CAST(op)->state.ascii \
&& _PyASCIIObject_CAST(op)->state.compact)), \
_Py_CAST(PyCompactUnicodeObject*, (op)))
#define _PyUnicodeObject_CAST(op) \
(assert(PyUnicode_Check(op)), \
(assert(!_PyASCIIObject_CAST(op)->state.compact), \
_Py_CAST(PyUnicodeObject*, (op)))


Expand Down
21 changes: 21 additions & 0 deletions Include/internal/pycore_unicodeobject.h
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ extern "C" {
#include "pycore_fileutils.h" // _Py_error_handler
#include "pycore_ucnhash.h" // _PyUnicode_Name_CAPI
#include "pycore_runtime.h" // _Py_LATIN1_CHR()
#include "pycore_pyatomic_ft_wrappers.h" // FT_ATOMIC_LOAD_PTR_ACQUIRE


// Maximum code point of Unicode 6.0: 0x10ffff (1,114,111).
Expand Down Expand Up @@ -108,6 +109,21 @@ _PyUnicode_EnsureUnicode(PyObject *obj)
return 0;
}

static inline char*
_PyUnicode_UTF8(PyObject *op)
{
return FT_ATOMIC_LOAD_PTR_ACQUIRE(_PyCompactUnicodeObject_CAST(op)->utf8);
}

/* true if the Unicode object has an allocated UTF-8 memory block
(not shared with other data) */
static inline int _PyUnicode_HAS_UTF8_MEMORY(PyObject *op)
{
return (!PyUnicode_IS_COMPACT_ASCII(op)
&& _PyUnicode_UTF8(op) != NULL
&& _PyUnicode_UTF8(op) != PyUnicode_DATA(op));
}

#ifndef NDEBUG
static inline int
_PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer)
Expand All @@ -125,6 +141,7 @@ _PyUnicodeWriter_CanWrite(_PyUnicodeWriter *writer)
assert(PyUnstable_Unicode_GET_CACHED_HASH(buffer) == -1);
assert(!PyUnicode_CHECK_INTERNED(buffer));
assert(!_Py_IsImmortal(buffer));
assert(!_PyUnicode_HAS_UTF8_MEMORY(buffer));
return 1;
}
#endif
Expand Down Expand Up @@ -441,6 +458,10 @@ extern int _PyUnicode_WideCharString_Opt_Converter(PyObject *, void *);
// Export for test_peg_generator
PyAPI_FUNC(Py_ssize_t) _PyUnicode_ScanIdentifier(PyObject *);

#ifdef Py_DEBUG
PyAPI_FUNC(void) _PyUnicode_Dump(PyObject *op);
#endif

/* --- Runtime lifecycle -------------------------------------------------- */

extern void _PyUnicode_InitState(PyInterpreterState *);
Expand Down
43 changes: 42 additions & 1 deletion Lib/test/test_capi/test_unicode.py
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
from test import support
from test.support import import_helper
from test.support import threading_helper
from test.support.script_helper import assert_python_failure
from test.support.script_helper import assert_python_failure, assert_python_ok
from threading import Thread

try:
Expand Down Expand Up @@ -1966,6 +1966,47 @@ def copy(text):
# CRASHES unicode_equal("abc", NULL)
# CRASHES unicode_equal(NULL, "abc")

def test_pyunicode_dump(self):
try:
ctypes.pythonapi._PyUnicode_Dump
except AttributeError:
self.skipTest("_PyUnicode_Dump not available")
proc = assert_python_ok('-c', """
import sys, ctypes
_PyUnicode_Dump = ctypes.pythonapi._PyUnicode_Dump
_PyUnicode_Dump.argtypes = [ctypes.py_object]
_PyUnicode_Dump.restype = None
PyUnicode_AsUTF8 = ctypes.pythonapi.PyUnicode_AsUTF8
PyUnicode_AsUTF8.argtypes = [ctypes.py_object]
PyUnicode_AsUTF8.restype = ctypes.c_char_p
for s in (
"ASCII parrot", "latin1 møøse",
"UCS2 half‐a‐bee", "UCS4 \N{RABBIT}"
):
print(s, flush=True)
_PyUnicode_Dump(s)
PyUnicode_AsUTF8(s)
_PyUnicode_Dump(s)
sys.stdout.flush()
""".encode(), PYTHONIOENCODING='UTF-8')
stdout = proc.out.decode().replace('\r', '')
self.assertRegex(stdout, textwrap.dedent(r"""
\A
ASCII parrot\n
ascii: len=12, data=[^\n]*\n
ascii: len=12, data=[^\n]*\n
latin1 møøse\n
latin1: len=12, utf8=NULL \(0\), data=[^\n]*\n
latin1: len=12, utf8=[^\n]* \(14\), data=[^\n]*\n
UCS2 half‐a‐bee\n
UCS2: len=15, utf8=NULL \(0\), data=[^\n]*\n
UCS2: len=15, utf8=[^\n]* \(19\), data=[^\n]*\n
UCS4 \N{RABBIT}\n
UCS4: len=6, utf8=NULL \(0\), data=[^\n]*\n
UCS4: len=6, utf8=[^\n]* \(9\), data=[^\n]*\n
\Z
""").strip().replace('\n', ''))


class PyUnicodeWriterTest(unittest.TestCase):
def create_writer(self, size):
Expand Down
49 changes: 18 additions & 31 deletions Objects/unicodeobject.c
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -114,11 +114,6 @@ NOTE: In the interpreter's initialization phase, some globals are currently
# define _PyUnicode_CHECK(op) PyUnicode_Check(op)
#endif

static inline char* _PyUnicode_UTF8(PyObject *op)
{
return FT_ATOMIC_LOAD_PTR_ACQUIRE(_PyCompactUnicodeObject_CAST(op)->utf8);
}

static inline char* PyUnicode_UTF8(PyObject *op)
{
assert(_PyUnicode_CHECK(op));
Expand Down Expand Up @@ -175,15 +170,6 @@ static inline int _PyUnicode_SHARE_UTF8(PyObject *op)
return (_PyUnicode_UTF8(op) == PyUnicode_DATA(op));
}

/* true if the Unicode object has an allocated UTF-8 memory block
(not shared with other data) */
static inline int _PyUnicode_HAS_UTF8_MEMORY(PyObject *op)
{
return (!PyUnicode_IS_COMPACT_ASCII(op)
&& _PyUnicode_UTF8(op) != NULL
&& _PyUnicode_UTF8(op) != PyUnicode_DATA(op));
}


#define LATIN1 _Py_LATIN1_CHR

Expand Down Expand Up @@ -1251,34 +1237,32 @@ const void *_PyUnicode_data(void *unicode_raw) {
printf("compact %d\n", PyUnicode_IS_COMPACT(unicode));
printf("compact ascii %d\n", PyUnicode_IS_COMPACT_ASCII(unicode));
printf("ascii op %p\n", (void*)(_PyASCIIObject_CAST(unicode) + 1));
printf("compact op %p\n", (void*)(_PyCompactUnicodeObject_CAST(unicode) + 1));
printf("compact data %p\n", _PyUnicode_COMPACT_DATA(unicode));
if (!PyUnicode_IS_COMPACT_ASCII(unicode)) {
printf("compact op %p\n", (void*)(_PyCompactUnicodeObject_CAST(unicode) + 1));
printf("compact data %p\n", _PyUnicode_COMPACT_DATA(unicode));
}
return PyUnicode_DATA(unicode);
}

void
_PyUnicode_Dump(PyObject *op)
{
PyASCIIObject *ascii = _PyASCIIObject_CAST(op);
PyCompactUnicodeObject *compact = _PyCompactUnicodeObject_CAST(op);
PyUnicodeObject *unicode = _PyUnicodeObject_CAST(op);
const void *data;

if (ascii->state.compact)
{
if (ascii->state.ascii)
data = (ascii + 1);
else
data = (compact + 1);
}
else
data = unicode->data.any;
printf("%s: len=%zu, ", unicode_kind_name(op), ascii->length);
printf("%s: len=%zu", unicode_kind_name(op), ascii->length);

if (!ascii->state.ascii) {
printf("utf8=%p (%zu)", (void *)compact->utf8, compact->utf8_length);
PyCompactUnicodeObject *compact = _PyCompactUnicodeObject_CAST(op);
if (compact->utf8 == NULL) {
printf(", utf8=NULL");
}
else {
printf(", utf8=%p", (void *)compact->utf8);
}
printf(" (%zu)", compact->utf8_length);
}
printf(", data=%p\n", data);
printf(", data=%p\n", PyUnicode_DATA(op));
fflush(stdout);
}
#endif

Expand Down Expand Up @@ -1767,6 +1751,9 @@ _PyUnicode_IsModifiable(PyObject *unicode)
return 0;
if (PyUnicode_CHECK_INTERNED(unicode))
return 0;
if (_PyUnicode_HAS_UTF8_MEMORY(unicode)) {
return 0;
}
#ifdef Py_DEBUG
/* singleton refcount is greater than 1 */
assert(!unicode_is_singleton(unicode));
Expand Down
Loading

Back | FazBrowse Home | New Git URL