| FazBrowse GitHub Viewer | Trending | | Home |
| Tools: [Download Repo ZIP] [Original HTTPS Page] |
1 parent 182f323 commit 84b0669
6 files changed
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -168,8 +168,14 @@ access to internal read-only data of Unicode objects: | |||
| 168 | 168 | The function performs no checks for any of its requirements, | |
| 169 | 169 | and is intended for usage in loops. | |
| 170 | 170 | ||
| 171 | + While :class:`str` objects are usually immutable in Python, this special C API allows | ||
| 172 | + mutating a fresh :class:`str` object if the string has not been "used" yet. | ||
| 173 | + | ||
| 171 | 174 | .. versionadded:: 3.3 | |
| 172 | 175 | ||
| 176 | + .. soft-deprecated:: next | ||
| 177 | + Use the :c:type:`PyUnicodeWriter` API instead. | ||
| 178 | + | ||
| 173 | 179 | ||
| 174 | 180 | .. c:function:: Py_UCS4 PyUnicode_READ(int kind, void *data, Py_ssize_t index) | |
| 175 | 181 | ||
@@ -407,9 +413,15 @@ APIs: | |||
| 407 | 413 | using the :c:type:`PyUnicodeWriter` API, or one of the ``PyUnicode_From*`` | |
| 408 | 414 | functions below. | |
| 409 | 415 | ||
| 416 | + While :class:`str` objects are usually immutable in Python, this special C API | ||
| 417 | + returns a :class:`str` object that can be mutated, except if *size* is zero, in which | ||
| 418 | + case it returns the immutable empty string constant. | ||
| 410 | 419 | ||
| 411 | 420 | .. versionadded:: 3.3 | |
| 412 | 421 | ||
| 422 | + .. soft-deprecated:: next | ||
| 423 | + Use the :c:type:`PyUnicodeWriter` API instead. | ||
| 424 | + | ||
| 413 | 425 | ||
| 414 | 426 | .. c:function:: PyObject* PyUnicode_FromKindAndData(int kind, const void *buffer, \ | |
| 415 | 427 | Py_ssize_t size) | |
@@ -754,11 +766,16 @@ APIs: | |||
| 754 | 766 | possible. Returns ``-1`` and sets an exception on error, otherwise returns | |
| 755 | 767 | the number of copied characters. | |
| 756 | 768 | ||
| 757 | - The string must not have been “used” yet. | ||
| 769 | + While :class:`str` objects are usually immutable in Python, this special C API allows | ||
| 770 | + mutating a fresh :class:`str` object if the string has not been "used" yet. | ||
| 771 | + | ||
| 758 | 772 | See :c:func:`PyUnicode_New` for details. | |
| 759 | 773 | ||
| 760 | 774 | .. versionadded:: 3.3 | |
| 761 | 775 | ||
| 776 | + .. soft-deprecated:: next | ||
| 777 | + Use the :c:type:`PyUnicodeWriter` API instead. | ||
| 778 | + | ||
| 762 | 779 | ||
| 763 | 780 | .. c:function:: int PyUnicode_Resize(PyObject **unicode, Py_ssize_t length); | |
| 764 | 781 | ||
@@ -774,6 +791,14 @@ APIs: | |||
| 774 | 791 | The function doesn't check string content, the result may not be a | |
| 775 | 792 | string in canonical representation. | |
| 776 | 793 | ||
| 794 | + While :class:`str` objects are usually immutable in Python, this special C API | ||
| 795 | + can resize a :class:`str` object in-place if the string has not been "used" yet. | ||
| 796 | + It returns a :class:`str` object which can be mutated, except if *size* is zero, in | ||
| 797 | + which case it returns the immutable empty string constant. | ||
| 798 | + | ||
| 799 | + .. soft-deprecated:: next | ||
| 800 | + Use the :c:type:`PyUnicodeWriter` API instead. | ||
| 801 | + | ||
| 777 | 802 | ||
| 778 | 803 | .. c:function:: Py_ssize_t PyUnicode_Fill(PyObject *unicode, Py_ssize_t start, \ | |
| 779 | 804 | Py_ssize_t length, Py_UCS4 fill_char) | |
@@ -784,14 +809,19 @@ APIs: | |||
| 784 | 809 | Fail if *fill_char* is bigger than the string maximum character, or if the | |
| 785 | 810 | string has more than 1 reference. | |
| 786 | 811 | ||
| 787 | - The string must not have been “used” yet. | ||
| 788 | - See :c:func:`PyUnicode_New` for details. | ||
| 789 | - | ||
| 790 | 812 | Return the number of written characters, or return ``-1`` and raise an | |
| 791 | 813 | exception on error. | |
| 792 | 814 | ||
| 815 | + While :class:`str` objects are usually immutable in Python, this special C API allows | ||
| 816 | + mutating a fresh :class:`str` object if the string has not been "used" yet. | ||
| 817 | + | ||
| 818 | + See :c:func:`PyUnicode_New` for details. | ||
| 819 | + | ||
| 793 | 820 | .. versionadded:: 3.3 | |
| 794 | 821 | ||
| 822 | + .. soft-deprecated:: next | ||
| 823 | + Use the :c:type:`PyUnicodeWriter` API instead. | ||
| 824 | + | ||
| 795 | 825 | ||
| 796 | 826 | .. c:function:: int PyUnicode_WriteChar(PyObject *unicode, Py_ssize_t index, \ | |
| 797 | 827 | Py_UCS4 character) | |
@@ -804,11 +834,16 @@ APIs: | |||
| 804 | 834 | See :c:func:`PyUnicode_WRITE` for a version that skips these checks, | |
| 805 | 835 | making them your responsibility. | |
| 806 | 836 | ||
| 807 | - The string must not have been “used” yet. | ||
| 837 | + While :class:`str` objects are usually immutable in Python, this special C API allows | ||
| 838 | + mutating a fresh :class:`str` object if the string has not been "used" yet. | ||
| 839 | + | ||
| 808 | 840 | See :c:func:`PyUnicode_New` for details. | |
| 809 | 841 | ||
| 810 | 842 | .. versionadded:: 3.3 | |
| 811 | 843 | ||
| 844 | + .. soft-deprecated:: next | ||
| 845 | + Use the :c:type:`PyUnicodeWriter` API instead. | ||
| 846 | + | ||
| 812 | 847 | ||
| 813 | 848 | .. c:function:: Py_UCS4 PyUnicode_ReadChar(PyObject *unicode, Py_ssize_t index) | |
| 814 | 849 | ||
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -1224,6 +1224,13 @@ Deprecated C APIs | |||
| 1224 | 1224 | :c:func:`PyModule_GetFilenameObject` instead is still recommended. | |
| 1225 | 1225 | (Contributed by Victor Stinner in :gh:`154757`.) | |
| 1226 | 1226 | ||
| 1227 | + * Soft deprecate functions modifying Unicode strings: | ||
| 1228 | + :c:func:`PyUnicode_New`, :c:func:`PyUnicode_CopyCharacters`, | ||
| 1229 | + :c:func:`PyUnicode_Fill`, :c:func:`PyUnicode_Resize`, | ||
| 1230 | + :c:func:`PyUnicode_WRITE` and :c:func:`PyUnicode_WriteChar`. | ||
| 1231 | + Use the safer :c:type:`PyUnicodeWriter` API instead. | ||
| 1232 | + (Contributed by Victor Stinner in :gh:`157710`.) | ||
| 1233 | + | ||
| 1227 | 1234 | .. Add C API deprecations above alphabetically, not here at the end. | |
| 1228 | 1235 | ||
| 1229 | 1236 | .. include:: ../deprecations/c-api-pending-removal-in-3.18.rst | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -25,6 +25,7 @@ | |||
| 25 | 25 | # Maximum invalid character which fits into 32-bit Py_UCS4 | |
| 26 | 26 | MAX_INVALID_CHAR = 0xFFFF_FFFF | |
| 27 | 27 | NULL = None | |
| 28 | + USED_STR_ERROR = 'Cannot modify a string currently used' | ||
| 28 | 29 | ||
| 29 | 30 | class Str(str): | |
| 30 | 31 | pass | |
@@ -76,9 +77,27 @@ def test_checkexact(self): | |||
| 76 | 77 | # Test PyUnicode_CheckExact() | |
| 77 | 78 | self._test_check(_testlimitedcapi.unicode_checkexact, exact=True) | |
| 78 | 79 | ||
| 80 | + def assert_is_mutable(self, result, refcnt): | ||
| 81 | + # Check that result is a "mutable" Unicode string | ||
| 82 | + self.assertEqual(refcnt, 1) | ||
| 83 | + self.assertFalse(sys._is_immortal(result)) | ||
| 84 | + | ||
| 85 | + def assert_is_empty_singleton(self, result): | ||
| 86 | + # Check that result is the empty string singleton | ||
| 87 | + self.assertEqual(result, '') | ||
| 88 | + self.assertTrue(sys._is_immortal(result)) | ||
| 89 | + | ||
| 79 | 90 | def test_new(self): | |
| 80 | 91 | """Test PyUnicode_New()""" | |
| 81 | - new = _testcapi.unicode_new | ||
| 92 | + _unicode_new = _testcapi.unicode_new | ||
| 93 | + | ||
| 94 | + def new(size, maxchar): | ||
| 95 | + result = _unicode_new(size, maxchar) | ||
| 96 | + if size != 0: | ||
| 97 | + self.assert_is_mutable(result, sys.getrefcount(result)) | ||
| 98 | + else: | ||
| 99 | + self.assert_is_empty_singleton(result) | ||
| 100 | + return result | ||
| 82 | 101 | ||
| 83 | 102 | for maxchar in 0, 0x61, 0xa1, 0x4f60, 0x1f600, 0x10ffff: | |
| 84 | 103 | self.assertEqual(new(0, maxchar), '') | |
@@ -123,6 +142,10 @@ def test_fill(self): | |||
| 123 | 142 | self.assertEqual(fill(to, start, length, fill_char), | |
| 124 | 143 | (expected, filled)) | |
| 125 | 144 | ||
| 145 | + # A string with 2 references cannot be modified | ||
| 146 | + with self.assertRaisesRegex(SystemError, USED_STR_ERROR): | ||
| 147 | + fill('abc', 0, 3, ord('x'), incref=True) | ||
| 148 | + | ||
| 126 | 149 | s = strings[0] | |
| 127 | 150 | self.assertRaises(IndexError, fill, s, -1, 0, 0x78) | |
| 128 | 151 | self.assertRaises(IndexError, fill, s, PY_SSIZE_T_MIN, 0, 0x78) | |
@@ -162,7 +185,12 @@ def _test_writechar(self, writechar, *, check): | |||
| 162 | 185 | ||
| 163 | 186 | def test_writechar(self): | |
| 164 | 187 | """Test PyUnicode_WriteChar()""" | |
| 165 | - self._test_writechar(_testlimitedcapi.unicode_writechar, check=True) | ||
| 188 | + writechar = _testlimitedcapi.unicode_writechar | ||
| 189 | + self._test_writechar(writechar, check=True) | ||
| 190 | + | ||
| 191 | + # A string with 2 references cannot be modified | ||
| 192 | + with self.assertRaisesRegex(SystemError, USED_STR_ERROR): | ||
| 193 | + writechar('abc', 1, ord('x'), incref=True) | ||
| 166 | 194 | ||
| 167 | 195 | def test_write_macro(self): | |
| 168 | 196 | """Test PyUnicode_WRITE()""" | |
@@ -187,18 +215,16 @@ def resize(s, length, new=True, compute_hash=False): | |||
| 187 | 215 | self.assertFalse(is_new_obj) | |
| 188 | 216 | elif length == 0: | |
| 189 | 217 | # Get the empty Unicode string | |
| 190 | - self.assertEqual(result, '') | ||
| 191 | - self.assertTrue(sys._is_immortal(result)) | ||
| 218 | + self.assert_is_empty_singleton(result) | ||
| 192 | 219 | self.assertTrue(is_new_obj) | |
| 193 | 220 | elif (not new) or compute_hash: | |
| 194 | 221 | # Get a fresh copy | |
| 195 | - self.assertEqual(refcnt, 1) | ||
| 222 | + self.assert_is_mutable(result, refcnt) | ||
| 196 | 223 | self.assertTrue(is_new_obj) | |
| 197 | - self.assertFalse(sys._is_immortal(result)) | ||
| 198 | 224 | else: | |
| 199 | 225 | # In-size replace can return the same address, or not. | |
| 200 | 226 | # So 'is_new_obj' cannot be tested. | |
| 201 | - self.assertFalse(sys._is_immortal(result)) | ||
| 227 | + self.assert_is_mutable(result, refcnt) | ||
| 202 | 228 | ||
| 203 | 229 | return result | |
| 204 | 230 | ||
@@ -1793,6 +1819,11 @@ def test_copycharacters(self): | |||
| 1793 | 1819 | self.assertRaises(SystemError, unicode_copycharacters, s, 0, s, 0, PY_SSIZE_T_MIN) | |
| 1794 | 1820 | self.assertRaises(SystemError, unicode_copycharacters, s, 0, b'', 0, 0) | |
| 1795 | 1821 | self.assertRaises(SystemError, unicode_copycharacters, s, 0, [], 0, 0) | |
| 1822 | + | ||
| 1823 | + # A string with 2 references cannot be modified | ||
| 1824 | + with self.assertRaisesRegex(SystemError, USED_STR_ERROR): | ||
| 1825 | + unicode_copycharacters('abc', 0, 'abc', 0, 1, incref=True) | ||
| 1826 | + | ||
| 1796 | 1827 | # CRASHES unicode_copycharacters(s, 0, NULL, 0, 0) | |
| 1797 | 1828 | # TODO: Test PyUnicode_CopyCharacters() with non-unicode and | |
| 1798 | 1829 | # non-modifiable unicode as "to". | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -0,0 +1,5 @@ | |||
| 1 | + Soft deprecate functions modifying Unicode strings: | ||
| 2 | + :c:func:`PyUnicode_New`, :c:func:`PyUnicode_CopyCharacters`, | ||
| 3 | + :c:func:`PyUnicode_Fill`, :c:func:`PyUnicode_Resize`, | ||
| 4 | + :c:func:`PyUnicode_WRITE` and :c:func:`PyUnicode_WriteChar`. Use | ||
| 5 | + the :c:type:`PyUnicodeWriter` API instead. Patch by Victor Stinner. | ||
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -59,13 +59,20 @@ unicode_copy(PyObject *unicode) | |||
| 59 | 59 | ||
| 60 | 60 | /* Test PyUnicode_Fill() */ | |
| 61 | 61 | static PyObject * | |
| 62 | - unicode_fill(PyObject *self, PyObject *args) | ||
| 62 | + unicode_fill(PyObject *self, PyObject *args, PyObject *kwargs) | ||
| 63 | 63 | { | |
| 64 | + static char *kwlist[] = {"to", "start", "length", "fill_char", | ||
| 65 | + "incref", NULL}; | ||
| 64 | 66 | PyObject *to, *to_copy; | |
| 65 | 67 | Py_ssize_t start, length, filled; | |
| 66 | 68 | unsigned int fill_char; | |
| 69 | + int incref = 0; | ||
| 67 | 70 | ||
| 68 | - if (!PyArg_ParseTuple(args, "OnnI", &to, &start, &length, &fill_char)) { | ||
| 71 | + if (!PyArg_ParseTupleAndKeywords(args, kwargs, | ||
| 72 | + "OnnI|p", kwlist, | ||
| 73 | + &to, &start, &length, | ||
| 74 | + &fill_char, &incref)) | ||
| 75 | + { | ||
| 69 | 76 | return NULL; | |
| 70 | 77 | } | |
| 71 | 78 | ||
@@ -74,7 +81,14 @@ unicode_fill(PyObject *self, PyObject *args) | |||
| 74 | 81 | return NULL; | |
| 75 | 82 | } | |
| 76 | 83 | ||
| 84 | + if (incref) { | ||
| 85 | + Py_INCREF(to_copy); | ||
| 86 | + } | ||
| 77 | 87 | filled = PyUnicode_Fill(to_copy, start, length, (Py_UCS4)fill_char); | |
| 88 | + if (incref) { | ||
| 89 | + Py_DECREF(to_copy); | ||
| 90 | + } | ||
| 91 | + | ||
| 78 | 92 | if (filled == -1 && PyErr_Occurred()) { | |
| 79 | 93 | Py_DECREF(to_copy); | |
| 80 | 94 | return NULL; | |
@@ -190,13 +204,18 @@ unicode_asutf8(PyObject *self, PyObject *args) | |||
| 190 | 204 | ||
| 191 | 205 | /* Test PyUnicode_CopyCharacters() */ | |
| 192 | 206 | static PyObject * | |
| 193 | - unicode_copycharacters(PyObject *self, PyObject *args) | ||
| 207 | + unicode_copycharacters(PyObject *self, PyObject *args, PyObject *kwargs) | ||
| 194 | 208 | { | |
| 209 | + static char *kwlist[] = {"to", "to_start", "from", "from_start", | ||
| 210 | + "howmany", "incref", NULL}; | ||
| 195 | 211 | PyObject *from, *to, *to_copy; | |
| 196 | 212 | Py_ssize_t from_start, to_start, how_many, copied; | |
| 213 | + int incref = 0; | ||
| 197 | 214 | ||
| 198 | - if (!PyArg_ParseTuple(args, "UnOnn", &to, &to_start, | ||
| 199 | - &from, &from_start, &how_many)) { | ||
| 215 | + if (!PyArg_ParseTupleAndKeywords(args, kwargs, | ||
| 216 | + "UnOnn|p", kwlist, | ||
| 217 | + &to, &to_start, &from, &from_start, | ||
| 218 | + &how_many, &incref)) { | ||
| 200 | 219 | return NULL; | |
| 201 | 220 | } | |
| 202 | 221 | ||
@@ -210,8 +229,15 @@ unicode_copycharacters(PyObject *self, PyObject *args) | |||
| 210 | 229 | return NULL; | |
| 211 | 230 | } | |
| 212 | 231 | ||
| 232 | + if (incref) { | ||
| 233 | + Py_INCREF(to_copy); | ||
| 234 | + } | ||
| 213 | 235 | copied = PyUnicode_CopyCharacters(to_copy, to_start, from, | |
| 214 | 236 | from_start, how_many); | |
| 237 | + if (incref) { | ||
| 238 | + Py_DECREF(to_copy); | ||
| 239 | + } | ||
| 240 | + | ||
| 215 | 241 | if (copied == -1 && PyErr_Occurred()) { | |
| 216 | 242 | Py_DECREF(to_copy); | |
| 217 | 243 | return NULL; | |
@@ -856,12 +882,12 @@ static PyType_Spec Writer_spec = { | |||
| 856 | 882 | ||
| 857 | 883 | static PyMethodDef TestMethods[] = { | |
| 858 | 884 | {"unicode_new", unicode_new, METH_VARARGS}, | |
| 859 | - {"unicode_fill", unicode_fill, METH_VARARGS}, | ||
| 885 | + {"unicode_fill", _PyCFunction_CAST(unicode_fill), METH_VARARGS | METH_KEYWORDS}, | ||
| 860 | 886 | {"unicode_fromkindanddata", unicode_fromkindanddata, METH_VARARGS}, | |
| 861 | 887 | {"unicode_asucs4", unicode_asucs4, METH_VARARGS}, | |
| 862 | 888 | {"unicode_asucs4copy", unicode_asucs4copy, METH_VARARGS}, | |
| 863 | 889 | {"unicode_asutf8", unicode_asutf8, METH_VARARGS}, | |
| 864 | - {"unicode_copycharacters", unicode_copycharacters, METH_VARARGS}, | ||
| 890 | + {"unicode_copycharacters", _PyCFunction_CAST(unicode_copycharacters), METH_VARARGS | METH_KEYWORDS}, | ||
| 865 | 891 | {"unicode_GET_CACHED_HASH", unicode_GET_CACHED_HASH, METH_O}, | |
| 866 | 892 | {"test_py_identifier", test_py_identifier, METH_NOARGS}, | |
| 867 | 893 | {"corrupt_unicode", corrupt_unicode, METH_VARARGS}, | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -140,14 +140,19 @@ unicode_copy(PyObject *unicode) | |||
| 140 | 140 | ||
| 141 | 141 | /* Test PyUnicode_WriteChar() */ | |
| 142 | 142 | static PyObject * | |
| 143 | - unicode_writechar(PyObject *self, PyObject *args) | ||
| 143 | + unicode_writechar(PyObject *self, PyObject *args, PyObject *kwargs) | ||
| 144 | 144 | { | |
| 145 | + static char *kwlist[] = {"to", "index", "character", "incref", NULL}; | ||
| 145 | 146 | PyObject *to, *to_copy; | |
| 146 | 147 | Py_ssize_t index; | |
| 147 | 148 | unsigned int character; | |
| 148 | 149 | int result; | |
| 150 | + int incref = 0; | ||
| 149 | 151 | ||
| 150 | - if (!PyArg_ParseTuple(args, "OnI", &to, &index, &character)) { | ||
| 152 | + if (!PyArg_ParseTupleAndKeywords(args, kwargs, | ||
| 153 | + "OnI|p", kwlist, | ||
| 154 | + &to, &index, &character, &incref)) | ||
| 155 | + { | ||
| 151 | 156 | return NULL; | |
| 152 | 157 | } | |
| 153 | 158 | ||
@@ -156,7 +161,14 @@ unicode_writechar(PyObject *self, PyObject *args) | |||
| 156 | 161 | return NULL; | |
| 157 | 162 | } | |
| 158 | 163 | ||
| 164 | + if (incref) { | ||
| 165 | + Py_INCREF(to_copy); | ||
| 166 | + } | ||
| 159 | 167 | result = PyUnicode_WriteChar(to_copy, index, (Py_UCS4)character); | |
| 168 | + if (incref) { | ||
| 169 | + Py_DECREF(to_copy); | ||
| 170 | + } | ||
| 171 | + | ||
| 160 | 172 | if (result == -1 && PyErr_Occurred()) { | |
| 161 | 173 | Py_DECREF(to_copy); | |
| 162 | 174 | return NULL; | |
@@ -1915,7 +1927,7 @@ static PyMethodDef TestMethods[] = { | |||
| 1915 | 1927 | test_unicode_compare_with_ascii, METH_NOARGS}, | |
| 1916 | 1928 | {"test_string_from_format", test_string_from_format, METH_NOARGS}, | |
| 1917 | 1929 | {"test_widechar", test_widechar, METH_NOARGS}, | |
| 1918 | - {"unicode_writechar", unicode_writechar, METH_VARARGS}, | ||
| 1930 | + {"unicode_writechar", _PyCFunction_CAST(unicode_writechar), METH_VARARGS | METH_KEYWORDS}, | ||
| 1919 | 1931 | {"unicode_resize", unicode_resize, METH_VARARGS}, | |
| 1920 | 1932 | {"unicode_resize_null", unicode_resize_null, METH_VARARGS}, | |
| 1921 | 1933 | {"unicode_append", unicode_append, METH_VARARGS}, | |
| Back | FazBrowse Home | New Git URL |
0 commit comments