| FazBrowse GitHub Viewer | Trending | | Home |
| Tools: [Download Repo ZIP] [Original HTTPS Page] |
1 parent c6cab4c commit 0a07cd9
23 files changed
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -869,6 +869,11 @@ def test_bug691291(self): | |||
| 869 | 869 | with reader: | |
| 870 | 870 | self.assertEqual(reader.read(), s1) | |
| 871 | 871 | ||
| 872 | + # TODO: RUSTPYTHON | ||
| 873 | + @unittest.expectedFailure | ||
| 874 | + def test_incremental_surrogatepass(self): | ||
| 875 | + super().test_incremental_surrogatepass() | ||
| 876 | + | ||
| 872 | 877 | class UTF16LETest(ReadTest, unittest.TestCase): | |
| 873 | 878 | encoding = "utf-16-le" | |
| 874 | 879 | ill_formed_sequence = b"\x80\xdc" | |
@@ -917,6 +922,11 @@ def test_nonbmp(self): | |||
| 917 | 922 | self.assertEqual(b'\x00\xd8\x03\xde'.decode(self.encoding), | |
| 918 | 923 | "\U00010203") | |
| 919 | 924 | ||
| 925 | + # TODO: RUSTPYTHON | ||
| 926 | + @unittest.expectedFailure | ||
| 927 | + def test_incremental_surrogatepass(self): | ||
| 928 | + super().test_incremental_surrogatepass() | ||
| 929 | + | ||
| 920 | 930 | class UTF16BETest(ReadTest, unittest.TestCase): | |
| 921 | 931 | encoding = "utf-16-be" | |
| 922 | 932 | ill_formed_sequence = b"\xdc\x80" | |
@@ -965,6 +975,11 @@ def test_nonbmp(self): | |||
| 965 | 975 | self.assertEqual(b'\xd8\x00\xde\x03'.decode(self.encoding), | |
| 966 | 976 | "\U00010203") | |
| 967 | 977 | ||
| 978 | + # TODO: RUSTPYTHON | ||
| 979 | + @unittest.expectedFailure | ||
| 980 | + def test_incremental_surrogatepass(self): | ||
| 981 | + super().test_incremental_surrogatepass() | ||
| 982 | + | ||
| 968 | 983 | class UTF8Test(ReadTest, unittest.TestCase): | |
| 969 | 984 | encoding = "utf-8" | |
| 970 | 985 | ill_formed_sequence = b"\xed\xb2\x80" | |
@@ -998,8 +1013,6 @@ def test_decoder_state(self): | |||
| 998 | 1013 | self.check_state_handling_decode(self.encoding, | |
| 999 | 1014 | u, u.encode(self.encoding)) | |
| 1000 | 1015 | ||
| 1001 | - # TODO: RUSTPYTHON | ||
| 1002 | - @unittest.expectedFailure | ||
| 1003 | 1016 | def test_decode_error(self): | |
| 1004 | 1017 | for data, error_handler, expected in ( | |
| 1005 | 1018 | (b'[\x80\xff]', 'ignore', '[]'), | |
@@ -1026,8 +1039,6 @@ def test_lone_surrogates(self): | |||
| 1026 | 1039 | exc = cm.exception | |
| 1027 | 1040 | self.assertEqual(exc.object[exc.start:exc.end], '\uD800\uDFFF') | |
| 1028 | 1041 | ||
| 1029 | - # TODO: RUSTPYTHON | ||
| 1030 | - @unittest.expectedFailure | ||
| 1031 | 1042 | def test_surrogatepass_handler(self): | |
| 1032 | 1043 | self.assertEqual("abc\ud800def".encode(self.encoding, "surrogatepass"), | |
| 1033 | 1044 | self.BOM + b"abc\xed\xa0\x80def") | |
@@ -2884,8 +2895,6 @@ def test_escape_encode(self): | |||
| 2884 | 2895 | ||
| 2885 | 2896 | class SurrogateEscapeTest(unittest.TestCase): | |
| 2886 | 2897 | ||
| 2887 | - # TODO: RUSTPYTHON | ||
| 2888 | - @unittest.expectedFailure | ||
| 2889 | 2898 | def test_utf8(self): | |
| 2890 | 2899 | # Bad byte | |
| 2891 | 2900 | self.assertEqual(b"foo\x80bar".decode("utf-8", "surrogateescape"), | |
@@ -2898,8 +2907,6 @@ def test_utf8(self): | |||
| 2898 | 2907 | self.assertEqual("\udced\udcb0\udc80".encode("utf-8", "surrogateescape"), | |
| 2899 | 2908 | b"\xed\xb0\x80") | |
| 2900 | 2909 | ||
| 2901 | - # TODO: RUSTPYTHON | ||
| 2902 | - @unittest.expectedFailure | ||
| 2903 | 2910 | def test_ascii(self): | |
| 2904 | 2911 | # bad byte | |
| 2905 | 2912 | self.assertEqual(b"foo\x80bar".decode("ascii", "surrogateescape"), | |
@@ -2916,8 +2923,6 @@ def test_charmap(self): | |||
| 2916 | 2923 | self.assertEqual("foo\udca5bar".encode("iso-8859-3", "surrogateescape"), | |
| 2917 | 2924 | b"foo\xa5bar") | |
| 2918 | 2925 | ||
| 2919 | - # TODO: RUSTPYTHON | ||
| 2920 | - @unittest.expectedFailure | ||
| 2921 | 2926 | def test_latin1(self): | |
| 2922 | 2927 | # Issue6373 | |
| 2923 | 2928 | self.assertEqual("\udce4\udceb\udcef\udcf6\udcfc".encode("latin-1", "surrogateescape"), | |
@@ -3561,8 +3566,6 @@ class ASCIITest(unittest.TestCase): | |||
| 3561 | 3566 | def test_encode(self): | |
| 3562 | 3567 | self.assertEqual('abc123'.encode('ascii'), b'abc123') | |
| 3563 | 3568 | ||
| 3564 | - # TODO: RUSTPYTHON | ||
| 3565 | - @unittest.expectedFailure | ||
| 3566 | 3569 | def test_encode_error(self): | |
| 3567 | 3570 | for data, error_handler, expected in ( | |
| 3568 | 3571 | ('[\x80\xff\u20ac]', 'ignore', b'[]'), | |
@@ -3585,8 +3588,6 @@ def test_encode_surrogateescape_error(self): | |||
| 3585 | 3588 | def test_decode(self): | |
| 3586 | 3589 | self.assertEqual(b'abc'.decode('ascii'), 'abc') | |
| 3587 | 3590 | ||
| 3588 | - # TODO: RUSTPYTHON | ||
| 3589 | - @unittest.expectedFailure | ||
| 3590 | 3591 | def test_decode_error(self): | |
| 3591 | 3592 | for data, error_handler, expected in ( | |
| 3592 | 3593 | (b'[\x80\xff]', 'ignore', '[]'), | |
@@ -3609,8 +3610,6 @@ def test_encode(self): | |||
| 3609 | 3610 | with self.subTest(data=data, expected=expected): | |
| 3610 | 3611 | self.assertEqual(data.encode('latin1'), expected) | |
| 3611 | 3612 | ||
| 3612 | - # TODO: RUSTPYTHON | ||
| 3613 | - @unittest.expectedFailure | ||
| 3614 | 3613 | def test_encode_errors(self): | |
| 3615 | 3614 | for data, error_handler, expected in ( | |
| 3616 | 3615 | ('[\u20ac\udc80]', 'ignore', b'[]'), | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -86,8 +86,6 @@ def test_scanstring(self): | |||
| 86 | 86 | scanstring('["Bad value", truth]', 2, True), | |
| 87 | 87 | ('Bad value', 12)) | |
| 88 | 88 | ||
| 89 | - # TODO: RUSTPYTHON | ||
| 90 | - @unittest.expectedFailure | ||
| 91 | 89 | def test_surrogates(self): | |
| 92 | 90 | scanstring = self.json.decoder.scanstring | |
| 93 | 91 | def assertScan(given, expect): | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -945,15 +945,13 @@ def test_leak(self): | |||
| 945 | 945 | """) | |
| 946 | 946 | self.check_leak(code, 'file descriptors') | |
| 947 | 947 | ||
| 948 | - @unittest.expectedFailureIfWindows('TODO: RUSTPYTHON Windows') | ||
| 949 | 948 | def test_list_tests(self): | |
| 950 | 949 | # test --list-tests | |
| 951 | 950 | tests = [self.create_test() for i in range(5)] | |
| 952 | 951 | output = self.run_tests('--list-tests', *tests) | |
| 953 | 952 | self.assertEqual(output.rstrip().splitlines(), | |
| 954 | 953 | tests) | |
| 955 | 954 | ||
| 956 | - @unittest.expectedFailureIfWindows('TODO: RUSTPYTHON Windows') | ||
| 957 | 955 | def test_list_cases(self): | |
| 958 | 956 | # test --list-cases | |
| 959 | 957 | code = textwrap.dedent(""" | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -6,8 +6,6 @@ | |||
| 6 | 6 | from stringprep import * | |
| 7 | 7 | ||
| 8 | 8 | class StringprepTests(unittest.TestCase): | |
| 9 | - # TODO: RUSTPYTHON | ||
| 10 | - @unittest.expectedFailure | ||
| 11 | 9 | def test(self): | |
| 12 | 10 | self.assertTrue(in_table_a1("\u0221")) | |
| 13 | 11 | self.assertFalse(in_table_a1("\u0222")) | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -1198,8 +1198,6 @@ def test_universal_newlines_communicate_encodings(self): | |||
| 1198 | 1198 | stdout, stderr = popen.communicate(input='') | |
| 1199 | 1199 | self.assertEqual(stdout, '1\n2\n3\n4') | |
| 1200 | 1200 | ||
| 1201 | - # TODO: RUSTPYTHON | ||
| 1202 | - @unittest.expectedFailure | ||
| 1203 | 1201 | def test_communicate_errors(self): | |
| 1204 | 1202 | for errors, expected in [ | |
| 1205 | 1203 | ('ignore', ''), | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -2086,11 +2086,6 @@ class UstarUnicodeTest(UnicodeTest, unittest.TestCase): | |||
| 2086 | 2086 | ||
| 2087 | 2087 | format = tarfile.USTAR_FORMAT | |
| 2088 | 2088 | ||
| 2089 | - # TODO: RUSTPYTHON | ||
| 2090 | - @unittest.expectedFailure | ||
| 2091 | - def test_uname_unicode(self): | ||
| 2092 | - super().test_uname_unicode() | ||
| 2093 | - | ||
| 2094 | 2089 | # Test whether the utf-8 encoded version of a filename exceeds the 100 | |
| 2095 | 2090 | # bytes name field limit (every occurrence of '\xff' will be expanded to 2 | |
| 2096 | 2091 | # bytes). | |
@@ -2170,13 +2165,6 @@ class GNUUnicodeTest(UnicodeTest, unittest.TestCase): | |||
| 2170 | 2165 | ||
| 2171 | 2166 | format = tarfile.GNU_FORMAT | |
| 2172 | 2167 | ||
| 2173 | - # TODO: RUSTPYTHON | ||
| 2174 | - @unittest.expectedFailure | ||
| 2175 | - def test_uname_unicode(self): | ||
| 2176 | - super().test_uname_unicode() | ||
| 2177 | - | ||
| 2178 | - # TODO: RUSTPYTHON | ||
| 2179 | - @unittest.expectedFailure | ||
| 2180 | 2168 | def test_bad_pax_header(self): | |
| 2181 | 2169 | # Test for issue #8633. GNU tar <= 1.23 creates raw binary fields | |
| 2182 | 2170 | # without a hdrcharset=BINARY header. | |
@@ -2198,8 +2186,6 @@ class PAXUnicodeTest(UnicodeTest, unittest.TestCase): | |||
| 2198 | 2186 | # PAX_FORMAT ignores encoding in write mode. | |
| 2199 | 2187 | test_unicode_filename_error = None | |
| 2200 | 2188 | ||
| 2201 | - # TODO: RUSTPYTHON | ||
| 2202 | - @unittest.expectedFailure | ||
| 2203 | 2189 | def test_binary_header(self): | |
| 2204 | 2190 | # Test a POSIX.1-2008 compatible header with a hdrcharset=BINARY field. | |
| 2205 | 2191 | for encoding, name in ( | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -608,8 +608,6 @@ def test_bytes_comparison(self): | |||
| 608 | 608 | self.assertEqual('abc' == bytearray(b'abc'), False) | |
| 609 | 609 | self.assertEqual('abc' != bytearray(b'abc'), True) | |
| 610 | 610 | ||
| 611 | - # TODO: RUSTPYTHON | ||
| 612 | - @unittest.expectedFailure | ||
| 613 | 611 | def test_comparison(self): | |
| 614 | 612 | # Comparisons: | |
| 615 | 613 | self.assertEqual('abc', 'abc') | |
@@ -830,8 +828,6 @@ def test_isidentifier_legacy(self): | |||
| 830 | 828 | warnings.simplefilter('ignore', DeprecationWarning) | |
| 831 | 829 | self.assertTrue(_testcapi.unicode_legacy_string(u).isidentifier()) | |
| 832 | 830 | ||
| 833 | - # TODO: RUSTPYTHON | ||
| 834 | - @unittest.expectedFailure | ||
| 835 | 831 | def test_isprintable(self): | |
| 836 | 832 | self.assertTrue("".isprintable()) | |
| 837 | 833 | self.assertTrue(" ".isprintable()) | |
@@ -847,8 +843,6 @@ def test_isprintable(self): | |||
| 847 | 843 | self.assertTrue('\U0001F46F'.isprintable()) | |
| 848 | 844 | self.assertFalse('\U000E0020'.isprintable()) | |
| 849 | 845 | ||
| 850 | - # TODO: RUSTPYTHON | ||
| 851 | - @unittest.expectedFailure | ||
| 852 | 846 | def test_surrogates(self): | |
| 853 | 847 | for s in ('a\uD800b\uDFFF', 'a\uDFFFb\uD800', | |
| 854 | 848 | 'a\uD800b\uDFFFa', 'a\uDFFFb\uD800a'): | |
@@ -1827,8 +1821,6 @@ def test_codecs_utf7(self): | |||
| 1827 | 1821 | 'ill-formed sequence'): | |
| 1828 | 1822 | b'+@'.decode('utf-7') | |
| 1829 | 1823 | ||
| 1830 | - # TODO: RUSTPYTHON | ||
| 1831 | - @unittest.expectedFailure | ||
| 1832 | 1824 | def test_codecs_utf8(self): | |
| 1833 | 1825 | self.assertEqual(''.encode('utf-8'), b'') | |
| 1834 | 1826 | self.assertEqual('\u20ac'.encode('utf-8'), b'\xe2\x82\xac') | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -53,17 +53,13 @@ def __rmod__(self, other): | |||
| 53 | 53 | str3 = ustr3('TEST') | |
| 54 | 54 | self.assertEqual(fmt2 % str3, 'value is TEST') | |
| 55 | 55 | ||
| 56 | - # TODO: RUSTPYTHON | ||
| 57 | - @unittest.expectedFailure | ||
| 58 | 56 | def test_encode_default_args(self): | |
| 59 | 57 | self.checkequal(b'hello', 'hello', 'encode') | |
| 60 | 58 | # Check that encoding defaults to utf-8 | |
| 61 | 59 | self.checkequal(b'\xf0\xa3\x91\x96', '\U00023456', 'encode') | |
| 62 | 60 | # Check that errors defaults to 'strict' | |
| 63 | 61 | self.checkraises(UnicodeError, '\ud800', 'encode') | |
| 64 | 62 | ||
| 65 | - # TODO: RUSTPYTHON | ||
| 66 | - @unittest.expectedFailure | ||
| 67 | 63 | def test_encode_explicit_none_args(self): | |
| 68 | 64 | self.checkequal(b'hello', 'hello', 'encode', None, None) | |
| 69 | 65 | # Check that encoding defaults to utf-8 | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -730,6 +730,7 @@ def testTraceback(self): | |||
| 730 | 730 | ||
| 731 | 731 | @unittest.skipIf(os_helper.TESTFN_UNENCODABLE is None, | |
| 732 | 732 | "need an unencodable filename") | |
| 733 | + @unittest.expectedFailureIfWindows("TODO: RUSTPYTHON") | ||
| 733 | 734 | def testUnencodable(self): | |
| 734 | 735 | filename = os_helper.TESTFN_UNENCODABLE + ".zip" | |
| 735 | 736 | self.addCleanup(os_helper.unlink, filename) | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -122,18 +122,18 @@ impl CodePoint { | |||
| 122 | 122 | ||
| 123 | 123 | /// Returns the numeric value of the code point if it is a leading surrogate. | |
| 124 | 124 | #[inline] | |
| 125 | - pub fn to_lead_surrogate(self) -> Option<u16> { | ||
| 125 | + pub fn to_lead_surrogate(self) -> Option<LeadSurrogate> { | ||
| 126 | 126 | match self.value { | |
| 127 | - lead @ 0xD800..=0xDBFF => Some(lead as u16), | ||
| 127 | + lead @ 0xD800..=0xDBFF => Some(LeadSurrogate(lead as u16)), | ||
| 128 | 128 | _ => None, | |
| 129 | 129 | } | |
| 130 | 130 | } | |
| 131 | 131 | ||
| 132 | 132 | /// Returns the numeric value of the code point if it is a trailing surrogate. | |
| 133 | 133 | #[inline] | |
| 134 | - pub fn to_trail_surrogate(self) -> Option<u16> { | ||
| 134 | + pub fn to_trail_surrogate(self) -> Option<TrailSurrogate> { | ||
| 135 | 135 | match self.value { | |
| 136 | - trail @ 0xDC00..=0xDFFF => Some(trail as u16), | ||
| 136 | + trail @ 0xDC00..=0xDFFF => Some(TrailSurrogate(trail as u16)), | ||
| 137 | 137 | _ => None, | |
| 138 | 138 | } | |
| 139 | 139 | } | |
@@ -216,6 +216,18 @@ impl PartialEq<CodePoint> for char { | |||
| 216 | 216 | } | |
| 217 | 217 | } | |
| 218 | 218 | ||
| 219 | + #[derive(Clone, Copy)] | ||
| 220 | + pub struct LeadSurrogate(u16); | ||
| 221 | + | ||
| 222 | + #[derive(Clone, Copy)] | ||
| 223 | + pub struct TrailSurrogate(u16); | ||
| 224 | + | ||
| 225 | + impl LeadSurrogate { | ||
| 226 | + pub fn merge(self, trail: TrailSurrogate) -> char { | ||
| 227 | + decode_surrogate_pair(self.0, trail.0) | ||
| 228 | + } | ||
| 229 | + } | ||
| 230 | + | ||
| 219 | 231 | /// An owned, growable string of well-formed WTF-8 data. | |
| 220 | 232 | /// | |
| 221 | 233 | /// Similar to `String`, but can additionally contain surrogate code points | |
@@ -291,6 +303,14 @@ impl Wtf8Buf { | |||
| 291 | 303 | Wtf8Buf { bytes: value } | |
| 292 | 304 | } | |
| 293 | 305 | ||
| 306 | + /// Create a WTF-8 string from a WTF-8 byte vec. | ||
| 307 | + pub fn from_bytes(value: Vec<u8>) -> Result<Self, Vec<u8>> { | ||
| 308 | + match Wtf8::from_bytes(&value) { | ||
| 309 | + Some(_) => Ok(unsafe { Self::from_bytes_unchecked(value) }), | ||
| 310 | + None => Err(value), | ||
| 311 | + } | ||
| 312 | + } | ||
| 313 | + | ||
| 294 | 314 | /// Creates a WTF-8 string from a UTF-8 `String`. | |
| 295 | 315 | /// | |
| 296 | 316 | /// This takes ownership of the `String` and does not copy. | |
@@ -750,15 +770,10 @@ impl Wtf8 { | |||
| 750 | 770 | } | |
| 751 | 771 | ||
| 752 | 772 | fn decode_surrogate(b: &[u8]) -> Option<CodePoint> { | |
| 753 | - let [a, b, c, ..] = *b else { return None }; | ||
| 754 | - if (a & 0xf0) == 0xe0 && (b & 0xc0) == 0x80 && (c & 0xc0) == 0x80 { | ||
| 755 | - // it's a three-byte code | ||
| 756 | - let c = ((a as u32 & 0x0f) << 12) + ((b as u32 & 0x3f) << 6) + (c as u32 & 0x3f); | ||
| 757 | - let 0xD800..=0xDFFF = c else { return None }; | ||
| 758 | - Some(CodePoint { value: c }) | ||
| 759 | - } else { | ||
| 760 | - None | ||
| 761 | - } | ||
| 773 | + let [0xed, b2 @ (0xa0..), b3, ..] = *b else { | ||
| 774 | + return None; | ||
| 775 | + }; | ||
| 776 | + Some(decode_surrogate(b2, b3).into()) | ||
| 762 | 777 | } | |
| 763 | 778 | ||
| 764 | 779 | /// Returns the length, in WTF-8 bytes. | |
@@ -914,14 +929,6 @@ impl Wtf8 { | |||
| 914 | 929 | } | |
| 915 | 930 | } | |
| 916 | 931 | ||
| 917 | - #[inline] | ||
| 918 | - fn final_lead_surrogate(&self) -> Option<u16> { | ||
| 919 | - match self.bytes { | ||
| 920 | - [.., 0xED, b2 @ 0xA0..=0xAF, b3] => Some(decode_surrogate(b2, b3)), | ||
| 921 | - _ => None, | ||
| 922 | - } | ||
| 923 | - } | ||
| 924 | - | ||
| 925 | 932 | pub fn is_code_point_boundary(&self, index: usize) -> bool { | |
| 926 | 933 | is_code_point_boundary(self, index) | |
| 927 | 934 | } | |
@@ -1222,6 +1229,12 @@ fn decode_surrogate(second_byte: u8, third_byte: u8) -> u16 { | |||
| 1222 | 1229 | 0xD800 | (second_byte as u16 & 0x3F) << 6 | third_byte as u16 & 0x3F | |
| 1223 | 1230 | } | |
| 1224 | 1231 | ||
| 1232 | + #[inline] | ||
| 1233 | + fn decode_surrogate_pair(lead: u16, trail: u16) -> char { | ||
| 1234 | + let code_point = 0x10000 + ((((lead - 0xD800) as u32) << 10) | (trail - 0xDC00) as u32); | ||
| 1235 | + unsafe { char::from_u32_unchecked(code_point) } | ||
| 1236 | + } | ||
| 1237 | + | ||
| 1225 | 1238 | /// Copied from str::is_char_boundary | |
| 1226 | 1239 | #[inline] | |
| 1227 | 1240 | fn is_code_point_boundary(slice: &Wtf8, index: usize) -> bool { | |
| Back | FazBrowse Home | New Git URL |
0 commit comments