| FazBrowse GitHub Viewer | Trending | | Home |
| Tools: [Download Repo ZIP] [Original HTTPS Page] |
1 parent 425065b commit f48a96a
4 files changed
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -159,6 +159,10 @@ or on combining URL components into a URL string. | |||
| 159 | 159 | ParseResult(scheme='http', netloc='www.cwi.nl:80', path='/%7Eguido/Python.html', | |
| 160 | 160 | params='', query='', fragment='') | |
| 161 | 161 | ||
| 162 | + .. warning:: | ||
| 163 | + | ||
| 164 | + :func:`urlparse` does not perform validation. See :ref:`URL parsing | ||
| 165 | + security <url-parsing-security>` for details. | ||
| 162 | 166 | ||
| 163 | 167 | .. versionchanged:: 3.2 | |
| 164 | 168 | Added IPv6 URL parsing capabilities. | |
@@ -324,8 +328,14 @@ or on combining URL components into a URL string. | |||
| 324 | 328 | ``#``, ``@``, or ``:`` will raise a :exc:`ValueError`. If the URL is | |
| 325 | 329 | decomposed before parsing, no error will be raised. | |
| 326 | 330 | ||
| 327 | - Following the `WHATWG spec`_ that updates RFC 3986, ASCII newline | ||
| 328 | - ``\n``, ``\r`` and tab ``\t`` characters are stripped from the URL. | ||
| 331 | + Following some of the `WHATWG spec`_ that updates RFC 3986, leading C0 | ||
| 332 | + control and space characters are stripped from the URL. ``\n``, | ||
| 333 | + ``\r`` and tab ``\t`` characters are removed from the URL at any position. | ||
| 334 | + | ||
| 335 | + .. warning:: | ||
| 336 | + | ||
| 337 | + :func:`urlsplit` does not perform validation. See :ref:`URL parsing | ||
| 338 | + security <url-parsing-security>` for details. | ||
| 329 | 339 | ||
| 330 | 340 | .. versionchanged:: 3.6 | |
| 331 | 341 | Out-of-range port numbers now raise :exc:`ValueError`, instead of | |
@@ -338,6 +348,9 @@ or on combining URL components into a URL string. | |||
| 338 | 348 | .. versionchanged:: 3.10 | |
| 339 | 349 | ASCII newline and tab characters are stripped from the URL. | |
| 340 | 350 | ||
| 351 | + .. versionchanged:: 3.10.12 | ||
| 352 | + Leading WHATWG C0 control and space characters are stripped from the URL. | ||
| 353 | + | ||
| 341 | 354 | .. _WHATWG spec: https://url.spec.whatwg.org/#concept-basic-url-parser | |
| 342 | 355 | ||
| 343 | 356 | .. function:: urlunsplit(parts) | |
@@ -414,6 +427,27 @@ or on combining URL components into a URL string. | |||
| 414 | 427 | or ``scheme://host/path``). If *url* is not a wrapped URL, it is returned | |
| 415 | 428 | without changes. | |
| 416 | 429 | ||
| 430 | + .. _url-parsing-security: | ||
| 431 | + | ||
| 432 | + URL parsing security | ||
| 433 | + -------------------- | ||
| 434 | + | ||
| 435 | + The :func:`urlsplit` and :func:`urlparse` APIs do not perform **validation** of | ||
| 436 | + inputs. They may not raise errors on inputs that other applications consider | ||
| 437 | + invalid. They may also succeed on some inputs that might not be considered | ||
| 438 | + URLs elsewhere. Their purpose is for practical functionality rather than | ||
| 439 | + purity. | ||
| 440 | + | ||
| 441 | + Instead of raising an exception on unusual input, they may instead return some | ||
| 442 | + component parts as empty strings. Or components may contain more than perhaps | ||
| 443 | + they should. | ||
| 444 | + | ||
| 445 | + We recommend that users of these APIs where the values may be used anywhere | ||
| 446 | + with security implications code defensively. Do some verification within your | ||
| 447 | + code before trusting a returned component part. Does that ``scheme`` make | ||
| 448 | + sense? Is that a sensible ``path``? Is there anything strange about that | ||
| 449 | + ``hostname``? etc. | ||
| 450 | + | ||
| 417 | 451 | .. _parsing-ascii-encoded-bytes: | |
| 418 | 452 | ||
| 419 | 453 | Parsing ASCII Encoded Bytes | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -649,14 +649,73 @@ def test_urlsplit_remove_unsafe_bytes(self): | |||
| 649 | 649 | self.assertEqual(p.scheme, "http") | |
| 650 | 650 | self.assertEqual(p.geturl(), "http://www.python.org/javascript:alert('msg')/?query=something#fragment") | |
| 651 | 651 | ||
| 652 | + def test_urlsplit_strip_url(self): | ||
| 653 | + noise = bytes(range(0, 0x20 + 1)) | ||
| 654 | + base_url = "http://User:Pass@www.python.org:080/doc/?query=yes#frag" | ||
| 655 | + | ||
| 656 | + url = noise.decode("utf-8") + base_url | ||
| 657 | + p = urllib.parse.urlsplit(url) | ||
| 658 | + self.assertEqual(p.scheme, "http") | ||
| 659 | + self.assertEqual(p.netloc, "User:Pass@www.python.org:080") | ||
| 660 | + self.assertEqual(p.path, "/doc/") | ||
| 661 | + self.assertEqual(p.query, "query=yes") | ||
| 662 | + self.assertEqual(p.fragment, "frag") | ||
| 663 | + self.assertEqual(p.username, "User") | ||
| 664 | + self.assertEqual(p.password, "Pass") | ||
| 665 | + self.assertEqual(p.hostname, "www.python.org") | ||
| 666 | + self.assertEqual(p.port, 80) | ||
| 667 | + self.assertEqual(p.geturl(), base_url) | ||
| 668 | + | ||
| 669 | + url = noise + base_url.encode("utf-8") | ||
| 670 | + p = urllib.parse.urlsplit(url) | ||
| 671 | + self.assertEqual(p.scheme, b"http") | ||
| 672 | + self.assertEqual(p.netloc, b"User:Pass@www.python.org:080") | ||
| 673 | + self.assertEqual(p.path, b"/doc/") | ||
| 674 | + self.assertEqual(p.query, b"query=yes") | ||
| 675 | + self.assertEqual(p.fragment, b"frag") | ||
| 676 | + self.assertEqual(p.username, b"User") | ||
| 677 | + self.assertEqual(p.password, b"Pass") | ||
| 678 | + self.assertEqual(p.hostname, b"www.python.org") | ||
| 679 | + self.assertEqual(p.port, 80) | ||
| 680 | + self.assertEqual(p.geturl(), base_url.encode("utf-8")) | ||
| 681 | + | ||
| 682 | + # Test that trailing space is preserved as some applications rely on | ||
| 683 | + # this within query strings. | ||
| 684 | + query_spaces_url = "https://www.python.org:88/doc/?query= " | ||
| 685 | + p = urllib.parse.urlsplit(noise.decode("utf-8") + query_spaces_url) | ||
| 686 | + self.assertEqual(p.scheme, "https") | ||
| 687 | + self.assertEqual(p.netloc, "www.python.org:88") | ||
| 688 | + self.assertEqual(p.path, "/doc/") | ||
| 689 | + self.assertEqual(p.query, "query= ") | ||
| 690 | + self.assertEqual(p.port, 88) | ||
| 691 | + self.assertEqual(p.geturl(), query_spaces_url) | ||
| 692 | + | ||
| 693 | + p = urllib.parse.urlsplit("www.pypi.org ") | ||
| 694 | + # That "hostname" gets considered a "path" due to the | ||
| 695 | + # trailing space and our existing logic... YUCK... | ||
| 696 | + # and re-assembles via geturl aka unurlsplit into the original. | ||
| 697 | + # django.core.validators.URLValidator (at least through v3.2) relies on | ||
| 698 | + # this, for better or worse, to catch it in a ValidationError via its | ||
| 699 | + # regular expressions. | ||
| 700 | + # Here we test the basic round trip concept of such a trailing space. | ||
| 701 | + self.assertEqual(urllib.parse.urlunsplit(p), "www.pypi.org ") | ||
| 702 | + | ||
| 703 | + # with scheme as cache-key | ||
| 704 | + url = "//www.python.org/" | ||
| 705 | + scheme = noise.decode("utf-8") + "https" + noise.decode("utf-8") | ||
| 706 | + for _ in range(2): | ||
| 707 | + p = urllib.parse.urlsplit(url, scheme=scheme) | ||
| 708 | + self.assertEqual(p.scheme, "https") | ||
| 709 | + self.assertEqual(p.geturl(), "https://www.python.org/") | ||
| 710 | + | ||
| 652 | 711 | def test_attributes_bad_port(self): | |
| 653 | 712 | """Check handling of invalid ports.""" | |
| 654 | 713 | for bytes in (False, True): | |
| 655 | 714 | for parse in (urllib.parse.urlsplit, urllib.parse.urlparse): | |
| 656 | 715 | for port in ("foo", "1.5", "-1", "0x10", "-0", "1_1", " 1", "1 ", "६"): | |
| 657 | 716 | with self.subTest(bytes=bytes, parse=parse, port=port): | |
| 658 | 717 | netloc = "www.example.net:" + port | |
| 659 | - url = "http://" + netloc | ||
| 718 | + url = "http://" + netloc + "/" | ||
| 660 | 719 | if bytes: | |
| 661 | 720 | if netloc.isascii() and port.isascii(): | |
| 662 | 721 | netloc = netloc.encode("ascii") | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -25,6 +25,10 @@ | |||
| 25 | 25 | scenarios for parsing, and for backward compatibility purposes, some | |
| 26 | 26 | parsing quirks from older RFCs are retained. The testcases in | |
| 27 | 27 | test_urlparse.py provides a good indicator of parsing behavior. | |
| 28 | + | ||
| 29 | + The WHATWG URL Parser spec should also be considered. We are not compliant with | ||
| 30 | + it either due to existing user code API behavior expectations (Hyrum's Law). | ||
| 31 | + It serves as a useful guide when making changes. | ||
| 28 | 32 | """ | |
| 29 | 33 | ||
| 30 | 34 | import re | |
@@ -78,6 +82,10 @@ | |||
| 78 | 82 | '0123456789' | |
| 79 | 83 | '+-.') | |
| 80 | 84 | ||
| 85 | + # Leading and trailing C0 control and space to be stripped per WHATWG spec. | ||
| 86 | + # == "".join([chr(i) for i in range(0, 0x20 + 1)]) | ||
| 87 | + _WHATWG_C0_CONTROL_OR_SPACE = '\x00\x01\x02\x03\x04\x05\x06\x07\x08\t\n\x0b\x0c\r\x0e\x0f\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1a\x1b\x1c\x1d\x1e\x1f ' | ||
| 88 | + | ||
| 81 | 89 | # Unsafe bytes to be removed per WHATWG spec | |
| 82 | 90 | _UNSAFE_URL_BYTES_TO_REMOVE = ['\t', '\r', '\n'] | |
| 83 | 91 | ||
@@ -455,6 +463,10 @@ def urlsplit(url, scheme='', allow_fragments=True): | |||
| 455 | 463 | """ | |
| 456 | 464 | ||
| 457 | 465 | url, scheme, _coerce_result = _coerce_args(url, scheme) | |
| 466 | + # Only lstrip url as some applications rely on preserving trailing space. | ||
| 467 | + # (https://url.spec.whatwg.org/#concept-basic-url-parser would strip both) | ||
| 468 | + url = url.lstrip(_WHATWG_C0_CONTROL_OR_SPACE) | ||
| 469 | + scheme = scheme.strip(_WHATWG_C0_CONTROL_OR_SPACE) | ||
| 458 | 470 | ||
| 459 | 471 | for b in _UNSAFE_URL_BYTES_TO_REMOVE: | |
| 460 | 472 | url = url.replace(b, "") | |
| Original file line number | Diff line number | Diff line change | |
|---|---|---|---|
@@ -0,0 +1,3 @@ | |||
| 1 | + :func:`urllib.parse.urlsplit` now strips leading C0 control and space | ||
| 2 | + characters following the specification for URLs defined by WHATWG in | ||
| 3 | + response to CVE-2023-24329. Patch by Illia Volochii. | ||
| Back | FazBrowse Home | New Git URL |
0 commit comments