FazBrowse GitHub Viewer | Trending |
URL:
| Home
Tools: [Download Repo ZIP]   [Original HTTPS Page]

gh-152204: Validate date and time fields in `_pydatetime.date{time}.fromisoformat` by tonghuaroot · Pull Request #152205 · python/cpython · GitHub

/ cpython Public
30 changes: 20 additions & 10 deletions Lib/_pydatetime.py
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -355,19 +355,30 @@ def _find_isoformat_datetime_separator(dtstr):
return 8


def _read_isoformat_component(s, n):
# The caller has verified the string is ASCII, so isdigit() matches only
# the ASCII digits accepted by the C parser.
if len(s) != n or not s.isdigit():
raise ValueError("Invalid isoformat string")
return int(s)


def _parse_isoformat_date(dtstr):
# It is assumed that this is an ASCII-only string of lengths 7, 8 or 10,
# see the comment on Modules/_datetimemodule.c:_find_isoformat_datetime_separator
if len(dtstr) not in (7, 8, 10):
raise ValueError("Invalid isoformat string")
year = int(dtstr[0:4])
if not dtstr.isascii():
raise ValueError("Invalid isoformat string")

year = _read_isoformat_component(dtstr[0:4], 4)
has_sep = dtstr[4] == '-'

pos = 4 + has_sep
if dtstr[pos:pos + 1] == "W":
# YYYY-?Www-?D?
pos += 1
weekno = int(dtstr[pos:pos + 2])
weekno = _read_isoformat_component(dtstr[pos:pos + 2], 2)
pos += 2

dayno = 1
Expand All @@ -377,17 +388,17 @@ def _parse_isoformat_date(dtstr):

pos += has_sep

dayno = int(dtstr[pos:pos + 1])
dayno = _read_isoformat_component(dtstr[pos:pos + 1], 1)

return list(_isoweek_to_gregorian(year, weekno, dayno))
else:
month = int(dtstr[pos:pos + 2])
month = _read_isoformat_component(dtstr[pos:pos + 2], 2)
pos += 2
if (dtstr[pos:pos + 1] == "-") != has_sep:
raise ValueError("Inconsistent use of dash separator")

pos += has_sep
day = int(dtstr[pos:pos + 2])
day = _read_isoformat_component(dtstr[pos:pos + 2], 2)

return [year, month, day]

Expand All @@ -402,10 +413,7 @@ def _parse_hh_mm_ss_ff(tstr):
time_comps = [0, 0, 0, 0]
pos = 0
for comp in range(0, 3):
if (len_str - pos) < 2:
raise ValueError("Incomplete time component")

time_comps[comp] = int(tstr[pos:pos+2])
time_comps[comp] = _read_isoformat_component(tstr[pos:pos+2], 2)

pos += 2
next_char = tstr[pos:pos+1]
Expand All @@ -426,7 +434,7 @@ def _parse_hh_mm_ss_ff(tstr):
raise ValueError("Invalid microsecond separator")
else:
pos += 1
if not all(map(_is_ascii_digit, tstr[pos:])):
if not tstr[pos:].isdigit():
raise ValueError("Non-digit values in fraction")

len_remainder = len_str - pos
Expand All @@ -447,6 +455,8 @@ def _parse_isoformat_time(tstr):
len_str = len(tstr)
if len_str < 2:
raise ValueError("Isoformat time too short")
if not tstr.isascii():
raise ValueError("Invalid isoformat string")

# This is equivalent to re.search('[+-Z]', tstr), but faster
tz_pos = (tstr.find('-') + 1 or tstr.find('+') + 1 or tstr.find('Z') + 1)
Expand Down
19 changes: 18 additions & 1 deletion Lib/test/datetimetester.py
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -2106,7 +2106,15 @@ def test_fromisoformat_fails(self):
'10000-W25-1', # Invalid year
'2020-W25-0', # Invalid day-of-week
'2020-W25-8', # Invalid day-of-week
'٢025-03-09' # Unicode characters
# gh-152204: each fixed-width field must be exactly N ASCII digits
'2020+12', # '+' in a basic-format field
'2020 12', # space in a basic-format field
'+020-06-15', # leading sign in the year
'202012+9', # '+' in the day field
'2020-W 5', # space in the week number
'2020061', # 7 chars: day slice reads a 1-character tail
'2020-W2', # 1-digit week number
'٢025-03-09', # Unicode characters
'2009\ud80002\ud80028', # Separators are surrogate codepoints
]

Expand Down Expand Up @@ -3758,6 +3766,15 @@ def test_fromisoformat_fails_datetime(self):
'2009-04-19T12:30:45-00:90:00', # Time zone field out from range
'2009-04-19T12:30:45-00:00:90', # Time zone field out from range
'2020-2020', # Ambiguous 9-char date portion
# gh-152204: each time field must be exactly N ASCII digits
'2020-12-12T0٥:02:03', # Unicode digit in the hour
'2020-12-12T01:0٥:03', # Unicode digit in the minute
'2020-12-12T01:02:0٥', # Unicode digit in the second
'2020-12-12T01:02:03.٥', # Unicode digit in the fraction
'2020-12-12T01:02:03.4_6', # underscore in the fraction
'2020-12-12T01:02:03+0٥:00', # Unicode digit in the tz hour
'2020-12-12T01:02:03+01:0٥', # Unicode digit in the tz minute
'20201212T0102٣٤', # Unicode digits in the basic-format time
'2009-04-19T12:30:45.+05:00', # Empty fraction before offset
'2009-04-19T12:30:45.-05:00', # Empty fraction before offset
'2009-04-19T12:30:45.Z', # Empty fraction before Z
Expand Down
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
Fix the pure-Python implementations of :meth:`datetime.date.fromisoformat`,
:meth:`datetime.time.fromisoformat` and :meth:`datetime.datetime.fromisoformat`
silently accepting some malformed ISO 8601 strings, such as non-ASCII digits or
a sign in a fixed-width field (for example ``'2020+12'`` or ``'20201212T0102٣٤'``).
Each field is now required to be exactly *N* ASCII digits, matching the C
implementation.
Loading

Back | FazBrowse Home | New Git URL