Skip to content

Commit 762b906

Browse files
committed
Fix ParserError on subsecond precision beyond nanoseconds
The pure-Python datetime parsers (used as a fallback when the compiled Rust extension is unavailable, e.g. PENDULUM_EXTENSIONS=0, and for the "common" format that has no Rust equivalent) capped the fractional seconds group at 9 digits (\\d{1,9}). A 10th digit made the whole string fail to match, raising ParserError instead of parsing it. The value is already truncated to 6 digits (microseconds) a few lines below, so the cap served no purpose beyond rejecting otherwise valid input. Python's datetime.fromisoformat() has no such limit and just truncates, which is the behavior restored here by widening both regexes to \\d+. Fixes ParserError on inputs like: pendulum.parse("2001-01-01T12:34:56.1234567890Z") pendulum.parse("2016/10/06 12:34:56.1234567890")
1 parent d3f44aa commit 762b906

4 files changed

Lines changed: 52 additions & 2 deletions

File tree

‎src/pendulum/parsing/__init__.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -49,7 +49,7 @@
4949
# Subsecond part (optional)
5050
" (?P<subsecondsection>"
5151
" (?:[.|,])" # Subsecond separator (optional)
52-
r" (?P<subsecond>\d{1,9})" # Subsecond
52+
r" (?P<subsecond>\d+)" # Subsecond (any number of digits; truncated to microseconds below)
5353
" )?"
5454
")?"
5555
"$",

‎src/pendulum/parsing/iso8601.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -49,7 +49,7 @@
4949
# Subsecond part (optional)
5050
" (?P<subsecondsection>"
5151
" (?:[.,])" # Subsecond separator (optional)
52-
r" (?P<subsecond>\d{1,9})" # Subsecond
52+
r" (?P<subsecond>\d+)" # Subsecond (any number of digits; truncated to microseconds below)
5353
" )?"
5454
# Timezone offset
5555
" (?P<tz>"

‎tests/parsing/test_parse_iso8601.py‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -214,3 +214,18 @@ def test_parse_iso8601_duration_invalid():
214214
# Must include at least one element
215215
with pytest.raises(ValueError):
216216
parse_iso8601("P")
217+
218+
219+
def test_parse_iso8601_subsecond_beyond_nanoseconds_pure_python():
220+
# Regression test for the pure-Python ISO 8601 parser (used as a fallback
221+
# when the compiled extension is unavailable): a subsecond part longer
222+
# than 9 digits used to raise a ParserError instead of being truncated to
223+
# microsecond precision like `datetime.fromisoformat` does. Imported
224+
# directly so the check runs regardless of whether the Rust extension is
225+
# built, since `pendulum.parsing.parse_iso8601` prefers the extension
226+
# when it's available.
227+
from pendulum.parsing.iso8601 import parse_iso8601 as py_parse_iso8601
228+
229+
parsed = py_parse_iso8601("2016-10-06T12:34:56.1234567890123+05:30")
230+
231+
assert parsed == datetime(2016, 10, 6, 12, 34, 56, 123456, FixedTimezone(19800))

‎tests/parsing/test_parsing.py‎

Lines changed: 35 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -156,6 +156,41 @@ def test_rfc_3339_extended_nanoseconds():
156156
assert parsed.utcoffset().total_seconds() == 19800
157157

158158

159+
def test_rfc_3339_extended_beyond_nanoseconds():
160+
# A subsecond part longer than 9 digits (i.e. finer than nanoseconds) is
161+
# unusual but not invalid ISO 8601: it should still be accepted and
162+
# truncated to microsecond precision, the same as `datetime.fromisoformat`
163+
# does, instead of raising a ParserError.
164+
text = "2016-10-06T12:34:56.1234567890123+05:30"
165+
166+
parsed = parse(text)
167+
168+
assert parsed.year == 2016
169+
assert parsed.month == 10
170+
assert parsed.day == 6
171+
assert parsed.hour == 12
172+
assert parsed.minute == 34
173+
assert parsed.second == 56
174+
assert parsed.microsecond == 123456
175+
assert parsed.utcoffset().total_seconds() == 19800
176+
177+
178+
def test_common_format_extended_beyond_nanoseconds():
179+
# Same as `test_rfc_3339_extended_beyond_nanoseconds()` but for the
180+
# "common" datetime format (handled separately from the ISO 8601 parser).
181+
text = "2016/10/06 12:34:56.1234567890123"
182+
183+
parsed = parse(text)
184+
185+
assert parsed.year == 2016
186+
assert parsed.month == 10
187+
assert parsed.day == 6
188+
assert parsed.hour == 12
189+
assert parsed.minute == 34
190+
assert parsed.second == 56
191+
assert parsed.microsecond == 123456
192+
193+
159194
def test_iso_8601_date():
160195
text = "2012"
161196

0 commit comments

Comments
 (0)