codekingpro/portable-devtools
114k
1# -*- coding: utf-8 -*-2"""3This module offers a generic date/time string parser which is able to parse4most known formats to represent a date and/or time.5 6This module attempts to be forgiving with regards to unlikely input formats,7returning a datetime object even for dates which are ambiguous. If an element8of a date/time stamp is omitted, the following rules are applied:9 10- If AM or PM is left unspecified, a 24-hour clock is assumed, however, an hour11 on a 12-hour clock (``0 <= hour <= 12``) *must* be specified if AM or PM is12 specified.13- If a time zone is omitted, a timezone-naive datetime is returned.14 15If any other elements are missing, they are taken from the16:class:`datetime.datetime` object passed to the parameter ``default``. If this17results in a day number exceeding the valid number of days per month, the18value falls back to the end of the month.19 20Additional resources about date/time string formats can be found below:21 22- `A summary of the international standard date and time notation23 <https://www.cl.cam.ac.uk/~mgk25/iso-time.html>`_24- `W3C Date and Time Formats <https://www.w3.org/TR/NOTE-datetime>`_25- `Time Formats (Planetary Rings Node) <https://pds-rings.seti.org:443/tools/time_formats.html>`_26- `CPAN ParseDate module27 <https://metacpan.org/pod/release/MUIR/Time-modules-2013.0912/lib/Time/ParseDate.pm>`_28- `Java SimpleDateFormat Class29 <https://docs.oracle.com/javase/6/docs/api/java/text/SimpleDateFormat.html>`_30"""31from __future__ import unicode_literals32 33import datetime34import re35import string36import time37import warnings38 39from calendar import monthrange40from io import StringIO41 42import six43from six import integer_types, text_type44 45from decimal import Decimal46 47from warnings import warn48 49from .. import relativedelta50from .. import tz51 52__all__ = ["parse", "parserinfo", "ParserError"]53 54 55# TODO: pandas.core.tools.datetimes imports this explicitly. Might be worth56# making public and/or figuring out if there is something we can57# take off their plate.58class _timelex(object):59 # Fractional seconds are sometimes split by a comma60 _split_decimal = re.compile("([.,])")61 62 def __init__(self, instream):63 if isinstance(instream, (bytes, bytearray)):64 instream = instream.decode()65 66 if isinstance(instream, text_type):67 instream = StringIO(instream)68 elif getattr(instream, 'read', None) is None:69 raise TypeError('Parser must be a string or character stream, not '70 '{itype}'.format(itype=instream.__class__.__name__))71 72 self.instream = instream73 self.charstack = []74 self.tokenstack = []75 self.eof = False76 77 def get_token(self):78 """79 This function breaks the time string into lexical units (tokens), which80 can be parsed by the parser. Lexical units are demarcated by changes in81 the character set, so any continuous string of letters is considered82 one unit, any continuous string of numbers is considered one unit.83 84 The main complication arises from the fact that dots ('.') can be used85 both as separators (e.g. "Sep.20.2009") or decimal points (e.g.86 "4:30:21.447"). As such, it is necessary to read the full context of87 any dot-separated strings before breaking it into tokens; as such, this88 function maintains a "token stack", for when the ambiguous context89 demands that multiple tokens be parsed at once.90 """91 if self.tokenstack:92 return self.tokenstack.pop(0)93 94 seenletters = False95 token = None96 state = None97 98 while not self.eof:99 # We only realize that we've reached the end of a token when we100 # find a character that's not part of the current token - since101 # that character may be part of the next token, it's stored in the102 # charstack.103 if self.charstack:104 nextchar = self.charstack.pop(0)105 else:106 nextchar = self.instream.read(1)107 while nextchar == '\x00':108 nextchar = self.instream.read(1)109 110 if not nextchar:111 self.eof = True112 break113 elif not state:114 # First character of the token - determines if we're starting115 # to parse a word, a number or something else.116 token = nextchar117 if self.isword(nextchar):118 state = 'a'119 elif self.isnum(nextchar):120 state = '0'121 elif self.isspace(nextchar):122 token = ' '123 break # emit token124 else:125 break # emit token126 elif state == 'a':127 # If we've already started reading a word, we keep reading128 # letters until we find something that's not part of a word.129 seenletters = True130 if self.isword(nextchar):131 token += nextchar132 elif nextchar == '.':133 token += nextchar134 state = 'a.'135 else:136 self.charstack.append(nextchar)137 break # emit token138 elif state == '0':139 # If we've already started reading a number, we keep reading140 # numbers until we find something that doesn't fit.141 if self.isnum(nextchar):142 token += nextchar143 elif nextchar == '.' or (nextchar == ',' and len(token) >= 2):144 token += nextchar145 state = '0.'146 else:147 self.charstack.append(nextchar)148 break # emit token149 elif state == 'a.':150 # If we've seen some letters and a dot separator, continue151 # parsing, and the tokens will be broken up later.152 seenletters = True153 if nextchar == '.' or self.isword(nextchar):154 token += nextchar155 elif self.isnum(nextchar) and token[-1] == '.':156 token += nextchar157 state = '0.'158 else:159 self.charstack.append(nextchar)160 break # emit token161 elif state == '0.':162 # If we've seen at least one dot separator, keep going, we'll163 # break up the tokens later.164 if nextchar == '.' or self.isnum(nextchar):165 token += nextchar166 elif self.isword(nextchar) and token[-1] == '.':167 token += nextchar168 state = 'a.'169 else:170 self.charstack.append(nextchar)171 break # emit token172 173 if (state in ('a.', '0.') and (seenletters or token.count('.') > 1 or174 token[-1] in '.,')):175 l = self._split_decimal.split(token)176 token = l[0]177 for tok in l[1:]:178 if tok:179 self.tokenstack.append(tok)180 181 if state == '0.' and token.count('.') == 0:182 token = token.replace(',', '.')183 184 return token185 186 def __iter__(self):187 return self188 189 def __next__(self):190 token = self.get_token()191 if token is None:192 raise StopIteration193 194 return token195 196 def next(self):197 return self.__next__() # Python 2.x support198 199 @classmethod200 def split(cls, s):201 return list(cls(s))202 203 @classmethod204 def isword(cls, nextchar):205 """ Whether or not the next character is part of a word """206 return nextchar.isalpha()207 208 @classmethod209 def isnum(cls, nextchar):210 """ Whether the next character is part of a number """211 return nextchar.isdigit()212 213 @classmethod214 def isspace(cls, nextchar):215 """ Whether the next character is whitespace """216 return nextchar.isspace()217 218 219class _resultbase(object):220 221 def __init__(self):222 for attr in self.__slots__:223 setattr(self, attr, None)224 225 def _repr(self, classname):226 l = []227 for attr in self.__slots__:228 value = getattr(self, attr)229 if value is not None:230 l.append("%s=%s" % (attr, repr(value)))231 return "%s(%s)" % (classname, ", ".join(l))232 233 def __len__(self):234 return (sum(getattr(self, attr) is not None235 for attr in self.__slots__))236 237 def __repr__(self):238 return self._repr(self.__class__.__name__)239 240 241class parserinfo(object):242 """243 Class which handles what inputs are accepted. Subclass this to customize244 the language and acceptable values for each parameter.245 246 :param dayfirst:247 Whether to interpret the first value in an ambiguous 3-integer date248 (e.g. 01/05/09) as the day (``True``) or month (``False``). If249 ``yearfirst`` is set to ``True``, this distinguishes between YDM250 and YMD. Default is ``False``.251 252 :param yearfirst:253 Whether to interpret the first value in an ambiguous 3-integer date254 (e.g. 01/05/09) as the year. If ``True``, the first number is taken255 to be the year, otherwise the last number is taken to be the year.256 Default is ``False``.257 """258 259 # m from a.m/p.m, t from ISO T separator260 JUMP = [" ", ".", ",", ";", "-", "/", "'",261 "at", "on", "and", "ad", "m", "t", "of",262 "st", "nd", "rd", "th"]263 264 WEEKDAYS = [("Mon", "Monday"),265 ("Tue", "Tuesday"), # TODO: "Tues"266 ("Wed", "Wednesday"),267 ("Thu", "Thursday"), # TODO: "Thurs"268 ("Fri", "Friday"),269 ("Sat", "Saturday"),270 ("Sun", "Sunday")]271 MONTHS = [("Jan", "January"),272 ("Feb", "February"), # TODO: "Febr"273 ("Mar", "March"),274 ("Apr", "April"),275 ("May", "May"),276 ("Jun", "June"),277 ("Jul", "July"),278 ("Aug", "August"),279 ("Sep", "Sept", "September"),280 ("Oct", "October"),281 ("Nov", "November"),282 ("Dec", "December")]283 HMS = [("h", "hour", "hours"),284 ("m", "minute", "minutes"),285 ("s", "second", "seconds")]286 AMPM = [("am", "a"),287 ("pm", "p")]288 UTCZONE = ["UTC", "GMT", "Z", "z"]289 PERTAIN = ["of"]290 TZOFFSET = {}291 # TODO: ERA = ["AD", "BC", "CE", "BCE", "Stardate",292 # "Anno Domini", "Year of Our Lord"]293 294 def __init__(self, dayfirst=False, yearfirst=False):295 self._jump = self._convert(self.JUMP)296 self._weekdays = self._convert(self.WEEKDAYS)297 self._months = self._convert(self.MONTHS)298 self._hms = self._convert(self.HMS)299 self._ampm = self._convert(self.AMPM)300 self._utczone = self._convert(self.UTCZONE)301 self._pertain = self._convert(self.PERTAIN)302 303 self.dayfirst = dayfirst304 self.yearfirst = yearfirst305 306 self._year = time.localtime().tm_year307 self._century = self._year // 100 * 100308 309 def _convert(self, lst):310 dct = {}311 for i, v in enumerate(lst):312 if isinstance(v, tuple):313 for v in v:314 dct[v.lower()] = i315 else:316 dct[v.lower()] = i317 return dct318 319 def jump(self, name):320 return name.lower() in self._jump321 322 def weekday(self, name):323 try:324 return self._weekdays[name.lower()]325 except KeyError:326 pass327 return None328 329 def month(self, name):330 try:331 return self._months[name.lower()] + 1332 except KeyError:333 pass334 return None335 336 def hms(self, name):337 try:338 return self._hms[name.lower()]339 except KeyError:340 return None341 342 def ampm(self, name):343 try:344 return self._ampm[name.lower()]345 except KeyError:346 return None347 348 def pertain(self, name):349 return name.lower() in self._pertain350 351 def utczone(self, name):352 return name.lower() in self._utczone353 354 def tzoffset(self, name):355 if name in self._utczone:356 return 0357 358 return self.TZOFFSET.get(name)359 360 def convertyear(self, year, century_specified=False):361 """362 Converts two-digit years to year within [-50, 49]363 range of self._year (current local time)364 """365 366 # Function contract is that the year is always positive367 assert year >= 0368 369 if year < 100 and not century_specified:370 # assume current century to start371 year += self._century372 373 if year >= self._year + 50: # if too far in future374 year -= 100375 elif year < self._year - 50: # if too far in past376 year += 100377 378 return year379 380 def validate(self, res):381 # move to info382 if res.year is not None:383 res.year = self.convertyear(res.year, res.century_specified)384 385 if ((res.tzoffset == 0 and not res.tzname) or386 (res.tzname == 'Z' or res.tzname == 'z')):387 res.tzname = "UTC"388 res.tzoffset = 0389 elif res.tzoffset != 0 and res.tzname and self.utczone(res.tzname):390 res.tzoffset = 0391 return True392 393 394class _ymd(list):395 def __init__(self, *args, **kwargs):396 super(self.__class__, self).__init__(*args, **kwargs)397 self.century_specified = False398 self.dstridx = None399 self.mstridx = None400 self.ystridx = None401 402 @property403 def has_year(self):404 return self.ystridx is not None405 406 @property407 def has_month(self):408 return self.mstridx is not None409 410 @property411 def has_day(self):412 return self.dstridx is not None413 414 def could_be_day(self, value):415 if self.has_day:416 return False417 elif not self.has_month:418 return 1 <= value <= 31419 elif not self.has_year:420 # Be permissive, assume leap year421 month = self[self.mstridx]422 return 1 <= value <= monthrange(2000, month)[1]423 else:424 month = self[self.mstridx]425 year = self[self.ystridx]426 return 1 <= value <= monthrange(year, month)[1]427 428 def append(self, val, label=None):429 if hasattr(val, '__len__'):430 if val.isdigit() and len(val) > 2:431 self.century_specified = True432 if label not in [None, 'Y']: # pragma: no cover433 raise ValueError(label)434 label = 'Y'435 elif val > 100:436 self.century_specified = True437 if label not in [None, 'Y']: # pragma: no cover438 raise ValueError(label)439 label = 'Y'440 441 super(self.__class__, self).append(int(val))442 443 if label == 'M':444 if self.has_month:445 raise ValueError('Month is already set')446 self.mstridx = len(self) - 1447 elif label == 'D':448 if self.has_day:449 raise ValueError('Day is already set')450 self.dstridx = len(self) - 1451 elif label == 'Y':452 if self.has_year:453 raise ValueError('Year is already set')454 self.ystridx = len(self) - 1455 456 def _resolve_from_stridxs(self, strids):457 """458 Try to resolve the identities of year/month/day elements using459 ystridx, mstridx, and dstridx, if enough of these are specified.460 """461 if len(self) == 3 and len(strids) == 2:462 # we can back out the remaining stridx value463 missing = [x for x in range(3) if x not in strids.values()]464 key = [x for x in ['y', 'm', 'd'] if x not in strids]465 assert len(missing) == len(key) == 1466 key = key[0]467 val = missing[0]468 strids[key] = val469 470 assert len(self) == len(strids) # otherwise this should not be called471 out = {key: self[strids[key]] for key in strids}472 return (out.get('y'), out.get('m'), out.get('d'))473 474 def resolve_ymd(self, yearfirst, dayfirst):475 len_ymd = len(self)476 year, month, day = (None, None, None)477 478 strids = (('y', self.ystridx),479 ('m', self.mstridx),480 ('d', self.dstridx))481 482 strids = {key: val for key, val in strids if val is not None}483 if (len(self) == len(strids) > 0 or484 (len(self) == 3 and len(strids) == 2)):485 return self._resolve_from_stridxs(strids)486 487 mstridx = self.mstridx488 489 if len_ymd > 3:490 raise ValueError("More than three YMD values")491 elif len_ymd == 1 or (mstridx is not None and len_ymd == 2):492 # One member, or two members with a month string493 if mstridx is not None:494 month = self[mstridx]495 # since mstridx is 0 or 1, self[mstridx-1] always496 # looks up the other element497 other = self[mstridx - 1]498 else:499 other = self[0]500 501 if len_ymd > 1 or mstridx is None:502 if other > 31:503 year = other504 else:505 day = other506 507 elif len_ymd == 2:508 # Two members with numbers509 if self[0] > 31:510 # 99-01511 year, month = self512 elif self[1] > 31:513 # 01-99514 month, year = self515 elif dayfirst and self[1] <= 12:516 # 13-01517 day, month = self518 else:519 # 01-13520 month, day = self521 522 elif len_ymd == 3:523 # Three members524 if mstridx == 0:525 if self[1] > 31:526 # Apr-2003-25527 month, year, day = self528 else:529 month, day, year = self530 elif mstridx == 1:531 if self[0] > 31 or (yearfirst and self[2] <= 31):532 # 99-Jan-01533 year, month, day = self534 else:535 # 01-Jan-01536 # Give precedence to day-first, since537 # two-digit years is usually hand-written.538 day, month, year = self539 540 elif mstridx == 2:541 # WTF!?542 if self[1] > 31:543 # 01-99-Jan544 day, year, month = self545 else:546 # 99-01-Jan547 year, day, month = self548 549 else:550 if (self[0] > 31 or551 self.ystridx == 0 or552 (yearfirst and self[1] <= 12 and self[2] <= 31)):553 # 99-01-01554 if dayfirst and self[2] <= 12:555 year, day, month = self556 else:557 year, month, day = self558 elif self[0] > 12 or (dayfirst and self[1] <= 12):559 # 13-01-01560 day, month, year = self561 else:562 # 01-13-01563 month, day, year = self564 565 return year, month, day566 567 568class parser(object):569 def __init__(self, info=None):570 self.info = info or parserinfo()571 572 def parse(self, timestr, default=None,573 ignoretz=False, tzinfos=None, **kwargs):574 """575 Parse the date/time string into a :class:`datetime.datetime` object.576 577 :param timestr:578 Any date/time string using the supported formats.579 580 :param default:581 The default datetime object, if this is a datetime object and not582 ``None``, elements specified in ``timestr`` replace elements in the583 default object.584 585 :param ignoretz:586 If set ``True``, time zones in parsed strings are ignored and a587 naive :class:`datetime.datetime` object is returned.588 589 :param tzinfos:590 Additional time zone names / aliases which may be present in the591 string. This argument maps time zone names (and optionally offsets592 from those time zones) to time zones. This parameter can be a593 dictionary with timezone aliases mapping time zone names to time594 zones or a function taking two parameters (``tzname`` and595 ``tzoffset``) and returning a time zone.596 597 The timezones to which the names are mapped can be an integer598 offset from UTC in seconds or a :class:`tzinfo` object.599 600 .. doctest::601 :options: +NORMALIZE_WHITESPACE602 603 >>> from dateutil.parser import parse604 >>> from dateutil.tz import gettz605 >>> tzinfos = {"BRST": -7200, "CST": gettz("America/Chicago")}606 >>> parse("2012-01-19 17:21:00 BRST", tzinfos=tzinfos)607 datetime.datetime(2012, 1, 19, 17, 21, tzinfo=tzoffset(u'BRST', -7200))608 >>> parse("2012-01-19 17:21:00 CST", tzinfos=tzinfos)609 datetime.datetime(2012, 1, 19, 17, 21,610 tzinfo=tzfile('/usr/share/zoneinfo/America/Chicago'))611 612 This parameter is ignored if ``ignoretz`` is set.613 614 :param \\*\\*kwargs:615 Keyword arguments as passed to ``_parse()``.616 617 :return:618 Returns a :class:`datetime.datetime` object or, if the619 ``fuzzy_with_tokens`` option is ``True``, returns a tuple, the620 first element being a :class:`datetime.datetime` object, the second621 a tuple containing the fuzzy tokens.622 623 :raises ParserError:624 Raised for invalid or unknown string format, if the provided625 :class:`tzinfo` is not in a valid format, or if an invalid date626 would be created.627 628 :raises TypeError:629 Raised for non-string or character stream input.630 631 :raises OverflowError:632 Raised if the parsed date exceeds the largest valid C integer on633 your system.634 """635 636 if default is None:637 default = datetime.datetime.now().replace(hour=0, minute=0,638 second=0, microsecond=0)639 640 res, skipped_tokens = self._parse(timestr, **kwargs)641 642 if res is None:643 raise ParserError("Unknown string format: %s", timestr)644 645 if len(res) == 0:646 raise ParserError("String does not contain a date: %s", timestr)647 648 try:649 ret = self._build_naive(res, default)650 except ValueError as e:651 six.raise_from(ParserError(str(e) + ": %s", timestr), e)652 653 if not ignoretz:654 ret = self._build_tzaware(ret, res, tzinfos)655 656 if kwargs.get('fuzzy_with_tokens', False):657 return ret, skipped_tokens658 else:659 return ret660 661 class _result(_resultbase):662 __slots__ = ["year", "month", "day", "weekday",663 "hour", "minute", "second", "microsecond",664 "tzname", "tzoffset", "ampm","any_unused_tokens"]665 666 def _parse(self, timestr, dayfirst=None, yearfirst=None, fuzzy=False,667 fuzzy_with_tokens=False):668 """669 Private method which performs the heavy lifting of parsing, called from670 ``parse()``, which passes on its ``kwargs`` to this function.671 672 :param timestr:673 The string to parse.674 675 :param dayfirst:676 Whether to interpret the first value in an ambiguous 3-integer date677 (e.g. 01/05/09) as the day (``True``) or month (``False``). If678 ``yearfirst`` is set to ``True``, this distinguishes between YDM679 and YMD. If set to ``None``, this value is retrieved from the680 current :class:`parserinfo` object (which itself defaults to681 ``False``).682 683 :param yearfirst:684 Whether to interpret the first value in an ambiguous 3-integer date685 (e.g. 01/05/09) as the year. If ``True``, the first number is taken686 to be the year, otherwise the last number is taken to be the year.687 If this is set to ``None``, the value is retrieved from the current688 :class:`parserinfo` object (which itself defaults to ``False``).689 690 :param fuzzy:691 Whether to allow fuzzy parsing, allowing for string like "Today is692 January 1, 2047 at 8:21:00AM".693 694 :param fuzzy_with_tokens:695 If ``True``, ``fuzzy`` is automatically set to True, and the parser696 will return a tuple where the first element is the parsed697 :class:`datetime.datetime` datetimestamp and the second element is698 a tuple containing the portions of the string which were ignored:699 700 .. doctest::701 702 >>> from dateutil.parser import parse703 >>> parse("Today is January 1, 2047 at 8:21:00AM", fuzzy_with_tokens=True)704 (datetime.datetime(2047, 1, 1, 8, 21), (u'Today is ', u' ', u'at '))705 706 """707 if fuzzy_with_tokens:708 fuzzy = True709 710 info = self.info711 712 if dayfirst is None:713 dayfirst = info.dayfirst714 715 if yearfirst is None:716 yearfirst = info.yearfirst717 718 res = self._result()719 l = _timelex.split(timestr) # Splits the timestr into tokens720 721 skipped_idxs = []722 723 # year/month/day list724 ymd = _ymd()725 726 len_l = len(l)727 i = 0728 try:729 while i < len_l:730 731 # Check if it's a number732 value_repr = l[i]733 try:734 value = float(value_repr)735 except ValueError:736 value = None737 738 if value is not None:739 # Numeric token740 i = self._parse_numeric_token(l, i, info, ymd, res, fuzzy)741 742 # Check weekday743 elif info.weekday(l[i]) is not None:744 value = info.weekday(l[i])745 res.weekday = value746 747 # Check month name748 elif info.month(l[i]) is not None:749 value = info.month(l[i])750 ymd.append(value, 'M')751 752 if i + 1 < len_l:753 if l[i + 1] in ('-', '/'):754 # Jan-01[-99]755 sep = l[i + 1]756 ymd.append(l[i + 2])757 758 if i + 3 < len_l and l[i + 3] == sep:759 # Jan-01-99760 ymd.append(l[i + 4])761 i += 2762 763 i += 2764 765 elif (i + 4 < len_l and l[i + 1] == l[i + 3] == ' ' and766 info.pertain(l[i + 2])):767 # Jan of 01768 # In this case, 01 is clearly year769 if l[i + 4].isdigit():770 # Convert it here to become unambiguous771 value = int(l[i + 4])772 year = str(info.convertyear(value))773 ymd.append(year, 'Y')774 else:775 # Wrong guess776 pass777 # TODO: not hit in tests778 i += 4779 780 # Check am/pm781 elif info.ampm(l[i]) is not None:782 value = info.ampm(l[i])783 val_is_ampm = self._ampm_valid(res.hour, res.ampm, fuzzy)784 785 if val_is_ampm:786 res.hour = self._adjust_ampm(res.hour, value)787 res.ampm = value788 789 elif fuzzy:790 skipped_idxs.append(i)791 792 # Check for a timezone name793 elif self._could_be_tzname(res.hour, res.tzname, res.tzoffset, l[i]):794 res.tzname = l[i]795 res.tzoffset = info.tzoffset(res.tzname)796 797 # Check for something like GMT+3, or BRST+3. Notice798 # that it doesn't mean "I am 3 hours after GMT", but799 # "my time +3 is GMT". If found, we reverse the800 # logic so that timezone parsing code will get it801 # right.802 if i + 1 < len_l and l[i + 1] in ('+', '-'):803 l[i + 1] = ('+', '-')[l[i + 1] == '+']804 res.tzoffset = None805 if info.utczone(res.tzname):806 # With something like GMT+3, the timezone807 # is *not* GMT.808 res.tzname = None809 810 # Check for a numbered timezone811 elif res.hour is not None and l[i] in ('+', '-'):812 signal = (-1, 1)[l[i] == '+']813 len_li = len(l[i + 1])814 815 # TODO: check that l[i + 1] is integer?816 if len_li == 4:817 # -0300818 hour_offset = int(l[i + 1][:2])819 min_offset = int(l[i + 1][2:])820 elif i + 2 < len_l and l[i + 2] == ':':821 # -03:00822 hour_offset = int(l[i + 1])823 min_offset = int(l[i + 3]) # TODO: Check that l[i+3] is minute-like?824 i += 2825 elif len_li <= 2:826 # -[0]3827 hour_offset = int(l[i + 1][:2])828 min_offset = 0829 else:830 raise ValueError(timestr)831 832 res.tzoffset = signal * (hour_offset * 3600 + min_offset * 60)833 834 # Look for a timezone name between parenthesis835 if (i + 5 < len_l and836 info.jump(l[i + 2]) and l[i + 3] == '(' and837 l[i + 5] == ')' and838 3 <= len(l[i + 4]) and839 self._could_be_tzname(res.hour, res.tzname,840 None, l[i + 4])):841 # -0300 (BRST)842 res.tzname = l[i + 4]843 i += 4844 845 i += 1846 847 # Check jumps848 elif not (info.jump(l[i]) or fuzzy):849 raise ValueError(timestr)850 851 else:852 skipped_idxs.append(i)853 i += 1854 855 # Process year/month/day856 year, month, day = ymd.resolve_ymd(yearfirst, dayfirst)857 858 res.century_specified = ymd.century_specified859 res.year = year860 res.month = month861 res.day = day862 863 except (IndexError, ValueError):864 return None, None865 866 if not info.validate(res):867 return None, None868 869 if fuzzy_with_tokens:870 skipped_tokens = self._recombine_skipped(l, skipped_idxs)871 return res, tuple(skipped_tokens)872 else:873 return res, None874 875 def _parse_numeric_token(self, tokens, idx, info, ymd, res, fuzzy):876 # Token is a number877 value_repr = tokens[idx]878 try:879 value = self._to_decimal(value_repr)880 except Exception as e:881 six.raise_from(ValueError('Unknown numeric token'), e)882 883 len_li = len(value_repr)884 885 len_l = len(tokens)886 887 if (len(ymd) == 3 and len_li in (2, 4) and888 res.hour is None and889 (idx + 1 >= len_l or890 (tokens[idx + 1] != ':' and891 info.hms(tokens[idx + 1]) is None))):892 # 19990101T23[59]893 s = tokens[idx]894 res.hour = int(s[:2])895 896 if len_li == 4:897 res.minute = int(s[2:])898 899 elif len_li == 6 or (len_li > 6 and tokens[idx].find('.') == 6):900 # YYMMDD or HHMMSS[.ss]901 s = tokens[idx]902 903 if not ymd and '.' not in tokens[idx]:904 ymd.append(s[:2])905 ymd.append(s[2:4])906 ymd.append(s[4:])907 else:908 # 19990101T235959[.59]909 910 # TODO: Check if res attributes already set.911 res.hour = int(s[:2])912 res.minute = int(s[2:4])913 res.second, res.microsecond = self._parsems(s[4:])914 915 elif len_li in (8, 12, 14):916 # YYYYMMDD917 s = tokens[idx]918 ymd.append(s[:4], 'Y')919 ymd.append(s[4:6])920 ymd.append(s[6:8])921 922 if len_li > 8:923 res.hour = int(s[8:10])924 res.minute = int(s[10:12])925 926 if len_li > 12:927 res.second = int(s[12:])928 929 elif self._find_hms_idx(idx, tokens, info, allow_jump=True) is not None:930 # HH[ ]h or MM[ ]m or SS[.ss][ ]s931 hms_idx = self._find_hms_idx(idx, tokens, info, allow_jump=True)932 (idx, hms) = self._parse_hms(idx, tokens, info, hms_idx)933 if hms is not None:934 # TODO: checking that hour/minute/second are not935 # already set?936 self._assign_hms(res, value_repr, hms)937 938 elif idx + 2 < len_l and tokens[idx + 1] == ':':939 # HH:MM[:SS[.ss]]940 res.hour = int(value)941 value = self._to_decimal(tokens[idx + 2]) # TODO: try/except for this?942 (res.minute, res.second) = self._parse_min_sec(value)943 944 if idx + 4 < len_l and tokens[idx + 3] == ':':945 res.second, res.microsecond = self._parsems(tokens[idx + 4])946 947 idx += 2948 949 idx += 2950 951 elif idx + 1 < len_l and tokens[idx + 1] in ('-', '/', '.'):952 sep = tokens[idx + 1]953 ymd.append(value_repr)954 955 if idx + 2 < len_l and not info.jump(tokens[idx + 2]):956 if tokens[idx + 2].isdigit():957 # 01-01[-01]958 ymd.append(tokens[idx + 2])959 else:960 # 01-Jan[-01]961 value = info.month(tokens[idx + 2])962 963 if value is not None:964 ymd.append(value, 'M')965 else:966 raise ValueError()967 968 if idx + 3 < len_l and tokens[idx + 3] == sep:969 # We have three members970 value = info.month(tokens[idx + 4])971 972 if value is not None:973 ymd.append(value, 'M')974 else:975 ymd.append(tokens[idx + 4])976 idx += 2977 978 idx += 1979 idx += 1980 981 elif idx + 1 >= len_l or info.jump(tokens[idx + 1]):982 if idx + 2 < len_l and info.ampm(tokens[idx + 2]) is not None:983 # 12 am984 hour = int(value)985 res.hour = self._adjust_ampm(hour, info.ampm(tokens[idx + 2]))986 idx += 1987 else:988 # Year, month or day989 ymd.append(value)990 idx += 1991 992 elif info.ampm(tokens[idx + 1]) is not None and (0 <= value < 24):993 # 12am994 hour = int(value)995 res.hour = self._adjust_ampm(hour, info.ampm(tokens[idx + 1]))996 idx += 1997 998 elif ymd.could_be_day(value):999 ymd.append(value)1000 1001 elif not fuzzy:1002 raise ValueError()1003 1004 return idx1005 1006 def _find_hms_idx(self, idx, tokens, info, allow_jump):1007 len_l = len(tokens)1008 1009 if idx+1 < len_l and info.hms(tokens[idx+1]) is not None:1010 # There is an "h", "m", or "s" label following this token. We take1011 # assign the upcoming label to the current token.1012 # e.g. the "12" in 12h"1013 hms_idx = idx + 11014 1015 elif (allow_jump and idx+2 < len_l and tokens[idx+1] == ' ' and1016 info.hms(tokens[idx+2]) is not None):1017 # There is a space and then an "h", "m", or "s" label.1018 # e.g. the "12" in "12 h"1019 hms_idx = idx + 21020 1021 elif idx > 0 and info.hms(tokens[idx-1]) is not None:1022 # There is a "h", "m", or "s" preceding this token. Since neither1023 # of the previous cases was hit, there is no label following this1024 # token, so we use the previous label.1025 # e.g. the "04" in "12h04"1026 hms_idx = idx-11027 1028 elif (1 < idx == len_l-1 and tokens[idx-1] == ' ' and1029 info.hms(tokens[idx-2]) is not None):1030 # If we are looking at the final token, we allow for a1031 # backward-looking check to skip over a space.1032 # TODO: Are we sure this is the right condition here?1033 hms_idx = idx - 21034 1035 else:1036 hms_idx = None1037 1038 return hms_idx1039 1040 def _assign_hms(self, res, value_repr, hms):1041 # See GH issue #427, fixing float rounding1042 value = self._to_decimal(value_repr)1043 1044 if hms == 0:1045 # Hour1046 res.hour = int(value)1047 if value % 1:1048 res.minute = int(60*(value % 1))1049 1050 elif hms == 1:1051 (res.minute, res.second) = self._parse_min_sec(value)1052 1053 elif hms == 2:1054 (res.second, res.microsecond) = self._parsems(value_repr)1055 1056 def _could_be_tzname(self, hour, tzname, tzoffset, token):1057 return (hour is not None and1058 tzname is None and1059 tzoffset is None and1060 len(token) <= 5 and1061 (all(x in string.ascii_uppercase for x in token)1062 or token in self.info.UTCZONE))1063 1064 def _ampm_valid(self, hour, ampm, fuzzy):1065 """1066 For fuzzy parsing, 'a' or 'am' (both valid English words)1067 may erroneously trigger the AM/PM flag. Deal with that1068 here.1069 """1070 val_is_ampm = True1071 1072 # If there's already an AM/PM flag, this one isn't one.1073 if fuzzy and ampm is not None:1074 val_is_ampm = False1075 1076 # If AM/PM is found and hour is not, raise a ValueError1077 if hour is None:1078 if fuzzy:1079 val_is_ampm = False1080 else:1081 raise ValueError('No hour specified with AM or PM flag.')1082 elif not 0 <= hour <= 12:1083 # If AM/PM is found, it's a 12 hour clock, so raise1084 # an error for invalid range1085 if fuzzy:1086 val_is_ampm = False1087 else:1088 raise ValueError('Invalid hour specified for 12-hour clock.')1089 1090 return val_is_ampm1091 1092 def _adjust_ampm(self, hour, ampm):1093 if hour < 12 and ampm == 1:1094 hour += 121095 elif hour == 12 and ampm == 0:1096 hour = 01097 return hour1098 1099 def _parse_min_sec(self, value):1100 # TODO: Every usage of this function sets res.second to the return1101 # value. Are there any cases where second will be returned as None and1102 # we *don't* want to set res.second = None?1103 minute = int(value)1104 second = None1105 1106 sec_remainder = value % 11107 if sec_remainder:1108 second = int(60 * sec_remainder)1109 return (minute, second)1110 1111 def _parse_hms(self, idx, tokens, info, hms_idx):1112 # TODO: Is this going to admit a lot of false-positives for when we1113 # just happen to have digits and "h", "m" or "s" characters in non-date1114 # text? I guess hex hashes won't have that problem, but there's plenty1115 # of random junk out there.1116 if hms_idx is None:1117 hms = None1118 new_idx = idx1119 elif hms_idx > idx:1120 hms = info.hms(tokens[hms_idx])1121 new_idx = hms_idx1122 else:1123 # Looking backwards, increment one.1124 hms = info.hms(tokens[hms_idx]) + 11125 new_idx = idx1126 1127 return (new_idx, hms)1128 1129 # ------------------------------------------------------------------1130 # Handling for individual tokens. These are kept as methods instead1131 # of functions for the sake of customizability via subclassing.1132 1133 def _parsems(self, value):1134 """Parse a I[.F] seconds value into (seconds, microseconds)."""1135 if "." not in value:1136 return int(value), 01137 else:1138 i, f = value.split(".")1139 return int(i), int(f.ljust(6, "0")[:6])1140 1141 def _to_decimal(self, val):1142 try:1143 decimal_value = Decimal(val)1144 # See GH 662, edge case, infinite value should not be converted1145 # via `_to_decimal`1146 if not decimal_value.is_finite():1147 raise ValueError("Converted decimal value is infinite or NaN")1148 except Exception as e:1149 msg = "Could not convert %s to decimal" % val1150 six.raise_from(ValueError(msg), e)1151 else:1152 return decimal_value1153 1154 # ------------------------------------------------------------------1155 # Post-Parsing construction of datetime output. These are kept as1156 # methods instead of functions for the sake of customizability via1157 # subclassing.1158 1159 def _build_tzinfo(self, tzinfos, tzname, tzoffset):1160 if callable(tzinfos):1161 tzdata = tzinfos(tzname, tzoffset)1162 else:1163 tzdata = tzinfos.get(tzname)1164 # handle case where tzinfo is paased an options that returns None1165 # eg tzinfos = {'BRST' : None}1166 if isinstance(tzdata, datetime.tzinfo) or tzdata is None:1167 tzinfo = tzdata1168 elif isinstance(tzdata, text_type):1169 tzinfo = tz.tzstr(tzdata)1170 elif isinstance(tzdata, integer_types):1171 tzinfo = tz.tzoffset(tzname, tzdata)1172 else:1173 raise TypeError("Offset must be tzinfo subclass, tz string, "1174 "or int offset.")1175 return tzinfo1176 1177 def _build_tzaware(self, naive, res, tzinfos):1178 if (callable(tzinfos) or (tzinfos and res.tzname in tzinfos)):1179 tzinfo = self._build_tzinfo(tzinfos, res.tzname, res.tzoffset)1180 aware = naive.replace(tzinfo=tzinfo)1181 aware = self._assign_tzname(aware, res.tzname)1182 1183 elif res.tzname and res.tzname in time.tzname:1184 aware = naive.replace(tzinfo=tz.tzlocal())1185 1186 # Handle ambiguous local datetime1187 aware = self._assign_tzname(aware, res.tzname)1188 1189 # This is mostly relevant for winter GMT zones parsed in the UK1190 if (aware.tzname() != res.tzname and1191 res.tzname in self.info.UTCZONE):1192 aware = aware.replace(tzinfo=tz.UTC)1193 1194 elif res.tzoffset == 0:1195 aware = naive.replace(tzinfo=tz.UTC)1196 1197 elif res.tzoffset:1198 aware = naive.replace(tzinfo=tz.tzoffset(res.tzname, res.tzoffset))1199 1200 elif not res.tzname and not res.tzoffset: