Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
__init__.py648 linesDownload Raw Back to text
1import functools2import itertools3import re4import textwrap5 6from typing import Iterable7 8try:9    from importlib.resources import files  # type: ignore10except ImportError:  # pragma: nocover11    from importlib_resources import files  # type: ignore12 13from jaraco.context import ExceptionTrap14from jaraco.functools import compose, method_cache15 16 17def substitution(old, new):18    """19    Return a function that will perform a substitution on a string20    """21    return lambda s: s.replace(old, new)22 23 24def multi_substitution(*substitutions):25    """26    Take a sequence of pairs specifying substitutions, and create27    a function that performs those substitutions.28 29    >>> multi_substitution(('foo', 'bar'), ('bar', 'baz'))('foo')30    'baz'31    """32    substitutions = itertools.starmap(substitution, substitutions)33    # compose function applies last function first, so reverse the34    #  substitutions to get the expected order.35    substitutions = reversed(tuple(substitutions))36    return compose(*substitutions)37 38 39class FoldedCase(str):40    """41    A case insensitive string class; behaves just like str42    except compares equal when the only variation is case.43 44    >>> s = FoldedCase('hello world')45 46    >>> s == 'Hello World'47    True48 49    >>> 'Hello World' == s50    True51 52    >>> s != 'Hello World'53    False54 55    >>> s.index('O')56    457 58    >>> s.split('O')59    ['hell', ' w', 'rld']60 61    >>> sorted(map(FoldedCase, ['GAMMA', 'alpha', 'Beta']))62    ['alpha', 'Beta', 'GAMMA']63 64    Sequence membership is straightforward.65 66    >>> "Hello World" in [s]67    True68    >>> s in ["Hello World"]69    True70 71    Allows testing for set inclusion, but candidate and elements72    must both be folded.73 74    >>> FoldedCase("Hello World") in {s}75    True76    >>> s in {FoldedCase("Hello World")}77    True78 79    String inclusion works as long as the FoldedCase object80    is on the right.81 82    >>> "hello" in FoldedCase("Hello World")83    True84 85    But not if the FoldedCase object is on the left:86 87    >>> FoldedCase('hello') in 'Hello World'88    False89 90    In that case, use ``in_``:91 92    >>> FoldedCase('hello').in_('Hello World')93    True94 95    >>> FoldedCase('hello') > FoldedCase('Hello')96    False97 98    >>> FoldedCase('ß') == FoldedCase('ss')99    True100    """101 102    def __lt__(self, other):103        return self.casefold() < other.casefold()104 105    def __gt__(self, other):106        return self.casefold() > other.casefold()107 108    def __eq__(self, other):109        return self.casefold() == other.casefold()110 111    def __ne__(self, other):112        return self.casefold() != other.casefold()113 114    def __hash__(self):115        return hash(self.casefold())116 117    def __contains__(self, other):118        return super().casefold().__contains__(other.casefold())119 120    def in_(self, other):121        "Does self appear in other?"122        return self in FoldedCase(other)123 124    # cache casefold since it's likely to be called frequently.125    @method_cache126    def casefold(self):127        return super().casefold()128 129    def index(self, sub):130        return self.casefold().index(sub.casefold())131 132    def split(self, splitter=' ', maxsplit=0):133        pattern = re.compile(re.escape(splitter), re.I)134        return pattern.split(self, maxsplit)135 136 137# Python 3.8 compatibility138_unicode_trap = ExceptionTrap(UnicodeDecodeError)139 140 141@_unicode_trap.passes142def is_decodable(value):143    r"""144    Return True if the supplied value is decodable (using the default145    encoding).146 147    >>> is_decodable(b'\xff')148    False149    >>> is_decodable(b'\x32')150    True151    """152    value.decode()153 154 155def is_binary(value):156    r"""157    Return True if the value appears to be binary (that is, it's a byte158    string and isn't decodable).159 160    >>> is_binary(b'\xff')161    True162    >>> is_binary('\xff')163    False164    """165    return isinstance(value, bytes) and not is_decodable(value)166 167 168def trim(s):169    r"""170    Trim something like a docstring to remove the whitespace that171    is common due to indentation and formatting.172 173    >>> trim("\n\tfoo = bar\n\t\tbar = baz\n")174    'foo = bar\n\tbar = baz'175    """176    return textwrap.dedent(s).strip()177 178 179def wrap(s):180    """181    Wrap lines of text, retaining existing newlines as182    paragraph markers.183 184    >>> print(wrap(lorem_ipsum))185    Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do186    eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad187    minim veniam, quis nostrud exercitation ullamco laboris nisi ut188    aliquip ex ea commodo consequat. Duis aute irure dolor in189    reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla190    pariatur. Excepteur sint occaecat cupidatat non proident, sunt in191    culpa qui officia deserunt mollit anim id est laborum.192    <BLANKLINE>193    Curabitur pretium tincidunt lacus. Nulla gravida orci a odio. Nullam194    varius, turpis et commodo pharetra, est eros bibendum elit, nec luctus195    magna felis sollicitudin mauris. Integer in mauris eu nibh euismod196    gravida. Duis ac tellus et risus vulputate vehicula. Donec lobortis197    risus a elit. Etiam tempor. Ut ullamcorper, ligula eu tempor congue,198    eros est euismod turpis, id tincidunt sapien risus a quam. Maecenas199    fermentum consequat mi. Donec fermentum. Pellentesque malesuada nulla200    a mi. Duis sapien sem, aliquet nec, commodo eget, consequat quis,201    neque. Aliquam faucibus, elit ut dictum aliquet, felis nisl adipiscing202    sapien, sed malesuada diam lacus eget erat. Cras mollis scelerisque203    nunc. Nullam arcu. Aliquam consequat. Curabitur augue lorem, dapibus204    quis, laoreet et, pretium ac, nisi. Aenean magna nisl, mollis quis,205    molestie eu, feugiat in, orci. In hac habitasse platea dictumst.206    """207    paragraphs = s.splitlines()208    wrapped = ('\n'.join(textwrap.wrap(para)) for para in paragraphs)209    return '\n\n'.join(wrapped)210 211 212def unwrap(s):213    r"""214    Given a multi-line string, return an unwrapped version.215 216    >>> wrapped = wrap(lorem_ipsum)217    >>> wrapped.count('\n')218    20219    >>> unwrapped = unwrap(wrapped)220    >>> unwrapped.count('\n')221    1222    >>> print(unwrapped)223    Lorem ipsum dolor sit amet, consectetur adipiscing ...224    Curabitur pretium tincidunt lacus. Nulla gravida orci ...225 226    """227    paragraphs = re.split(r'\n\n+', s)228    cleaned = (para.replace('\n', ' ') for para in paragraphs)229    return '\n'.join(cleaned)230 231 232lorem_ipsum: str = (233    files(__name__).joinpath('Lorem ipsum.txt').read_text(encoding='utf-8')234)235 236 237class Splitter:238    """object that will split a string with the given arguments for each call239 240    >>> s = Splitter(',')241    >>> s('hello, world, this is your, master calling')242    ['hello', ' world', ' this is your', ' master calling']243    """244 245    def __init__(self, *args):246        self.args = args247 248    def __call__(self, s):249        return s.split(*self.args)250 251 252def indent(string, prefix=' ' * 4):253    """254    >>> indent('foo')255    '    foo'256    """257    return prefix + string258 259 260class WordSet(tuple):261    """262    Given an identifier, return the words that identifier represents,263    whether in camel case, underscore-separated, etc.264 265    >>> WordSet.parse("camelCase")266    ('camel', 'Case')267 268    >>> WordSet.parse("under_sep")269    ('under', 'sep')270 271    Acronyms should be retained272 273    >>> WordSet.parse("firstSNL")274    ('first', 'SNL')275 276    >>> WordSet.parse("you_and_I")277    ('you', 'and', 'I')278 279    >>> WordSet.parse("A simple test")280    ('A', 'simple', 'test')281 282    Multiple caps should not interfere with the first cap of another word.283 284    >>> WordSet.parse("myABCClass")285    ('my', 'ABC', 'Class')286 287    The result is a WordSet, providing access to other forms.288 289    >>> WordSet.parse("myABCClass").underscore_separated()290    'my_ABC_Class'291 292    >>> WordSet.parse('a-command').camel_case()293    'ACommand'294 295    >>> WordSet.parse('someIdentifier').lowered().space_separated()296    'some identifier'297 298    Slices of the result should return another WordSet.299 300    >>> WordSet.parse('taken-out-of-context')[1:].underscore_separated()301    'out_of_context'302 303    >>> WordSet.from_class_name(WordSet()).lowered().space_separated()304    'word set'305 306    >>> example = WordSet.parse('figured it out')307    >>> example.headless_camel_case()308    'figuredItOut'309    >>> example.dash_separated()310    'figured-it-out'311 312    """313 314    _pattern = re.compile('([A-Z]?[a-z]+)|([A-Z]+(?![a-z]))')315 316    def capitalized(self):317        return WordSet(word.capitalize() for word in self)318 319    def lowered(self):320        return WordSet(word.lower() for word in self)321 322    def camel_case(self):323        return ''.join(self.capitalized())324 325    def headless_camel_case(self):326        words = iter(self)327        first = next(words).lower()328        new_words = itertools.chain((first,), WordSet(words).camel_case())329        return ''.join(new_words)330 331    def underscore_separated(self):332        return '_'.join(self)333 334    def dash_separated(self):335        return '-'.join(self)336 337    def space_separated(self):338        return ' '.join(self)339 340    def trim_right(self, item):341        """342        Remove the item from the end of the set.343 344        >>> WordSet.parse('foo bar').trim_right('foo')345        ('foo', 'bar')346        >>> WordSet.parse('foo bar').trim_right('bar')347        ('foo',)348        >>> WordSet.parse('').trim_right('bar')349        ()350        """351        return self[:-1] if self and self[-1] == item else self352 353    def trim_left(self, item):354        """355        Remove the item from the beginning of the set.356 357        >>> WordSet.parse('foo bar').trim_left('foo')358        ('bar',)359        >>> WordSet.parse('foo bar').trim_left('bar')360        ('foo', 'bar')361        >>> WordSet.parse('').trim_left('bar')362        ()363        """364        return self[1:] if self and self[0] == item else self365 366    def trim(self, item):367        """368        >>> WordSet.parse('foo bar').trim('foo')369        ('bar',)370        """371        return self.trim_left(item).trim_right(item)372 373    def __getitem__(self, item):374        result = super().__getitem__(item)375        if isinstance(item, slice):376            result = WordSet(result)377        return result378 379    @classmethod380    def parse(cls, identifier):381        matches = cls._pattern.finditer(identifier)382        return WordSet(match.group(0) for match in matches)383 384    @classmethod385    def from_class_name(cls, subject):386        return cls.parse(subject.__class__.__name__)387 388 389# for backward compatibility390words = WordSet.parse391 392 393def simple_html_strip(s):394    r"""395    Remove HTML from the string `s`.396 397    >>> str(simple_html_strip(''))398    ''399 400    >>> print(simple_html_strip('A <bold>stormy</bold> day in paradise'))401    A stormy day in paradise402 403    >>> print(simple_html_strip('Somebody <!-- do not --> tell the truth.'))404    Somebody  tell the truth.405 406    >>> print(simple_html_strip('What about<br/>\nmultiple lines?'))407    What about408    multiple lines?409    """410    html_stripper = re.compile('(<!--.*?-->)|(<[^>]*>)|([^<]+)', re.DOTALL)411    texts = (match.group(3) or '' for match in html_stripper.finditer(s))412    return ''.join(texts)413 414 415class SeparatedValues(str):416    """417    A string separated by a separator. Overrides __iter__ for getting418    the values.419 420    >>> list(SeparatedValues('a,b,c'))421    ['a', 'b', 'c']422 423    Whitespace is stripped and empty values are discarded.424 425    >>> list(SeparatedValues(' a,   b   , c,  '))426    ['a', 'b', 'c']427    """428 429    separator = ','430 431    def __iter__(self):432        parts = self.split(self.separator)433        return filter(None, (part.strip() for part in parts))434 435 436class Stripper:437    r"""438    Given a series of lines, find the common prefix and strip it from them.439 440    >>> lines = [441    ...     'abcdefg\n',442    ...     'abc\n',443    ...     'abcde\n',444    ... ]445    >>> res = Stripper.strip_prefix(lines)446    >>> res.prefix447    'abc'448    >>> list(res.lines)449    ['defg\n', '\n', 'de\n']450 451    If no prefix is common, nothing should be stripped.452 453    >>> lines = [454    ...     'abcd\n',455    ...     '1234\n',456    ... ]457    >>> res = Stripper.strip_prefix(lines)458    >>> res.prefix = ''459    >>> list(res.lines)460    ['abcd\n', '1234\n']461    """462 463    def __init__(self, prefix, lines):464        self.prefix = prefix465        self.lines = map(self, lines)466 467    @classmethod468    def strip_prefix(cls, lines):469        prefix_lines, lines = itertools.tee(lines)470        prefix = functools.reduce(cls.common_prefix, prefix_lines)471        return cls(prefix, lines)472 473    def __call__(self, line):474        if not self.prefix:475            return line476        null, prefix, rest = line.partition(self.prefix)477        return rest478 479    @staticmethod480    def common_prefix(s1, s2):481        """482        Return the common prefix of two lines.483        """484        index = min(len(s1), len(s2))485        while s1[:index] != s2[:index]:486            index -= 1487        return s1[:index]488 489 490def remove_prefix(text, prefix):491    """492    Remove the prefix from the text if it exists.493 494    >>> remove_prefix('underwhelming performance', 'underwhelming ')495    'performance'496 497    >>> remove_prefix('something special', 'sample')498    'something special'499    """500    null, prefix, rest = text.rpartition(prefix)501    return rest502 503 504def remove_suffix(text, suffix):505    """506    Remove the suffix from the text if it exists.507 508    >>> remove_suffix('name.git', '.git')509    'name'510 511    >>> remove_suffix('something special', 'sample')512    'something special'513    """514    rest, suffix, null = text.partition(suffix)515    return rest516 517 518def normalize_newlines(text):519    r"""520    Replace alternate newlines with the canonical newline.521 522    >>> normalize_newlines('Lorem Ipsum\u2029')523    'Lorem Ipsum\n'524    >>> normalize_newlines('Lorem Ipsum\r\n')525    'Lorem Ipsum\n'526    >>> normalize_newlines('Lorem Ipsum\x85')527    'Lorem Ipsum\n'528    """529    newlines = ['\r\n', '\r', '\n', '\u0085', '\u2028', '\u2029']530    pattern = '|'.join(newlines)531    return re.sub(pattern, '\n', text)532 533 534def _nonblank(str):535    return str and not str.startswith('#')536 537 538@functools.singledispatch539def yield_lines(iterable):540    r"""541    Yield valid lines of a string or iterable.542 543    >>> list(yield_lines(''))544    []545    >>> list(yield_lines(['foo', 'bar']))546    ['foo', 'bar']547    >>> list(yield_lines('foo\nbar'))548    ['foo', 'bar']549    >>> list(yield_lines('\nfoo\n#bar\nbaz #comment'))550    ['foo', 'baz #comment']551    >>> list(yield_lines(['foo\nbar', 'baz', 'bing\n\n\n']))552    ['foo', 'bar', 'baz', 'bing']553    """554    return itertools.chain.from_iterable(map(yield_lines, iterable))555 556 557@yield_lines.register(str)558def _(text):559    return clean(text.splitlines())560 561 562def clean(lines: Iterable[str]):563    """564    Yield non-blank, non-comment elements from lines.565    """566    return filter(_nonblank, map(str.strip, lines))567 568 569def drop_comment(line):570    """571    Drop comments.572 573    >>> drop_comment('foo # bar')574    'foo'575 576    A hash without a space may be in a URL.577 578    >>> drop_comment('http://example.com/foo#bar')579    'http://example.com/foo#bar'580    """581    return line.partition(' #')[0]582 583 584def join_continuation(lines):585    r"""586    Join lines continued by a trailing backslash.587 588    >>> list(join_continuation(['foo \\', 'bar', 'baz']))589    ['foobar', 'baz']590    >>> list(join_continuation(['foo \\', 'bar', 'baz']))591    ['foobar', 'baz']592    >>> list(join_continuation(['foo \\', 'bar \\', 'baz']))593    ['foobarbaz']594 595    Not sure why, but...596    The character preceding the backslash is also elided.597 598    >>> list(join_continuation(['goo\\', 'dly']))599    ['godly']600 601    A terrible idea, but...602    If no line is available to continue, suppress the lines.603 604    >>> list(join_continuation(['foo', 'bar\\', 'baz\\']))605    ['foo']606    """607    lines = iter(lines)608    for item in lines:609        while item.endswith('\\'):610            try:611                item = item[:-2].strip() + next(lines)612            except StopIteration:613                return614        yield item615 616 617def read_newlines(filename, limit=1024):618    r"""619    >>> tmp_path = getfixture('tmp_path')620    >>> filename = tmp_path / 'out.txt'621    >>> _ = filename.write_text('foo\n', newline='', encoding='utf-8')622    >>> read_newlines(filename)623    '\n'624    >>> _ = filename.write_text('foo\r\n', newline='', encoding='utf-8')625    >>> read_newlines(filename)626    '\r\n'627    >>> _ = filename.write_text('foo\r\nbar\nbing\r', newline='', encoding='utf-8')628    >>> read_newlines(filename)629    ('\r', '\n', '\r\n')630    """631    with open(filename, encoding='utf-8') as fp:632        fp.read(limit)633    return fp.newlines634 635 636def lines_from(input):637    """638    Generate lines from a :class:`importlib.resources.abc.Traversable` path.639 640    >>> lines = lines_from(files(__name__).joinpath('Lorem ipsum.txt'))641    >>> next(lines)642    'Lorem ipsum...'643    >>> next(lines)644    'Curabitur pretium...'645    """646    with input.open(encoding='utf-8') as stream:647        yield from stream648 
codekingpro/portable-devtools · Team Ai