codekingpro/portable-devtools
114k
1import functools2import itertools3import re4import textwrap5 6from typing import Iterable7 8try:9 from importlib.resources import files # type: ignore10except ImportError: # pragma: nocover11 from importlib_resources import files # type: ignore12 13from jaraco.context import ExceptionTrap14from jaraco.functools import compose, method_cache15 16 17def substitution(old, new):18 """19 Return a function that will perform a substitution on a string20 """21 return lambda s: s.replace(old, new)22 23 24def multi_substitution(*substitutions):25 """26 Take a sequence of pairs specifying substitutions, and create27 a function that performs those substitutions.28 29 >>> multi_substitution(('foo', 'bar'), ('bar', 'baz'))('foo')30 'baz'31 """32 substitutions = itertools.starmap(substitution, substitutions)33 # compose function applies last function first, so reverse the34 # substitutions to get the expected order.35 substitutions = reversed(tuple(substitutions))36 return compose(*substitutions)37 38 39class FoldedCase(str):40 """41 A case insensitive string class; behaves just like str42 except compares equal when the only variation is case.43 44 >>> s = FoldedCase('hello world')45 46 >>> s == 'Hello World'47 True48 49 >>> 'Hello World' == s50 True51 52 >>> s != 'Hello World'53 False54 55 >>> s.index('O')56 457 58 >>> s.split('O')59 ['hell', ' w', 'rld']60 61 >>> sorted(map(FoldedCase, ['GAMMA', 'alpha', 'Beta']))62 ['alpha', 'Beta', 'GAMMA']63 64 Sequence membership is straightforward.65 66 >>> "Hello World" in [s]67 True68 >>> s in ["Hello World"]69 True70 71 Allows testing for set inclusion, but candidate and elements72 must both be folded.73 74 >>> FoldedCase("Hello World") in {s}75 True76 >>> s in {FoldedCase("Hello World")}77 True78 79 String inclusion works as long as the FoldedCase object80 is on the right.81 82 >>> "hello" in FoldedCase("Hello World")83 True84 85 But not if the FoldedCase object is on the left:86 87 >>> FoldedCase('hello') in 'Hello World'88 False89 90 In that case, use ``in_``:91 92 >>> FoldedCase('hello').in_('Hello World')93 True94 95 >>> FoldedCase('hello') > FoldedCase('Hello')96 False97 98 >>> FoldedCase('ß') == FoldedCase('ss')99 True100 """101 102 def __lt__(self, other):103 return self.casefold() < other.casefold()104 105 def __gt__(self, other):106 return self.casefold() > other.casefold()107 108 def __eq__(self, other):109 return self.casefold() == other.casefold()110 111 def __ne__(self, other):112 return self.casefold() != other.casefold()113 114 def __hash__(self):115 return hash(self.casefold())116 117 def __contains__(self, other):118 return super().casefold().__contains__(other.casefold())119 120 def in_(self, other):121 "Does self appear in other?"122 return self in FoldedCase(other)123 124 # cache casefold since it's likely to be called frequently.125 @method_cache126 def casefold(self):127 return super().casefold()128 129 def index(self, sub):130 return self.casefold().index(sub.casefold())131 132 def split(self, splitter=' ', maxsplit=0):133 pattern = re.compile(re.escape(splitter), re.I)134 return pattern.split(self, maxsplit)135 136 137# Python 3.8 compatibility138_unicode_trap = ExceptionTrap(UnicodeDecodeError)139 140 141@_unicode_trap.passes142def is_decodable(value):143 r"""144 Return True if the supplied value is decodable (using the default145 encoding).146 147 >>> is_decodable(b'\xff')148 False149 >>> is_decodable(b'\x32')150 True151 """152 value.decode()153 154 155def is_binary(value):156 r"""157 Return True if the value appears to be binary (that is, it's a byte158 string and isn't decodable).159 160 >>> is_binary(b'\xff')161 True162 >>> is_binary('\xff')163 False164 """165 return isinstance(value, bytes) and not is_decodable(value)166 167 168def trim(s):169 r"""170 Trim something like a docstring to remove the whitespace that171 is common due to indentation and formatting.172 173 >>> trim("\n\tfoo = bar\n\t\tbar = baz\n")174 'foo = bar\n\tbar = baz'175 """176 return textwrap.dedent(s).strip()177 178 179def wrap(s):180 """181 Wrap lines of text, retaining existing newlines as182 paragraph markers.183 184 >>> print(wrap(lorem_ipsum))185 Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do186 eiusmod tempor incididunt ut labore et dolore magna aliqua. Ut enim ad187 minim veniam, quis nostrud exercitation ullamco laboris nisi ut188 aliquip ex ea commodo consequat. Duis aute irure dolor in189 reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla190 pariatur. Excepteur sint occaecat cupidatat non proident, sunt in191 culpa qui officia deserunt mollit anim id est laborum.192 <BLANKLINE>193 Curabitur pretium tincidunt lacus. Nulla gravida orci a odio. Nullam194 varius, turpis et commodo pharetra, est eros bibendum elit, nec luctus195 magna felis sollicitudin mauris. Integer in mauris eu nibh euismod196 gravida. Duis ac tellus et risus vulputate vehicula. Donec lobortis197 risus a elit. Etiam tempor. Ut ullamcorper, ligula eu tempor congue,198 eros est euismod turpis, id tincidunt sapien risus a quam. Maecenas199 fermentum consequat mi. Donec fermentum. Pellentesque malesuada nulla200 a mi. Duis sapien sem, aliquet nec, commodo eget, consequat quis,201 neque. Aliquam faucibus, elit ut dictum aliquet, felis nisl adipiscing202 sapien, sed malesuada diam lacus eget erat. Cras mollis scelerisque203 nunc. Nullam arcu. Aliquam consequat. Curabitur augue lorem, dapibus204 quis, laoreet et, pretium ac, nisi. Aenean magna nisl, mollis quis,205 molestie eu, feugiat in, orci. In hac habitasse platea dictumst.206 """207 paragraphs = s.splitlines()208 wrapped = ('\n'.join(textwrap.wrap(para)) for para in paragraphs)209 return '\n\n'.join(wrapped)210 211 212def unwrap(s):213 r"""214 Given a multi-line string, return an unwrapped version.215 216 >>> wrapped = wrap(lorem_ipsum)217 >>> wrapped.count('\n')218 20219 >>> unwrapped = unwrap(wrapped)220 >>> unwrapped.count('\n')221 1222 >>> print(unwrapped)223 Lorem ipsum dolor sit amet, consectetur adipiscing ...224 Curabitur pretium tincidunt lacus. Nulla gravida orci ...225 226 """227 paragraphs = re.split(r'\n\n+', s)228 cleaned = (para.replace('\n', ' ') for para in paragraphs)229 return '\n'.join(cleaned)230 231 232lorem_ipsum: str = (233 files(__name__).joinpath('Lorem ipsum.txt').read_text(encoding='utf-8')234)235 236 237class Splitter:238 """object that will split a string with the given arguments for each call239 240 >>> s = Splitter(',')241 >>> s('hello, world, this is your, master calling')242 ['hello', ' world', ' this is your', ' master calling']243 """244 245 def __init__(self, *args):246 self.args = args247 248 def __call__(self, s):249 return s.split(*self.args)250 251 252def indent(string, prefix=' ' * 4):253 """254 >>> indent('foo')255 ' foo'256 """257 return prefix + string258 259 260class WordSet(tuple):261 """262 Given an identifier, return the words that identifier represents,263 whether in camel case, underscore-separated, etc.264 265 >>> WordSet.parse("camelCase")266 ('camel', 'Case')267 268 >>> WordSet.parse("under_sep")269 ('under', 'sep')270 271 Acronyms should be retained272 273 >>> WordSet.parse("firstSNL")274 ('first', 'SNL')275 276 >>> WordSet.parse("you_and_I")277 ('you', 'and', 'I')278 279 >>> WordSet.parse("A simple test")280 ('A', 'simple', 'test')281 282 Multiple caps should not interfere with the first cap of another word.283 284 >>> WordSet.parse("myABCClass")285 ('my', 'ABC', 'Class')286 287 The result is a WordSet, providing access to other forms.288 289 >>> WordSet.parse("myABCClass").underscore_separated()290 'my_ABC_Class'291 292 >>> WordSet.parse('a-command').camel_case()293 'ACommand'294 295 >>> WordSet.parse('someIdentifier').lowered().space_separated()296 'some identifier'297 298 Slices of the result should return another WordSet.299 300 >>> WordSet.parse('taken-out-of-context')[1:].underscore_separated()301 'out_of_context'302 303 >>> WordSet.from_class_name(WordSet()).lowered().space_separated()304 'word set'305 306 >>> example = WordSet.parse('figured it out')307 >>> example.headless_camel_case()308 'figuredItOut'309 >>> example.dash_separated()310 'figured-it-out'311 312 """313 314 _pattern = re.compile('([A-Z]?[a-z]+)|([A-Z]+(?![a-z]))')315 316 def capitalized(self):317 return WordSet(word.capitalize() for word in self)318 319 def lowered(self):320 return WordSet(word.lower() for word in self)321 322 def camel_case(self):323 return ''.join(self.capitalized())324 325 def headless_camel_case(self):326 words = iter(self)327 first = next(words).lower()328 new_words = itertools.chain((first,), WordSet(words).camel_case())329 return ''.join(new_words)330 331 def underscore_separated(self):332 return '_'.join(self)333 334 def dash_separated(self):335 return '-'.join(self)336 337 def space_separated(self):338 return ' '.join(self)339 340 def trim_right(self, item):341 """342 Remove the item from the end of the set.343 344 >>> WordSet.parse('foo bar').trim_right('foo')345 ('foo', 'bar')346 >>> WordSet.parse('foo bar').trim_right('bar')347 ('foo',)348 >>> WordSet.parse('').trim_right('bar')349 ()350 """351 return self[:-1] if self and self[-1] == item else self352 353 def trim_left(self, item):354 """355 Remove the item from the beginning of the set.356 357 >>> WordSet.parse('foo bar').trim_left('foo')358 ('bar',)359 >>> WordSet.parse('foo bar').trim_left('bar')360 ('foo', 'bar')361 >>> WordSet.parse('').trim_left('bar')362 ()363 """364 return self[1:] if self and self[0] == item else self365 366 def trim(self, item):367 """368 >>> WordSet.parse('foo bar').trim('foo')369 ('bar',)370 """371 return self.trim_left(item).trim_right(item)372 373 def __getitem__(self, item):374 result = super().__getitem__(item)375 if isinstance(item, slice):376 result = WordSet(result)377 return result378 379 @classmethod380 def parse(cls, identifier):381 matches = cls._pattern.finditer(identifier)382 return WordSet(match.group(0) for match in matches)383 384 @classmethod385 def from_class_name(cls, subject):386 return cls.parse(subject.__class__.__name__)387 388 389# for backward compatibility390words = WordSet.parse391 392 393def simple_html_strip(s):394 r"""395 Remove HTML from the string `s`.396 397 >>> str(simple_html_strip(''))398 ''399 400 >>> print(simple_html_strip('A <bold>stormy</bold> day in paradise'))401 A stormy day in paradise402 403 >>> print(simple_html_strip('Somebody <!-- do not --> tell the truth.'))404 Somebody tell the truth.405 406 >>> print(simple_html_strip('What about<br/>\nmultiple lines?'))407 What about408 multiple lines?409 """410 html_stripper = re.compile('(<!--.*?-->)|(<[^>]*>)|([^<]+)', re.DOTALL)411 texts = (match.group(3) or '' for match in html_stripper.finditer(s))412 return ''.join(texts)413 414 415class SeparatedValues(str):416 """417 A string separated by a separator. Overrides __iter__ for getting418 the values.419 420 >>> list(SeparatedValues('a,b,c'))421 ['a', 'b', 'c']422 423 Whitespace is stripped and empty values are discarded.424 425 >>> list(SeparatedValues(' a, b , c, '))426 ['a', 'b', 'c']427 """428 429 separator = ','430 431 def __iter__(self):432 parts = self.split(self.separator)433 return filter(None, (part.strip() for part in parts))434 435 436class Stripper:437 r"""438 Given a series of lines, find the common prefix and strip it from them.439 440 >>> lines = [441 ... 'abcdefg\n',442 ... 'abc\n',443 ... 'abcde\n',444 ... ]445 >>> res = Stripper.strip_prefix(lines)446 >>> res.prefix447 'abc'448 >>> list(res.lines)449 ['defg\n', '\n', 'de\n']450 451 If no prefix is common, nothing should be stripped.452 453 >>> lines = [454 ... 'abcd\n',455 ... '1234\n',456 ... ]457 >>> res = Stripper.strip_prefix(lines)458 >>> res.prefix = ''459 >>> list(res.lines)460 ['abcd\n', '1234\n']461 """462 463 def __init__(self, prefix, lines):464 self.prefix = prefix465 self.lines = map(self, lines)466 467 @classmethod468 def strip_prefix(cls, lines):469 prefix_lines, lines = itertools.tee(lines)470 prefix = functools.reduce(cls.common_prefix, prefix_lines)471 return cls(prefix, lines)472 473 def __call__(self, line):474 if not self.prefix:475 return line476 null, prefix, rest = line.partition(self.prefix)477 return rest478 479 @staticmethod480 def common_prefix(s1, s2):481 """482 Return the common prefix of two lines.483 """484 index = min(len(s1), len(s2))485 while s1[:index] != s2[:index]:486 index -= 1487 return s1[:index]488 489 490def remove_prefix(text, prefix):491 """492 Remove the prefix from the text if it exists.493 494 >>> remove_prefix('underwhelming performance', 'underwhelming ')495 'performance'496 497 >>> remove_prefix('something special', 'sample')498 'something special'499 """500 null, prefix, rest = text.rpartition(prefix)501 return rest502 503 504def remove_suffix(text, suffix):505 """506 Remove the suffix from the text if it exists.507 508 >>> remove_suffix('name.git', '.git')509 'name'510 511 >>> remove_suffix('something special', 'sample')512 'something special'513 """514 rest, suffix, null = text.partition(suffix)515 return rest516 517 518def normalize_newlines(text):519 r"""520 Replace alternate newlines with the canonical newline.521 522 >>> normalize_newlines('Lorem Ipsum\u2029')523 'Lorem Ipsum\n'524 >>> normalize_newlines('Lorem Ipsum\r\n')525 'Lorem Ipsum\n'526 >>> normalize_newlines('Lorem Ipsum\x85')527 'Lorem Ipsum\n'528 """529 newlines = ['\r\n', '\r', '\n', '\u0085', '\u2028', '\u2029']530 pattern = '|'.join(newlines)531 return re.sub(pattern, '\n', text)532 533 534def _nonblank(str):535 return str and not str.startswith('#')536 537 538@functools.singledispatch539def yield_lines(iterable):540 r"""541 Yield valid lines of a string or iterable.542 543 >>> list(yield_lines(''))544 []545 >>> list(yield_lines(['foo', 'bar']))546 ['foo', 'bar']547 >>> list(yield_lines('foo\nbar'))548 ['foo', 'bar']549 >>> list(yield_lines('\nfoo\n#bar\nbaz #comment'))550 ['foo', 'baz #comment']551 >>> list(yield_lines(['foo\nbar', 'baz', 'bing\n\n\n']))552 ['foo', 'bar', 'baz', 'bing']553 """554 return itertools.chain.from_iterable(map(yield_lines, iterable))555 556 557@yield_lines.register(str)558def _(text):559 return clean(text.splitlines())560 561 562def clean(lines: Iterable[str]):563 """564 Yield non-blank, non-comment elements from lines.565 """566 return filter(_nonblank, map(str.strip, lines))567 568 569def drop_comment(line):570 """571 Drop comments.572 573 >>> drop_comment('foo # bar')574 'foo'575 576 A hash without a space may be in a URL.577 578 >>> drop_comment('http://example.com/foo#bar')579 'http://example.com/foo#bar'580 """581 return line.partition(' #')[0]582 583 584def join_continuation(lines):585 r"""586 Join lines continued by a trailing backslash.587 588 >>> list(join_continuation(['foo \\', 'bar', 'baz']))589 ['foobar', 'baz']590 >>> list(join_continuation(['foo \\', 'bar', 'baz']))591 ['foobar', 'baz']592 >>> list(join_continuation(['foo \\', 'bar \\', 'baz']))593 ['foobarbaz']594 595 Not sure why, but...596 The character preceding the backslash is also elided.597 598 >>> list(join_continuation(['goo\\', 'dly']))599 ['godly']600 601 A terrible idea, but...602 If no line is available to continue, suppress the lines.603 604 >>> list(join_continuation(['foo', 'bar\\', 'baz\\']))605 ['foo']606 """607 lines = iter(lines)608 for item in lines:609 while item.endswith('\\'):610 try:611 item = item[:-2].strip() + next(lines)612 except StopIteration:613 return614 yield item615 616 617def read_newlines(filename, limit=1024):618 r"""619 >>> tmp_path = getfixture('tmp_path')620 >>> filename = tmp_path / 'out.txt'621 >>> _ = filename.write_text('foo\n', newline='', encoding='utf-8')622 >>> read_newlines(filename)623 '\n'624 >>> _ = filename.write_text('foo\r\n', newline='', encoding='utf-8')625 >>> read_newlines(filename)626 '\r\n'627 >>> _ = filename.write_text('foo\r\nbar\nbing\r', newline='', encoding='utf-8')628 >>> read_newlines(filename)629 ('\r', '\n', '\r\n')630 """631 with open(filename, encoding='utf-8') as fp:632 fp.read(limit)633 return fp.newlines634 635 636def lines_from(input):637 """638 Generate lines from a :class:`importlib.resources.abc.Traversable` path.639 640 >>> lines = lines_from(files(__name__).joinpath('Lorem ipsum.txt'))641 >>> next(lines)642 'Lorem ipsum...'643 >>> next(lines)644 'Curabitur pretium...'645 """646 with input.open(encoding='utf-8') as stream:647 yield from stream648 