Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
tnetstring.py262 linesDownload Raw Back to io
1"""2tnetstring:  data serialization using typed netstrings3======================================================4 5This is a custom Python 3 implementation of tnetstrings.6Compared to other implementations, the main difference7is that this implementation supports a custom unicode datatype.8 9An ordinary tnetstring is a blob of data prefixed with its length and postfixed10with its type. Here are some examples:11 12    >>> tnetstring.dumps("hello world")13    11:hello world,14    >>> tnetstring.dumps(12345)15    5:12345#16    >>> tnetstring.dumps([12345, True, 0])17    19:5:12345#4:true!1:0#]18 19This module gives you the following functions:20 21    :dump:    dump an object as a tnetstring to a file22    :dumps:   dump an object as a tnetstring to a string23    :load:    load a tnetstring-encoded object from a file24    :loads:   load a tnetstring-encoded object from a string25 26Note that since parsing a tnetstring requires reading all the data into memory27at once, there's no efficiency gain from using the file-based versions of these28functions.  They're only here so you can use load() to read precisely one29item from a file or socket without consuming any extra data.30 31The tnetstrings specification explicitly states that strings are binary blobs32and forbids the use of unicode at the protocol level.33**This implementation decodes dictionary keys as surrogate-escaped ASCII**,34all other strings are returned as plain bytes.35 36:Copyright: (c) 2012-2013 by Ryan Kelly <ryan@rfk.id.au>.37:Copyright: (c) 2014 by Carlo Pires <carlopires@gmail.com>.38:Copyright: (c) 2016 by Maximilian Hils <tnetstring3@maximilianhils.com>.39 40:License: MIT41"""42 43import collections44from typing import BinaryIO45from typing import Union46 47TSerializable = Union[None, str, bool, int, float, bytes, list, tuple, dict]48 49 50def dumps(value: TSerializable) -> bytes:51    """52    This function dumps a python object as a tnetstring.53    """54    #  This uses a deque to collect output fragments in reverse order,55    #  then joins them together at the end.  It's measurably faster56    #  than creating all the intermediate strings.57    q: collections.deque = collections.deque()58    _rdumpq(q, 0, value)59    return b"".join(q)60 61 62def dump(value: TSerializable, file_handle: BinaryIO) -> None:63    """64    This function dumps a python object as a tnetstring and65    writes it to the given file.66    """67    file_handle.write(dumps(value))68 69 70def _rdumpq(q: collections.deque, size: int, value: TSerializable) -> int:71    """72    Dump value as a tnetstring, to a deque instance, last chunks first.73 74    This function generates the tnetstring representation of the given value,75    pushing chunks of the output onto the given deque instance.  It pushes76    the last chunk first, then recursively generates more chunks.77 78    When passed in the current size of the string in the queue, it will return79    the new size of the string in the queue.80 81    Operating last-chunk-first makes it easy to calculate the size written82    for recursive structures without having to build their representation as83    a string.  This is measurably faster than generating the intermediate84    strings, especially on deeply nested structures.85    """86    write = q.appendleft87    if value is None:88        write(b"0:~")89        return size + 390    elif value is True:91        write(b"4:true!")92        return size + 793    elif value is False:94        write(b"5:false!")95        return size + 896    elif isinstance(value, int):97        data = str(value).encode()98        ldata = len(data)99        span = str(ldata).encode()100        write(b"%s:%s#" % (span, data))101        return size + 2 + len(span) + ldata102    elif isinstance(value, float):103        #  Use repr() for float rather than str().104        #  It round-trips more accurately.105        #  Probably unnecessary in later python versions that106        #  use David Gay's ftoa routines.107        data = repr(value).encode()108        ldata = len(data)109        span = str(ldata).encode()110        write(b"%s:%s^" % (span, data))111        return size + 2 + len(span) + ldata112    elif isinstance(value, bytes):113        data = value114        ldata = len(data)115        span = str(ldata).encode()116        write(b",")117        write(data)118        write(b":")119        write(span)120        return size + 2 + len(span) + ldata121    elif isinstance(value, str):122        data = value.encode("utf8")123        ldata = len(data)124        span = str(ldata).encode()125        write(b";")126        write(data)127        write(b":")128        write(span)129        return size + 2 + len(span) + ldata130    elif isinstance(value, (list, tuple)):131        write(b"]")132        init_size = size = size + 1133        for item in reversed(value):134            size = _rdumpq(q, size, item)135        span = str(size - init_size).encode()136        write(b":")137        write(span)138        return size + 1 + len(span)139    elif isinstance(value, dict):140        write(b"}")141        init_size = size = size + 1142        for k, v in value.items():143            size = _rdumpq(q, size, v)144            size = _rdumpq(q, size, k)145        span = str(size - init_size).encode()146        write(b":")147        write(span)148        return size + 1 + len(span)149    else:150        raise ValueError(f"unserializable object: {value} ({type(value)})")151 152 153def loads(string: bytes) -> TSerializable:154    """155    This function parses a tnetstring into a python object.156    """157    return pop(memoryview(string))[0]158 159 160def load(file_handle: BinaryIO) -> TSerializable:161    """load(file) -> object162 163    This function reads a tnetstring from a file and parses it into a164    python object.  The file must support the read() method, and this165    function promises not to read more data than necessary.166    """167    #  Read the length prefix one char at a time.168    #  Note that the netstring spec explicitly forbids padding zeros.169    c = file_handle.read(1)170    if c == b"":  # we want to detect this special case.171        raise ValueError("not a tnetstring: empty file")172    data_length = b""173    while c.isdigit():174        data_length += c175        if len(data_length) > 12:176            raise ValueError("not a tnetstring: absurdly large length prefix")177        c = file_handle.read(1)178    if c != b":":179        raise ValueError("not a tnetstring: missing or invalid length prefix")180 181    data = memoryview(file_handle.read(int(data_length)))182    data_type = file_handle.read(1)[0]183 184    return parse(data_type, data)185 186 187def parse(data_type: int, data: memoryview) -> TSerializable:188    if data_type == ord(b","):189        return data.tobytes()190    if data_type == ord(b";"):191        return str(data, "utf8")192    if data_type == ord(b"#"):193        try:194            return int(data)195        except ValueError:196            raise ValueError(f"not a tnetstring: invalid integer literal: {data!r}")197    if data_type == ord(b"^"):198        try:199            return float(data)200        except ValueError:201            raise ValueError(f"not a tnetstring: invalid float literal: {data!r}")202    if data_type == ord(b"!"):203        if data == b"true":204            return True205        elif data == b"false":206            return False207        else:208            raise ValueError(f"not a tnetstring: invalid boolean literal: {data!r}")209    if data_type == ord(b"~"):210        if data:211            raise ValueError(f"not a tnetstring: invalid null literal: {data!r}")212        return None213    if data_type == ord(b"]"):214        lst = []215        while data:216            item, data = pop(data)217            lst.append(item)  # type: ignore218        return lst219    if data_type == ord(b"}"):220        d = {}221        while data:222            key, data = pop(data)223            val, data = pop(data)224            d[key] = val  # type: ignore225        return d226    raise ValueError(f"unknown type tag: {data_type}")227 228 229def split(data: memoryview, sep: bytes) -> tuple[int, memoryview]:230    i = 0231    try:232        ord_sep = ord(sep)233        while data[i] != ord_sep:234            i += 1235        # here i is the position of b":" in the memoryview236        return int(data[:i]), data[i + 1 :]237    except (IndexError, ValueError):238        raise ValueError(239            f"not a tnetstring: missing or invalid length prefix: {data.tobytes()!r}"240        )241 242 243def pop(data: memoryview) -> tuple[TSerializable, memoryview]:244    """245    This function parses a tnetstring into a python object.246    It returns a tuple giving the parsed object and a string247    containing any unparsed data from the end of the string.248    """249    # Parse out data length, type and remaining string.250    length, data = split(data, b":")251    try:252        data, data_type, remain = data[:length], data[length], data[length + 1 :]253    except IndexError:254        #  This fires if len(data) < dlen, meaning we don't need255        #  to further validate that data is the right length.256        raise ValueError(f"not a tnetstring: invalid length prefix: {length}")257    # Parse the data based on the type tag.258    return parse(data_type, data), remain259 260 261__all__ = ["dump", "dumps", "load", "loads", "pop"]262 
codekingpro/portable-devtools · Team Ai