codekingpro/portable-devtools
115k
1"""2tnetstring: data serialization using typed netstrings3======================================================4 5This is a custom Python 3 implementation of tnetstrings.6Compared to other implementations, the main difference7is that this implementation supports a custom unicode datatype.8 9An ordinary tnetstring is a blob of data prefixed with its length and postfixed10with its type. Here are some examples:11 12 >>> tnetstring.dumps("hello world")13 11:hello world,14 >>> tnetstring.dumps(12345)15 5:12345#16 >>> tnetstring.dumps([12345, True, 0])17 19:5:12345#4:true!1:0#]18 19This module gives you the following functions:20 21 :dump: dump an object as a tnetstring to a file22 :dumps: dump an object as a tnetstring to a string23 :load: load a tnetstring-encoded object from a file24 :loads: load a tnetstring-encoded object from a string25 26Note that since parsing a tnetstring requires reading all the data into memory27at once, there's no efficiency gain from using the file-based versions of these28functions. They're only here so you can use load() to read precisely one29item from a file or socket without consuming any extra data.30 31The tnetstrings specification explicitly states that strings are binary blobs32and forbids the use of unicode at the protocol level.33**This implementation decodes dictionary keys as surrogate-escaped ASCII**,34all other strings are returned as plain bytes.35 36:Copyright: (c) 2012-2013 by Ryan Kelly <ryan@rfk.id.au>.37:Copyright: (c) 2014 by Carlo Pires <carlopires@gmail.com>.38:Copyright: (c) 2016 by Maximilian Hils <tnetstring3@maximilianhils.com>.39 40:License: MIT41"""42 43import collections44from typing import BinaryIO45from typing import Union46 47TSerializable = Union[None, str, bool, int, float, bytes, list, tuple, dict]48 49 50def dumps(value: TSerializable) -> bytes:51 """52 This function dumps a python object as a tnetstring.53 """54 # This uses a deque to collect output fragments in reverse order,55 # then joins them together at the end. It's measurably faster56 # than creating all the intermediate strings.57 q: collections.deque = collections.deque()58 _rdumpq(q, 0, value)59 return b"".join(q)60 61 62def dump(value: TSerializable, file_handle: BinaryIO) -> None:63 """64 This function dumps a python object as a tnetstring and65 writes it to the given file.66 """67 file_handle.write(dumps(value))68 69 70def _rdumpq(q: collections.deque, size: int, value: TSerializable) -> int:71 """72 Dump value as a tnetstring, to a deque instance, last chunks first.73 74 This function generates the tnetstring representation of the given value,75 pushing chunks of the output onto the given deque instance. It pushes76 the last chunk first, then recursively generates more chunks.77 78 When passed in the current size of the string in the queue, it will return79 the new size of the string in the queue.80 81 Operating last-chunk-first makes it easy to calculate the size written82 for recursive structures without having to build their representation as83 a string. This is measurably faster than generating the intermediate84 strings, especially on deeply nested structures.85 """86 write = q.appendleft87 if value is None:88 write(b"0:~")89 return size + 390 elif value is True:91 write(b"4:true!")92 return size + 793 elif value is False:94 write(b"5:false!")95 return size + 896 elif isinstance(value, int):97 data = str(value).encode()98 ldata = len(data)99 span = str(ldata).encode()100 write(b"%s:%s#" % (span, data))101 return size + 2 + len(span) + ldata102 elif isinstance(value, float):103 # Use repr() for float rather than str().104 # It round-trips more accurately.105 # Probably unnecessary in later python versions that106 # use David Gay's ftoa routines.107 data = repr(value).encode()108 ldata = len(data)109 span = str(ldata).encode()110 write(b"%s:%s^" % (span, data))111 return size + 2 + len(span) + ldata112 elif isinstance(value, bytes):113 data = value114 ldata = len(data)115 span = str(ldata).encode()116 write(b",")117 write(data)118 write(b":")119 write(span)120 return size + 2 + len(span) + ldata121 elif isinstance(value, str):122 data = value.encode("utf8")123 ldata = len(data)124 span = str(ldata).encode()125 write(b";")126 write(data)127 write(b":")128 write(span)129 return size + 2 + len(span) + ldata130 elif isinstance(value, (list, tuple)):131 write(b"]")132 init_size = size = size + 1133 for item in reversed(value):134 size = _rdumpq(q, size, item)135 span = str(size - init_size).encode()136 write(b":")137 write(span)138 return size + 1 + len(span)139 elif isinstance(value, dict):140 write(b"}")141 init_size = size = size + 1142 for k, v in value.items():143 size = _rdumpq(q, size, v)144 size = _rdumpq(q, size, k)145 span = str(size - init_size).encode()146 write(b":")147 write(span)148 return size + 1 + len(span)149 else:150 raise ValueError(f"unserializable object: {value} ({type(value)})")151 152 153def loads(string: bytes) -> TSerializable:154 """155 This function parses a tnetstring into a python object.156 """157 return pop(memoryview(string))[0]158 159 160def load(file_handle: BinaryIO) -> TSerializable:161 """load(file) -> object162 163 This function reads a tnetstring from a file and parses it into a164 python object. The file must support the read() method, and this165 function promises not to read more data than necessary.166 """167 # Read the length prefix one char at a time.168 # Note that the netstring spec explicitly forbids padding zeros.169 c = file_handle.read(1)170 if c == b"": # we want to detect this special case.171 raise ValueError("not a tnetstring: empty file")172 data_length = b""173 while c.isdigit():174 data_length += c175 if len(data_length) > 12:176 raise ValueError("not a tnetstring: absurdly large length prefix")177 c = file_handle.read(1)178 if c != b":":179 raise ValueError("not a tnetstring: missing or invalid length prefix")180 181 data = memoryview(file_handle.read(int(data_length)))182 data_type = file_handle.read(1)[0]183 184 return parse(data_type, data)185 186 187def parse(data_type: int, data: memoryview) -> TSerializable:188 if data_type == ord(b","):189 return data.tobytes()190 if data_type == ord(b";"):191 return str(data, "utf8")192 if data_type == ord(b"#"):193 try:194 return int(data)195 except ValueError:196 raise ValueError(f"not a tnetstring: invalid integer literal: {data!r}")197 if data_type == ord(b"^"):198 try:199 return float(data)200 except ValueError:201 raise ValueError(f"not a tnetstring: invalid float literal: {data!r}")202 if data_type == ord(b"!"):203 if data == b"true":204 return True205 elif data == b"false":206 return False207 else:208 raise ValueError(f"not a tnetstring: invalid boolean literal: {data!r}")209 if data_type == ord(b"~"):210 if data:211 raise ValueError(f"not a tnetstring: invalid null literal: {data!r}")212 return None213 if data_type == ord(b"]"):214 lst = []215 while data:216 item, data = pop(data)217 lst.append(item) # type: ignore218 return lst219 if data_type == ord(b"}"):220 d = {}221 while data:222 key, data = pop(data)223 val, data = pop(data)224 d[key] = val # type: ignore225 return d226 raise ValueError(f"unknown type tag: {data_type}")227 228 229def split(data: memoryview, sep: bytes) -> tuple[int, memoryview]:230 i = 0231 try:232 ord_sep = ord(sep)233 while data[i] != ord_sep:234 i += 1235 # here i is the position of b":" in the memoryview236 return int(data[:i]), data[i + 1 :]237 except (IndexError, ValueError):238 raise ValueError(239 f"not a tnetstring: missing or invalid length prefix: {data.tobytes()!r}"240 )241 242 243def pop(data: memoryview) -> tuple[TSerializable, memoryview]:244 """245 This function parses a tnetstring into a python object.246 It returns a tuple giving the parsed object and a string247 containing any unparsed data from the end of the string.248 """249 # Parse out data length, type and remaining string.250 length, data = split(data, b":")251 try:252 data, data_type, remain = data[:length], data[length], data[length + 1 :]253 except IndexError:254 # This fires if len(data) < dlen, meaning we don't need255 # to further validate that data is the right length.256 raise ValueError(f"not a tnetstring: invalid length prefix: {length}")257 # Parse the data based on the type tag.258 return parse(data_type, data), remain259 260 261__all__ = ["dump", "dumps", "load", "loads", "pop"]262 