Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
_view_xml_html.py279 linesDownload Raw Back to contentviews
1import io2import re3import textwrap4from collections.abc import Iterable5 6from mitmproxy.contentviews._api import Contentview7from mitmproxy.contentviews._api import Metadata8from mitmproxy.utils import sliding_window9from mitmproxy.utils import strutils10 11"""12A custom XML/HTML prettifier. Compared to other prettifiers, its main features are:13 14- Implemented in pure Python.15- Modifies whitespace only.16- Works with any input.17- Lazy evaluation.18 19The implementation is split into two main parts: tokenization and formatting of tokens.20"""21 22# http://www.xml.com/pub/a/2001/07/25/namingparts.html - this is close enough for what we do.23REGEX_TAG = re.compile(r"[a-zA-Z0-9._:\-]+(?!=)")24# https://www.w3.org/TR/html5/syntax.html#void-elements25HTML_VOID_ELEMENTS = {26    "area",27    "base",28    "br",29    "col",30    "embed",31    "hr",32    "img",33    "input",34    "keygen",35    "link",36    "meta",37    "param",38    "source",39    "track",40    "wbr",41}42NO_INDENT_TAGS = {"xml", "doctype", "html"}43INDENT = 244 45 46class Token:47    def __init__(self, data):48        self.data = data49 50    def __repr__(self):51        return f"{type(self).__name__}({self.data})"52 53 54class Text(Token):55    @property56    def text(self):57        return self.data.strip()58 59 60class Tag(Token):61    @property62    def tag(self):63        t = REGEX_TAG.search(self.data)64        if t is not None:65            return t.group(0).lower()66        return "<empty>"67 68    @property69    def is_comment(self) -> bool:70        return self.data.startswith("<!--")71 72    @property73    def is_cdata(self) -> bool:74        return self.data.startswith("<![CDATA[")75 76    @property77    def is_closing(self):78        return self.data.startswith("</")79 80    @property81    def is_self_closing(self):82        return (83            self.is_comment84            or self.is_cdata85            or self.data.endswith("/>")86            or self.tag in HTML_VOID_ELEMENTS87        )88 89    @property90    def is_opening(self):91        return not self.is_closing and not self.is_self_closing92 93    @property94    def done(self):95        if self.is_comment:96            return self.data.endswith("-->")97        elif self.is_cdata:98            return self.data.endswith("]]>")99        else:100            # This fails for attributes that contain an unescaped ">"101            return self.data.endswith(">")102 103 104def tokenize(data: str) -> Iterable[Token]:105    token: Token = Text("")106 107    i = 0108 109    def readuntil(char, start, include=1):110        nonlocal i111        end = data.find(char, start)112        if end == -1:113            end = len(data)114        ret = data[i : end + include]115        i = end + include116        return ret117 118    while i < len(data):119        if isinstance(token, Text):120            token.data = readuntil("<", i, 0)121            if token.text:122                yield token123            token = Tag("")124        elif isinstance(token, Tag):125            token.data += readuntil(">", i, 1)126            if token.done:127                yield token128                token = Text("")129    if token.data.strip():130        yield token131 132 133def indent_text(data: str, prefix: str) -> str:134    # Add spacing to first line so that we dedent in cases like this:135    # <li>This is136    #     example text137    #     over multiple lines138    # </li>139    dedented = textwrap.dedent(" " * 32 + data).strip()140    return textwrap.indent(dedented, prefix[:32])141 142 143def is_inline_text(a: Token | None, b: Token | None, c: Token | None) -> bool:144    if isinstance(a, Tag) and isinstance(b, Text) and isinstance(c, Tag):145        if a.is_opening and "\n" not in b.data and c.is_closing and a.tag == c.tag:146            return True147    return False148 149 150def is_inline(151    prev2: Token | None,152    prev1: Token | None,153    t: Token | None,154    next1: Token | None,155    next2: Token | None,156) -> bool:157    if isinstance(t, Text):158        return is_inline_text(prev1, t, next1)159    elif isinstance(t, Tag):160        if is_inline_text(prev2, prev1, t) or is_inline_text(t, next1, next2):161            return True162        if (163            isinstance(next1, Tag)164            and t.is_opening165            and next1.is_closing166            and t.tag == next1.tag167        ):168            return True  # <div></div> (start tag)169        if (170            isinstance(prev1, Tag)171            and prev1.is_opening172            and t.is_closing173            and prev1.tag == t.tag174        ):175            return True  # <div></div> (end tag)176    return False177 178 179class ElementStack:180    """181    Keep track of how deeply nested our document is.182    """183 184    def __init__(self):185        self.open_tags = []186        self.indent = ""187 188    def push_tag(self, tag: str):189        if len(self.open_tags) > 16:190            return191        self.open_tags.append(tag)192        if tag not in NO_INDENT_TAGS:193            self.indent += " " * INDENT194 195    def pop_tag(self, tag: str):196        if tag in self.open_tags:197            remove_indent = 0198            while True:199                t = self.open_tags.pop()200                if t not in NO_INDENT_TAGS:201                    remove_indent += INDENT202                if t == tag:203                    break204            self.indent = self.indent[:-remove_indent]205        else:206            pass  # this closing tag has no start tag. let's keep indentation as-is.207 208 209def format_xml(tokens: Iterable[Token]) -> str:210    out = io.StringIO()211 212    context = ElementStack()213 214    for prev2, prev1, token, next1, next2 in sliding_window.window(tokens, 2, 2):215        if isinstance(token, Tag):216            if token.is_opening:217                out.write(indent_text(token.data, context.indent))218 219                if not is_inline(prev2, prev1, token, next1, next2):220                    out.write("\n")221 222                context.push_tag(token.tag)223            elif token.is_closing:224                context.pop_tag(token.tag)225 226                if is_inline(prev2, prev1, token, next1, next2):227                    out.write(token.data)228                else:229                    out.write(indent_text(token.data, context.indent))230                out.write("\n")231 232            else:  # self-closing233                out.write(indent_text(token.data, context.indent))234                out.write("\n")235        elif isinstance(token, Text):236            if is_inline(prev2, prev1, token, next1, next2):237                out.write(token.text)238            else:239                out.write(indent_text(token.data, context.indent))240                out.write("\n")241        else:  # pragma: no cover242            raise RuntimeError()243 244    return out.getvalue()245 246 247class XmlHtmlContentview(Contentview):248    __content_types = ("text/xml", "text/html")249    name = "XML/HTML"250    syntax_highlight = "xml"251 252    def prettify(253        self,254        data: bytes,255        metadata: Metadata,256    ) -> str:257        if metadata.http_message:258            data_str = metadata.http_message.get_text(strict=False) or ""259        else:260            data_str = data.decode("utf8", "backslashreplace")261        tokens = tokenize(data_str)262        return format_xml(tokens)263 264    def render_priority(265        self,266        data: bytes,267        metadata: Metadata,268    ) -> float:269        if not data:270            return 0271        if metadata.content_type in self.__content_types:272            return 1273        elif strutils.is_xml(data):274            return 0.4275        return 0276 277 278xml_html = XmlHtmlContentview()279 
codekingpro/portable-devtools · Team Ai