Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
state_inline.py167 linesDownload Raw Back to rules_inline
1from __future__ import annotations2 3from collections import namedtuple4from dataclasses import dataclass5from typing import TYPE_CHECKING, Any, Literal6 7from .._compat import DATACLASS_KWARGS8from ..common.utils import isMdAsciiPunct, isPunctChar, isWhiteSpace9from ..ruler import StateBase10from ..token import Token11from ..utils import EnvType12 13if TYPE_CHECKING:14    from markdown_it import MarkdownIt15 16 17@dataclass(**DATACLASS_KWARGS)18class Delimiter:19    # Char code of the starting marker (number).20    marker: int21 22    # Total length of these series of delimiters.23    length: int24 25    # A position of the token this delimiter corresponds to.26    token: int27 28    # If this delimiter is matched as a valid opener, `end` will be29    # equal to its position, otherwise it's `-1`.30    end: int31 32    # Boolean flags that determine if this delimiter could open or close33    # an emphasis.34    open: bool35    close: bool36 37    level: bool | None = None38 39 40Scanned = namedtuple("Scanned", ["can_open", "can_close", "length"])41 42 43class StateInline(StateBase):44    def __init__(45        self, src: str, md: MarkdownIt, env: EnvType, outTokens: list[Token]46    ) -> None:47        self.src = src48        self.env = env49        self.md = md50        self.tokens = outTokens51        self.tokens_meta: list[dict[str, Any] | None] = [None] * len(outTokens)52 53        self.pos = 054        self.posMax = len(self.src)55        self.level = 056        self.pending = ""57        self.pendingLevel = 058 59        # Stores { start: end } pairs. Useful for backtrack60        # optimization of pairs parse (emphasis, strikes).61        self.cache: dict[int, int] = {}62 63        # List of emphasis-like delimiters for current tag64        self.delimiters: list[Delimiter] = []65 66        # Stack of delimiter lists for upper level tags67        self._prev_delimiters: list[list[Delimiter]] = []68 69        # backticklength => last seen position70        self.backticks: dict[int, int] = {}71        self.backticksScanned = False72 73        # Counter used to disable inline linkify-it execution74        # inside <a> and markdown links75        self.linkLevel = 076 77    def __repr__(self) -> str:78        return (79            f"{self.__class__.__name__}"80            f"(pos=[{self.pos} of {self.posMax}], token={len(self.tokens)})"81        )82 83    def pushPending(self) -> Token:84        token = Token("text", "", 0)85        token.content = self.pending86        token.level = self.pendingLevel87        self.tokens.append(token)88        self.pending = ""89        return token90 91    def push(self, ttype: str, tag: str, nesting: Literal[-1, 0, 1]) -> Token:92        """Push new token to "stream".93        If pending text exists - flush it as text token94        """95        if self.pending:96            self.pushPending()97 98        token = Token(ttype, tag, nesting)99        token_meta = None100 101        if nesting < 0:102            # closing tag103            self.level -= 1104            self.delimiters = self._prev_delimiters.pop()105 106        token.level = self.level107 108        if nesting > 0:109            # opening tag110            self.level += 1111            self._prev_delimiters.append(self.delimiters)112            self.delimiters = []113            token_meta = {"delimiters": self.delimiters}114 115        self.pendingLevel = self.level116        self.tokens.append(token)117        self.tokens_meta.append(token_meta)118        return token119 120    def scanDelims(self, start: int, canSplitWord: bool) -> Scanned:121        """122        Scan a sequence of emphasis-like markers, and determine whether123        it can start an emphasis sequence or end an emphasis sequence.124 125         - start - position to scan from (it should point at a valid marker);126         - canSplitWord - determine if these markers can be found inside a word127 128        """129        pos = start130        maximum = self.posMax131        marker = self.src[start]132 133        # treat beginning of the line as a whitespace134        lastChar = self.src[start - 1] if start > 0 else " "135 136        while pos < maximum and self.src[pos] == marker:137            pos += 1138 139        count = pos - start140 141        # treat end of the line as a whitespace142        nextChar = self.src[pos] if pos < maximum else " "143 144        isLastPunctChar = isMdAsciiPunct(ord(lastChar)) or isPunctChar(lastChar)145        isNextPunctChar = isMdAsciiPunct(ord(nextChar)) or isPunctChar(nextChar)146 147        isLastWhiteSpace = isWhiteSpace(ord(lastChar))148        isNextWhiteSpace = isWhiteSpace(ord(nextChar))149 150        left_flanking = not (151            isNextWhiteSpace152            or (isNextPunctChar and not (isLastWhiteSpace or isLastPunctChar))153        )154        right_flanking = not (155            isLastWhiteSpace156            or (isLastPunctChar and not (isNextWhiteSpace or isNextPunctChar))157        )158 159        if not canSplitWord:160            can_open = left_flanking and ((not right_flanking) or isLastPunctChar)161            can_close = right_flanking and ((not left_flanking) or isNextPunctChar)162        else:163            can_open = left_flanking164            can_close = right_flanking165 166        return Scanned(can_open, can_close, count)167 
codekingpro/portable-devtools · Team Ai