codekingpro/portable-devtools
115k
1from __future__ import annotations2 3from dataclasses import dataclass4from typing import TYPE_CHECKING, Any, Literal, NamedTuple5 6from ..common.utils import isMdAsciiPunct, isPunctChar, isWhiteSpace7from ..ruler import StateBase8from ..token import Token9from ..utils import EnvType10 11if TYPE_CHECKING:12 from markdown_it import MarkdownIt13 14 15@dataclass(slots=True)16class Delimiter:17 # Char code of the starting marker (number).18 marker: int19 20 # Total length of these series of delimiters.21 length: int22 23 # A position of the token this delimiter corresponds to.24 token: int25 26 # If this delimiter is matched as a valid opener, `end` will be27 # equal to its position, otherwise it's `-1`.28 end: int29 30 # Boolean flags that determine if this delimiter could open or close31 # an emphasis.32 open: bool33 close: bool34 35 level: bool | None = None36 37 38class Scanned(NamedTuple):39 can_open: bool40 can_close: bool41 length: int42 43 44class StateInline(StateBase):45 def __init__(46 self, src: str, md: MarkdownIt, env: EnvType, outTokens: list[Token]47 ) -> None:48 self.src = src49 self.env = env50 self.md = md51 self.tokens = outTokens52 self.tokens_meta: list[dict[str, Any] | None] = [None] * len(outTokens)53 54 self.pos = 055 self.posMax = len(self.src)56 self.level = 057 self.pending = ""58 self.pendingLevel = 059 60 # Stores { start: end } pairs. Useful for backtrack61 # optimization of pairs parse (emphasis, strikes).62 self.cache: dict[int, int] = {}63 64 # List of emphasis-like delimiters for current tag65 self.delimiters: list[Delimiter] = []66 67 # Stack of delimiter lists for upper level tags68 self._prev_delimiters: list[list[Delimiter]] = []69 70 # backticklength => last seen position71 self.backticks: dict[int, int] = {}72 self.backticksScanned = False73 74 # Counter used to disable inline linkify-it execution75 # inside <a> and markdown links76 self.linkLevel = 077 78 def __repr__(self) -> str:79 return (80 f"{self.__class__.__name__}"81 f"(pos=[{self.pos} of {self.posMax}], token={len(self.tokens)})"82 )83 84 def pushPending(self) -> Token:85 token = Token("text", "", 0)86 token.content = self.pending87 token.level = self.pendingLevel88 self.tokens.append(token)89 self.pending = ""90 return token91 92 def push(self, ttype: str, tag: str, nesting: Literal[-1, 0, 1]) -> Token:93 """Push new token to "stream".94 If pending text exists - flush it as text token95 """96 if self.pending:97 self.pushPending()98 99 token = Token(ttype, tag, nesting)100 token_meta = None101 102 if nesting < 0:103 # closing tag104 self.level -= 1105 self.delimiters = self._prev_delimiters.pop()106 107 token.level = self.level108 109 if nesting > 0:110 # opening tag111 self.level += 1112 self._prev_delimiters.append(self.delimiters)113 self.delimiters = []114 token_meta = {"delimiters": self.delimiters}115 116 self.pendingLevel = self.level117 self.tokens.append(token)118 self.tokens_meta.append(token_meta)119 return token120 121 def scanDelims(self, start: int, canSplitWord: bool) -> Scanned:122 """123 Scan a sequence of emphasis-like markers, and determine whether124 it can start an emphasis sequence or end an emphasis sequence.125 126 - start - position to scan from (it should point at a valid marker);127 - canSplitWord - determine if these markers can be found inside a word128 129 """130 pos = start131 maximum = self.posMax132 marker = self.src[start]133 134 # treat beginning of the line as a whitespace135 lastChar = self.src[start - 1] if start > 0 else " "136 137 while pos < maximum and self.src[pos] == marker:138 pos += 1139 140 count = pos - start141 142 # treat end of the line as a whitespace143 nextChar = self.src[pos] if pos < maximum else " "144 145 isLastPunctChar = isMdAsciiPunct(ord(lastChar)) or isPunctChar(lastChar)146 isNextPunctChar = isMdAsciiPunct(ord(nextChar)) or isPunctChar(nextChar)147 148 isLastWhiteSpace = isWhiteSpace(ord(lastChar))149 isNextWhiteSpace = isWhiteSpace(ord(nextChar))150 151 left_flanking = not (152 isNextWhiteSpace153 or (isNextPunctChar and not (isLastWhiteSpace or isLastPunctChar))154 )155 right_flanking = not (156 isLastWhiteSpace157 or (isLastPunctChar and not (isNextWhiteSpace or isNextPunctChar))158 )159 160 can_open = left_flanking and (161 canSplitWord or (not right_flanking) or isLastPunctChar162 )163 can_close = right_flanking and (164 canSplitWord or (not left_flanking) or isNextPunctChar165 )166 167 return Scanned(can_open, can_close, count)168 