codekingpro/portable-devtools
114k
1"""Convert straight quotation marks to typographic ones"""2 3from __future__ import annotations4 5import re6from typing import Any7 8from ..common.utils import charCodeAt, isMdAsciiPunct, isPunctChar, isWhiteSpace9from ..token import Token10from .state_core import StateCore11 12QUOTE_TEST_RE = re.compile(r"['\"]")13QUOTE_RE = re.compile(r"['\"]")14APOSTROPHE = "\u2019" # ’15 16 17def replaceAt(string: str, index: int, ch: str) -> str:18 # When the index is negative, the behavior is different from the js version.19 # But basically, the index will not be negative.20 assert index >= 021 return string[:index] + ch + string[index + 1 :]22 23 24def process_inlines(tokens: list[Token], state: StateCore) -> None:25 stack: list[dict[str, Any]] = []26 27 for i, token in enumerate(tokens):28 thisLevel = token.level29 30 j = 031 for j in range(len(stack))[::-1]:32 if stack[j]["level"] <= thisLevel:33 break34 else:35 # When the loop is terminated without a "break".36 # Subtract 1 to get the same index as the js version.37 j -= 138 39 stack = stack[: j + 1]40 41 if token.type != "text":42 continue43 44 text = token.content45 pos = 046 maximum = len(text)47 48 while pos < maximum:49 goto_outer = False50 lastIndex = pos51 t = QUOTE_RE.search(text[lastIndex:])52 if not t:53 break54 55 canOpen = canClose = True56 pos = t.start(0) + lastIndex + 157 isSingle = t.group(0) == "'"58 59 # Find previous character,60 # default to space if it's the beginning of the line61 lastChar: None | int = 0x2062 63 if t.start(0) + lastIndex - 1 >= 0:64 lastChar = charCodeAt(text, t.start(0) + lastIndex - 1)65 else:66 for j in range(i)[::-1]:67 if tokens[j].type == "softbreak" or tokens[j].type == "hardbreak":68 break69 # should skip all tokens except 'text', 'html_inline' or 'code_inline'70 if not tokens[j].content:71 continue72 73 lastChar = charCodeAt(tokens[j].content, len(tokens[j].content) - 1)74 break75 76 # Find next character,77 # default to space if it's the end of the line78 nextChar: None | int = 0x2079 80 if pos < maximum:81 nextChar = charCodeAt(text, pos)82 else:83 for j in range(i + 1, len(tokens)):84 # nextChar defaults to 0x2085 if tokens[j].type == "softbreak" or tokens[j].type == "hardbreak":86 break87 # should skip all tokens except 'text', 'html_inline' or 'code_inline'88 if not tokens[j].content:89 continue90 91 nextChar = charCodeAt(tokens[j].content, 0)92 break93 94 isLastPunctChar = lastChar is not None and (95 isMdAsciiPunct(lastChar) or isPunctChar(chr(lastChar))96 )97 isNextPunctChar = nextChar is not None and (98 isMdAsciiPunct(nextChar) or isPunctChar(chr(nextChar))99 )100 101 isLastWhiteSpace = lastChar is not None and isWhiteSpace(lastChar)102 isNextWhiteSpace = nextChar is not None and isWhiteSpace(nextChar)103 104 if isNextWhiteSpace: # noqa: SIM114105 canOpen = False106 elif isNextPunctChar and not (isLastWhiteSpace or isLastPunctChar):107 canOpen = False108 109 if isLastWhiteSpace: # noqa: SIM114110 canClose = False111 elif isLastPunctChar and not (isNextWhiteSpace or isNextPunctChar):112 canClose = False113 114 if nextChar == 0x22 and t.group(0) == '"': # 0x22: " # noqa: SIM102115 if (116 lastChar is not None and lastChar >= 0x30 and lastChar <= 0x39117 ): # 0x30: 0, 0x39: 9118 # special case: 1"" - count first quote as an inch119 canClose = canOpen = False120 121 if canOpen and canClose:122 # Replace quotes in the middle of punctuation sequence, but not123 # in the middle of the words, i.e.:124 #125 # 1. foo " bar " baz - not replaced126 # 2. foo-"-bar-"-baz - replaced127 # 3. foo"bar"baz - not replaced128 canOpen = isLastPunctChar129 canClose = isNextPunctChar130 131 if not canOpen and not canClose:132 # middle of word133 if isSingle:134 token.content = replaceAt(135 token.content, t.start(0) + lastIndex, APOSTROPHE136 )137 continue138 139 if canClose:140 # this could be a closing quote, rewind the stack to get a match141 for j in range(len(stack))[::-1]:142 item = stack[j]143 if stack[j]["level"] < thisLevel:144 break145 if item["single"] == isSingle and stack[j]["level"] == thisLevel:146 item = stack[j]147 148 if isSingle:149 openQuote = state.md.options.quotes[2]150 closeQuote = state.md.options.quotes[3]151 else:152 openQuote = state.md.options.quotes[0]153 closeQuote = state.md.options.quotes[1]154 155 # replace token.content *before* tokens[item.token].content,156 # because, if they are pointing at the same token, replaceAt157 # could mess up indices when quote length != 1158 token.content = replaceAt(159 token.content, t.start(0) + lastIndex, closeQuote160 )161 tokens[item["token"]].content = replaceAt(162 tokens[item["token"]].content, item["pos"], openQuote163 )164 165 pos += len(closeQuote) - 1166 if item["token"] == i:167 pos += len(openQuote) - 1168 169 text = token.content170 maximum = len(text)171 172 stack = stack[:j]173 goto_outer = True174 break175 if goto_outer:176 goto_outer = False177 continue178 179 if canOpen:180 stack.append(181 {182 "token": i,183 "pos": t.start(0) + lastIndex,184 "single": isSingle,185 "level": thisLevel,186 }187 )188 elif canClose and isSingle:189 token.content = replaceAt(190 token.content, t.start(0) + lastIndex, APOSTROPHE191 )192 193 194def smartquotes(state: StateCore) -> None:195 if not state.md.options.typographer:196 return197 198 for token in state.tokens:199 if token.type != "inline" or not QUOTE_RE.search(token.content):200 continue201 if token.children is not None:202 process_inlines(token.children, state)203 