Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
smartquotes.py203 linesDownload Raw Back to rules_core
1"""Convert straight quotation marks to typographic ones2"""3from __future__ import annotations4 5import re6from typing import Any7 8from ..common.utils import charCodeAt, isMdAsciiPunct, isPunctChar, isWhiteSpace9from ..token import Token10from .state_core import StateCore11 12QUOTE_TEST_RE = re.compile(r"['\"]")13QUOTE_RE = re.compile(r"['\"]")14APOSTROPHE = "\u2019"  # ’15 16 17def replaceAt(string: str, index: int, ch: str) -> str:18    # When the index is negative, the behavior is different from the js version.19    # But basically, the index will not be negative.20    assert index >= 021    return string[:index] + ch + string[index + 1 :]22 23 24def process_inlines(tokens: list[Token], state: StateCore) -> None:25    stack: list[dict[str, Any]] = []26 27    for i, token in enumerate(tokens):28        thisLevel = token.level29 30        j = 031        for j in range(len(stack))[::-1]:32            if stack[j]["level"] <= thisLevel:33                break34        else:35            # When the loop is terminated without a "break".36            # Subtract 1 to get the same index as the js version.37            j -= 138 39        stack = stack[: j + 1]40 41        if token.type != "text":42            continue43 44        text = token.content45        pos = 046        maximum = len(text)47 48        while pos < maximum:49            goto_outer = False50            lastIndex = pos51            t = QUOTE_RE.search(text[lastIndex:])52            if not t:53                break54 55            canOpen = canClose = True56            pos = t.start(0) + lastIndex + 157            isSingle = t.group(0) == "'"58 59            # Find previous character,60            # default to space if it's the beginning of the line61            lastChar: None | int = 0x2062 63            if t.start(0) + lastIndex - 1 >= 0:64                lastChar = charCodeAt(text, t.start(0) + lastIndex - 1)65            else:66                for j in range(i)[::-1]:67                    if tokens[j].type == "softbreak" or tokens[j].type == "hardbreak":68                        break69                    # should skip all tokens except 'text', 'html_inline' or 'code_inline'70                    if not tokens[j].content:71                        continue72 73                    lastChar = charCodeAt(tokens[j].content, len(tokens[j].content) - 1)74                    break75 76            # Find next character,77            # default to space if it's the end of the line78            nextChar: None | int = 0x2079 80            if pos < maximum:81                nextChar = charCodeAt(text, pos)82            else:83                for j in range(i + 1, len(tokens)):84                    # nextChar defaults to 0x2085                    if tokens[j].type == "softbreak" or tokens[j].type == "hardbreak":86                        break87                    # should skip all tokens except 'text', 'html_inline' or 'code_inline'88                    if not tokens[j].content:89                        continue90 91                    nextChar = charCodeAt(tokens[j].content, 0)92                    break93 94            isLastPunctChar = lastChar is not None and (95                isMdAsciiPunct(lastChar) or isPunctChar(chr(lastChar))96            )97            isNextPunctChar = nextChar is not None and (98                isMdAsciiPunct(nextChar) or isPunctChar(chr(nextChar))99            )100 101            isLastWhiteSpace = lastChar is not None and isWhiteSpace(lastChar)102            isNextWhiteSpace = nextChar is not None and isWhiteSpace(nextChar)103 104            if isNextWhiteSpace:  # noqa: SIM114105                canOpen = False106            elif isNextPunctChar and not (isLastWhiteSpace or isLastPunctChar):107                canOpen = False108 109            if isLastWhiteSpace:  # noqa: SIM114110                canClose = False111            elif isLastPunctChar and not (isNextWhiteSpace or isNextPunctChar):112                canClose = False113 114            if nextChar == 0x22 and t.group(0) == '"':  # 0x22: "  # noqa: SIM102115                if (116                    lastChar is not None and lastChar >= 0x30 and lastChar <= 0x39117                ):  # 0x30: 0, 0x39: 9118                    # special case: 1"" - count first quote as an inch119                    canClose = canOpen = False120 121            if canOpen and canClose:122                # Replace quotes in the middle of punctuation sequence, but not123                # in the middle of the words, i.e.:124                #125                # 1. foo " bar " baz - not replaced126                # 2. foo-"-bar-"-baz - replaced127                # 3. foo"bar"baz     - not replaced128                canOpen = isLastPunctChar129                canClose = isNextPunctChar130 131            if not canOpen and not canClose:132                # middle of word133                if isSingle:134                    token.content = replaceAt(135                        token.content, t.start(0) + lastIndex, APOSTROPHE136                    )137                continue138 139            if canClose:140                # this could be a closing quote, rewind the stack to get a match141                for j in range(len(stack))[::-1]:142                    item = stack[j]143                    if stack[j]["level"] < thisLevel:144                        break145                    if item["single"] == isSingle and stack[j]["level"] == thisLevel:146                        item = stack[j]147 148                        if isSingle:149                            openQuote = state.md.options.quotes[2]150                            closeQuote = state.md.options.quotes[3]151                        else:152                            openQuote = state.md.options.quotes[0]153                            closeQuote = state.md.options.quotes[1]154 155                        # replace token.content *before* tokens[item.token].content,156                        # because, if they are pointing at the same token, replaceAt157                        # could mess up indices when quote length != 1158                        token.content = replaceAt(159                            token.content, t.start(0) + lastIndex, closeQuote160                        )161                        tokens[item["token"]].content = replaceAt(162                            tokens[item["token"]].content, item["pos"], openQuote163                        )164 165                        pos += len(closeQuote) - 1166                        if item["token"] == i:167                            pos += len(openQuote) - 1168 169                        text = token.content170                        maximum = len(text)171 172                        stack = stack[:j]173                        goto_outer = True174                        break175                if goto_outer:176                    goto_outer = False177                    continue178 179            if canOpen:180                stack.append(181                    {182                        "token": i,183                        "pos": t.start(0) + lastIndex,184                        "single": isSingle,185                        "level": thisLevel,186                    }187                )188            elif canClose and isSingle:189                token.content = replaceAt(190                    token.content, t.start(0) + lastIndex, APOSTROPHE191                )192 193 194def smartquotes(state: StateCore) -> None:195    if not state.md.options.typographer:196        return197 198    for token in state.tokens:199        if token.type != "inline" or not QUOTE_RE.search(token.content):200            continue201        if token.children is not None:202            process_inlines(token.children, state)203 
codekingpro/portable-devtools · Team Ai