Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
latex.py519 linesDownload Raw Back to formatters
1"""2    pygments.formatters.latex3    ~~~~~~~~~~~~~~~~~~~~~~~~~4 5    Formatter for LaTeX fancyvrb output.6 7    :copyright: Copyright 2006-present by the Pygments team, see AUTHORS.8    :license: BSD, see LICENSE for details.9"""10 11from io import StringIO12 13from pygments.formatter import Formatter14from pygments.lexer import Lexer, do_insertions15from pygments.token import Token, STANDARD_TYPES16from pygments.util import get_bool_opt, get_int_opt17 18 19__all__ = ['LatexFormatter']20 21 22def escape_tex(text, commandprefix):23    return text.replace('\\', '\x00'). \24                replace('{', '\x01'). \25                replace('}', '\x02'). \26                replace('\x00', rf'\{commandprefix}Zbs{{}}'). \27                replace('\x01', rf'\{commandprefix}Zob{{}}'). \28                replace('\x02', rf'\{commandprefix}Zcb{{}}'). \29                replace('^', rf'\{commandprefix}Zca{{}}'). \30                replace('_', rf'\{commandprefix}Zus{{}}'). \31                replace('&', rf'\{commandprefix}Zam{{}}'). \32                replace('<', rf'\{commandprefix}Zlt{{}}'). \33                replace('>', rf'\{commandprefix}Zgt{{}}'). \34                replace('#', rf'\{commandprefix}Zsh{{}}'). \35                replace('%', rf'\{commandprefix}Zpc{{}}'). \36                replace('$', rf'\{commandprefix}Zdl{{}}'). \37                replace('-', rf'\{commandprefix}Zhy{{}}'). \38                replace("'", rf'\{commandprefix}Zsq{{}}'). \39                replace('"', rf'\{commandprefix}Zdq{{}}'). \40                replace('~', rf'\{commandprefix}Zti{{}}')41 42 43DOC_TEMPLATE = r'''44\documentclass{%(docclass)s}45\usepackage{fancyvrb}46\usepackage{color}47\usepackage[%(encoding)s]{inputenc}48%(preamble)s49 50%(styledefs)s51 52\begin{document}53 54\section*{%(title)s}55 56%(code)s57\end{document}58'''59 60## Small explanation of the mess below :)61#62# The previous version of the LaTeX formatter just assigned a command to63# each token type defined in the current style.  That obviously is64# problematic if the highlighted code is produced for a different style65# than the style commands themselves.66#67# This version works much like the HTML formatter which assigns multiple68# CSS classes to each <span> tag, from the most specific to the least69# specific token type, thus falling back to the parent token type if one70# is not defined.  Here, the classes are there too and use the same short71# forms given in token.STANDARD_TYPES.72#73# Highlighted code now only uses one custom command, which by default is74# \PY and selectable by the commandprefix option (and in addition the75# escapes \PYZat, \PYZlb and \PYZrb which haven't been renamed for76# backwards compatibility purposes).77#78# \PY has two arguments: the classes, separated by +, and the text to79# render in that style.  The classes are resolved into the respective80# style commands by magic, which serves to ignore unknown classes.81#82# The magic macros are:83# * \PY@it, \PY@bf, etc. are unconditionally wrapped around the text84#   to render in \PY@do.  Their definition determines the style.85# * \PY@reset resets \PY@it etc. to do nothing.86# * \PY@toks parses the list of classes, using magic inspired by the87#   keyval package (but modified to use plusses instead of commas88#   because fancyvrb redefines commas inside its environments).89# * \PY@tok processes one class, calling the \PY@tok@classname command90#   if it exists.91# * \PY@tok@classname sets the \PY@it etc. to reflect the chosen style92#   for its class.93# * \PY resets the style, parses the classnames and then calls \PY@do.94#95# Tip: to read this code, print it out in substituted form using e.g.96# >>> print STYLE_TEMPLATE % {'cp': 'PY'}97 98STYLE_TEMPLATE = r'''99\makeatletter100\def\%(cp)s@reset{\let\%(cp)s@it=\relax \let\%(cp)s@bf=\relax%%101    \let\%(cp)s@ul=\relax \let\%(cp)s@tc=\relax%%102    \let\%(cp)s@bc=\relax \let\%(cp)s@ff=\relax}103\def\%(cp)s@tok#1{\csname %(cp)s@tok@#1\endcsname}104\def\%(cp)s@toks#1+{\ifx\relax#1\empty\else%%105    \%(cp)s@tok{#1}\expandafter\%(cp)s@toks\fi}106\def\%(cp)s@do#1{\%(cp)s@bc{\%(cp)s@tc{\%(cp)s@ul{%%107    \%(cp)s@it{\%(cp)s@bf{\%(cp)s@ff{#1}}}}}}}108\def\%(cp)s#1#2{\%(cp)s@reset\%(cp)s@toks#1+\relax+\%(cp)s@do{#2}}109 110%(styles)s111 112\def\%(cp)sZbs{\char`\\}113\def\%(cp)sZus{\char`\_}114\def\%(cp)sZob{\char`\{}115\def\%(cp)sZcb{\char`\}}116\def\%(cp)sZca{\char`\^}117\def\%(cp)sZam{\char`\&}118\def\%(cp)sZlt{\char`\<}119\def\%(cp)sZgt{\char`\>}120\def\%(cp)sZsh{\char`\#}121\def\%(cp)sZpc{\char`\%%}122\def\%(cp)sZdl{\char`\$}123\def\%(cp)sZhy{\char`\-}124\def\%(cp)sZsq{\char`\'}125\def\%(cp)sZdq{\char`\"}126\def\%(cp)sZti{\char`\~}127%% for compatibility with earlier versions128\def\%(cp)sZat{@}129\def\%(cp)sZlb{[}130\def\%(cp)sZrb{]}131\makeatother132'''133 134 135def _get_ttype_name(ttype):136    fname = STANDARD_TYPES.get(ttype)137    if fname:138        return fname139    aname = ''140    while fname is None:141        aname = ttype[-1] + aname142        ttype = ttype.parent143        fname = STANDARD_TYPES.get(ttype)144    return fname + aname145 146 147class LatexFormatter(Formatter):148    r"""149    Format tokens as LaTeX code. This needs the `fancyvrb` and `color`150    standard packages.151 152    Without the `full` option, code is formatted as one ``Verbatim``153    environment, like this:154 155    .. sourcecode:: latex156 157        \begin{Verbatim}[commandchars=\\\{\}]158        \PY{k}{def }\PY{n+nf}{foo}(\PY{n}{bar}):159            \PY{k}{pass}160        \end{Verbatim}161 162    Wrapping can be disabled using the `nowrap` option.163 164    The special command used here (``\PY``) and all the other macros it needs165    are output by the `get_style_defs` method.166 167    With the `full` option, a complete LaTeX document is output, including168    the command definitions in the preamble.169 170    The `get_style_defs()` method of a `LatexFormatter` returns a string171    containing ``\def`` commands defining the macros needed inside the172    ``Verbatim`` environments.173 174    Additional options accepted:175 176    `nowrap`177        If set to ``True``, don't wrap the tokens at all, not even inside a178        ``\begin{Verbatim}`` environment. This disables most other options179        (default: ``False``).180 181    `style`182        The style to use, can be a string or a Style subclass (default:183        ``'default'``).184 185    `full`186        Tells the formatter to output a "full" document, i.e. a complete187        self-contained document (default: ``False``).188 189    `title`190        If `full` is true, the title that should be used to caption the191        document (default: ``''``).192 193    `docclass`194        If the `full` option is enabled, this is the document class to use195        (default: ``'article'``).196 197    `preamble`198        If the `full` option is enabled, this can be further preamble commands,199        e.g. ``\usepackage`` (default: ``''``).200 201    `linenos`202        If set to ``True``, output line numbers (default: ``False``).203 204    `linenostart`205        The line number for the first line (default: ``1``).206 207    `linenostep`208        If set to a number n > 1, only every nth line number is printed.209 210    `verboptions`211        Additional options given to the Verbatim environment (see the *fancyvrb*212        docs for possible values) (default: ``''``).213 214    `commandprefix`215        The LaTeX commands used to produce colored output are constructed216        using this prefix and some letters (default: ``'PY'``).217 218        .. versionadded:: 0.7219        .. versionchanged:: 0.10220           The default is now ``'PY'`` instead of ``'C'``.221 222    `texcomments`223        If set to ``True``, enables LaTeX comment lines.  That is, LaTex markup224        in comment tokens is not escaped so that LaTeX can render it (default:225        ``False``).226 227        .. versionadded:: 1.2228 229    `mathescape`230        If set to ``True``, enables LaTeX math mode escape in comments. That231        is, ``'$...$'`` inside a comment will trigger math mode (default:232        ``False``).233 234        .. versionadded:: 1.2235 236    `escapeinside`237        If set to a string of length 2, enables escaping to LaTeX. Text238        delimited by these 2 characters is read as LaTeX code and239        typeset accordingly. It has no effect in string literals. It has240        no effect in comments if `texcomments` or `mathescape` is241        set. (default: ``''``).242 243        .. versionadded:: 2.0244 245    `envname`246        Allows you to pick an alternative environment name replacing Verbatim.247        The alternate environment still has to support Verbatim's option syntax.248        (default: ``'Verbatim'``).249 250        .. versionadded:: 2.0251    """252    name = 'LaTeX'253    aliases = ['latex', 'tex']254    filenames = ['*.tex']255 256    def __init__(self, **options):257        Formatter.__init__(self, **options)258        self.nowrap = get_bool_opt(options, 'nowrap', False)259        self.docclass = options.get('docclass', 'article')260        self.preamble = options.get('preamble', '')261        self.linenos = get_bool_opt(options, 'linenos', False)262        self.linenostart = abs(get_int_opt(options, 'linenostart', 1))263        self.linenostep = abs(get_int_opt(options, 'linenostep', 1))264        self.verboptions = options.get('verboptions', '')265        self.nobackground = get_bool_opt(options, 'nobackground', False)266        self.commandprefix = options.get('commandprefix', 'PY')267        self.texcomments = get_bool_opt(options, 'texcomments', False)268        self.mathescape = get_bool_opt(options, 'mathescape', False)269        self.escapeinside = options.get('escapeinside', '')270        if len(self.escapeinside) == 2:271            self.left = self.escapeinside[0]272            self.right = self.escapeinside[1]273        else:274            self.escapeinside = ''275        self.envname = options.get('envname', 'Verbatim')276 277        self._create_stylesheet()278 279    def _create_stylesheet(self):280        t2n = self.ttype2name = {Token: ''}281        c2d = self.cmd2def = {}282        cp = self.commandprefix283 284        def rgbcolor(col):285            if col:286                return ','.join(['%.2f' % (int(col[i] + col[i + 1], 16) / 255.0)287                                 for i in (0, 2, 4)])288            else:289                return '1,1,1'290 291        for ttype, ndef in self.style:292            name = _get_ttype_name(ttype)293            cmndef = ''294            if ndef['bold']:295                cmndef += r'\let\$$@bf=\textbf'296            if ndef['italic']:297                cmndef += r'\let\$$@it=\textit'298            if ndef['underline']:299                cmndef += r'\let\$$@ul=\underline'300            if ndef['roman']:301                cmndef += r'\let\$$@ff=\textrm'302            if ndef['sans']:303                cmndef += r'\let\$$@ff=\textsf'304            if ndef['mono']:305                cmndef += r'\let\$$@ff=\textsf'306            if ndef['color']:307                cmndef += (r'\def\$$@tc##1{{\textcolor[rgb]{{{}}}{{##1}}}}'.format(rgbcolor(ndef['color'])))308            if ndef['border']:309                cmndef += (r'\def\$$@bc##1{{{{\setlength{{\fboxsep}}{{\string -\fboxrule}}'310                           r'\fcolorbox[rgb]{{{}}}{{{}}}{{\strut ##1}}}}}}'.format(rgbcolor(ndef['border']),311                            rgbcolor(ndef['bgcolor'])))312            elif ndef['bgcolor']:313                cmndef += (r'\def\$$@bc##1{{{{\setlength{{\fboxsep}}{{0pt}}'314                           r'\colorbox[rgb]{{{}}}{{\strut ##1}}}}}}'.format(rgbcolor(ndef['bgcolor'])))315            if cmndef == '':316                continue317            cmndef = cmndef.replace('$$', cp)318            t2n[ttype] = name319            c2d[name] = cmndef320 321    def get_style_defs(self, arg=''):322        """323        Return the command sequences needed to define the commands324        used to format text in the verbatim environment. ``arg`` is ignored.325        """326        cp = self.commandprefix327        styles = []328        for name, definition in self.cmd2def.items():329            styles.append(rf'\@namedef{{{cp}@tok@{name}}}{{{definition}}}')330        return STYLE_TEMPLATE % {'cp': self.commandprefix,331                                 'styles': '\n'.join(styles)}332 333    def format_unencoded(self, tokensource, outfile):334        # TODO: add support for background colors335        t2n = self.ttype2name336        cp = self.commandprefix337 338        if self.full:339            realoutfile = outfile340            outfile = StringIO()341 342        if not self.nowrap:343            outfile.write('\\begin{' + self.envname + '}[commandchars=\\\\\\{\\}')344            if self.linenos:345                start, step = self.linenostart, self.linenostep346                outfile.write(',numbers=left' +347                              (start and ',firstnumber=%d' % start or '') +348                              (step and ',stepnumber=%d' % step or ''))349            if self.mathescape or self.texcomments or self.escapeinside:350                outfile.write(',codes={\\catcode`\\$=3\\catcode`\\^=7'351                              '\\catcode`\\_=8\\relax}')352            if self.verboptions:353                outfile.write(',' + self.verboptions)354            outfile.write(']\n')355 356        for ttype, value in tokensource:357            if ttype in Token.Comment:358                if self.texcomments:359                    # Try to guess comment starting lexeme and escape it ...360                    start = value[0:1]361                    for i in range(1, len(value)):362                        if start[0] != value[i]:363                            break364                        start += value[i]365 366                    value = value[len(start):]367                    start = escape_tex(start, cp)368 369                    # ... but do not escape inside comment.370                    value = start + value371                elif self.mathescape:372                    # Only escape parts not inside a math environment.373                    parts = value.split('$')374                    in_math = False375                    for i, part in enumerate(parts):376                        if not in_math:377                            parts[i] = escape_tex(part, cp)378                        in_math = not in_math379                    value = '$'.join(parts)380                elif self.escapeinside:381                    text = value382                    value = ''383                    while text:384                        a, sep1, text = text.partition(self.left)385                        if sep1:386                            b, sep2, text = text.partition(self.right)387                            if sep2:388                                value += escape_tex(a, cp) + b389                            else:390                                value += escape_tex(a + sep1 + b, cp)391                        else:392                            value += escape_tex(a, cp)393                else:394                    value = escape_tex(value, cp)395            elif ttype not in Token.Escape:396                value = escape_tex(value, cp)397            styles = []398            while ttype is not Token:399                try:400                    styles.append(t2n[ttype])401                except KeyError:402                    # not in current style403                    styles.append(_get_ttype_name(ttype))404                ttype = ttype.parent405            styleval = '+'.join(reversed(styles))406            if styleval:407                spl = value.split('\n')408                for line in spl[:-1]:409                    if line:410                        outfile.write(f"\\{cp}{{{styleval}}}{{{line}}}")411                    outfile.write('\n')412                if spl[-1]:413                    outfile.write(f"\\{cp}{{{styleval}}}{{{spl[-1]}}}")414            else:415                outfile.write(value)416 417        if not self.nowrap:418            outfile.write('\\end{' + self.envname + '}\n')419 420        if self.full:421            encoding = self.encoding or 'utf8'422            # map known existings encodings from LaTeX distribution423            encoding = {424                'utf_8': 'utf8',425                'latin_1': 'latin1',426                'iso_8859_1': 'latin1',427            }.get(encoding.replace('-', '_'), encoding)428            realoutfile.write(DOC_TEMPLATE %429                dict(docclass  = self.docclass,430                     preamble  = self.preamble,431                     title     = self.title,432                     encoding  = encoding,433                     styledefs = self.get_style_defs(),434                     code      = outfile.getvalue()))435 436 437class LatexEmbeddedLexer(Lexer):438    """439    This lexer takes one lexer as argument, the lexer for the language440    being formatted, and the left and right delimiters for escaped text.441 442    First everything is scanned using the language lexer to obtain443    strings and comments. All other consecutive tokens are merged and444    the resulting text is scanned for escaped segments, which are given445    the Token.Escape type. Finally text that is not escaped is scanned446    again with the language lexer.447    """448    def __init__(self, left, right, lang, **options):449        self.left = left450        self.right = right451        self.lang = lang452        Lexer.__init__(self, **options)453 454    def get_tokens_unprocessed(self, text):455        # find and remove all the escape tokens (replace with an empty string)456        # this is very similar to DelegatingLexer.get_tokens_unprocessed.457        buffered = ''458        insertions = []459        insertion_buf = []460        for i, t, v in self._find_safe_escape_tokens(text):461            if t is None:462                if insertion_buf:463                    insertions.append((len(buffered), insertion_buf))464                    insertion_buf = []465                buffered += v466            else:467                insertion_buf.append((i, t, v))468        if insertion_buf:469            insertions.append((len(buffered), insertion_buf))470        return do_insertions(insertions,471                             self.lang.get_tokens_unprocessed(buffered))472 473    def _find_safe_escape_tokens(self, text):474        """ find escape tokens that are not in strings or comments """475        for i, t, v in self._filter_to(476            self.lang.get_tokens_unprocessed(text),477            lambda t: t in Token.Comment or t in Token.String478        ):479            if t is None:480                for i2, t2, v2 in self._find_escape_tokens(v):481                    yield i + i2, t2, v2482            else:483                yield i, None, v484 485    def _filter_to(self, it, pred):486        """ Keep only the tokens that match `pred`, merge the others together """487        buf = ''488        idx = 0489        for i, t, v in it:490            if pred(t):491                if buf:492                    yield idx, None, buf493                    buf = ''494                yield i, t, v495            else:496                if not buf:497                    idx = i498                buf += v499        if buf:500            yield idx, None, buf501 502    def _find_escape_tokens(self, text):503        """ Find escape tokens within text, give token=None otherwise """504        index = 0505        while text:506            a, sep1, text = text.partition(self.left)507            if a:508                yield index, None, a509                index += len(a)510            if sep1:511                b, sep2, text = text.partition(self.right)512                if sep2:513                    yield index + len(sep1), Token.Escape, b514                    index += len(sep1) + len(b) + len(sep2)515                else:516                    yield index, Token.Error, sep1517                    index += len(sep1)518                    text = b519 
codekingpro/portable-devtools · Team Ai