codekingpro/portable-devtools
114k
1"""2 pygments.formatters.latex3 ~~~~~~~~~~~~~~~~~~~~~~~~~4 5 Formatter for LaTeX fancyvrb output.6 7 :copyright: Copyright 2006-present by the Pygments team, see AUTHORS.8 :license: BSD, see LICENSE for details.9"""10 11from io import StringIO12 13from pygments.formatter import Formatter14from pygments.lexer import Lexer, do_insertions15from pygments.token import Token, STANDARD_TYPES16from pygments.util import get_bool_opt, get_int_opt17 18 19__all__ = ['LatexFormatter']20 21 22def escape_tex(text, commandprefix):23 return text.replace('\\', '\x00'). \24 replace('{', '\x01'). \25 replace('}', '\x02'). \26 replace('\x00', rf'\{commandprefix}Zbs{{}}'). \27 replace('\x01', rf'\{commandprefix}Zob{{}}'). \28 replace('\x02', rf'\{commandprefix}Zcb{{}}'). \29 replace('^', rf'\{commandprefix}Zca{{}}'). \30 replace('_', rf'\{commandprefix}Zus{{}}'). \31 replace('&', rf'\{commandprefix}Zam{{}}'). \32 replace('<', rf'\{commandprefix}Zlt{{}}'). \33 replace('>', rf'\{commandprefix}Zgt{{}}'). \34 replace('#', rf'\{commandprefix}Zsh{{}}'). \35 replace('%', rf'\{commandprefix}Zpc{{}}'). \36 replace('$', rf'\{commandprefix}Zdl{{}}'). \37 replace('-', rf'\{commandprefix}Zhy{{}}'). \38 replace("'", rf'\{commandprefix}Zsq{{}}'). \39 replace('"', rf'\{commandprefix}Zdq{{}}'). \40 replace('~', rf'\{commandprefix}Zti{{}}')41 42 43DOC_TEMPLATE = r'''44\documentclass{%(docclass)s}45\usepackage{fancyvrb}46\usepackage{color}47\usepackage[%(encoding)s]{inputenc}48%(preamble)s49 50%(styledefs)s51 52\begin{document}53 54\section*{%(title)s}55 56%(code)s57\end{document}58'''59 60## Small explanation of the mess below :)61#62# The previous version of the LaTeX formatter just assigned a command to63# each token type defined in the current style. That obviously is64# problematic if the highlighted code is produced for a different style65# than the style commands themselves.66#67# This version works much like the HTML formatter which assigns multiple68# CSS classes to each <span> tag, from the most specific to the least69# specific token type, thus falling back to the parent token type if one70# is not defined. Here, the classes are there too and use the same short71# forms given in token.STANDARD_TYPES.72#73# Highlighted code now only uses one custom command, which by default is74# \PY and selectable by the commandprefix option (and in addition the75# escapes \PYZat, \PYZlb and \PYZrb which haven't been renamed for76# backwards compatibility purposes).77#78# \PY has two arguments: the classes, separated by +, and the text to79# render in that style. The classes are resolved into the respective80# style commands by magic, which serves to ignore unknown classes.81#82# The magic macros are:83# * \PY@it, \PY@bf, etc. are unconditionally wrapped around the text84# to render in \PY@do. Their definition determines the style.85# * \PY@reset resets \PY@it etc. to do nothing.86# * \PY@toks parses the list of classes, using magic inspired by the87# keyval package (but modified to use plusses instead of commas88# because fancyvrb redefines commas inside its environments).89# * \PY@tok processes one class, calling the \PY@tok@classname command90# if it exists.91# * \PY@tok@classname sets the \PY@it etc. to reflect the chosen style92# for its class.93# * \PY resets the style, parses the classnames and then calls \PY@do.94#95# Tip: to read this code, print it out in substituted form using e.g.96# >>> print STYLE_TEMPLATE % {'cp': 'PY'}97 98STYLE_TEMPLATE = r'''99\makeatletter100\def\%(cp)s@reset{\let\%(cp)s@it=\relax \let\%(cp)s@bf=\relax%%101 \let\%(cp)s@ul=\relax \let\%(cp)s@tc=\relax%%102 \let\%(cp)s@bc=\relax \let\%(cp)s@ff=\relax}103\def\%(cp)s@tok#1{\csname %(cp)s@tok@#1\endcsname}104\def\%(cp)s@toks#1+{\ifx\relax#1\empty\else%%105 \%(cp)s@tok{#1}\expandafter\%(cp)s@toks\fi}106\def\%(cp)s@do#1{\%(cp)s@bc{\%(cp)s@tc{\%(cp)s@ul{%%107 \%(cp)s@it{\%(cp)s@bf{\%(cp)s@ff{#1}}}}}}}108\def\%(cp)s#1#2{\%(cp)s@reset\%(cp)s@toks#1+\relax+\%(cp)s@do{#2}}109 110%(styles)s111 112\def\%(cp)sZbs{\char`\\}113\def\%(cp)sZus{\char`\_}114\def\%(cp)sZob{\char`\{}115\def\%(cp)sZcb{\char`\}}116\def\%(cp)sZca{\char`\^}117\def\%(cp)sZam{\char`\&}118\def\%(cp)sZlt{\char`\<}119\def\%(cp)sZgt{\char`\>}120\def\%(cp)sZsh{\char`\#}121\def\%(cp)sZpc{\char`\%%}122\def\%(cp)sZdl{\char`\$}123\def\%(cp)sZhy{\char`\-}124\def\%(cp)sZsq{\char`\'}125\def\%(cp)sZdq{\char`\"}126\def\%(cp)sZti{\char`\~}127%% for compatibility with earlier versions128\def\%(cp)sZat{@}129\def\%(cp)sZlb{[}130\def\%(cp)sZrb{]}131\makeatother132'''133 134 135def _get_ttype_name(ttype):136 fname = STANDARD_TYPES.get(ttype)137 if fname:138 return fname139 aname = ''140 while fname is None:141 aname = ttype[-1] + aname142 ttype = ttype.parent143 fname = STANDARD_TYPES.get(ttype)144 return fname + aname145 146 147class LatexFormatter(Formatter):148 r"""149 Format tokens as LaTeX code. This needs the `fancyvrb` and `color`150 standard packages.151 152 Without the `full` option, code is formatted as one ``Verbatim``153 environment, like this:154 155 .. sourcecode:: latex156 157 \begin{Verbatim}[commandchars=\\\{\}]158 \PY{k}{def }\PY{n+nf}{foo}(\PY{n}{bar}):159 \PY{k}{pass}160 \end{Verbatim}161 162 Wrapping can be disabled using the `nowrap` option.163 164 The special command used here (``\PY``) and all the other macros it needs165 are output by the `get_style_defs` method.166 167 With the `full` option, a complete LaTeX document is output, including168 the command definitions in the preamble.169 170 The `get_style_defs()` method of a `LatexFormatter` returns a string171 containing ``\def`` commands defining the macros needed inside the172 ``Verbatim`` environments.173 174 Additional options accepted:175 176 `nowrap`177 If set to ``True``, don't wrap the tokens at all, not even inside a178 ``\begin{Verbatim}`` environment. This disables most other options179 (default: ``False``).180 181 `style`182 The style to use, can be a string or a Style subclass (default:183 ``'default'``).184 185 `full`186 Tells the formatter to output a "full" document, i.e. a complete187 self-contained document (default: ``False``).188 189 `title`190 If `full` is true, the title that should be used to caption the191 document (default: ``''``).192 193 `docclass`194 If the `full` option is enabled, this is the document class to use195 (default: ``'article'``).196 197 `preamble`198 If the `full` option is enabled, this can be further preamble commands,199 e.g. ``\usepackage`` (default: ``''``).200 201 `linenos`202 If set to ``True``, output line numbers (default: ``False``).203 204 `linenostart`205 The line number for the first line (default: ``1``).206 207 `linenostep`208 If set to a number n > 1, only every nth line number is printed.209 210 `verboptions`211 Additional options given to the Verbatim environment (see the *fancyvrb*212 docs for possible values) (default: ``''``).213 214 `commandprefix`215 The LaTeX commands used to produce colored output are constructed216 using this prefix and some letters (default: ``'PY'``).217 218 .. versionadded:: 0.7219 .. versionchanged:: 0.10220 The default is now ``'PY'`` instead of ``'C'``.221 222 `texcomments`223 If set to ``True``, enables LaTeX comment lines. That is, LaTex markup224 in comment tokens is not escaped so that LaTeX can render it (default:225 ``False``).226 227 .. versionadded:: 1.2228 229 `mathescape`230 If set to ``True``, enables LaTeX math mode escape in comments. That231 is, ``'$...$'`` inside a comment will trigger math mode (default:232 ``False``).233 234 .. versionadded:: 1.2235 236 `escapeinside`237 If set to a string of length 2, enables escaping to LaTeX. Text238 delimited by these 2 characters is read as LaTeX code and239 typeset accordingly. It has no effect in string literals. It has240 no effect in comments if `texcomments` or `mathescape` is241 set. (default: ``''``).242 243 .. versionadded:: 2.0244 245 `envname`246 Allows you to pick an alternative environment name replacing Verbatim.247 The alternate environment still has to support Verbatim's option syntax.248 (default: ``'Verbatim'``).249 250 .. versionadded:: 2.0251 """252 name = 'LaTeX'253 aliases = ['latex', 'tex']254 filenames = ['*.tex']255 256 def __init__(self, **options):257 Formatter.__init__(self, **options)258 self.nowrap = get_bool_opt(options, 'nowrap', False)259 self.docclass = options.get('docclass', 'article')260 self.preamble = options.get('preamble', '')261 self.linenos = get_bool_opt(options, 'linenos', False)262 self.linenostart = abs(get_int_opt(options, 'linenostart', 1))263 self.linenostep = abs(get_int_opt(options, 'linenostep', 1))264 self.verboptions = options.get('verboptions', '')265 self.nobackground = get_bool_opt(options, 'nobackground', False)266 self.commandprefix = options.get('commandprefix', 'PY')267 self.texcomments = get_bool_opt(options, 'texcomments', False)268 self.mathescape = get_bool_opt(options, 'mathescape', False)269 self.escapeinside = options.get('escapeinside', '')270 if len(self.escapeinside) == 2:271 self.left = self.escapeinside[0]272 self.right = self.escapeinside[1]273 else:274 self.escapeinside = ''275 self.envname = options.get('envname', 'Verbatim')276 277 self._create_stylesheet()278 279 def _create_stylesheet(self):280 t2n = self.ttype2name = {Token: ''}281 c2d = self.cmd2def = {}282 cp = self.commandprefix283 284 def rgbcolor(col):285 if col:286 return ','.join(['%.2f' % (int(col[i] + col[i + 1], 16) / 255.0)287 for i in (0, 2, 4)])288 else:289 return '1,1,1'290 291 for ttype, ndef in self.style:292 name = _get_ttype_name(ttype)293 cmndef = ''294 if ndef['bold']:295 cmndef += r'\let\$$@bf=\textbf'296 if ndef['italic']:297 cmndef += r'\let\$$@it=\textit'298 if ndef['underline']:299 cmndef += r'\let\$$@ul=\underline'300 if ndef['roman']:301 cmndef += r'\let\$$@ff=\textrm'302 if ndef['sans']:303 cmndef += r'\let\$$@ff=\textsf'304 if ndef['mono']:305 cmndef += r'\let\$$@ff=\textsf'306 if ndef['color']:307 cmndef += (r'\def\$$@tc##1{{\textcolor[rgb]{{{}}}{{##1}}}}'.format(rgbcolor(ndef['color'])))308 if ndef['border']:309 cmndef += (r'\def\$$@bc##1{{{{\setlength{{\fboxsep}}{{\string -\fboxrule}}'310 r'\fcolorbox[rgb]{{{}}}{{{}}}{{\strut ##1}}}}}}'.format(rgbcolor(ndef['border']),311 rgbcolor(ndef['bgcolor'])))312 elif ndef['bgcolor']:313 cmndef += (r'\def\$$@bc##1{{{{\setlength{{\fboxsep}}{{0pt}}'314 r'\colorbox[rgb]{{{}}}{{\strut ##1}}}}}}'.format(rgbcolor(ndef['bgcolor'])))315 if cmndef == '':316 continue317 cmndef = cmndef.replace('$$', cp)318 t2n[ttype] = name319 c2d[name] = cmndef320 321 def get_style_defs(self, arg=''):322 """323 Return the command sequences needed to define the commands324 used to format text in the verbatim environment. ``arg`` is ignored.325 """326 cp = self.commandprefix327 styles = []328 for name, definition in self.cmd2def.items():329 styles.append(rf'\@namedef{{{cp}@tok@{name}}}{{{definition}}}')330 return STYLE_TEMPLATE % {'cp': self.commandprefix,331 'styles': '\n'.join(styles)}332 333 def format_unencoded(self, tokensource, outfile):334 # TODO: add support for background colors335 t2n = self.ttype2name336 cp = self.commandprefix337 338 if self.full:339 realoutfile = outfile340 outfile = StringIO()341 342 if not self.nowrap:343 outfile.write('\\begin{' + self.envname + '}[commandchars=\\\\\\{\\}')344 if self.linenos:345 start, step = self.linenostart, self.linenostep346 outfile.write(',numbers=left' +347 (start and ',firstnumber=%d' % start or '') +348 (step and ',stepnumber=%d' % step or ''))349 if self.mathescape or self.texcomments or self.escapeinside:350 outfile.write(',codes={\\catcode`\\$=3\\catcode`\\^=7'351 '\\catcode`\\_=8\\relax}')352 if self.verboptions:353 outfile.write(',' + self.verboptions)354 outfile.write(']\n')355 356 for ttype, value in tokensource:357 if ttype in Token.Comment:358 if self.texcomments:359 # Try to guess comment starting lexeme and escape it ...360 start = value[0:1]361 for i in range(1, len(value)):362 if start[0] != value[i]:363 break364 start += value[i]365 366 value = value[len(start):]367 start = escape_tex(start, cp)368 369 # ... but do not escape inside comment.370 value = start + value371 elif self.mathescape:372 # Only escape parts not inside a math environment.373 parts = value.split('$')374 in_math = False375 for i, part in enumerate(parts):376 if not in_math:377 parts[i] = escape_tex(part, cp)378 in_math = not in_math379 value = '$'.join(parts)380 elif self.escapeinside:381 text = value382 value = ''383 while text:384 a, sep1, text = text.partition(self.left)385 if sep1:386 b, sep2, text = text.partition(self.right)387 if sep2:388 value += escape_tex(a, cp) + b389 else:390 value += escape_tex(a + sep1 + b, cp)391 else:392 value += escape_tex(a, cp)393 else:394 value = escape_tex(value, cp)395 elif ttype not in Token.Escape:396 value = escape_tex(value, cp)397 styles = []398 while ttype is not Token:399 try:400 styles.append(t2n[ttype])401 except KeyError:402 # not in current style403 styles.append(_get_ttype_name(ttype))404 ttype = ttype.parent405 styleval = '+'.join(reversed(styles))406 if styleval:407 spl = value.split('\n')408 for line in spl[:-1]:409 if line:410 outfile.write(f"\\{cp}{{{styleval}}}{{{line}}}")411 outfile.write('\n')412 if spl[-1]:413 outfile.write(f"\\{cp}{{{styleval}}}{{{spl[-1]}}}")414 else:415 outfile.write(value)416 417 if not self.nowrap:418 outfile.write('\\end{' + self.envname + '}\n')419 420 if self.full:421 encoding = self.encoding or 'utf8'422 # map known existings encodings from LaTeX distribution423 encoding = {424 'utf_8': 'utf8',425 'latin_1': 'latin1',426 'iso_8859_1': 'latin1',427 }.get(encoding.replace('-', '_'), encoding)428 realoutfile.write(DOC_TEMPLATE %429 dict(docclass = self.docclass,430 preamble = self.preamble,431 title = self.title,432 encoding = encoding,433 styledefs = self.get_style_defs(),434 code = outfile.getvalue()))435 436 437class LatexEmbeddedLexer(Lexer):438 """439 This lexer takes one lexer as argument, the lexer for the language440 being formatted, and the left and right delimiters for escaped text.441 442 First everything is scanned using the language lexer to obtain443 strings and comments. All other consecutive tokens are merged and444 the resulting text is scanned for escaped segments, which are given445 the Token.Escape type. Finally text that is not escaped is scanned446 again with the language lexer.447 """448 def __init__(self, left, right, lang, **options):449 self.left = left450 self.right = right451 self.lang = lang452 Lexer.__init__(self, **options)453 454 def get_tokens_unprocessed(self, text):455 # find and remove all the escape tokens (replace with an empty string)456 # this is very similar to DelegatingLexer.get_tokens_unprocessed.457 buffered = ''458 insertions = []459 insertion_buf = []460 for i, t, v in self._find_safe_escape_tokens(text):461 if t is None:462 if insertion_buf:463 insertions.append((len(buffered), insertion_buf))464 insertion_buf = []465 buffered += v466 else:467 insertion_buf.append((i, t, v))468 if insertion_buf:469 insertions.append((len(buffered), insertion_buf))470 return do_insertions(insertions,471 self.lang.get_tokens_unprocessed(buffered))472 473 def _find_safe_escape_tokens(self, text):474 """ find escape tokens that are not in strings or comments """475 for i, t, v in self._filter_to(476 self.lang.get_tokens_unprocessed(text),477 lambda t: t in Token.Comment or t in Token.String478 ):479 if t is None:480 for i2, t2, v2 in self._find_escape_tokens(v):481 yield i + i2, t2, v2482 else:483 yield i, None, v484 485 def _filter_to(self, it, pred):486 """ Keep only the tokens that match `pred`, merge the others together """487 buf = ''488 idx = 0489 for i, t, v in it:490 if pred(t):491 if buf:492 yield idx, None, buf493 buf = ''494 yield i, t, v495 else:496 if not buf:497 idx = i498 buf += v499 if buf:500 yield idx, None, buf501 502 def _find_escape_tokens(self, text):503 """ Find escape tokens within text, give token=None otherwise """504 index = 0505 while text:506 a, sep1, text = text.partition(self.left)507 if a:508 yield index, None, a509 index += len(a)510 if sep1:511 b, sep2, text = text.partition(self.right)512 if sep2:513 yield index + len(sep1), Token.Escape, b514 index += len(sep1) + len(b) + len(sep2)515 else:516 yield index, Token.Error, sep1517 index += len(sep1)518 text = b519 