codekingpro/portable-devtools
114k
1"""2 pygments.lexers.sql3 ~~~~~~~~~~~~~~~~~~~4 5 Lexers for various SQL dialects and related interactive sessions.6 7 Postgres specific lexers:8 9 `PostgresLexer`10 A SQL lexer for the PostgreSQL dialect. Differences w.r.t. the SQL11 lexer are:12 13 - keywords and data types list parsed from the PG docs (run the14 `_postgres_builtins` module to update them);15 - Content of $-strings parsed using a specific lexer, e.g. the content16 of a PL/Python function is parsed using the Python lexer;17 - parse PG specific constructs: E-strings, $-strings, U&-strings,18 different operators and punctuation.19 20 `PlPgsqlLexer`21 A lexer for the PL/pgSQL language. Adds a few specific construct on22 top of the PG SQL lexer (such as <<label>>).23 24 `PostgresConsoleLexer`25 A lexer to highlight an interactive psql session:26 27 - identifies the prompt and does its best to detect the end of command28 in multiline statement where not all the lines are prefixed by a29 prompt, telling them apart from the output;30 - highlights errors in the output and notification levels;31 - handles psql backslash commands.32 33 `PostgresExplainLexer`34 A lexer to highlight Postgres execution plan.35 36 The ``tests/examplefiles`` contains a few test files with data to be37 parsed by these lexers.38 39 :copyright: Copyright 2006-2024 by the Pygments team, see AUTHORS.40 :license: BSD, see LICENSE for details.41"""42 43import re44 45from pygments.lexer import Lexer, RegexLexer, do_insertions, bygroups, words46from pygments.token import Punctuation, Whitespace, Text, Comment, Operator, \47 Keyword, Name, String, Number, Generic, Literal48from pygments.lexers import get_lexer_by_name, ClassNotFound49 50from pygments.lexers._postgres_builtins import KEYWORDS, DATATYPES, \51 PSEUDO_TYPES, PLPGSQL_KEYWORDS, EXPLAIN_KEYWORDS52from pygments.lexers._mysql_builtins import \53 MYSQL_CONSTANTS, \54 MYSQL_DATATYPES, \55 MYSQL_FUNCTIONS, \56 MYSQL_KEYWORDS, \57 MYSQL_OPTIMIZER_HINTS58 59from pygments.lexers import _tsql_builtins60 61 62__all__ = ['PostgresLexer', 'PlPgsqlLexer', 'PostgresConsoleLexer',63 'PostgresExplainLexer', 'SqlLexer', 'TransactSqlLexer',64 'MySqlLexer', 'SqliteConsoleLexer', 'RqlLexer']65 66line_re = re.compile('.*?\n')67sqlite_prompt_re = re.compile(r'^(?:sqlite| ...)>(?= )')68 69language_re = re.compile(r"\s+LANGUAGE\s+'?(\w+)'?", re.IGNORECASE)70 71do_re = re.compile(r'\bDO\b', re.IGNORECASE)72 73# Regular expressions for analyse_text()74name_between_bracket_re = re.compile(r'\[[a-zA-Z_]\w*\]')75name_between_backtick_re = re.compile(r'`[a-zA-Z_]\w*`')76tsql_go_re = re.compile(r'\bgo\b', re.IGNORECASE)77tsql_declare_re = re.compile(r'\bdeclare\s+@', re.IGNORECASE)78tsql_variable_re = re.compile(r'@[a-zA-Z_]\w*\b')79 80 81def language_callback(lexer, match):82 """Parse the content of a $-string using a lexer83 84 The lexer is chosen looking for a nearby LANGUAGE or assumed as85 plpgsql if inside a DO statement and no LANGUAGE has been found.86 """87 lx = None88 m = language_re.match(lexer.text[match.end():match.end()+100])89 if m is not None:90 lx = lexer._get_lexer(m.group(1))91 else:92 m = list(language_re.finditer(93 lexer.text[max(0, match.start()-100):match.start()]))94 if m:95 lx = lexer._get_lexer(m[-1].group(1))96 else:97 m = list(do_re.finditer(98 lexer.text[max(0, match.start()-25):match.start()]))99 if m:100 lx = lexer._get_lexer('plpgsql')101 102 # 1 = $, 2 = delimiter, 3 = $103 yield (match.start(1), String, match.group(1))104 yield (match.start(2), String.Delimiter, match.group(2))105 yield (match.start(3), String, match.group(3))106 # 4 = string contents107 if lx:108 yield from lx.get_tokens_unprocessed(match.group(4))109 else:110 yield (match.start(4), String, match.group(4))111 # 5 = $, 6 = delimiter, 7 = $112 yield (match.start(5), String, match.group(5))113 yield (match.start(6), String.Delimiter, match.group(6))114 yield (match.start(7), String, match.group(7))115 116 117class PostgresBase:118 """Base class for Postgres-related lexers.119 120 This is implemented as a mixin to avoid the Lexer metaclass kicking in.121 this way the different lexer don't have a common Lexer ancestor. If they122 had, _tokens could be created on this ancestor and not updated for the123 other classes, resulting e.g. in PL/pgSQL parsed as SQL. This shortcoming124 seem to suggest that regexp lexers are not really subclassable.125 """126 def get_tokens_unprocessed(self, text, *args):127 # Have a copy of the entire text to be used by `language_callback`.128 self.text = text129 yield from super().get_tokens_unprocessed(text, *args)130 131 def _get_lexer(self, lang):132 if lang.lower() == 'sql':133 return get_lexer_by_name('postgresql', **self.options)134 135 tries = [lang]136 if lang.startswith('pl'):137 tries.append(lang[2:])138 if lang.endswith('u'):139 tries.append(lang[:-1])140 if lang.startswith('pl') and lang.endswith('u'):141 tries.append(lang[2:-1])142 143 for lx in tries:144 try:145 return get_lexer_by_name(lx, **self.options)146 except ClassNotFound:147 pass148 else:149 # TODO: better logging150 # print >>sys.stderr, "language not found:", lang151 return None152 153 154class PostgresLexer(PostgresBase, RegexLexer):155 """156 Lexer for the PostgreSQL dialect of SQL.157 """158 159 name = 'PostgreSQL SQL dialect'160 aliases = ['postgresql', 'postgres']161 mimetypes = ['text/x-postgresql']162 url = 'https://www.postgresql.org'163 version_added = '1.5'164 165 flags = re.IGNORECASE166 tokens = {167 'root': [168 (r'\s+', Whitespace),169 (r'--.*\n?', Comment.Single),170 (r'/\*', Comment.Multiline, 'multiline-comments'),171 (r'(' + '|'.join(s.replace(" ", r"\s+")172 for s in DATATYPES + PSEUDO_TYPES) + r')\b',173 Name.Builtin),174 (words(KEYWORDS, suffix=r'\b'), Keyword),175 (r'[+*/<>=~!@#%^&|`?-]+', Operator),176 (r'::', Operator), # cast177 (r'\$\d+', Name.Variable),178 (r'([0-9]*\.[0-9]*|[0-9]+)(e[+-]?[0-9]+)?', Number.Float),179 (r'[0-9]+', Number.Integer),180 (r"((?:E|U&)?)(')", bygroups(String.Affix, String.Single), 'string'),181 # quoted identifier182 (r'((?:U&)?)(")', bygroups(String.Affix, String.Name), 'quoted-ident'),183 (r'(?s)(\$)([^$]*)(\$)(.*?)(\$)(\2)(\$)', language_callback),184 (r'[a-z_]\w*', Name),185 186 # psql variable in SQL187 (r""":(['"]?)[a-z]\w*\b\1""", Name.Variable),188 189 (r'[;:()\[\]{},.]', Punctuation),190 ],191 'multiline-comments': [192 (r'/\*', Comment.Multiline, 'multiline-comments'),193 (r'\*/', Comment.Multiline, '#pop'),194 (r'[^/*]+', Comment.Multiline),195 (r'[/*]', Comment.Multiline)196 ],197 'string': [198 (r"[^']+", String.Single),199 (r"''", String.Single),200 (r"'", String.Single, '#pop'),201 ],202 'quoted-ident': [203 (r'[^"]+', String.Name),204 (r'""', String.Name),205 (r'"', String.Name, '#pop'),206 ],207 }208 209 210class PlPgsqlLexer(PostgresBase, RegexLexer):211 """212 Handle the extra syntax in Pl/pgSQL language.213 """214 name = 'PL/pgSQL'215 aliases = ['plpgsql']216 mimetypes = ['text/x-plpgsql']217 url = 'https://www.postgresql.org/docs/current/plpgsql.html'218 version_added = '1.5'219 220 flags = re.IGNORECASE221 # FIXME: use inheritance222 tokens = {name: state[:] for (name, state) in PostgresLexer.tokens.items()}223 224 # extend the keywords list225 for i, pattern in enumerate(tokens['root']):226 if pattern[1] == Keyword:227 tokens['root'][i] = (228 words(KEYWORDS + PLPGSQL_KEYWORDS, suffix=r'\b'),229 Keyword)230 del i231 break232 else:233 assert 0, "SQL keywords not found"234 235 # Add specific PL/pgSQL rules (before the SQL ones)236 tokens['root'][:0] = [237 (r'\%[a-z]\w*\b', Name.Builtin), # actually, a datatype238 (r':=', Operator),239 (r'\<\<[a-z]\w*\>\>', Name.Label),240 (r'\#[a-z]\w*\b', Keyword.Pseudo), # #variable_conflict241 ]242 243 244class PsqlRegexLexer(PostgresBase, RegexLexer):245 """246 Extend the PostgresLexer adding support specific for psql commands.247 248 This is not a complete psql lexer yet as it lacks prompt support249 and output rendering.250 """251 252 name = 'PostgreSQL console - regexp based lexer'253 aliases = [] # not public254 255 flags = re.IGNORECASE256 tokens = {name: state[:] for (name, state) in PostgresLexer.tokens.items()}257 258 tokens['root'].append(259 (r'\\[^\s]+', Keyword.Pseudo, 'psql-command'))260 tokens['psql-command'] = [261 (r'\n', Text, 'root'),262 (r'\s+', Whitespace),263 (r'\\[^\s]+', Keyword.Pseudo),264 (r""":(['"]?)[a-z]\w*\b\1""", Name.Variable),265 (r"'(''|[^'])*'", String.Single),266 (r"`([^`])*`", String.Backtick),267 (r"[^\s]+", String.Symbol),268 ]269 270 271re_prompt = re.compile(r'^(\S.*?)??[=\-\(\$\'\"][#>]')272re_psql_command = re.compile(r'\s*\\')273re_end_command = re.compile(r';\s*(--.*?)?$')274re_psql_command = re.compile(r'(\s*)(\\.+?)(\s+)$')275re_error = re.compile(r'(ERROR|FATAL):')276re_message = re.compile(277 r'((?:DEBUG|INFO|NOTICE|WARNING|ERROR|'278 r'FATAL|HINT|DETAIL|CONTEXT|LINE [0-9]+):)(.*?\n)')279 280 281class lookahead:282 """Wrap an iterator and allow pushing back an item."""283 def __init__(self, x):284 self.iter = iter(x)285 self._nextitem = None286 287 def __iter__(self):288 return self289 290 def send(self, i):291 self._nextitem = i292 return i293 294 def __next__(self):295 if self._nextitem is not None:296 ni = self._nextitem297 self._nextitem = None298 return ni299 return next(self.iter)300 next = __next__301 302 303class PostgresConsoleLexer(Lexer):304 """305 Lexer for psql sessions.306 """307 308 name = 'PostgreSQL console (psql)'309 aliases = ['psql', 'postgresql-console', 'postgres-console']310 mimetypes = ['text/x-postgresql-psql']311 url = 'https://www.postgresql.org'312 version_added = '1.5'313 314 def get_tokens_unprocessed(self, data):315 sql = PsqlRegexLexer(**self.options)316 317 lines = lookahead(line_re.findall(data))318 319 # prompt-output cycle320 while 1:321 322 # consume the lines of the command: start with an optional prompt323 # and continue until the end of command is detected324 curcode = ''325 insertions = []326 for line in lines:327 # Identify a shell prompt in case of psql commandline example328 if line.startswith('$') and not curcode:329 lexer = get_lexer_by_name('console', **self.options)330 yield from lexer.get_tokens_unprocessed(line)331 break332 333 # Identify a psql prompt334 mprompt = re_prompt.match(line)335 if mprompt is not None:336 insertions.append((len(curcode),337 [(0, Generic.Prompt, mprompt.group())]))338 curcode += line[len(mprompt.group()):]339 else:340 curcode += line341 342 # Check if this is the end of the command343 # TODO: better handle multiline comments at the end with344 # a lexer with an external state?345 if re_psql_command.match(curcode) \346 or re_end_command.search(curcode):347 break348 349 # Emit the combined stream of command and prompt(s)350 yield from do_insertions(insertions,351 sql.get_tokens_unprocessed(curcode))352 353 # Emit the output lines354 out_token = Generic.Output355 for line in lines:356 mprompt = re_prompt.match(line)357 if mprompt is not None:358 # push the line back to have it processed by the prompt359 lines.send(line)360 break361 362 mmsg = re_message.match(line)363 if mmsg is not None:364 if mmsg.group(1).startswith("ERROR") \365 or mmsg.group(1).startswith("FATAL"):366 out_token = Generic.Error367 yield (mmsg.start(1), Generic.Strong, mmsg.group(1))368 yield (mmsg.start(2), out_token, mmsg.group(2))369 else:370 yield (0, out_token, line)371 else:372 return373 374 375class PostgresExplainLexer(RegexLexer):376 """377 Handle PostgreSQL EXPLAIN output378 """379 380 name = 'PostgreSQL EXPLAIN dialect'381 aliases = ['postgres-explain']382 filenames = ['*.explain']383 mimetypes = ['text/x-postgresql-explain']384 url = 'https://www.postgresql.org/docs/current/using-explain.html'385 version_added = '2.15'386 387 tokens = {388 'root': [389 (r'(:|\(|\)|ms|kB|->|\.\.|\,)', Punctuation),390 (r'(\s+)', Whitespace),391 392 # This match estimated cost and effectively measured counters with ANALYZE393 # Then, we move to instrumentation state394 (r'(cost)(=?)', bygroups(Name.Class, Punctuation), 'instrumentation'),395 (r'(actual)( )(=?)', bygroups(Name.Class, Whitespace, Punctuation), 'instrumentation'),396 397 # Misc keywords398 (words(('actual', 'Memory Usage', 'Memory', 'Buckets', 'Batches',399 'originally', 'row', 'rows', 'Hits', 'Misses',400 'Evictions', 'Overflows'), suffix=r'\b'),401 Comment.Single),402 403 (r'(hit|read|dirtied|written|write|time|calls)(=)', bygroups(Comment.Single, Operator)),404 (r'(shared|temp|local)', Keyword.Pseudo),405 406 # We move to sort state in order to emphasize specific keywords (especially disk access)407 (r'(Sort Method)(: )', bygroups(Comment.Preproc, Punctuation), 'sort'),408 409 # These keywords can be followed by an object, like a table410 (r'(Sort Key|Group Key|Presorted Key|Hash Key)(:)( )',411 bygroups(Comment.Preproc, Punctuation, Whitespace), 'object_name'),412 (r'(Cache Key|Cache Mode)(:)( )', bygroups(Comment, Punctuation, Whitespace), 'object_name'),413 414 # These keywords can be followed by a predicate415 (words(('Join Filter', 'Subplans Removed', 'Filter', 'Merge Cond',416 'Hash Cond', 'Index Cond', 'Recheck Cond', 'Heap Blocks',417 'TID Cond', 'Run Condition', 'Order By', 'Function Call',418 'Table Function Call', 'Inner Unique', 'Params Evaluated',419 'Single Copy', 'Sampling', 'One-Time Filter', 'Output',420 'Relations', 'Remote SQL'), suffix=r'\b'),421 Comment.Preproc, 'predicate'),422 423 # Special keyword to handle ON CONFLICT424 (r'Conflict ', Comment.Preproc, 'conflict'),425 426 # Special keyword for InitPlan or SubPlan427 (r'(InitPlan|SubPlan)( )(\d+)( )',428 bygroups(Keyword, Whitespace, Number.Integer, Whitespace),429 'init_plan'),430 431 (words(('Sort Method', 'Join Filter', 'Planning time',432 'Planning Time', 'Execution time', 'Execution Time',433 'Workers Planned', 'Workers Launched', 'Buffers',434 'Planning', 'Worker', 'Query Identifier', 'Time',435 'Full-sort Groups', 'Pre-sorted Groups'), suffix=r'\b'), Comment.Preproc),436 437 # Emphasize these keywords438 439 (words(('Rows Removed by Join Filter', 'Rows Removed by Filter',440 'Rows Removed by Index Recheck',441 'Heap Fetches', 'never executed'),442 suffix=r'\b'), Name.Exception),443 (r'(I/O Timings)(:)( )', bygroups(Name.Exception, Punctuation, Whitespace)),444 445 (words(EXPLAIN_KEYWORDS, suffix=r'\b'), Keyword),446 447 # join keywords448 (r'((Right|Left|Full|Semi|Anti) Join)', Keyword.Type),449 (r'(Parallel |Async |Finalize |Partial )', Comment.Preproc),450 (r'Backward', Comment.Preproc),451 (r'(Intersect|Except|Hash)', Comment.Preproc),452 453 (r'(CTE)( )(\w*)?', bygroups(Comment, Whitespace, Name.Variable)),454 455 456 # Treat "on" and "using" as a punctuation457 (r'(on|using)', Punctuation, 'object_name'),458 459 460 # strings461 (r"'(''|[^'])*'", String.Single),462 # numbers463 (r'-?\d+\.\d+', Number.Float),464 (r'(-?\d+)', Number.Integer),465 466 # boolean467 (r'(true|false)', Name.Constant),468 # explain header469 (r'\s*QUERY PLAN\s*\n\s*-+', Comment.Single),470 # Settings471 (r'(Settings)(:)( )', bygroups(Comment.Preproc, Punctuation, Whitespace), 'setting'),472 473 # Handle JIT counters474 (r'(JIT|Functions|Options|Timing)(:)', bygroups(Comment.Preproc, Punctuation)),475 (r'(Inlining|Optimization|Expressions|Deforming|Generation|Emission|Total)', Keyword.Pseudo),476 477 # Handle Triggers counters478 (r'(Trigger)( )(\S*)(:)( )',479 bygroups(Comment.Preproc, Whitespace, Name.Variable, Punctuation, Whitespace)),480 481 ],482 'expression': [483 # matches any kind of parenthesized expression484 # the first opening paren is matched by the 'caller'485 (r'\(', Punctuation, '#push'),486 (r'\)', Punctuation, '#pop'),487 (r'(never executed)', Name.Exception),488 (r'[^)(]+', Comment),489 ],490 'object_name': [491 492 # This is a cost or analyze measure493 (r'(\(cost)(=?)', bygroups(Name.Class, Punctuation), 'instrumentation'),494 (r'(\(actual)( )(=?)', bygroups(Name.Class, Whitespace, Punctuation), 'instrumentation'),495 496 # if object_name is parenthesized, mark opening paren as497 # punctuation, call 'expression', and exit state498 (r'\(', Punctuation, 'expression'),499 (r'(on)', Punctuation),500 # matches possibly schema-qualified table and column names501 (r'\w+(\.\w+)*( USING \S+| \w+ USING \S+)', Name.Variable),502 (r'\"?\w+\"?(?:\.\"?\w+\"?)?', Name.Variable),503 (r'\'\S*\'', Name.Variable),504 505 # if we encounter a comma, another object is listed506 (r',\n', Punctuation, 'object_name'),507 (r',', Punctuation, 'object_name'),508 509 # special case: "*SELECT*"510 (r'"\*SELECT\*( \d+)?"(.\w+)?', Name.Variable),511 (r'"\*VALUES\*(_\d+)?"(.\w+)?', Name.Variable),512 (r'"ANY_subquery"', Name.Variable),513 514 # Variable $1 ...515 (r'\$\d+', Name.Variable),516 # cast517 (r'::\w+', Name.Variable),518 (r' +', Whitespace),519 (r'"', Punctuation),520 (r'\[\.\.\.\]', Punctuation),521 (r'\)', Punctuation, '#pop'),522 ],523 'predicate': [524 # if predicate is parenthesized, mark paren as punctuation525 (r'(\()([^\n]*)(\))', bygroups(Punctuation, Name.Variable, Punctuation), '#pop'),526 # otherwise color until newline527 (r'[^\n]*', Name.Variable, '#pop'),528 ],529 'instrumentation': [530 (r'=|\.\.', Punctuation),531 (r' +', Whitespace),532 (r'(rows|width|time|loops)', Name.Class),533 (r'\d+\.\d+', Number.Float),534 (r'(\d+)', Number.Integer),535 (r'\)', Punctuation, '#pop'),536 ],537 'conflict': [538 (r'(Resolution: )(\w+)', bygroups(Comment.Preproc, Name.Variable)),539 (r'(Arbiter \w+:)', Comment.Preproc, 'object_name'),540 (r'(Filter: )', Comment.Preproc, 'predicate'),541 ],542 'setting': [543 (r'([a-z_]*?)(\s*)(=)(\s*)(\'.*?\')', bygroups(Name.Attribute, Whitespace, Operator, Whitespace, String)),544 (r'\, ', Punctuation),545 ],546 'init_plan': [547 (r'\(', Punctuation),548 (r'returns \$\d+(,\$\d+)?', Name.Variable),549 (r'\)', Punctuation, '#pop'),550 ],551 'sort': [552 (r':|kB', Punctuation),553 (r'(quicksort|top-N|heapsort|Average|Memory|Peak)', Comment.Prepoc),554 (r'(external|merge|Disk|sort)', Name.Exception),555 (r'(\d+)', Number.Integer),556 (r' +', Whitespace),557 ],558 }559 560 561class SqlLexer(RegexLexer):562 """563 Lexer for Structured Query Language. Currently, this lexer does564 not recognize any special syntax except ANSI SQL.565 """566 567 name = 'SQL'568 aliases = ['sql']569 filenames = ['*.sql']570 mimetypes = ['text/x-sql']571 url = 'https://en.wikipedia.org/wiki/SQL'572 version_added = ''573 574 flags = re.IGNORECASE575 tokens = {576 'root': [577 (r'\s+', Whitespace),578 (r'--.*\n?', Comment.Single),579 (r'/\*', Comment.Multiline, 'multiline-comments'),580 (words((581 'ABORT', 'ABS', 'ABSOLUTE', 'ACCESS', 'ADA', 'ADD', 'ADMIN', 'AFTER',582 'AGGREGATE', 'ALIAS', 'ALL', 'ALLOCATE', 'ALTER', 'ANALYSE', 'ANALYZE',583 'AND', 'ANY', 'ARE', 'AS', 'ASC', 'ASENSITIVE', 'ASSERTION', 'ASSIGNMENT',584 'ASYMMETRIC', 'AT', 'ATOMIC', 'AUTHORIZATION', 'AVG', 'BACKWARD',585 'BEFORE', 'BEGIN', 'BETWEEN', 'BITVAR', 'BIT_LENGTH', 'BOTH', 'BREADTH',586 'BY', 'C', 'CACHE', 'CALL', 'CALLED', 'CARDINALITY', 'CASCADE',587 'CASCADED', 'CASE', 'CAST', 'CATALOG', 'CATALOG_NAME', 'CHAIN',588 'CHARACTERISTICS', 'CHARACTER_LENGTH', 'CHARACTER_SET_CATALOG',589 'CHARACTER_SET_NAME', 'CHARACTER_SET_SCHEMA', 'CHAR_LENGTH', 'CHECK',590 'CHECKED', 'CHECKPOINT', 'CLASS', 'CLASS_ORIGIN', 'CLOB', 'CLOSE',591 'CLUSTER', 'COALESCE', 'COBOL', 'COLLATE', 'COLLATION',592 'COLLATION_CATALOG', 'COLLATION_NAME', 'COLLATION_SCHEMA', 'COLUMN',593 'COLUMN_NAME', 'COMMAND_FUNCTION', 'COMMAND_FUNCTION_CODE', 'COMMENT',594 'COMMIT', 'COMMITTED', 'COMPLETION', 'CONDITION_NUMBER', 'CONNECT',595 'CONNECTION', 'CONNECTION_NAME', 'CONSTRAINT', 'CONSTRAINTS',596 'CONSTRAINT_CATALOG', 'CONSTRAINT_NAME', 'CONSTRAINT_SCHEMA',597 'CONSTRUCTOR', 'CONTAINS', 'CONTINUE', 'CONVERSION', 'CONVERT',598 'COPY', 'CORRESPONDING', 'COUNT', 'CREATE', 'CREATEDB', 'CREATEUSER',599 'CROSS', 'CUBE', 'CURRENT', 'CURRENT_DATE', 'CURRENT_PATH',600 'CURRENT_ROLE', 'CURRENT_TIME', 'CURRENT_TIMESTAMP', 'CURRENT_USER',601 'CURSOR', 'CURSOR_NAME', 'CYCLE', 'DATA', 'DATABASE',602 'DATETIME_INTERVAL_CODE', 'DATETIME_INTERVAL_PRECISION', 'DAY',603 'DEALLOCATE', 'DECLARE', 'DEFAULT', 'DEFAULTS', 'DEFERRABLE',604 'DEFERRED', 'DEFINED', 'DEFINER', 'DELETE', 'DELIMITER', 'DELIMITERS',605 'DEREF', 'DESC', 'DESCRIBE', 'DESCRIPTOR', 'DESTROY', 'DESTRUCTOR',606 'DETERMINISTIC', 'DIAGNOSTICS', 'DICTIONARY', 'DISCONNECT', 'DISPATCH',607 'DISTINCT', 'DO', 'DOMAIN', 'DROP', 'DYNAMIC', 'DYNAMIC_FUNCTION',608 'DYNAMIC_FUNCTION_CODE', 'EACH', 'ELSE', 'ELSIF', 'ENCODING',609 'ENCRYPTED', 'END', 'END-EXEC', 'EQUALS', 'ESCAPE', 'EVERY', 'EXCEPTION',610 'EXCEPT', 'EXCLUDING', 'EXCLUSIVE', 'EXEC', 'EXECUTE', 'EXISTING',611 'EXISTS', 'EXPLAIN', 'EXTERNAL', 'EXTRACT', 'FALSE', 'FETCH', 'FINAL',612 'FIRST', 'FOR', 'FORCE', 'FOREIGN', 'FORTRAN', 'FORWARD', 'FOUND', 'FREE',613 'FREEZE', 'FROM', 'FULL', 'FUNCTION', 'G', 'GENERAL', 'GENERATED', 'GET',614 'GLOBAL', 'GO', 'GOTO', 'GRANT', 'GRANTED', 'GROUP', 'GROUPING',615 'HANDLER', 'HAVING', 'HIERARCHY', 'HOLD', 'HOST', 'IDENTITY', 'IF',616 'IGNORE', 'ILIKE', 'IMMEDIATE', 'IMMEDIATELY', 'IMMUTABLE', 'IMPLEMENTATION', 'IMPLICIT',617 'IN', 'INCLUDING', 'INCREMENT', 'INDEX', 'INDITCATOR', 'INFIX',618 'INHERITS', 'INITIALIZE', 'INITIALLY', 'INNER', 'INOUT', 'INPUT',619 'INSENSITIVE', 'INSERT', 'INSTANTIABLE', 'INSTEAD', 'INTERSECT', 'INTO',620 'INVOKER', 'IS', 'ISNULL', 'ISOLATION', 'ITERATE', 'JOIN', 'KEY',621 'KEY_MEMBER', 'KEY_TYPE', 'LANCOMPILER', 'LANGUAGE', 'LARGE', 'LAST',622 'LATERAL', 'LEADING', 'LEFT', 'LENGTH', 'LESS', 'LEVEL', 'LIKE', 'LIMIT',623 'LISTEN', 'LOAD', 'LOCAL', 'LOCALTIME', 'LOCALTIMESTAMP', 'LOCATION',624 'LOCATOR', 'LOCK', 'LOWER', 'MAP', 'MATCH', 'MAX', 'MAXVALUE',625 'MESSAGE_LENGTH', 'MESSAGE_OCTET_LENGTH', 'MESSAGE_TEXT', 'METHOD', 'MIN',626 'MINUTE', 'MINVALUE', 'MOD', 'MODE', 'MODIFIES', 'MODIFY', 'MONTH',627 'MORE', 'MOVE', 'MUMPS', 'NAMES', 'NATIONAL', 'NATURAL', 'NCHAR', 'NCLOB',628 'NEW', 'NEXT', 'NO', 'NOCREATEDB', 'NOCREATEUSER', 'NONE', 'NOT',629 'NOTHING', 'NOTIFY', 'NOTNULL', 'NULL', 'NULLABLE', 'NULLIF', 'OBJECT',630 'OCTET_LENGTH', 'OF', 'OFF', 'OFFSET', 'OIDS', 'OLD', 'ON', 'ONLY',631 'OPEN', 'OPERATION', 'OPERATOR', 'OPTION', 'OPTIONS', 'OR', 'ORDER',632 'ORDINALITY', 'OUT', 'OUTER', 'OUTPUT', 'OVERLAPS', 'OVERLAY',633 'OVERRIDING', 'OWNER', 'PAD', 'PARAMETER', 'PARAMETERS', 'PARAMETER_MODE',634 'PARAMETER_NAME', 'PARAMETER_ORDINAL_POSITION',635 'PARAMETER_SPECIFIC_CATALOG', 'PARAMETER_SPECIFIC_NAME',636 'PARAMETER_SPECIFIC_SCHEMA', 'PARTIAL', 'PASCAL', 'PENDANT', 'PERIOD', 'PLACING',637 'PLI', 'POSITION', 'POSTFIX', 'PRECEEDS', 'PRECISION', 'PREFIX', 'PREORDER',638 'PREPARE', 'PRESERVE', 'PRIMARY', 'PRIOR', 'PRIVILEGES', 'PROCEDURAL',639 'PROCEDURE', 'PUBLIC', 'READ', 'READS', 'RECHECK', 'RECURSIVE', 'REF',640 'REFERENCES', 'REFERENCING', 'REINDEX', 'RELATIVE', 'RENAME',641 'REPEATABLE', 'REPLACE', 'RESET', 'RESTART', 'RESTRICT', 'RESULT',642 'RETURN', 'RETURNED_LENGTH', 'RETURNED_OCTET_LENGTH', 'RETURNED_SQLSTATE',643 'RETURNS', 'REVOKE', 'RIGHT', 'ROLE', 'ROLLBACK', 'ROLLUP', 'ROUTINE',644 'ROUTINE_CATALOG', 'ROUTINE_NAME', 'ROUTINE_SCHEMA', 'ROW', 'ROWS',645 'ROW_COUNT', 'RULE', 'SAVE_POINT', 'SCALE', 'SCHEMA', 'SCHEMA_NAME',646 'SCOPE', 'SCROLL', 'SEARCH', 'SECOND', 'SECURITY', 'SELECT', 'SELF',647 'SENSITIVE', 'SERIALIZABLE', 'SERVER_NAME', 'SESSION', 'SESSION_USER',648 'SET', 'SETOF', 'SETS', 'SHARE', 'SHOW', 'SIMILAR', 'SIMPLE', 'SIZE',649 'SOME', 'SOURCE', 'SPACE', 'SPECIFIC', 'SPECIFICTYPE', 'SPECIFIC_NAME',650 'SQL', 'SQLCODE', 'SQLERROR', 'SQLEXCEPTION', 'SQLSTATE', 'SQLWARNINIG',651 'STABLE', 'START', 'STATE', 'STATEMENT', 'STATIC', 'STATISTICS', 'STDIN',652 'STDOUT', 'STORAGE', 'STRICT', 'STRUCTURE', 'STYPE', 'SUBCLASS_ORIGIN',653 'SUBLIST', 'SUBSTRING', 'SUCCEEDS', 'SUM', 'SYMMETRIC', 'SYSID', 'SYSTEM',654 'SYSTEM_USER', 'TABLE', 'TABLE_NAME', ' TEMP', 'TEMPLATE', 'TEMPORARY',655 'TERMINATE', 'THAN', 'THEN', 'TIME', 'TIMESTAMP', 'TIMEZONE_HOUR',656 'TIMEZONE_MINUTE', 'TO', 'TOAST', 'TRAILING', 'TRANSACTION',657 'TRANSACTIONS_COMMITTED', 'TRANSACTIONS_ROLLED_BACK', 'TRANSACTION_ACTIVE',658 'TRANSFORM', 'TRANSFORMS', 'TRANSLATE', 'TRANSLATION', 'TREAT', 'TRIGGER',659 'TRIGGER_CATALOG', 'TRIGGER_NAME', 'TRIGGER_SCHEMA', 'TRIM', 'TRUE',660 'TRUNCATE', 'TRUSTED', 'TYPE', 'UNCOMMITTED', 'UNDER', 'UNENCRYPTED',661 'UNION', 'UNIQUE', 'UNKNOWN', 'UNLISTEN', 'UNNAMED', 'UNNEST', 'UNTIL',662 'UPDATE', 'UPPER', 'USAGE', 'USER', 'USER_DEFINED_TYPE_CATALOG',663 'USER_DEFINED_TYPE_NAME', 'USER_DEFINED_TYPE_SCHEMA', 'USING', 'VACUUM',664 'VALID', 'VALIDATOR', 'VALUES', 'VARIABLE', 'VERBOSE',665 'VERSION', 'VERSIONS', 'VERSIONING', 'VIEW',666 'VOLATILE', 'WHEN', 'WHENEVER', 'WHERE', 'WITH', 'WITHOUT', 'WORK',667 'WRITE', 'YEAR', 'ZONE'), suffix=r'\b'),668 Keyword),669 (words((670 'ARRAY', 'BIGINT', 'BINARY', 'BIT', 'BLOB', 'BOOLEAN', 'CHAR',671 'CHARACTER', 'DATE', 'DEC', 'DECIMAL', 'FLOAT', 'INT', 'INTEGER',672 'INTERVAL', 'NUMBER', 'NUMERIC', 'REAL', 'SERIAL', 'SMALLINT',673 'VARCHAR', 'VARYING', 'INT8', 'SERIAL8', 'TEXT'), suffix=r'\b'),674 Name.Builtin),675 (r'[+*/<>=~!@#%^&|`?-]', Operator),676 (r'[0-9]+', Number.Integer),677 # TODO: Backslash escapes?678 (r"'(''|[^'])*'", String.Single),679 (r'"(""|[^"])*"', String.Symbol), # not a real string literal in ANSI SQL680 (r'[a-z_][\w$]*', Name), # allow $s in strings for Oracle681 (r'[;:()\[\],.]', Punctuation)682 ],683 'multiline-comments': [684 (r'/\*', Comment.Multiline, 'multiline-comments'),685 (r'\*/', Comment.Multiline, '#pop'),686 (r'[^/*]+', Comment.Multiline),687 (r'[/*]', Comment.Multiline)688 ]689 }690 691 def analyse_text(self, text):692 return693 694 695class TransactSqlLexer(RegexLexer):696 """697 Transact-SQL (T-SQL) is Microsoft's and Sybase's proprietary extension to698 SQL.699 700 The list of keywords includes ODBC and keywords reserved for future use..701 """702 703 name = 'Transact-SQL'704 aliases = ['tsql', 't-sql']705 filenames = ['*.sql']706 mimetypes = ['text/x-tsql']707 url = 'https://www.tsql.info'708 version_added = ''709 710 flags = re.IGNORECASE711 712 tokens = {713 'root': [714 (r'\s+', Whitespace),715 (r'--.*?$\n?', Comment.Single),716 (r'/\*', Comment.Multiline, 'multiline-comments'),717 (words(_tsql_builtins.OPERATORS), Operator),718 (words(_tsql_builtins.OPERATOR_WORDS, suffix=r'\b'), Operator.Word),719 (words(_tsql_builtins.TYPES, suffix=r'\b'), Name.Class),720 (words(_tsql_builtins.FUNCTIONS, suffix=r'\b'), Name.Function),721 (r'(goto)(\s+)(\w+\b)', bygroups(Keyword, Whitespace, Name.Label)),722 (words(_tsql_builtins.KEYWORDS, suffix=r'\b'), Keyword),723 (r'(\[)([^]]+)(\])', bygroups(Operator, Name, Operator)),724 (r'0x[0-9a-f]+', Number.Hex),725 # Float variant 1, for example: 1., 1.e2, 1.2e3726 (r'[0-9]+\.[0-9]*(e[+-]?[0-9]+)?', Number.Float),727 # Float variant 2, for example: .1, .1e2728 (r'\.[0-9]+(e[+-]?[0-9]+)?', Number.Float),729 # Float variant 3, for example: 123e45730 (r'[0-9]+e[+-]?[0-9]+', Number.Float),731 (r'[0-9]+', Number.Integer),732 (r"'(''|[^'])*'", String.Single),733 (r'"(""|[^"])*"', String.Symbol),734 (r'[;(),.]', Punctuation),735 # Below we use \w even for the first "real" character because736 # tokens starting with a digit have already been recognized737 # as Number above.738 (r'@@\w+', Name.Builtin),739 (r'@\w+', Name.Variable),740 (r'(\w+)(:)', bygroups(Name.Label, Punctuation)),741 (r'#?#?\w+', Name), # names for temp tables and anything else742 (r'\?', Name.Variable.Magic), # parameter for prepared statements743 ],744 'multiline-comments': [745 (r'/\*', Comment.Multiline, 'multiline-comments'),746 (r'\*/', Comment.Multiline, '#pop'),747 (r'[^/*]+', Comment.Multiline),748 (r'[/*]', Comment.Multiline)749 ]750 }751 752 def analyse_text(text):753 rating = 0754 if tsql_declare_re.search(text):755 # Found T-SQL variable declaration.756 rating = 1.0757 else:758 name_between_backtick_count = len(759 name_between_backtick_re.findall(text))760 name_between_bracket_count = len(761 name_between_bracket_re.findall(text))762 # We need to check if there are any names using763 # backticks or brackets, as otherwise both are 0764 # and 0 >= 2 * 0, so we would always assume it's true765 dialect_name_count = name_between_backtick_count + name_between_bracket_count766 if dialect_name_count >= 1 and \767 name_between_bracket_count >= 2 * name_between_backtick_count:768 # Found at least twice as many [name] as `name`.769 rating += 0.5770 elif name_between_bracket_count > name_between_backtick_count:771 rating += 0.2772 elif name_between_bracket_count > 0:773 rating += 0.1774 if tsql_variable_re.search(text) is not None:775 rating += 0.1776 if tsql_go_re.search(text) is not None:777 rating += 0.1778 return rating779 780 781class MySqlLexer(RegexLexer):782 """The Oracle MySQL lexer.783 784 This lexer does not attempt to maintain strict compatibility with785 MariaDB syntax or keywords. Although MySQL and MariaDB's common code786 history suggests there may be significant overlap between the two,787 compatibility between the two is not a target for this lexer.788 """789 790 name = 'MySQL'791 aliases = ['mysql']792 mimetypes = ['text/x-mysql']793 url = 'https://www.mysql.com'794 version_added = ''795 796 flags = re.IGNORECASE797 tokens = {798 'root': [799 (r'\s+', Whitespace),800 801 # Comments802 (r'(?:#|--\s+).*', Comment.Single),803 (r'/\*\+', Comment.Special, 'optimizer-hints'),804 (r'/\*', Comment.Multiline, 'multiline-comment'),805 806 # Hexadecimal literals807 (r"x'([0-9a-f]{2})+'", Number.Hex), # MySQL requires paired hex characters in this form.808 (r'0x[0-9a-f]+', Number.Hex),809 810 # Binary literals811 (r"b'[01]+'", Number.Bin),812 (r'0b[01]+', Number.Bin),813 814 # Numeric literals815 (r'[0-9]+\.[0-9]*(e[+-]?[0-9]+)?', Number.Float), # Mandatory integer, optional fraction and exponent816 (r'[0-9]*\.[0-9]+(e[+-]?[0-9]+)?', Number.Float), # Mandatory fraction, optional integer and exponent817 (r'[0-9]+e[+-]?[0-9]+', Number.Float), # Exponents with integer significands are still floats818 (r'[0-9]+(?=[^0-9a-z$_\u0080-\uffff])', Number.Integer), # Integers that are not in a schema object name819 820 # Date literals821 (r"\{\s*d\s*(?P<quote>['\"])\s*\d{2}(\d{2})?.?\d{2}.?\d{2}\s*(?P=quote)\s*\}",822 Literal.Date),823 824 # Time literals825 (r"\{\s*t\s*(?P<quote>['\"])\s*(?:\d+\s+)?\d{1,2}.?\d{1,2}.?\d{1,2}(\.\d*)?\s*(?P=quote)\s*\}",826 Literal.Date),827 828 # Timestamp literals829 (830 r"\{\s*ts\s*(?P<quote>['\"])\s*"831 r"\d{2}(?:\d{2})?.?\d{2}.?\d{2}" # Date part832 r"\s+" # Whitespace between date and time833 r"\d{1,2}.?\d{1,2}.?\d{1,2}(\.\d*)?" # Time part834 r"\s*(?P=quote)\s*\}",835 Literal.Date836 ),837 838 # String literals839 (r"'", String.Single, 'single-quoted-string'),840 (r'"', String.Double, 'double-quoted-string'),841 842 # Variables843 (r'@@(?:global\.|persist\.|persist_only\.|session\.)?[a-z_]+', Name.Variable),844 (r'@[a-z0-9_$.]+', Name.Variable),845 (r"@'", Name.Variable, 'single-quoted-variable'),846 (r'@"', Name.Variable, 'double-quoted-variable'),847 (r"@`", Name.Variable, 'backtick-quoted-variable'),848 (r'\?', Name.Variable), # For demonstrating prepared statements849 850 # Operators851 (r'[!%&*+/:<=>^|~-]+', Operator),852 853 # Exceptions; these words tokenize differently in different contexts.854 (r'\b(set)(?!\s*\()', Keyword),855 (r'\b(character)(\s+)(set)\b', bygroups(Keyword, Whitespace, Keyword)),856 # In all other known cases, "SET" is tokenized by MYSQL_DATATYPES.857 858 (words(MYSQL_CONSTANTS, prefix=r'\b', suffix=r'\b'), Name.Constant),859 (words(MYSQL_DATATYPES, prefix=r'\b', suffix=r'\b'), Keyword.Type),860 (words(MYSQL_KEYWORDS, prefix=r'\b', suffix=r'\b'), Keyword),861 (words(MYSQL_FUNCTIONS, prefix=r'\b', suffix=r'\b(\s*)(\()'),862 bygroups(Name.Function, Whitespace, Punctuation)),863 864 # Schema object names865 #866 # Note: Although the first regex supports unquoted all-numeric867 # identifiers, this will not be a problem in practice because868 # numeric literals have already been handled above.869 #870 ('[0-9a-z$_\u0080-\uffff]+', Name),871 (r'`', Name.Quoted, 'schema-object-name'),872 873 # Punctuation874 (r'[(),.;]', Punctuation),875 ],876 877 # Multiline comment substates878 # ---------------------------879 880 'optimizer-hints': [881 (r'[^*a-z]+', Comment.Special),882 (r'\*/', Comment.Special, '#pop'),883 (words(MYSQL_OPTIMIZER_HINTS, suffix=r'\b'), Comment.Preproc),884 ('[a-z]+', Comment.Special),885 (r'\*', Comment.Special),886 ],887 888 'multiline-comment': [889 (r'[^*]+', Comment.Multiline),890 (r'\*/', Comment.Multiline, '#pop'),891 (r'\*', Comment.Multiline),892 ],893 894 # String substates895 # ----------------896 897 'single-quoted-string': [898 (r"[^'\\]+", String.Single),899 (r"''", String.Escape),900 (r"""\\[0'"bnrtZ\\%_]""", String.Escape),901 (r"'", String.Single, '#pop'),902 ],903 904 'double-quoted-string': [905 (r'[^"\\]+', String.Double),906 (r'""', String.Escape),907 (r"""\\[0'"bnrtZ\\%_]""", String.Escape),908 (r'"', String.Double, '#pop'),909 ],910 911 # Variable substates912 # ------------------913 914 'single-quoted-variable': [915 (r"[^']+", Name.Variable),916 (r"''", Name.Variable),917 (r"'", Name.Variable, '#pop'),918 ],919 920 'double-quoted-variable': [921 (r'[^"]+', Name.Variable),922 (r'""', Name.Variable),923 (r'"', Name.Variable, '#pop'),924 ],925 926 'backtick-quoted-variable': [927 (r'[^`]+', Name.Variable),928 (r'``', Name.Variable),929 (r'`', Name.Variable, '#pop'),930 ],931 932 # Schema object name substates933 # ----------------------------934 #935 # "Name.Quoted" and "Name.Quoted.Escape" are non-standard but936 # formatters will style them as "Name" by default but add937 # additional styles based on the token name. This gives users938 # flexibility to add custom styles as desired.939 #940 'schema-object-name': [941 (r'[^`]+', Name.Quoted),942 (r'``', Name.Quoted.Escape),943 (r'`', Name.Quoted, '#pop'),944 ],945 }946 947 def analyse_text(text):948 rating = 0949 name_between_backtick_count = len(950 name_between_backtick_re.findall(text))951 name_between_bracket_count = len(952 name_between_bracket_re.findall(text))953 # Same logic as above in the TSQL analysis954 dialect_name_count = name_between_backtick_count + name_between_bracket_count955 if dialect_name_count >= 1 and \956 name_between_backtick_count >= 2 * name_between_bracket_count:957 # Found at least twice as many `name` as [name].958 rating += 0.5959 elif name_between_backtick_count > name_between_bracket_count:960 rating += 0.2961 elif name_between_backtick_count > 0:962 rating += 0.1963 return rating964 965 966class SqliteConsoleLexer(Lexer):967 """968 Lexer for example sessions using sqlite3.969 """970 971 name = 'sqlite3con'972 aliases = ['sqlite3']973 filenames = ['*.sqlite3-console']974 mimetypes = ['text/x-sqlite3-console']975 url = 'https://www.sqlite.org'976 version_added = '0.11'977 978 def get_tokens_unprocessed(self, data):979 sql = SqlLexer(**self.options)980 981 curcode = ''982 insertions = []983 for match in line_re.finditer(data):984 line = match.group()985 prompt_match = sqlite_prompt_re.match(line)986 if prompt_match is not None:987 insertions.append((len(curcode),988 [(0, Generic.Prompt, line[:7])]))989 insertions.append((len(curcode),990 [(7, Whitespace, ' ')]))991 curcode += line[8:]992 else:993 if curcode:994 yield from do_insertions(insertions,995 sql.get_tokens_unprocessed(curcode))996 curcode = ''997 insertions = []998 if line.startswith('SQL error: '):999 yield (match.start(), Generic.Traceback, line)1000 else:1001 yield (match.start(), Generic.Output, line)1002 if curcode:1003 yield from do_insertions(insertions,1004 sql.get_tokens_unprocessed(curcode))1005 1006 1007class RqlLexer(RegexLexer):1008 """1009 Lexer for Relation Query Language.1010 """1011 name = 'RQL'1012 url = 'http://www.logilab.org/project/rql'1013 aliases = ['rql']1014 filenames = ['*.rql']1015 mimetypes = ['text/x-rql']1016 version_added = '2.0'1017 1018 flags = re.IGNORECASE1019 tokens = {1020 'root': [1021 (r'\s+', Whitespace),1022 (r'(DELETE|SET|INSERT|UNION|DISTINCT|WITH|WHERE|BEING|OR'1023 r'|AND|NOT|GROUPBY|HAVING|ORDERBY|ASC|DESC|LIMIT|OFFSET'1024 r'|TODAY|NOW|TRUE|FALSE|NULL|EXISTS)\b', Keyword),1025 (r'[+*/<>=%-]', Operator),1026 (r'(Any|is|instance_of|CWEType|CWRelation)\b', Name.Builtin),1027 (r'[0-9]+', Number.Integer),1028 (r'[A-Z_]\w*\??', Name),1029 (r"'(''|[^'])*'", String.Single),1030 (r'"(""|[^"])*"', String.Single),1031 (r'[;:()\[\],.]', Punctuation)1032 ],1033 }1034 