Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1/**2 * The Lexer class handles tokenizing the input in various ways. Since our3 * parser expects us to be able to backtrack, the lexer allows lexing from any4 * given starting point.5 *6 * Its main exposed function is the `lex` function, which takes a position to7 * lex from and a type of token to lex. It defers to the appropriate `_innerLex`8 * function.9 *10 * The various `_innerLex` functions perform the actual lexing of different11 * kinds.12 */13 14import ParseError from "./ParseError";15import SourceLocation from "./SourceLocation";16import {Token} from "./Token";17 18import type {LexerInterface} from "./Token";19import type Settings from "./Settings";20 21/* The following tokenRegex22 * - matches typical whitespace (but not NBSP etc.) using its first group23 * - does not match any control character \x00-\x1f except whitespace24 * - does not match a bare backslash25 * - matches any ASCII character except those just mentioned26 * - does not match the BMP private use area \uE000-\uF8FF27 * - does not match bare surrogate code units28 * - matches any BMP character except for those just described29 * - matches any valid Unicode surrogate pair30 * - matches a backslash followed by one or more whitespace characters31 * - matches a backslash followed by one or more letters then whitespace32 * - matches a backslash followed by any BMP character33 * Capturing groups:34 * [1] regular whitespace35 * [2] backslash followed by whitespace36 * [3] anything else, which may include:37 * [4] left character of \verb*38 * [5] left character of \verb39 * [6] backslash followed by word, excluding any trailing whitespace40 * Just because the Lexer matches something doesn't mean it's valid input:41 * If there is no matching function or symbol definition, the Parser will42 * still reject the input.43 */44const spaceRegexString = "[ \r\n\t]";45const controlWordRegexString = "\\\\[a-zA-Z@]+";46const controlSymbolRegexString = "\\\\[^\uD800-\uDFFF]";47const controlWordWhitespaceRegexString =48 `(${controlWordRegexString})${spaceRegexString}*`;49const controlSpaceRegexString = "\\\\(\n|[ \r\t]+\n?)[ \r\t]*";50const combiningDiacriticalMarkString = "[\u0300-\u036f]";51export const combiningDiacriticalMarksEndRegex: RegExp =52 new RegExp(`${combiningDiacriticalMarkString}+$`);53const tokenRegexString = `(${spaceRegexString}+)|` + // whitespace54 `${controlSpaceRegexString}|` + // \whitespace55 "([!-\\[\\]-\u2027\u202A-\uD7FF\uF900-\uFFFF]" + // single codepoint56 `${combiningDiacriticalMarkString}*` + // ...plus accents57 "|[\uD800-\uDBFF][\uDC00-\uDFFF]" + // surrogate pair58 `${combiningDiacriticalMarkString}*` + // ...plus accents59 "|\\\\verb\\*([^]).*?\\4" + // \verb*60 "|\\\\verb([^*a-zA-Z]).*?\\5" + // \verb unstarred61 `|${controlWordWhitespaceRegexString}` + // \macroName + spaces62 `|${controlSymbolRegexString})`; // \\, \', etc.63 64/** Main Lexer class */65export default class Lexer implements LexerInterface {66 input: string;67 settings: Settings;68 tokenRegex: RegExp;69 // Category codes. The lexer only supports comment characters (14) for now.70 // MacroExpander additionally distinguishes active (13).71 catcodes: Record<string, number>;72 73 constructor(input: string, settings: Settings) {74 // Separate accents from characters75 this.input = input;76 this.settings = settings;77 this.tokenRegex = new RegExp(tokenRegexString, 'g');78 this.catcodes = {79 "%": 14, // comment character80 "~": 13, // active character81 };82 }83 84 setCatcode(char: string, code: number) {85 this.catcodes[char] = code;86 }87 88 /**89 * This function lexes a single token.90 */91 lex(): Token {92 const input = this.input;93 const pos = this.tokenRegex.lastIndex;94 if (pos === input.length) {95 return new Token("EOF", new SourceLocation(this, pos, pos));96 }97 const match = this.tokenRegex.exec(input);98 if (match === null || match.index !== pos) {99 throw new ParseError(100 `Unexpected character: '${input[pos]}'`,101 new Token(input[pos], new SourceLocation(this, pos, pos + 1)));102 }103 const text = match[6] || match[3] || (match[2] ? "\\ " : " ");104 105 if (this.catcodes[text] === 14) { // comment character106 const nlIndex = input.indexOf('\n', this.tokenRegex.lastIndex);107 if (nlIndex === -1) {108 this.tokenRegex.lastIndex = input.length; // EOF109 this.settings.reportNonstrict("commentAtEnd",110 "% comment has no terminating newline; LaTeX would " +111 "fail because of commenting the end of math mode (e.g. $)");112 } else {113 this.tokenRegex.lastIndex = nlIndex + 1;114 }115 return this.lex();116 }117 118 return new Token(text, new SourceLocation(this, pos,119 this.tokenRegex.lastIndex));120 }121}122 