Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
Lexer.ts122 linesDownload Raw Back to src
1/**2 * The Lexer class handles tokenizing the input in various ways. Since our3 * parser expects us to be able to backtrack, the lexer allows lexing from any4 * given starting point.5 *6 * Its main exposed function is the `lex` function, which takes a position to7 * lex from and a type of token to lex. It defers to the appropriate `_innerLex`8 * function.9 *10 * The various `_innerLex` functions perform the actual lexing of different11 * kinds.12 */13 14import ParseError from "./ParseError";15import SourceLocation from "./SourceLocation";16import {Token} from "./Token";17 18import type {LexerInterface} from "./Token";19import type Settings from "./Settings";20 21/* The following tokenRegex22 * - matches typical whitespace (but not NBSP etc.) using its first group23 * - does not match any control character \x00-\x1f except whitespace24 * - does not match a bare backslash25 * - matches any ASCII character except those just mentioned26 * - does not match the BMP private use area \uE000-\uF8FF27 * - does not match bare surrogate code units28 * - matches any BMP character except for those just described29 * - matches any valid Unicode surrogate pair30 * - matches a backslash followed by one or more whitespace characters31 * - matches a backslash followed by one or more letters then whitespace32 * - matches a backslash followed by any BMP character33 * Capturing groups:34 *   [1] regular whitespace35 *   [2] backslash followed by whitespace36 *   [3] anything else, which may include:37 *     [4] left character of \verb*38 *     [5] left character of \verb39 *     [6] backslash followed by word, excluding any trailing whitespace40 * Just because the Lexer matches something doesn't mean it's valid input:41 * If there is no matching function or symbol definition, the Parser will42 * still reject the input.43 */44const spaceRegexString = "[ \r\n\t]";45const controlWordRegexString = "\\\\[a-zA-Z@]+";46const controlSymbolRegexString = "\\\\[^\uD800-\uDFFF]";47const controlWordWhitespaceRegexString =48    `(${controlWordRegexString})${spaceRegexString}*`;49const controlSpaceRegexString = "\\\\(\n|[ \r\t]+\n?)[ \r\t]*";50const combiningDiacriticalMarkString = "[\u0300-\u036f]";51export const combiningDiacriticalMarksEndRegex: RegExp =52    new RegExp(`${combiningDiacriticalMarkString}+$`);53const tokenRegexString = `(${spaceRegexString}+)|` +  // whitespace54    `${controlSpaceRegexString}|` +                   // \whitespace55    "([!-\\[\\]-\u2027\u202A-\uD7FF\uF900-\uFFFF]" +  // single codepoint56    `${combiningDiacriticalMarkString}*` +            // ...plus accents57    "|[\uD800-\uDBFF][\uDC00-\uDFFF]" +               // surrogate pair58    `${combiningDiacriticalMarkString}*` +            // ...plus accents59    "|\\\\verb\\*([^]).*?\\4" +                       // \verb*60    "|\\\\verb([^*a-zA-Z]).*?\\5" +                   // \verb unstarred61    `|${controlWordWhitespaceRegexString}` +          // \macroName + spaces62    `|${controlSymbolRegexString})`;                  // \\, \', etc.63 64/** Main Lexer class */65export default class Lexer implements LexerInterface {66    input: string;67    settings: Settings;68    tokenRegex: RegExp;69    // Category codes. The lexer only supports comment characters (14) for now.70    // MacroExpander additionally distinguishes active (13).71    catcodes: Record<string, number>;72 73    constructor(input: string, settings: Settings) {74        // Separate accents from characters75        this.input = input;76        this.settings = settings;77        this.tokenRegex = new RegExp(tokenRegexString, 'g');78        this.catcodes = {79            "%": 14, // comment character80            "~": 13, // active character81        };82    }83 84    setCatcode(char: string, code: number) {85        this.catcodes[char] = code;86    }87 88    /**89     * This function lexes a single token.90     */91    lex(): Token {92        const input = this.input;93        const pos = this.tokenRegex.lastIndex;94        if (pos === input.length) {95            return new Token("EOF", new SourceLocation(this, pos, pos));96        }97        const match = this.tokenRegex.exec(input);98        if (match === null || match.index !== pos) {99            throw new ParseError(100                `Unexpected character: '${input[pos]}'`,101                new Token(input[pos], new SourceLocation(this, pos, pos + 1)));102        }103        const text = match[6] || match[3] || (match[2] ? "\\ " : " ");104 105        if (this.catcodes[text] === 14) { // comment character106            const nlIndex = input.indexOf('\n', this.tokenRegex.lastIndex);107            if (nlIndex === -1) {108                this.tokenRegex.lastIndex = input.length; // EOF109                this.settings.reportNonstrict("commentAtEnd",110                    "% comment has no terminating newline; LaTeX would " +111                    "fail because of commenting the end of math mode (e.g. $)");112            } else {113                this.tokenRegex.lastIndex = nlIndex + 1;114            }115            return this.lex();116        }117 118        return new Token(text, new SourceLocation(this, pos,119            this.tokenRegex.lastIndex));120    }121}122 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai