Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
Parser.ts1054 linesDownload Raw Back to src
1/* eslint no-constant-condition:0 */2import functions from "./functions";3import MacroExpander, {implicitCommands} from "./MacroExpander";4import symbols, {extraLatin} from "./symbols";5import {isAtom} from "./atoms";6import {validUnit} from "./units";7import {supportedCodepoint} from "./unicodeScripts";8import ParseError from "./ParseError";9import {combiningDiacriticalMarksEndRegex} from "./Lexer";10import Settings from "./Settings";11import SourceLocation from "./SourceLocation";12import {uSubsAndSups, unicodeSubRegEx} from "./unicodeSupOrSub";13import {Token} from "./Token";14 15// Pre-evaluate both modules as unicodeSymbols require String.normalize()16import unicodeAccents from /*preval*/ "./unicodeAccents";17import unicodeSymbols from /*preval*/ "./unicodeSymbols";18 19import type {NodeType, ParseNode, AnyParseNode, SymbolParseNode,20    UnsupportedCmdParseNode} from "./parseNode";21import type {Group} from "./atoms";22import type {Mode, ArgType, BreakToken} from "./types";23import type {FunctionContext, FunctionSpec} from "./defineFunction";24import type {EnvSpec} from "./defineEnvironment";25 26/**27 * This file contains the parser used to parse out a TeX expression from the28 * input. Since TeX isn't context-free, standard parsers don't work particularly29 * well.30 *31 * The strategy of this parser is as such:32 *33 * The main functions (the `.parse...` ones) take a position in the current34 * parse string to parse tokens from. The lexer (found in Lexer.js, stored at35 * this.gullet.lexer) also supports pulling out tokens at arbitrary places. When36 * individual tokens are needed at a position, the lexer is called to pull out a37 * token, which is then used.38 *39 * The parser has a property called "mode" indicating the mode that40 * the parser is currently in. Currently it has to be one of "math" or41 * "text", which denotes whether the current environment is a math-y42 * one or a text-y one (e.g. inside \text). Currently, this serves to43 * limit the functions which can be used in text mode.44 *45 * The main functions then return an object which contains the useful data that46 * was parsed at its given point, and a new position at the end of the parsed47 * data. The main functions can call each other and continue the parsing by48 * using the returned position as a new starting point.49 *50 * There are also extra `.handle...` functions, which pull out some reused51 * functionality into self-contained functions.52 *53 * The functions return ParseNodes.54 */55 56export default class Parser {57    mode: Mode;58    gullet: MacroExpander;59    settings: Settings;60    leftrightDepth: number;61    nextToken: Token | null;62 63    constructor(input: string, settings: Settings) {64        // Start in math mode65        this.mode = "math";66        // Create a new macro expander (gullet) and (indirectly via that) also a67        // new lexer (mouth) for this parser (stomach, in the language of TeX)68        this.gullet = new MacroExpander(input, settings, this.mode);69        // Store the settings for use in parsing70        this.settings = settings;71        // Count leftright depth (for \middle errors)72        this.leftrightDepth = 0;73        this.nextToken = null;74    }75 76    /**77     * Checks a result to make sure it has the right type, and throws an78     * appropriate error otherwise.79     */80    expect(text: string, consume = true) {81        if (this.fetch().text !== text) {82            throw new ParseError(83                `Expected '${text}', got '${this.fetch().text}'`, this.fetch()84            );85        }86        if (consume) {87            this.consume();88        }89    }90 91    /**92     * Discards the current lookahead token, considering it consumed.93     */94    consume() {95        this.nextToken = null;96    }97 98    /**99     * Return the current lookahead token, or if there isn't one (at the100     * beginning, or if the previous lookahead token was consume()d),101     * fetch the next token as the new lookahead token and return it.102     */103    fetch(): Token {104        if (this.nextToken == null) {105            this.nextToken = this.gullet.expandNextToken();106        }107        return this.nextToken;108    }109 110    /**111     * Switches between "text" and "math" modes.112     */113    switchMode(newMode: Mode) {114        this.mode = newMode;115        this.gullet.switchMode(newMode);116    }117 118    /**119     * Main parsing function, which parses an entire input.120     */121    parse(): AnyParseNode[] {122        if (!this.settings.globalGroup) {123            // Create a group namespace for the math expression.124            // (LaTeX creates a new group for every $...$, $$...$$, \[...\].)125            this.gullet.beginGroup();126        }127 128        // Use old \color behavior (same as LaTeX's \textcolor) if requested.129        // We do this within the group for the math expression, so it doesn't130        // pollute settings.macros.131        if (this.settings.colorIsTextColor) {132            this.gullet.macros.set("\\color", "\\textcolor");133        }134 135        try {136            // Try to parse the input137            const parse = this.parseExpression(false);138 139            // If we succeeded, make sure there's an EOF at the end140            this.expect("EOF");141 142            // End the group namespace for the expression143            if (!this.settings.globalGroup) {144                this.gullet.endGroup();145            }146 147            return parse;148 149        // Close any leftover groups in case of a parse error.150        } finally {151            this.gullet.endGroups();152        }153    }154 155    /**156     * Fully parse a separate sequence of tokens as a separate job.157     * Tokens should be specified in reverse order, as in a MacroDefinition.158     */159    subparse(tokens: Token[]): AnyParseNode[] {160        // Save the next token from the current job.161        const oldToken = this.nextToken;162        this.consume();163 164        // Run the new job, terminating it with an excess '}'165        this.gullet.pushToken(new Token("}"));166        this.gullet.pushTokens(tokens);167        const parse = this.parseExpression(false);168        this.expect("}");169 170        // Restore the next token from the current job.171        this.nextToken = oldToken;172 173        return parse;174    }175 176    static endOfExpression: Set<string> =177        new Set(["}", "\\endgroup", "\\end", "\\right", "&"]);178 179    /**180     * Parses an "expression", which is a list of atoms.181     *182     * `breakOnInfix`: Should the parsing stop when we hit infix nodes? This183     *                 happens when functions have higher precedence than infix184     *                 nodes in implicit parses.185     *186     * `breakOnTokenText`: The text of the token that the expression should end187     *                     with, or `null` if something else should end the188     *                     expression.189     */190    parseExpression(191        breakOnInfix: boolean,192        breakOnTokenText?: BreakToken,193    ): AnyParseNode[] {194        const body = [];195        // Keep adding atoms to the body until we can't parse any more atoms (either196        // we reached the end, a }, or a \right)197        while (true) {198            // Ignore spaces in math mode199            if (this.mode === "math") {200                this.consumeSpaces();201            }202            const lex = this.fetch();203            if (Parser.endOfExpression.has(lex.text)) {204                break;205            }206            if (breakOnTokenText && lex.text === breakOnTokenText) {207                break;208            }209            if (breakOnInfix && functions[lex.text] && functions[lex.text].infix) {210                break;211            }212            const atom = this.parseAtom(breakOnTokenText);213            if (!atom) {214                break;215            } else if (atom.type === "internal") {216                // Internal nodes do not appear in parse tree217                continue;218            }219            body.push(atom);220        }221        if (this.mode === "text") {222            this.formLigatures(body);223        }224        return this.handleInfixNodes(body);225    }226 227    /**228     * Rewrites infix operators such as \over with corresponding commands such229     * as \frac.230     *231     * There can only be one infix operator per group.  If there's more than one232     * then the expression is ambiguous.  This can be resolved by adding {}.233     */234    handleInfixNodes(body: AnyParseNode[]): AnyParseNode[] {235        let overIndex = -1;236        let funcName;237 238        for (let i = 0; i < body.length; i++) {239            const node = body[i];240            if (node.type === "infix") {241                if (overIndex !== -1) {242                    throw new ParseError(243                        "only one infix operator per group",244                        node.token);245                }246                overIndex = i;247                funcName = node.replaceWith;248            }249        }250 251        if (overIndex !== -1 && funcName) {252            let numerNode: AnyParseNode;253            let denomNode: AnyParseNode;254 255            const numerBody = body.slice(0, overIndex);256            const denomBody = body.slice(overIndex + 1);257 258            if (numerBody.length === 1 && numerBody[0].type === "ordgroup") {259                numerNode = numerBody[0];260            } else {261                numerNode = {type: "ordgroup", mode: this.mode, body: numerBody};262            }263 264            if (denomBody.length === 1 && denomBody[0].type === "ordgroup") {265                denomNode = denomBody[0];266            } else {267                denomNode = {type: "ordgroup", mode: this.mode, body: denomBody};268            }269 270            let node;271            if (funcName === "\\\\abovefrac") {272                node = this.callFunction(funcName,273                    [numerNode, body[overIndex], denomNode], []);274            } else {275                node = this.callFunction(funcName, [numerNode, denomNode], []);276            }277            return [node];278        } else {279            return body;280        }281    }282 283    /**284     * Handle a subscript or superscript with nice errors.285     */286    handleSupSubscript(287        name: string,   // For error reporting.288    ): AnyParseNode {289        const symbolToken = this.fetch();290        const symbol = symbolToken.text;291        this.consume();292        this.consumeSpaces(); // ignore spaces before sup/subscript argument293 294        // Skip over allowed internal nodes such as \relax295        let group: AnyParseNode | null | undefined;296        do {297            group = this.parseGroup(name);298        } while (group?.type === "internal");299 300        if (!group) {301            throw new ParseError(302                "Expected group after '" + symbol + "'",303                symbolToken304            );305        }306 307        return group;308    }309 310    /**311     * Converts the textual input of an unsupported command into a text node312     * contained within a color node whose color is determined by errorColor313     */314    formatUnsupportedCmd(text: string): UnsupportedCmdParseNode {315        const textordArray: ParseNode<"textord">[] = [];316 317        for (let i = 0; i < text.length; i++) {318            textordArray.push({type: "textord", mode: "text", text: text[i]});319        }320 321        const textNode: ParseNode<"text"> = {322            type: "text",323            mode: this.mode,324            body: textordArray,325        };326        const colorNode: ParseNode<"color"> = {327            type: "color",328            mode: this.mode,329            color: this.settings.errorColor,330            body: [textNode],331        };332 333        return colorNode;334    }335 336    /**337     * Parses a group with optional super/subscripts.338     */339    parseAtom(breakOnTokenText?: BreakToken): AnyParseNode | null | undefined {340        // The body of an atom is an implicit group, so that things like341        // \left(x\right)^2 work correctly.342        const base = this.parseGroup("atom", breakOnTokenText);343 344        // Internal nodes (e.g. \relax) cannot support super/subscripts.345        // Instead we will pick up super/subscripts with blank base next round.346        if (base?.type === "internal") {347            return base;348        }349 350        // In text mode, we don't have superscripts or subscripts351        if (this.mode === "text") {352            return base;353        }354 355        // Note that base may be empty (i.e. null) at this point.356 357        let superscript: AnyParseNode | null | undefined;358        let subscript: AnyParseNode | null | undefined;359        while (true) {360            // Guaranteed in math mode, so eat any spaces first.361            this.consumeSpaces();362 363            // Lex the first token364            const lex = this.fetch();365 366            if (lex.text === "\\limits" || lex.text === "\\nolimits") {367                // We got a limit control368                if (base && base.type === "op") {369                    const limits = lex.text === "\\limits";370                    base.limits = limits;371                    base.alwaysHandleSupSub = true;372                } else if (base && base.type === "operatorname") {373                    if (base.alwaysHandleSupSub) {374                        base.limits = lex.text === "\\limits";375                    }376                } else {377                    throw new ParseError(378                        "Limit controls must follow a math operator",379                        lex);380                }381                this.consume();382            } else if (lex.text === "^") {383                // We got a superscript start384                if (superscript) {385                    throw new ParseError("Double superscript", lex);386                }387                superscript = this.handleSupSubscript("superscript");388            } else if (lex.text === "_") {389                // We got a subscript start390                if (subscript) {391                    throw new ParseError("Double subscript", lex);392                }393                subscript = this.handleSupSubscript("subscript");394            } else if (lex.text === "'") {395                // We got a prime396                if (superscript) {397                    throw new ParseError("Double superscript", lex);398                }399 400                const prime: ParseNode<"textord"> = {type: "textord", mode: this.mode, text: "\\prime"};401 402                // Many primes can be grouped together, so we handle this here403                const primes: AnyParseNode[] = [prime];404                this.consume();405                // Keep lexing tokens until we get something that's not a prime406                while (this.fetch().text === "'") {407                    // For each one, add another prime to the list408                    primes.push(prime);409                    this.consume();410                }411                // If there's a superscript following the primes, combine that412                // superscript in with the primes.413                if (this.fetch().text === "^") {414                    primes.push(this.handleSupSubscript("superscript"));415                }416                // Put everything into an ordgroup as the superscript417                superscript = {type: "ordgroup", mode: this.mode, body: primes};418            } else if (uSubsAndSups[lex.text]) {419                // A Unicode subscript or superscript character.420                // We treat these similarly to the unicode-math package.421                // So we render a string of Unicode (sub|super)scripts the422                // same as a (sub|super)script of regular characters.423                const isSub = unicodeSubRegEx.test(lex.text);424                const subsupTokens = [];425                subsupTokens.push(new Token(uSubsAndSups[lex.text]));426                this.consume();427                // Continue fetching tokens to fill out the string.428                while (true) {429                    const token = this.fetch().text;430                    if (!(uSubsAndSups[token])) { break; }431                    if (unicodeSubRegEx.test(token) !== isSub) { break; }432                    subsupTokens.unshift(new Token(uSubsAndSups[token]));433                    this.consume();434                }435                // Now create a (sub|super)script.436                const body = this.subparse(subsupTokens);437                if (isSub) {438                    subscript = {type: "ordgroup", mode: "math", body};439                } else {440                    superscript = {type: "ordgroup", mode: "math", body};441                }442            } else {443                // If it wasn't ^, _, or ', stop parsing super/subscripts444                break;445            }446        }447 448        // Base must be set if superscript or subscript are set per logic above,449        // but need to check here for type check to pass.450        if (superscript || subscript) {451            // If we got either a superscript or subscript, create a supsub452            return {453                type: "supsub",454                mode: this.mode,455                base: base,456                sup: superscript,457                sub: subscript,458            };459        } else {460            // Otherwise return the original body461            return base;462        }463    }464 465    /**466     * Parses an entire function, including its base and all of its arguments.467     */468    parseFunction(469        breakOnTokenText?: BreakToken,470        name?: string, // For determining its context471    ): AnyParseNode | null | undefined {472        const token = this.fetch();473        const func = token.text;474        const funcData = functions[func];475        if (!funcData) {476            return null;477        }478        this.consume(); // consume command token479 480        if (name && name !== "atom" && !funcData.allowedInArgument) {481            throw new ParseError(482                "Got function '" + func + "' with no arguments" +483                (name ? " as " + name : ""), token);484        } else if (this.mode === "text" && !funcData.allowedInText) {485            throw new ParseError(486                "Can't use function '" + func + "' in text mode", token);487        } else if (this.mode === "math" && funcData.allowedInMath === false) {488            throw new ParseError(489                "Can't use function '" + func + "' in math mode", token);490        }491 492        const {args, optArgs} = this.parseArguments(func, funcData);493        return this.callFunction(func, args, optArgs, token, breakOnTokenText);494    }495 496    /**497     * Call a function handler with a suitable context and arguments.498     */499    callFunction(500        name: string,501        args: AnyParseNode[],502        optArgs: (AnyParseNode | null | undefined)[],503        token?: Token,504        breakOnTokenText?: BreakToken,505    ): AnyParseNode {506        const context: FunctionContext = {507            funcName: name,508            parser: this,509            token,510            breakOnTokenText,511        };512        const func = functions[name];513        if (func && func.handler) {514            return func.handler(context, args, optArgs);515        } else {516            throw new ParseError(`No function handler for ${name}`);517        }518    }519 520    /**521     * Parses the arguments of a function or environment522     */523    parseArguments(524        func: string,   // Should look like "\name" or "\begin{name}".525        funcData: FunctionSpec<NodeType> | EnvSpec<NodeType>,526    ): {527        args: AnyParseNode[];528        optArgs: (AnyParseNode | null | undefined)[];529    } {530        const totalArgs = funcData.numArgs + funcData.numOptionalArgs;531        if (totalArgs === 0) {532            return {args: [], optArgs: []};533        }534 535        const args: AnyParseNode[] = [];536        const optArgs: (AnyParseNode | null | undefined)[] = [];537 538        for (let i = 0; i < totalArgs; i++) {539            let argType = funcData.argTypes && funcData.argTypes[i];540            const isOptional = i < funcData.numOptionalArgs;541 542            if (("primitive" in funcData && funcData.primitive && argType == null) ||543                // \sqrt expands into primitive if optional argument doesn't exist544                (funcData.type === "sqrt" && i === 1 && optArgs[0] == null)545            ) {546                argType = "primitive";547            }548 549            const arg = this.parseGroupOfType(`argument to '${func}'`,550                argType as ArgType | null | undefined, isOptional);551            if (isOptional) {552                optArgs.push(arg);553            } else if (arg != null) {554                args.push(arg);555            } else { // should be unreachable556                throw new ParseError("Null argument, please report this as a bug");557            }558        }559 560        return {args, optArgs};561    }562 563    /**564     * Parses a group when the mode is changing.565     */566    parseGroupOfType(567        name: string,568        type: ArgType | null | undefined,569        optional: boolean,570    ): AnyParseNode | null | undefined {571        switch (type) {572            case "color":573                return this.parseColorGroup(optional);574            case "size":575                return this.parseSizeGroup(optional);576            case "url":577                return this.parseUrlGroup(optional);578            case "math":579            case "text":580                return this.parseArgumentGroup(optional, type);581            case "hbox": {582                // hbox argument type wraps the argument in the equivalent of583                // \hbox, which is like \text but switching to \textstyle size584                // and resetting math font.585                const group = this.parseArgumentGroup(optional, "text");586                return group != null ? {587                    type: "styling",588                    mode: group.mode,589                    body: [group],590                    style: "text", // simulate \textstyle591                    resetFont: true,592                } : null;593            }594            case "raw": {595                const token = this.parseStringGroup("raw", optional);596                return token != null ? {597                    type: "raw",598                    mode: "text",599                    string: token.text,600                } : null;601            }602            case "primitive": {603                if (optional) {604                    throw new ParseError("A primitive argument cannot be optional");605                }606                const group = this.parseGroup(name);607                if (group == null) {608                    throw new ParseError("Expected group as " + name, this.fetch());609                }610                return group;611            }612            case "original":613            case null:614            case undefined:615                return this.parseArgumentGroup(optional);616            default:617                throw new ParseError(618                    "Unknown group type as " + name, this.fetch());619        }620    }621 622    /**623     * Discard any space tokens, fetching the next non-space token.624     */625    consumeSpaces() {626        while (this.fetch().text === " ") {627            this.consume();628        }629    }630 631    /**632     * Parses a group, essentially returning the string formed by the633     * brace-enclosed tokens plus some position information.634     */635    parseStringGroup(636        modeName: ArgType,  // Used to describe the mode in error messages.637        optional: boolean,638    ): Token | null | undefined {639        const argToken = this.gullet.scanArgument(optional);640        if (argToken == null) {641            return null;642        }643        let str = "";644        let nextToken: Token;645        while ((nextToken = this.fetch()).text !== "EOF") {646            str += nextToken.text;647            this.consume();648        }649        this.consume(); // consume the end of the argument650        argToken.text = str;651        return argToken;652    }653 654    /**655     * Parses a regex-delimited group: the largest sequence of tokens656     * whose concatenated strings match `regex`. Returns the string657     * formed by the tokens plus some position information.658     */659    parseRegexGroup(660        regex: RegExp,661        modeName: string,   // Used to describe the mode in error messages.662    ): Token {663        const firstToken = this.fetch();664        let lastToken = firstToken;665        let str = "";666        let nextToken: Token;667        while ((nextToken = this.fetch()).text !== "EOF" &&668               regex.test(str + nextToken.text)) {669            lastToken = nextToken;670            str += lastToken.text;671            this.consume();672        }673        if (str === "") {674            throw new ParseError(675                "Invalid " + modeName + ": '" + firstToken.text + "'",676                firstToken);677        }678        return firstToken.range(lastToken, str);679    }680 681    /**682     * Parses a color description.683     */684    parseColorGroup(optional: boolean): ParseNode<"color-token"> | null | undefined {685        const res = this.parseStringGroup("color", optional);686        if (res == null) {687            return null;688        }689        const match = (690            /^(#[a-f0-9]{3,4}|#[a-f0-9]{6}|#[a-f0-9]{8}|[a-f0-9]{6}|[a-z]+)$/i691        ).exec(res.text);692        if (!match) {693            throw new ParseError("Invalid color: '" + res.text + "'", res);694        }695        let color = match[0];696        if (/^[0-9a-f]{6}$/i.test(color)) {697            // We allow a 6-digit HTML color spec without a leading "#".698            // This follows the xcolor package's HTML color model.699            // Predefined color names are all missed by this RegEx pattern.700            color = "#" + color;701        }702 703        return {704            type: "color-token",705            mode: this.mode,706            color,707        };708    }709 710    /**711     * Parses a size specification, consisting of magnitude and unit.712     */713    parseSizeGroup(optional: boolean): ParseNode<"size"> | null | undefined {714        let res: Token | null | undefined;715        let isBlank = false;716        // don't expand before parseStringGroup717        this.gullet.consumeSpaces();718        if (!optional && this.gullet.future().text !== "{") {719            res = this.parseRegexGroup(720                /^[-+]? *(?:$|\d+|\d+\.\d*|\.\d*) *[a-z]{0,2} *$/, "size");721        } else {722            res = this.parseStringGroup("size", optional);723        }724        if (!res) {725            return null;726        }727        if (!optional && res.text.length === 0) {728            // Because we've tested for what is !optional, this block won't729            // affect \kern, \hspace, etc. It will capture the mandatory arguments730            // to \genfrac and \above.731            res.text = "0pt";    // Enable \above{}732            isBlank = true;      // This is here specifically for \genfrac733        }734        const match = (/([-+]?) *(\d+(?:\.\d*)?|\.\d+) *([a-z]{2})/).exec(res.text);735        if (!match) {736            throw new ParseError("Invalid size: '" + res.text + "'", res);737        }738        const data = {739            number: +(match[1] + match[2]), // sign + magnitude, cast to number740            unit: match[3],741        };742        if (!validUnit(data)) {743            throw new ParseError("Invalid unit: '" + data.unit + "'", res);744        }745 746        return {747            type: "size",748            mode: this.mode,749            value: data,750            isBlank,751        };752    }753 754    /**755     * Parses an URL, checking escaped letters and allowed protocols,756     * and setting the catcode of % as an active character (as in \hyperref).757     */758    parseUrlGroup(optional: boolean): ParseNode<"url"> | null | undefined {759        this.gullet.lexer.setCatcode("%", 13); // active character760        this.gullet.lexer.setCatcode("~", 12); // other character761        const res = this.parseStringGroup("url", optional);762        this.gullet.lexer.setCatcode("%", 14); // comment character763        this.gullet.lexer.setCatcode("~", 13); // active character764        if (res == null) {765            return null;766        }767        // hyperref package allows backslashes alone in href, but doesn't768        // generate valid links in such cases; we interpret this as769        // "undefined" behaviour, and keep them as-is. Some browser will770        // replace backslashes with forward slashes.771        const url = res.text.replace(/\\([#$%&~_^{}])/g, '$1');772        return {773            type: "url",774            mode: this.mode,775            url,776        };777    }778 779    /**780     * Parses an argument with the mode specified.781     */782    parseArgumentGroup(optional: boolean, mode?: Mode): ParseNode<"ordgroup"> | null | undefined {783        const argToken = this.gullet.scanArgument(optional);784        if (argToken == null) {785            return null;786        }787        const outerMode = this.mode;788        if (mode) { // Switch to specified mode789            this.switchMode(mode);790        }791 792        this.gullet.beginGroup();793        const expression = this.parseExpression(false, "EOF");794        // TODO: find an alternative way to denote the end795        this.expect("EOF"); // expect the end of the argument796        this.gullet.endGroup();797        const result: ParseNode<"ordgroup"> = {798            type: "ordgroup",799            mode: this.mode,800            loc: argToken.loc,801            body: expression,802        };803 804        if (mode) { // Switch mode back805            this.switchMode(outerMode);806        }807        return result;808    }809 810    /**811     * Parses an ordinary group, which is either a single nucleus (like "x")812     * or an expression in braces (like "{x+y}") or an implicit group, a group813     * that starts at the current position, and ends right before a higher explicit814     * group ends, or at EOF.815     */816    parseGroup(817        name: string, // For error reporting.818        breakOnTokenText?: BreakToken,819    ): AnyParseNode | null | undefined {820        const firstToken = this.fetch();821        const text = firstToken.text;822 823        let result: AnyParseNode | null | undefined;824        // Try to parse an open brace or \begingroup825        if (text === "{" || text === "\\begingroup") {826            this.consume();827            const groupEnd = text === "{" ? "}" : "\\endgroup";828 829            this.gullet.beginGroup();830            // If we get a brace, parse an expression831            const expression = this.parseExpression(false, groupEnd);832            const lastToken = this.fetch();833            this.expect(groupEnd); // Check that we got a matching closing brace834            this.gullet.endGroup();835            result = {836                type: "ordgroup",837                mode: this.mode,838                loc: SourceLocation.range(firstToken, lastToken),839                body: expression,840                // A group formed by \begingroup...\endgroup is a semi-simple group841                // which doesn't affect spacing in math mode, i.e., is transparent.842                // https://tex.stackexchange.com/questions/1930/when-should-one-843                // use-begingroup-instead-of-bgroup844                semisimple: text === "\\begingroup" || undefined,845            };846        } else {847            // If there exists a function with this name, parse the function.848            // Otherwise, just return a nucleus849            result = this.parseFunction(breakOnTokenText, name) ||850                this.parseSymbol();851            if (result == null && text[0] === "\\" &&852                    !implicitCommands.hasOwnProperty(text)) {853                if (this.settings.throwOnError) {854                    throw new ParseError(855                        "Undefined control sequence: " + text, firstToken);856                }857                result = this.formatUnsupportedCmd(text);858                this.consume();859            }860        }861        return result;862    }863 864    /**865     * Form ligature-like combinations of characters for text mode.866     * This includes inputs like "--", "---", "``" and "''".867     * The result will simply replace multiple textord nodes with a single868     * character in each value by a single textord node having multiple869     * characters in its value.  The representation is still ASCII source.870     * The group will be modified in place.871     */872    formLigatures(group: AnyParseNode[]) {873        let n = group.length - 1;874        for (let i = 0; i < n; ++i) {875            const a = group[i];876            if (a.type !== "textord") {877                continue;878            }879            const v = a.text;880            const next = group[i + 1];881            if (!next || next.type !== "textord") {882                continue;883            }884            if (v === "-" && next.text === "-") {885                const afterNext = group[i + 2];886                if (i + 1 < n && afterNext && afterNext.type === "textord" && afterNext.text === "-") {887                    group.splice(i, 3, {888                        type: "textord",889                        mode: "text",890                        loc: SourceLocation.range(a, afterNext),891                        text: "---",892                    });893                    n -= 2;894                } else {895                    group.splice(i, 2, {896                        type: "textord",897                        mode: "text",898                        loc: SourceLocation.range(a, next),899                        text: "--",900                    });901                    n -= 1;902                }903            }904            if ((v === "'" || v === "`") && next.text === v) {905                group.splice(i, 2, {906                    type: "textord",907                    mode: "text",908                    loc: SourceLocation.range(a, next),909                    text: v + v,910                });911                n -= 1;912            }913        }914    }915 916    /**917     * Parse a single symbol out of the string. Here, we handle single character918     * symbols and special functions like \verb.919     */920    parseSymbol(): AnyParseNode | null | undefined {921        const nucleus = this.fetch();922        let text = nucleus.text;923 924        if (/^\\verb[^a-zA-Z]/.test(text)) {925            this.consume();926            let arg = text.slice(5);927            const star = (arg.charAt(0) === "*");928            if (star) {929                arg = arg.slice(1);930            }931            // Lexer's tokenRegex is constructed to always have matching932            // first/last characters.933            if (arg.length < 2 || arg.charAt(0) !== arg.slice(-1)) {934                throw new ParseError(`\\verb assertion failed --935                    please report what input caused this bug`);936            }937            arg = arg.slice(1, -1);  // remove first and last char938 939            return {940                type: "verb",941                mode: "text",942                body: arg,943                star,944            };945        }946        // At this point, we should have a symbol, possibly with accents.947        // First expand any accented base symbol according to unicodeSymbols.948        if (unicodeSymbols.hasOwnProperty(text[0]) &&949            !symbols[this.mode][text[0]]) {950            // This behavior is not strict (XeTeX-compatible) in math mode.951            if (this.settings.strict && this.mode === "math") {952                this.settings.reportNonstrict("unicodeTextInMathMode",953                    `Accented Unicode text character "${text[0]}" used in ` +954                    `math mode`, nucleus);955            }956            text = unicodeSymbols[text[0]] + text.slice(1);957        }958        // Strip off any combining characters959        const match = combiningDiacriticalMarksEndRegex.exec(text);960        if (match) {961            text = text.substring(0, match.index);962            if (text === 'i') {963                text = '\u0131';  // dotless i, in math and text mode964            } else if (text === 'j') {965                text = '\u0237';  // dotless j, in math and text mode966            }967        }968        // Recognize base symbol969        let symbol: AnyParseNode;970        if (symbols[this.mode][text]) {971            if (this.settings.strict && this.mode === 'math' &&972                extraLatin.includes(text)) {973                this.settings.reportNonstrict("unicodeTextInMathMode",974                    `Latin-1/Unicode text character "${text[0]}" used in ` +975                    `math mode`, nucleus);976            }977            const group: Group = symbols[this.mode][text].group;978            const loc = SourceLocation.range(nucleus);979            let s: SymbolParseNode;980            if (isAtom(group)) {981                s = {982                    type: "atom",983                    mode: this.mode,984                    family: group,985                    loc,986                    text,987                };988            } else {989                s = {990                    type: group,991                    mode: this.mode,992                    loc,993                    text,994                };995            }996            symbol = s;997        } else if (text.charCodeAt(0) >= 0x80) { // no symbol for e.g. ^998            if (this.settings.strict) {999                if (!supportedCodepoint(text.charCodeAt(0))) {1000                    this.settings.reportNonstrict("unknownSymbol",1001                        `Unrecognized Unicode character "${text[0]}"` +1002                        ` (${text.charCodeAt(0)})`, nucleus);1003                } else if (this.mode === "math") {1004                    this.settings.reportNonstrict("unicodeTextInMathMode",1005                        `Unicode text character "${text[0]}" used in math mode`,1006                        nucleus);1007                }1008            }1009            // All nonmathematical Unicode characters are rendered as if they1010            // are in text mode (wrapped in \text) because that's what it1011            // takes to render them in LaTeX.  Setting `mode: this.mode` is1012            // another natural choice (the user requested math mode), but1013            // this makes it more difficult for getCharacterMetrics() to1014            // distinguish Unicode characters without metrics and those for1015            // which we want to simulate the letter M.1016            symbol = {1017                type: "textord",1018                mode: "text",1019                loc: SourceLocation.range(nucleus),1020                text,1021            };1022        } else {1023            return null;  // EOF, ^, _, {, }, etc.1024        }1025        this.consume();1026        // Transform combining characters into accents1027        if (match) {1028            for (let i = 0; i < match[0].length; i++) {1029                const accent: string = match[0][i];1030                if (!unicodeAccents[accent]) {1031                    throw new ParseError(`Unknown accent ' ${accent}'`, nucleus);1032                }1033                const command = unicodeAccents[accent][this.mode] ||1034                    unicodeAccents[accent].text;1035                if (!command) {1036                    throw new ParseError(1037                        `Accent ${accent} unsupported in ${this.mode} mode`,1038                        nucleus);1039                }1040                symbol = {1041                    type: "accent",1042                    mode: this.mode,1043                    loc: SourceLocation.range(nucleus),1044                    label: command,1045                    isStretchy: false,1046                    isShifty: true,1047                    base: symbol,1048                };1049            }1050        }1051        return symbol;1052    }1053}1054 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai