Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1/* eslint no-constant-condition:0 */2import functions from "./functions";3import MacroExpander, {implicitCommands} from "./MacroExpander";4import symbols, {extraLatin} from "./symbols";5import {isAtom} from "./atoms";6import {validUnit} from "./units";7import {supportedCodepoint} from "./unicodeScripts";8import ParseError from "./ParseError";9import {combiningDiacriticalMarksEndRegex} from "./Lexer";10import Settings from "./Settings";11import SourceLocation from "./SourceLocation";12import {uSubsAndSups, unicodeSubRegEx} from "./unicodeSupOrSub";13import {Token} from "./Token";14 15// Pre-evaluate both modules as unicodeSymbols require String.normalize()16import unicodeAccents from /*preval*/ "./unicodeAccents";17import unicodeSymbols from /*preval*/ "./unicodeSymbols";18 19import type {NodeType, ParseNode, AnyParseNode, SymbolParseNode,20 UnsupportedCmdParseNode} from "./parseNode";21import type {Group} from "./atoms";22import type {Mode, ArgType, BreakToken} from "./types";23import type {FunctionContext, FunctionSpec} from "./defineFunction";24import type {EnvSpec} from "./defineEnvironment";25 26/**27 * This file contains the parser used to parse out a TeX expression from the28 * input. Since TeX isn't context-free, standard parsers don't work particularly29 * well.30 *31 * The strategy of this parser is as such:32 *33 * The main functions (the `.parse...` ones) take a position in the current34 * parse string to parse tokens from. The lexer (found in Lexer.js, stored at35 * this.gullet.lexer) also supports pulling out tokens at arbitrary places. When36 * individual tokens are needed at a position, the lexer is called to pull out a37 * token, which is then used.38 *39 * The parser has a property called "mode" indicating the mode that40 * the parser is currently in. Currently it has to be one of "math" or41 * "text", which denotes whether the current environment is a math-y42 * one or a text-y one (e.g. inside \text). Currently, this serves to43 * limit the functions which can be used in text mode.44 *45 * The main functions then return an object which contains the useful data that46 * was parsed at its given point, and a new position at the end of the parsed47 * data. The main functions can call each other and continue the parsing by48 * using the returned position as a new starting point.49 *50 * There are also extra `.handle...` functions, which pull out some reused51 * functionality into self-contained functions.52 *53 * The functions return ParseNodes.54 */55 56export default class Parser {57 mode: Mode;58 gullet: MacroExpander;59 settings: Settings;60 leftrightDepth: number;61 nextToken: Token | null;62 63 constructor(input: string, settings: Settings) {64 // Start in math mode65 this.mode = "math";66 // Create a new macro expander (gullet) and (indirectly via that) also a67 // new lexer (mouth) for this parser (stomach, in the language of TeX)68 this.gullet = new MacroExpander(input, settings, this.mode);69 // Store the settings for use in parsing70 this.settings = settings;71 // Count leftright depth (for \middle errors)72 this.leftrightDepth = 0;73 this.nextToken = null;74 }75 76 /**77 * Checks a result to make sure it has the right type, and throws an78 * appropriate error otherwise.79 */80 expect(text: string, consume = true) {81 if (this.fetch().text !== text) {82 throw new ParseError(83 `Expected '${text}', got '${this.fetch().text}'`, this.fetch()84 );85 }86 if (consume) {87 this.consume();88 }89 }90 91 /**92 * Discards the current lookahead token, considering it consumed.93 */94 consume() {95 this.nextToken = null;96 }97 98 /**99 * Return the current lookahead token, or if there isn't one (at the100 * beginning, or if the previous lookahead token was consume()d),101 * fetch the next token as the new lookahead token and return it.102 */103 fetch(): Token {104 if (this.nextToken == null) {105 this.nextToken = this.gullet.expandNextToken();106 }107 return this.nextToken;108 }109 110 /**111 * Switches between "text" and "math" modes.112 */113 switchMode(newMode: Mode) {114 this.mode = newMode;115 this.gullet.switchMode(newMode);116 }117 118 /**119 * Main parsing function, which parses an entire input.120 */121 parse(): AnyParseNode[] {122 if (!this.settings.globalGroup) {123 // Create a group namespace for the math expression.124 // (LaTeX creates a new group for every $...$, $$...$$, \[...\].)125 this.gullet.beginGroup();126 }127 128 // Use old \color behavior (same as LaTeX's \textcolor) if requested.129 // We do this within the group for the math expression, so it doesn't130 // pollute settings.macros.131 if (this.settings.colorIsTextColor) {132 this.gullet.macros.set("\\color", "\\textcolor");133 }134 135 try {136 // Try to parse the input137 const parse = this.parseExpression(false);138 139 // If we succeeded, make sure there's an EOF at the end140 this.expect("EOF");141 142 // End the group namespace for the expression143 if (!this.settings.globalGroup) {144 this.gullet.endGroup();145 }146 147 return parse;148 149 // Close any leftover groups in case of a parse error.150 } finally {151 this.gullet.endGroups();152 }153 }154 155 /**156 * Fully parse a separate sequence of tokens as a separate job.157 * Tokens should be specified in reverse order, as in a MacroDefinition.158 */159 subparse(tokens: Token[]): AnyParseNode[] {160 // Save the next token from the current job.161 const oldToken = this.nextToken;162 this.consume();163 164 // Run the new job, terminating it with an excess '}'165 this.gullet.pushToken(new Token("}"));166 this.gullet.pushTokens(tokens);167 const parse = this.parseExpression(false);168 this.expect("}");169 170 // Restore the next token from the current job.171 this.nextToken = oldToken;172 173 return parse;174 }175 176 static endOfExpression: Set<string> =177 new Set(["}", "\\endgroup", "\\end", "\\right", "&"]);178 179 /**180 * Parses an "expression", which is a list of atoms.181 *182 * `breakOnInfix`: Should the parsing stop when we hit infix nodes? This183 * happens when functions have higher precedence than infix184 * nodes in implicit parses.185 *186 * `breakOnTokenText`: The text of the token that the expression should end187 * with, or `null` if something else should end the188 * expression.189 */190 parseExpression(191 breakOnInfix: boolean,192 breakOnTokenText?: BreakToken,193 ): AnyParseNode[] {194 const body = [];195 // Keep adding atoms to the body until we can't parse any more atoms (either196 // we reached the end, a }, or a \right)197 while (true) {198 // Ignore spaces in math mode199 if (this.mode === "math") {200 this.consumeSpaces();201 }202 const lex = this.fetch();203 if (Parser.endOfExpression.has(lex.text)) {204 break;205 }206 if (breakOnTokenText && lex.text === breakOnTokenText) {207 break;208 }209 if (breakOnInfix && functions[lex.text] && functions[lex.text].infix) {210 break;211 }212 const atom = this.parseAtom(breakOnTokenText);213 if (!atom) {214 break;215 } else if (atom.type === "internal") {216 // Internal nodes do not appear in parse tree217 continue;218 }219 body.push(atom);220 }221 if (this.mode === "text") {222 this.formLigatures(body);223 }224 return this.handleInfixNodes(body);225 }226 227 /**228 * Rewrites infix operators such as \over with corresponding commands such229 * as \frac.230 *231 * There can only be one infix operator per group. If there's more than one232 * then the expression is ambiguous. This can be resolved by adding {}.233 */234 handleInfixNodes(body: AnyParseNode[]): AnyParseNode[] {235 let overIndex = -1;236 let funcName;237 238 for (let i = 0; i < body.length; i++) {239 const node = body[i];240 if (node.type === "infix") {241 if (overIndex !== -1) {242 throw new ParseError(243 "only one infix operator per group",244 node.token);245 }246 overIndex = i;247 funcName = node.replaceWith;248 }249 }250 251 if (overIndex !== -1 && funcName) {252 let numerNode: AnyParseNode;253 let denomNode: AnyParseNode;254 255 const numerBody = body.slice(0, overIndex);256 const denomBody = body.slice(overIndex + 1);257 258 if (numerBody.length === 1 && numerBody[0].type === "ordgroup") {259 numerNode = numerBody[0];260 } else {261 numerNode = {type: "ordgroup", mode: this.mode, body: numerBody};262 }263 264 if (denomBody.length === 1 && denomBody[0].type === "ordgroup") {265 denomNode = denomBody[0];266 } else {267 denomNode = {type: "ordgroup", mode: this.mode, body: denomBody};268 }269 270 let node;271 if (funcName === "\\\\abovefrac") {272 node = this.callFunction(funcName,273 [numerNode, body[overIndex], denomNode], []);274 } else {275 node = this.callFunction(funcName, [numerNode, denomNode], []);276 }277 return [node];278 } else {279 return body;280 }281 }282 283 /**284 * Handle a subscript or superscript with nice errors.285 */286 handleSupSubscript(287 name: string, // For error reporting.288 ): AnyParseNode {289 const symbolToken = this.fetch();290 const symbol = symbolToken.text;291 this.consume();292 this.consumeSpaces(); // ignore spaces before sup/subscript argument293 294 // Skip over allowed internal nodes such as \relax295 let group: AnyParseNode | null | undefined;296 do {297 group = this.parseGroup(name);298 } while (group?.type === "internal");299 300 if (!group) {301 throw new ParseError(302 "Expected group after '" + symbol + "'",303 symbolToken304 );305 }306 307 return group;308 }309 310 /**311 * Converts the textual input of an unsupported command into a text node312 * contained within a color node whose color is determined by errorColor313 */314 formatUnsupportedCmd(text: string): UnsupportedCmdParseNode {315 const textordArray: ParseNode<"textord">[] = [];316 317 for (let i = 0; i < text.length; i++) {318 textordArray.push({type: "textord", mode: "text", text: text[i]});319 }320 321 const textNode: ParseNode<"text"> = {322 type: "text",323 mode: this.mode,324 body: textordArray,325 };326 const colorNode: ParseNode<"color"> = {327 type: "color",328 mode: this.mode,329 color: this.settings.errorColor,330 body: [textNode],331 };332 333 return colorNode;334 }335 336 /**337 * Parses a group with optional super/subscripts.338 */339 parseAtom(breakOnTokenText?: BreakToken): AnyParseNode | null | undefined {340 // The body of an atom is an implicit group, so that things like341 // \left(x\right)^2 work correctly.342 const base = this.parseGroup("atom", breakOnTokenText);343 344 // Internal nodes (e.g. \relax) cannot support super/subscripts.345 // Instead we will pick up super/subscripts with blank base next round.346 if (base?.type === "internal") {347 return base;348 }349 350 // In text mode, we don't have superscripts or subscripts351 if (this.mode === "text") {352 return base;353 }354 355 // Note that base may be empty (i.e. null) at this point.356 357 let superscript: AnyParseNode | null | undefined;358 let subscript: AnyParseNode | null | undefined;359 while (true) {360 // Guaranteed in math mode, so eat any spaces first.361 this.consumeSpaces();362 363 // Lex the first token364 const lex = this.fetch();365 366 if (lex.text === "\\limits" || lex.text === "\\nolimits") {367 // We got a limit control368 if (base && base.type === "op") {369 const limits = lex.text === "\\limits";370 base.limits = limits;371 base.alwaysHandleSupSub = true;372 } else if (base && base.type === "operatorname") {373 if (base.alwaysHandleSupSub) {374 base.limits = lex.text === "\\limits";375 }376 } else {377 throw new ParseError(378 "Limit controls must follow a math operator",379 lex);380 }381 this.consume();382 } else if (lex.text === "^") {383 // We got a superscript start384 if (superscript) {385 throw new ParseError("Double superscript", lex);386 }387 superscript = this.handleSupSubscript("superscript");388 } else if (lex.text === "_") {389 // We got a subscript start390 if (subscript) {391 throw new ParseError("Double subscript", lex);392 }393 subscript = this.handleSupSubscript("subscript");394 } else if (lex.text === "'") {395 // We got a prime396 if (superscript) {397 throw new ParseError("Double superscript", lex);398 }399 400 const prime: ParseNode<"textord"> = {type: "textord", mode: this.mode, text: "\\prime"};401 402 // Many primes can be grouped together, so we handle this here403 const primes: AnyParseNode[] = [prime];404 this.consume();405 // Keep lexing tokens until we get something that's not a prime406 while (this.fetch().text === "'") {407 // For each one, add another prime to the list408 primes.push(prime);409 this.consume();410 }411 // If there's a superscript following the primes, combine that412 // superscript in with the primes.413 if (this.fetch().text === "^") {414 primes.push(this.handleSupSubscript("superscript"));415 }416 // Put everything into an ordgroup as the superscript417 superscript = {type: "ordgroup", mode: this.mode, body: primes};418 } else if (uSubsAndSups[lex.text]) {419 // A Unicode subscript or superscript character.420 // We treat these similarly to the unicode-math package.421 // So we render a string of Unicode (sub|super)scripts the422 // same as a (sub|super)script of regular characters.423 const isSub = unicodeSubRegEx.test(lex.text);424 const subsupTokens = [];425 subsupTokens.push(new Token(uSubsAndSups[lex.text]));426 this.consume();427 // Continue fetching tokens to fill out the string.428 while (true) {429 const token = this.fetch().text;430 if (!(uSubsAndSups[token])) { break; }431 if (unicodeSubRegEx.test(token) !== isSub) { break; }432 subsupTokens.unshift(new Token(uSubsAndSups[token]));433 this.consume();434 }435 // Now create a (sub|super)script.436 const body = this.subparse(subsupTokens);437 if (isSub) {438 subscript = {type: "ordgroup", mode: "math", body};439 } else {440 superscript = {type: "ordgroup", mode: "math", body};441 }442 } else {443 // If it wasn't ^, _, or ', stop parsing super/subscripts444 break;445 }446 }447 448 // Base must be set if superscript or subscript are set per logic above,449 // but need to check here for type check to pass.450 if (superscript || subscript) {451 // If we got either a superscript or subscript, create a supsub452 return {453 type: "supsub",454 mode: this.mode,455 base: base,456 sup: superscript,457 sub: subscript,458 };459 } else {460 // Otherwise return the original body461 return base;462 }463 }464 465 /**466 * Parses an entire function, including its base and all of its arguments.467 */468 parseFunction(469 breakOnTokenText?: BreakToken,470 name?: string, // For determining its context471 ): AnyParseNode | null | undefined {472 const token = this.fetch();473 const func = token.text;474 const funcData = functions[func];475 if (!funcData) {476 return null;477 }478 this.consume(); // consume command token479 480 if (name && name !== "atom" && !funcData.allowedInArgument) {481 throw new ParseError(482 "Got function '" + func + "' with no arguments" +483 (name ? " as " + name : ""), token);484 } else if (this.mode === "text" && !funcData.allowedInText) {485 throw new ParseError(486 "Can't use function '" + func + "' in text mode", token);487 } else if (this.mode === "math" && funcData.allowedInMath === false) {488 throw new ParseError(489 "Can't use function '" + func + "' in math mode", token);490 }491 492 const {args, optArgs} = this.parseArguments(func, funcData);493 return this.callFunction(func, args, optArgs, token, breakOnTokenText);494 }495 496 /**497 * Call a function handler with a suitable context and arguments.498 */499 callFunction(500 name: string,501 args: AnyParseNode[],502 optArgs: (AnyParseNode | null | undefined)[],503 token?: Token,504 breakOnTokenText?: BreakToken,505 ): AnyParseNode {506 const context: FunctionContext = {507 funcName: name,508 parser: this,509 token,510 breakOnTokenText,511 };512 const func = functions[name];513 if (func && func.handler) {514 return func.handler(context, args, optArgs);515 } else {516 throw new ParseError(`No function handler for ${name}`);517 }518 }519 520 /**521 * Parses the arguments of a function or environment522 */523 parseArguments(524 func: string, // Should look like "\name" or "\begin{name}".525 funcData: FunctionSpec<NodeType> | EnvSpec<NodeType>,526 ): {527 args: AnyParseNode[];528 optArgs: (AnyParseNode | null | undefined)[];529 } {530 const totalArgs = funcData.numArgs + funcData.numOptionalArgs;531 if (totalArgs === 0) {532 return {args: [], optArgs: []};533 }534 535 const args: AnyParseNode[] = [];536 const optArgs: (AnyParseNode | null | undefined)[] = [];537 538 for (let i = 0; i < totalArgs; i++) {539 let argType = funcData.argTypes && funcData.argTypes[i];540 const isOptional = i < funcData.numOptionalArgs;541 542 if (("primitive" in funcData && funcData.primitive && argType == null) ||543 // \sqrt expands into primitive if optional argument doesn't exist544 (funcData.type === "sqrt" && i === 1 && optArgs[0] == null)545 ) {546 argType = "primitive";547 }548 549 const arg = this.parseGroupOfType(`argument to '${func}'`,550 argType as ArgType | null | undefined, isOptional);551 if (isOptional) {552 optArgs.push(arg);553 } else if (arg != null) {554 args.push(arg);555 } else { // should be unreachable556 throw new ParseError("Null argument, please report this as a bug");557 }558 }559 560 return {args, optArgs};561 }562 563 /**564 * Parses a group when the mode is changing.565 */566 parseGroupOfType(567 name: string,568 type: ArgType | null | undefined,569 optional: boolean,570 ): AnyParseNode | null | undefined {571 switch (type) {572 case "color":573 return this.parseColorGroup(optional);574 case "size":575 return this.parseSizeGroup(optional);576 case "url":577 return this.parseUrlGroup(optional);578 case "math":579 case "text":580 return this.parseArgumentGroup(optional, type);581 case "hbox": {582 // hbox argument type wraps the argument in the equivalent of583 // \hbox, which is like \text but switching to \textstyle size584 // and resetting math font.585 const group = this.parseArgumentGroup(optional, "text");586 return group != null ? {587 type: "styling",588 mode: group.mode,589 body: [group],590 style: "text", // simulate \textstyle591 resetFont: true,592 } : null;593 }594 case "raw": {595 const token = this.parseStringGroup("raw", optional);596 return token != null ? {597 type: "raw",598 mode: "text",599 string: token.text,600 } : null;601 }602 case "primitive": {603 if (optional) {604 throw new ParseError("A primitive argument cannot be optional");605 }606 const group = this.parseGroup(name);607 if (group == null) {608 throw new ParseError("Expected group as " + name, this.fetch());609 }610 return group;611 }612 case "original":613 case null:614 case undefined:615 return this.parseArgumentGroup(optional);616 default:617 throw new ParseError(618 "Unknown group type as " + name, this.fetch());619 }620 }621 622 /**623 * Discard any space tokens, fetching the next non-space token.624 */625 consumeSpaces() {626 while (this.fetch().text === " ") {627 this.consume();628 }629 }630 631 /**632 * Parses a group, essentially returning the string formed by the633 * brace-enclosed tokens plus some position information.634 */635 parseStringGroup(636 modeName: ArgType, // Used to describe the mode in error messages.637 optional: boolean,638 ): Token | null | undefined {639 const argToken = this.gullet.scanArgument(optional);640 if (argToken == null) {641 return null;642 }643 let str = "";644 let nextToken: Token;645 while ((nextToken = this.fetch()).text !== "EOF") {646 str += nextToken.text;647 this.consume();648 }649 this.consume(); // consume the end of the argument650 argToken.text = str;651 return argToken;652 }653 654 /**655 * Parses a regex-delimited group: the largest sequence of tokens656 * whose concatenated strings match `regex`. Returns the string657 * formed by the tokens plus some position information.658 */659 parseRegexGroup(660 regex: RegExp,661 modeName: string, // Used to describe the mode in error messages.662 ): Token {663 const firstToken = this.fetch();664 let lastToken = firstToken;665 let str = "";666 let nextToken: Token;667 while ((nextToken = this.fetch()).text !== "EOF" &&668 regex.test(str + nextToken.text)) {669 lastToken = nextToken;670 str += lastToken.text;671 this.consume();672 }673 if (str === "") {674 throw new ParseError(675 "Invalid " + modeName + ": '" + firstToken.text + "'",676 firstToken);677 }678 return firstToken.range(lastToken, str);679 }680 681 /**682 * Parses a color description.683 */684 parseColorGroup(optional: boolean): ParseNode<"color-token"> | null | undefined {685 const res = this.parseStringGroup("color", optional);686 if (res == null) {687 return null;688 }689 const match = (690 /^(#[a-f0-9]{3,4}|#[a-f0-9]{6}|#[a-f0-9]{8}|[a-f0-9]{6}|[a-z]+)$/i691 ).exec(res.text);692 if (!match) {693 throw new ParseError("Invalid color: '" + res.text + "'", res);694 }695 let color = match[0];696 if (/^[0-9a-f]{6}$/i.test(color)) {697 // We allow a 6-digit HTML color spec without a leading "#".698 // This follows the xcolor package's HTML color model.699 // Predefined color names are all missed by this RegEx pattern.700 color = "#" + color;701 }702 703 return {704 type: "color-token",705 mode: this.mode,706 color,707 };708 }709 710 /**711 * Parses a size specification, consisting of magnitude and unit.712 */713 parseSizeGroup(optional: boolean): ParseNode<"size"> | null | undefined {714 let res: Token | null | undefined;715 let isBlank = false;716 // don't expand before parseStringGroup717 this.gullet.consumeSpaces();718 if (!optional && this.gullet.future().text !== "{") {719 res = this.parseRegexGroup(720 /^[-+]? *(?:$|\d+|\d+\.\d*|\.\d*) *[a-z]{0,2} *$/, "size");721 } else {722 res = this.parseStringGroup("size", optional);723 }724 if (!res) {725 return null;726 }727 if (!optional && res.text.length === 0) {728 // Because we've tested for what is !optional, this block won't729 // affect \kern, \hspace, etc. It will capture the mandatory arguments730 // to \genfrac and \above.731 res.text = "0pt"; // Enable \above{}732 isBlank = true; // This is here specifically for \genfrac733 }734 const match = (/([-+]?) *(\d+(?:\.\d*)?|\.\d+) *([a-z]{2})/).exec(res.text);735 if (!match) {736 throw new ParseError("Invalid size: '" + res.text + "'", res);737 }738 const data = {739 number: +(match[1] + match[2]), // sign + magnitude, cast to number740 unit: match[3],741 };742 if (!validUnit(data)) {743 throw new ParseError("Invalid unit: '" + data.unit + "'", res);744 }745 746 return {747 type: "size",748 mode: this.mode,749 value: data,750 isBlank,751 };752 }753 754 /**755 * Parses an URL, checking escaped letters and allowed protocols,756 * and setting the catcode of % as an active character (as in \hyperref).757 */758 parseUrlGroup(optional: boolean): ParseNode<"url"> | null | undefined {759 this.gullet.lexer.setCatcode("%", 13); // active character760 this.gullet.lexer.setCatcode("~", 12); // other character761 const res = this.parseStringGroup("url", optional);762 this.gullet.lexer.setCatcode("%", 14); // comment character763 this.gullet.lexer.setCatcode("~", 13); // active character764 if (res == null) {765 return null;766 }767 // hyperref package allows backslashes alone in href, but doesn't768 // generate valid links in such cases; we interpret this as769 // "undefined" behaviour, and keep them as-is. Some browser will770 // replace backslashes with forward slashes.771 const url = res.text.replace(/\\([#$%&~_^{}])/g, '$1');772 return {773 type: "url",774 mode: this.mode,775 url,776 };777 }778 779 /**780 * Parses an argument with the mode specified.781 */782 parseArgumentGroup(optional: boolean, mode?: Mode): ParseNode<"ordgroup"> | null | undefined {783 const argToken = this.gullet.scanArgument(optional);784 if (argToken == null) {785 return null;786 }787 const outerMode = this.mode;788 if (mode) { // Switch to specified mode789 this.switchMode(mode);790 }791 792 this.gullet.beginGroup();793 const expression = this.parseExpression(false, "EOF");794 // TODO: find an alternative way to denote the end795 this.expect("EOF"); // expect the end of the argument796 this.gullet.endGroup();797 const result: ParseNode<"ordgroup"> = {798 type: "ordgroup",799 mode: this.mode,800 loc: argToken.loc,801 body: expression,802 };803 804 if (mode) { // Switch mode back805 this.switchMode(outerMode);806 }807 return result;808 }809 810 /**811 * Parses an ordinary group, which is either a single nucleus (like "x")812 * or an expression in braces (like "{x+y}") or an implicit group, a group813 * that starts at the current position, and ends right before a higher explicit814 * group ends, or at EOF.815 */816 parseGroup(817 name: string, // For error reporting.818 breakOnTokenText?: BreakToken,819 ): AnyParseNode | null | undefined {820 const firstToken = this.fetch();821 const text = firstToken.text;822 823 let result: AnyParseNode | null | undefined;824 // Try to parse an open brace or \begingroup825 if (text === "{" || text === "\\begingroup") {826 this.consume();827 const groupEnd = text === "{" ? "}" : "\\endgroup";828 829 this.gullet.beginGroup();830 // If we get a brace, parse an expression831 const expression = this.parseExpression(false, groupEnd);832 const lastToken = this.fetch();833 this.expect(groupEnd); // Check that we got a matching closing brace834 this.gullet.endGroup();835 result = {836 type: "ordgroup",837 mode: this.mode,838 loc: SourceLocation.range(firstToken, lastToken),839 body: expression,840 // A group formed by \begingroup...\endgroup is a semi-simple group841 // which doesn't affect spacing in math mode, i.e., is transparent.842 // https://tex.stackexchange.com/questions/1930/when-should-one-843 // use-begingroup-instead-of-bgroup844 semisimple: text === "\\begingroup" || undefined,845 };846 } else {847 // If there exists a function with this name, parse the function.848 // Otherwise, just return a nucleus849 result = this.parseFunction(breakOnTokenText, name) ||850 this.parseSymbol();851 if (result == null && text[0] === "\\" &&852 !implicitCommands.hasOwnProperty(text)) {853 if (this.settings.throwOnError) {854 throw new ParseError(855 "Undefined control sequence: " + text, firstToken);856 }857 result = this.formatUnsupportedCmd(text);858 this.consume();859 }860 }861 return result;862 }863 864 /**865 * Form ligature-like combinations of characters for text mode.866 * This includes inputs like "--", "---", "``" and "''".867 * The result will simply replace multiple textord nodes with a single868 * character in each value by a single textord node having multiple869 * characters in its value. The representation is still ASCII source.870 * The group will be modified in place.871 */872 formLigatures(group: AnyParseNode[]) {873 let n = group.length - 1;874 for (let i = 0; i < n; ++i) {875 const a = group[i];876 if (a.type !== "textord") {877 continue;878 }879 const v = a.text;880 const next = group[i + 1];881 if (!next || next.type !== "textord") {882 continue;883 }884 if (v === "-" && next.text === "-") {885 const afterNext = group[i + 2];886 if (i + 1 < n && afterNext && afterNext.type === "textord" && afterNext.text === "-") {887 group.splice(i, 3, {888 type: "textord",889 mode: "text",890 loc: SourceLocation.range(a, afterNext),891 text: "---",892 });893 n -= 2;894 } else {895 group.splice(i, 2, {896 type: "textord",897 mode: "text",898 loc: SourceLocation.range(a, next),899 text: "--",900 });901 n -= 1;902 }903 }904 if ((v === "'" || v === "`") && next.text === v) {905 group.splice(i, 2, {906 type: "textord",907 mode: "text",908 loc: SourceLocation.range(a, next),909 text: v + v,910 });911 n -= 1;912 }913 }914 }915 916 /**917 * Parse a single symbol out of the string. Here, we handle single character918 * symbols and special functions like \verb.919 */920 parseSymbol(): AnyParseNode | null | undefined {921 const nucleus = this.fetch();922 let text = nucleus.text;923 924 if (/^\\verb[^a-zA-Z]/.test(text)) {925 this.consume();926 let arg = text.slice(5);927 const star = (arg.charAt(0) === "*");928 if (star) {929 arg = arg.slice(1);930 }931 // Lexer's tokenRegex is constructed to always have matching932 // first/last characters.933 if (arg.length < 2 || arg.charAt(0) !== arg.slice(-1)) {934 throw new ParseError(`\\verb assertion failed --935 please report what input caused this bug`);936 }937 arg = arg.slice(1, -1); // remove first and last char938 939 return {940 type: "verb",941 mode: "text",942 body: arg,943 star,944 };945 }946 // At this point, we should have a symbol, possibly with accents.947 // First expand any accented base symbol according to unicodeSymbols.948 if (unicodeSymbols.hasOwnProperty(text[0]) &&949 !symbols[this.mode][text[0]]) {950 // This behavior is not strict (XeTeX-compatible) in math mode.951 if (this.settings.strict && this.mode === "math") {952 this.settings.reportNonstrict("unicodeTextInMathMode",953 `Accented Unicode text character "${text[0]}" used in ` +954 `math mode`, nucleus);955 }956 text = unicodeSymbols[text[0]] + text.slice(1);957 }958 // Strip off any combining characters959 const match = combiningDiacriticalMarksEndRegex.exec(text);960 if (match) {961 text = text.substring(0, match.index);962 if (text === 'i') {963 text = '\u0131'; // dotless i, in math and text mode964 } else if (text === 'j') {965 text = '\u0237'; // dotless j, in math and text mode966 }967 }968 // Recognize base symbol969 let symbol: AnyParseNode;970 if (symbols[this.mode][text]) {971 if (this.settings.strict && this.mode === 'math' &&972 extraLatin.includes(text)) {973 this.settings.reportNonstrict("unicodeTextInMathMode",974 `Latin-1/Unicode text character "${text[0]}" used in ` +975 `math mode`, nucleus);976 }977 const group: Group = symbols[this.mode][text].group;978 const loc = SourceLocation.range(nucleus);979 let s: SymbolParseNode;980 if (isAtom(group)) {981 s = {982 type: "atom",983 mode: this.mode,984 family: group,985 loc,986 text,987 };988 } else {989 s = {990 type: group,991 mode: this.mode,992 loc,993 text,994 };995 }996 symbol = s;997 } else if (text.charCodeAt(0) >= 0x80) { // no symbol for e.g. ^998 if (this.settings.strict) {999 if (!supportedCodepoint(text.charCodeAt(0))) {1000 this.settings.reportNonstrict("unknownSymbol",1001 `Unrecognized Unicode character "${text[0]}"` +1002 ` (${text.charCodeAt(0)})`, nucleus);1003 } else if (this.mode === "math") {1004 this.settings.reportNonstrict("unicodeTextInMathMode",1005 `Unicode text character "${text[0]}" used in math mode`,1006 nucleus);1007 }1008 }1009 // All nonmathematical Unicode characters are rendered as if they1010 // are in text mode (wrapped in \text) because that's what it1011 // takes to render them in LaTeX. Setting `mode: this.mode` is1012 // another natural choice (the user requested math mode), but1013 // this makes it more difficult for getCharacterMetrics() to1014 // distinguish Unicode characters without metrics and those for1015 // which we want to simulate the letter M.1016 symbol = {1017 type: "textord",1018 mode: "text",1019 loc: SourceLocation.range(nucleus),1020 text,1021 };1022 } else {1023 return null; // EOF, ^, _, {, }, etc.1024 }1025 this.consume();1026 // Transform combining characters into accents1027 if (match) {1028 for (let i = 0; i < match[0].length; i++) {1029 const accent: string = match[0][i];1030 if (!unicodeAccents[accent]) {1031 throw new ParseError(`Unknown accent ' ${accent}'`, nucleus);1032 }1033 const command = unicodeAccents[accent][this.mode] ||1034 unicodeAccents[accent].text;1035 if (!command) {1036 throw new ParseError(1037 `Accent ${accent} unsupported in ${this.mode} mode`,1038 nucleus);1039 }1040 symbol = {1041 type: "accent",1042 mode: this.mode,1043 loc: SourceLocation.range(nucleus),1044 label: command,1045 isStretchy: false,1046 isShifty: true,1047 base: symbol,1048 };1049 }1050 }1051 return symbol;1052 }1053}1054 