Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1import { htmlDecodeTree } from "./generated/decode-data-html.js";2import { xmlDecodeTree } from "./generated/decode-data-xml.js";3import { replaceCodePoint, fromCodePoint } from "./decode-codepoint.js";4 5const enum CharCodes {6 NUM = 35, // "#"7 SEMI = 59, // ";"8 EQUALS = 61, // "="9 ZERO = 48, // "0"10 NINE = 57, // "9"11 LOWER_A = 97, // "a"12 LOWER_F = 102, // "f"13 LOWER_X = 120, // "x"14 LOWER_Z = 122, // "z"15 UPPER_A = 65, // "A"16 UPPER_F = 70, // "F"17 UPPER_Z = 90, // "Z"18}19 20/** Bit that needs to be set to convert an upper case ASCII character to lower case */21const TO_LOWER_BIT = 0b10_0000;22 23export enum BinTrieFlags {24 VALUE_LENGTH = 0b1100_0000_0000_0000,25 BRANCH_LENGTH = 0b0011_1111_1000_0000,26 JUMP_TABLE = 0b0000_0000_0111_1111,27}28 29function isNumber(code: number): boolean {30 return code >= CharCodes.ZERO && code <= CharCodes.NINE;31}32 33function isHexadecimalCharacter(code: number): boolean {34 return (35 (code >= CharCodes.UPPER_A && code <= CharCodes.UPPER_F) ||36 (code >= CharCodes.LOWER_A && code <= CharCodes.LOWER_F)37 );38}39 40function isAsciiAlphaNumeric(code: number): boolean {41 return (42 (code >= CharCodes.UPPER_A && code <= CharCodes.UPPER_Z) ||43 (code >= CharCodes.LOWER_A && code <= CharCodes.LOWER_Z) ||44 isNumber(code)45 );46}47 48/**49 * Checks if the given character is a valid end character for an entity in an attribute.50 *51 * Attribute values that aren't terminated properly aren't parsed, and shouldn't lead to a parser error.52 * See the example in https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state53 */54function isEntityInAttributeInvalidEnd(code: number): boolean {55 return code === CharCodes.EQUALS || isAsciiAlphaNumeric(code);56}57 58const enum EntityDecoderState {59 EntityStart,60 NumericStart,61 NumericDecimal,62 NumericHex,63 NamedEntity,64}65 66export enum DecodingMode {67 /** Entities in text nodes that can end with any character. */68 Legacy = 0,69 /** Only allow entities terminated with a semicolon. */70 Strict = 1,71 /** Entities in attributes have limitations on ending characters. */72 Attribute = 2,73}74 75/**76 * Producers for character reference errors as defined in the HTML spec.77 */78export interface EntityErrorProducer {79 missingSemicolonAfterCharacterReference(): void;80 absenceOfDigitsInNumericCharacterReference(81 consumedCharacters: number,82 ): void;83 validateNumericCharacterReference(code: number): void;84}85 86/**87 * Token decoder with support of writing partial entities.88 */89export class EntityDecoder {90 constructor(91 /** The tree used to decode entities. */92 private readonly decodeTree: Uint16Array,93 /**94 * The function that is called when a codepoint is decoded.95 *96 * For multi-byte named entities, this will be called multiple times,97 * with the second codepoint, and the same `consumed` value.98 *99 * @param codepoint The decoded codepoint.100 * @param consumed The number of bytes consumed by the decoder.101 */102 private readonly emitCodePoint: (cp: number, consumed: number) => void,103 /** An object that is used to produce errors. */104 private readonly errors?: EntityErrorProducer | undefined,105 ) {}106 107 /** The current state of the decoder. */108 private state = EntityDecoderState.EntityStart;109 /** Characters that were consumed while parsing an entity. */110 private consumed = 1;111 /**112 * The result of the entity.113 *114 * Either the result index of a numeric entity, or the codepoint of a115 * numeric entity.116 */117 private result = 0;118 119 /** The current index in the decode tree. */120 private treeIndex = 0;121 /** The number of characters that were consumed in excess. */122 private excess = 1;123 /** The mode in which the decoder is operating. */124 private decodeMode = DecodingMode.Strict;125 126 /** Resets the instance to make it reusable. */127 startEntity(decodeMode: DecodingMode): void {128 this.decodeMode = decodeMode;129 this.state = EntityDecoderState.EntityStart;130 this.result = 0;131 this.treeIndex = 0;132 this.excess = 1;133 this.consumed = 1;134 }135 136 /**137 * Write an entity to the decoder. This can be called multiple times with partial entities.138 * If the entity is incomplete, the decoder will return -1.139 *140 * Mirrors the implementation of `getDecoder`, but with the ability to stop decoding if the141 * entity is incomplete, and resume when the next string is written.142 *143 * @param input The string containing the entity (or a continuation of the entity).144 * @param offset The offset at which the entity begins. Should be 0 if this is not the first call.145 * @returns The number of characters that were consumed, or -1 if the entity is incomplete.146 */147 write(input: string, offset: number): number {148 switch (this.state) {149 case EntityDecoderState.EntityStart: {150 if (input.charCodeAt(offset) === CharCodes.NUM) {151 this.state = EntityDecoderState.NumericStart;152 this.consumed += 1;153 return this.stateNumericStart(input, offset + 1);154 }155 this.state = EntityDecoderState.NamedEntity;156 return this.stateNamedEntity(input, offset);157 }158 159 case EntityDecoderState.NumericStart: {160 return this.stateNumericStart(input, offset);161 }162 163 case EntityDecoderState.NumericDecimal: {164 return this.stateNumericDecimal(input, offset);165 }166 167 case EntityDecoderState.NumericHex: {168 return this.stateNumericHex(input, offset);169 }170 171 case EntityDecoderState.NamedEntity: {172 return this.stateNamedEntity(input, offset);173 }174 }175 }176 177 /**178 * Switches between the numeric decimal and hexadecimal states.179 *180 * Equivalent to the `Numeric character reference state` in the HTML spec.181 *182 * @param input The string containing the entity (or a continuation of the entity).183 * @param offset The current offset.184 * @returns The number of characters that were consumed, or -1 if the entity is incomplete.185 */186 private stateNumericStart(input: string, offset: number): number {187 if (offset >= input.length) {188 return -1;189 }190 191 if ((input.charCodeAt(offset) | TO_LOWER_BIT) === CharCodes.LOWER_X) {192 this.state = EntityDecoderState.NumericHex;193 this.consumed += 1;194 return this.stateNumericHex(input, offset + 1);195 }196 197 this.state = EntityDecoderState.NumericDecimal;198 return this.stateNumericDecimal(input, offset);199 }200 201 private addToNumericResult(202 input: string,203 start: number,204 end: number,205 base: number,206 ): void {207 if (start !== end) {208 const digitCount = end - start;209 this.result =210 this.result * Math.pow(base, digitCount) +211 Number.parseInt(input.substr(start, digitCount), base);212 this.consumed += digitCount;213 }214 }215 216 /**217 * Parses a hexadecimal numeric entity.218 *219 * Equivalent to the `Hexademical character reference state` in the HTML spec.220 *221 * @param input The string containing the entity (or a continuation of the entity).222 * @param offset The current offset.223 * @returns The number of characters that were consumed, or -1 if the entity is incomplete.224 */225 private stateNumericHex(input: string, offset: number): number {226 const startIndex = offset;227 228 while (offset < input.length) {229 const char = input.charCodeAt(offset);230 if (isNumber(char) || isHexadecimalCharacter(char)) {231 offset += 1;232 } else {233 this.addToNumericResult(input, startIndex, offset, 16);234 return this.emitNumericEntity(char, 3);235 }236 }237 238 this.addToNumericResult(input, startIndex, offset, 16);239 240 return -1;241 }242 243 /**244 * Parses a decimal numeric entity.245 *246 * Equivalent to the `Decimal character reference state` in the HTML spec.247 *248 * @param input The string containing the entity (or a continuation of the entity).249 * @param offset The current offset.250 * @returns The number of characters that were consumed, or -1 if the entity is incomplete.251 */252 private stateNumericDecimal(input: string, offset: number): number {253 const startIndex = offset;254 255 while (offset < input.length) {256 const char = input.charCodeAt(offset);257 if (isNumber(char)) {258 offset += 1;259 } else {260 this.addToNumericResult(input, startIndex, offset, 10);261 return this.emitNumericEntity(char, 2);262 }263 }264 265 this.addToNumericResult(input, startIndex, offset, 10);266 267 return -1;268 }269 270 /**271 * Validate and emit a numeric entity.272 *273 * Implements the logic from the `Hexademical character reference start274 * state` and `Numeric character reference end state` in the HTML spec.275 *276 * @param lastCp The last code point of the entity. Used to see if the277 * entity was terminated with a semicolon.278 * @param expectedLength The minimum number of characters that should be279 * consumed. Used to validate that at least one digit280 * was consumed.281 * @returns The number of characters that were consumed.282 */283 private emitNumericEntity(lastCp: number, expectedLength: number): number {284 // Ensure we consumed at least one digit.285 if (this.consumed <= expectedLength) {286 this.errors?.absenceOfDigitsInNumericCharacterReference(287 this.consumed,288 );289 return 0;290 }291 292 // Figure out if this is a legit end of the entity293 if (lastCp === CharCodes.SEMI) {294 this.consumed += 1;295 } else if (this.decodeMode === DecodingMode.Strict) {296 return 0;297 }298 299 this.emitCodePoint(replaceCodePoint(this.result), this.consumed);300 301 if (this.errors) {302 if (lastCp !== CharCodes.SEMI) {303 this.errors.missingSemicolonAfterCharacterReference();304 }305 306 this.errors.validateNumericCharacterReference(this.result);307 }308 309 return this.consumed;310 }311 312 /**313 * Parses a named entity.314 *315 * Equivalent to the `Named character reference state` in the HTML spec.316 *317 * @param input The string containing the entity (or a continuation of the entity).318 * @param offset The current offset.319 * @returns The number of characters that were consumed, or -1 if the entity is incomplete.320 */321 private stateNamedEntity(input: string, offset: number): number {322 const { decodeTree } = this;323 let current = decodeTree[this.treeIndex];324 // The mask is the number of bytes of the value, including the current byte.325 let valueLength = (current & BinTrieFlags.VALUE_LENGTH) >> 14;326 327 for (; offset < input.length; offset++, this.excess++) {328 const char = input.charCodeAt(offset);329 330 this.treeIndex = determineBranch(331 decodeTree,332 current,333 this.treeIndex + Math.max(1, valueLength),334 char,335 );336 337 if (this.treeIndex < 0) {338 return this.result === 0 ||339 // If we are parsing an attribute340 (this.decodeMode === DecodingMode.Attribute &&341 // We shouldn't have consumed any characters after the entity,342 (valueLength === 0 ||343 // And there should be no invalid characters.344 isEntityInAttributeInvalidEnd(char)))345 ? 0346 : this.emitNotTerminatedNamedEntity();347 }348 349 current = decodeTree[this.treeIndex];350 valueLength = (current & BinTrieFlags.VALUE_LENGTH) >> 14;351 352 // If the branch is a value, store it and continue353 if (valueLength !== 0) {354 // If the entity is terminated by a semicolon, we are done.355 if (char === CharCodes.SEMI) {356 return this.emitNamedEntityData(357 this.treeIndex,358 valueLength,359 this.consumed + this.excess,360 );361 }362 363 // If we encounter a non-terminated (legacy) entity while parsing strictly, then ignore it.364 if (this.decodeMode !== DecodingMode.Strict) {365 this.result = this.treeIndex;366 this.consumed += this.excess;367 this.excess = 0;368 }369 }370 }371 372 return -1;373 }374 375 /**376 * Emit a named entity that was not terminated with a semicolon.377 *378 * @returns The number of characters consumed.379 */380 private emitNotTerminatedNamedEntity(): number {381 const { result, decodeTree } = this;382 383 const valueLength =384 (decodeTree[result] & BinTrieFlags.VALUE_LENGTH) >> 14;385 386 this.emitNamedEntityData(result, valueLength, this.consumed);387 this.errors?.missingSemicolonAfterCharacterReference();388 389 return this.consumed;390 }391 392 /**393 * Emit a named entity.394 *395 * @param result The index of the entity in the decode tree.396 * @param valueLength The number of bytes in the entity.397 * @param consumed The number of characters consumed.398 *399 * @returns The number of characters consumed.400 */401 private emitNamedEntityData(402 result: number,403 valueLength: number,404 consumed: number,405 ): number {406 const { decodeTree } = this;407 408 this.emitCodePoint(409 valueLength === 1410 ? decodeTree[result] & ~BinTrieFlags.VALUE_LENGTH411 : decodeTree[result + 1],412 consumed,413 );414 if (valueLength === 3) {415 // For multi-byte values, we need to emit the second byte.416 this.emitCodePoint(decodeTree[result + 2], consumed);417 }418 419 return consumed;420 }421 422 /**423 * Signal to the parser that the end of the input was reached.424 *425 * Remaining data will be emitted and relevant errors will be produced.426 *427 * @returns The number of characters consumed.428 */429 end(): number {430 switch (this.state) {431 case EntityDecoderState.NamedEntity: {432 // Emit a named entity if we have one.433 return this.result !== 0 &&434 (this.decodeMode !== DecodingMode.Attribute ||435 this.result === this.treeIndex)436 ? this.emitNotTerminatedNamedEntity()437 : 0;438 }439 // Otherwise, emit a numeric entity if we have one.440 case EntityDecoderState.NumericDecimal: {441 return this.emitNumericEntity(0, 2);442 }443 case EntityDecoderState.NumericHex: {444 return this.emitNumericEntity(0, 3);445 }446 case EntityDecoderState.NumericStart: {447 this.errors?.absenceOfDigitsInNumericCharacterReference(448 this.consumed,449 );450 return 0;451 }452 case EntityDecoderState.EntityStart: {453 // Return 0 if we have no entity.454 return 0;455 }456 }457 }458}459 460/**461 * Creates a function that decodes entities in a string.462 *463 * @param decodeTree The decode tree.464 * @returns A function that decodes entities in a string.465 */466function getDecoder(decodeTree: Uint16Array) {467 let returnValue = "";468 const decoder = new EntityDecoder(469 decodeTree,470 (data) => (returnValue += fromCodePoint(data)),471 );472 473 return function decodeWithTrie(474 input: string,475 decodeMode: DecodingMode,476 ): string {477 let lastIndex = 0;478 let offset = 0;479 480 while ((offset = input.indexOf("&", offset)) >= 0) {481 returnValue += input.slice(lastIndex, offset);482 483 decoder.startEntity(decodeMode);484 485 const length = decoder.write(486 input,487 // Skip the "&"488 offset + 1,489 );490 491 if (length < 0) {492 lastIndex = offset + decoder.end();493 break;494 }495 496 lastIndex = offset + length;497 // If `length` is 0, skip the current `&` and continue.498 offset = length === 0 ? lastIndex + 1 : lastIndex;499 }500 501 const result = returnValue + input.slice(lastIndex);502 503 // Make sure we don't keep a reference to the final string.504 returnValue = "";505 506 return result;507 };508}509 510/**511 * Determines the branch of the current node that is taken given the current512 * character. This function is used to traverse the trie.513 *514 * @param decodeTree The trie.515 * @param current The current node.516 * @param nodeIdx The index right after the current node and its value.517 * @param char The current character.518 * @returns The index of the next node, or -1 if no branch is taken.519 */520export function determineBranch(521 decodeTree: Uint16Array,522 current: number,523 nodeIndex: number,524 char: number,525): number {526 const branchCount = (current & BinTrieFlags.BRANCH_LENGTH) >> 7;527 const jumpOffset = current & BinTrieFlags.JUMP_TABLE;528 529 // Case 1: Single branch encoded in jump offset530 if (branchCount === 0) {531 return jumpOffset !== 0 && char === jumpOffset ? nodeIndex : -1;532 }533 534 // Case 2: Multiple branches encoded in jump table535 if (jumpOffset) {536 const value = char - jumpOffset;537 538 return value < 0 || value >= branchCount539 ? -1540 : decodeTree[nodeIndex + value] - 1;541 }542 543 // Case 3: Multiple branches encoded in dictionary544 545 // Binary search for the character.546 let lo = nodeIndex;547 let hi = lo + branchCount - 1;548 549 while (lo <= hi) {550 const mid = (lo + hi) >>> 1;551 const midValue = decodeTree[mid];552 553 if (midValue < char) {554 lo = mid + 1;555 } else if (midValue > char) {556 hi = mid - 1;557 } else {558 return decodeTree[mid + branchCount];559 }560 }561 562 return -1;563}564 565const htmlDecoder = /* #__PURE__ */ getDecoder(htmlDecodeTree);566const xmlDecoder = /* #__PURE__ */ getDecoder(xmlDecodeTree);567 568/**569 * Decodes an HTML string.570 *571 * @param htmlString The string to decode.572 * @param mode The decoding mode.573 * @returns The decoded string.574 */575export function decodeHTML(576 htmlString: string,577 mode: DecodingMode = DecodingMode.Legacy,578): string {579 return htmlDecoder(htmlString, mode);580}581 582/**583 * Decodes an HTML string in an attribute.584 *585 * @param htmlAttribute The string to decode.586 * @returns The decoded string.587 */588export function decodeHTMLAttribute(htmlAttribute: string): string {589 return htmlDecoder(htmlAttribute, DecodingMode.Attribute);590}591 592/**593 * Decodes an HTML string, requiring all entities to be terminated by a semicolon.594 *595 * @param htmlString The string to decode.596 * @returns The decoded string.597 */598export function decodeHTMLStrict(htmlString: string): string {599 return htmlDecoder(htmlString, DecodingMode.Strict);600}601 602/**603 * Decodes an XML string, requiring all entities to be terminated by a semicolon.604 *605 * @param xmlString The string to decode.606 * @returns The decoded string.607 */608export function decodeXML(xmlString: string): string {609 return xmlDecoder(xmlString, DecodingMode.Strict);610}611 612// Re-export for use by eg. htmlparser2613export { htmlDecodeTree } from "./generated/decode-data-html.js";614export { xmlDecodeTree } from "./generated/decode-data-xml.js";615 616export {617 decodeCodePoint,618 replaceCodePoint,619 fromCodePoint,620} from "./decode-codepoint.js";621 