Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
decode.ts621 linesDownload Raw Back to src
1import { htmlDecodeTree } from "./generated/decode-data-html.js";2import { xmlDecodeTree } from "./generated/decode-data-xml.js";3import { replaceCodePoint, fromCodePoint } from "./decode-codepoint.js";4 5const enum CharCodes {6    NUM = 35, // "#"7    SEMI = 59, // ";"8    EQUALS = 61, // "="9    ZERO = 48, // "0"10    NINE = 57, // "9"11    LOWER_A = 97, // "a"12    LOWER_F = 102, // "f"13    LOWER_X = 120, // "x"14    LOWER_Z = 122, // "z"15    UPPER_A = 65, // "A"16    UPPER_F = 70, // "F"17    UPPER_Z = 90, // "Z"18}19 20/** Bit that needs to be set to convert an upper case ASCII character to lower case */21const TO_LOWER_BIT = 0b10_0000;22 23export enum BinTrieFlags {24    VALUE_LENGTH = 0b1100_0000_0000_0000,25    BRANCH_LENGTH = 0b0011_1111_1000_0000,26    JUMP_TABLE = 0b0000_0000_0111_1111,27}28 29function isNumber(code: number): boolean {30    return code >= CharCodes.ZERO && code <= CharCodes.NINE;31}32 33function isHexadecimalCharacter(code: number): boolean {34    return (35        (code >= CharCodes.UPPER_A && code <= CharCodes.UPPER_F) ||36        (code >= CharCodes.LOWER_A && code <= CharCodes.LOWER_F)37    );38}39 40function isAsciiAlphaNumeric(code: number): boolean {41    return (42        (code >= CharCodes.UPPER_A && code <= CharCodes.UPPER_Z) ||43        (code >= CharCodes.LOWER_A && code <= CharCodes.LOWER_Z) ||44        isNumber(code)45    );46}47 48/**49 * Checks if the given character is a valid end character for an entity in an attribute.50 *51 * Attribute values that aren't terminated properly aren't parsed, and shouldn't lead to a parser error.52 * See the example in https://html.spec.whatwg.org/multipage/parsing.html#named-character-reference-state53 */54function isEntityInAttributeInvalidEnd(code: number): boolean {55    return code === CharCodes.EQUALS || isAsciiAlphaNumeric(code);56}57 58const enum EntityDecoderState {59    EntityStart,60    NumericStart,61    NumericDecimal,62    NumericHex,63    NamedEntity,64}65 66export enum DecodingMode {67    /** Entities in text nodes that can end with any character. */68    Legacy = 0,69    /** Only allow entities terminated with a semicolon. */70    Strict = 1,71    /** Entities in attributes have limitations on ending characters. */72    Attribute = 2,73}74 75/**76 * Producers for character reference errors as defined in the HTML spec.77 */78export interface EntityErrorProducer {79    missingSemicolonAfterCharacterReference(): void;80    absenceOfDigitsInNumericCharacterReference(81        consumedCharacters: number,82    ): void;83    validateNumericCharacterReference(code: number): void;84}85 86/**87 * Token decoder with support of writing partial entities.88 */89export class EntityDecoder {90    constructor(91        /** The tree used to decode entities. */92        private readonly decodeTree: Uint16Array,93        /**94         * The function that is called when a codepoint is decoded.95         *96         * For multi-byte named entities, this will be called multiple times,97         * with the second codepoint, and the same `consumed` value.98         *99         * @param codepoint The decoded codepoint.100         * @param consumed The number of bytes consumed by the decoder.101         */102        private readonly emitCodePoint: (cp: number, consumed: number) => void,103        /** An object that is used to produce errors. */104        private readonly errors?: EntityErrorProducer | undefined,105    ) {}106 107    /** The current state of the decoder. */108    private state = EntityDecoderState.EntityStart;109    /** Characters that were consumed while parsing an entity. */110    private consumed = 1;111    /**112     * The result of the entity.113     *114     * Either the result index of a numeric entity, or the codepoint of a115     * numeric entity.116     */117    private result = 0;118 119    /** The current index in the decode tree. */120    private treeIndex = 0;121    /** The number of characters that were consumed in excess. */122    private excess = 1;123    /** The mode in which the decoder is operating. */124    private decodeMode = DecodingMode.Strict;125 126    /** Resets the instance to make it reusable. */127    startEntity(decodeMode: DecodingMode): void {128        this.decodeMode = decodeMode;129        this.state = EntityDecoderState.EntityStart;130        this.result = 0;131        this.treeIndex = 0;132        this.excess = 1;133        this.consumed = 1;134    }135 136    /**137     * Write an entity to the decoder. This can be called multiple times with partial entities.138     * If the entity is incomplete, the decoder will return -1.139     *140     * Mirrors the implementation of `getDecoder`, but with the ability to stop decoding if the141     * entity is incomplete, and resume when the next string is written.142     *143     * @param input The string containing the entity (or a continuation of the entity).144     * @param offset The offset at which the entity begins. Should be 0 if this is not the first call.145     * @returns The number of characters that were consumed, or -1 if the entity is incomplete.146     */147    write(input: string, offset: number): number {148        switch (this.state) {149            case EntityDecoderState.EntityStart: {150                if (input.charCodeAt(offset) === CharCodes.NUM) {151                    this.state = EntityDecoderState.NumericStart;152                    this.consumed += 1;153                    return this.stateNumericStart(input, offset + 1);154                }155                this.state = EntityDecoderState.NamedEntity;156                return this.stateNamedEntity(input, offset);157            }158 159            case EntityDecoderState.NumericStart: {160                return this.stateNumericStart(input, offset);161            }162 163            case EntityDecoderState.NumericDecimal: {164                return this.stateNumericDecimal(input, offset);165            }166 167            case EntityDecoderState.NumericHex: {168                return this.stateNumericHex(input, offset);169            }170 171            case EntityDecoderState.NamedEntity: {172                return this.stateNamedEntity(input, offset);173            }174        }175    }176 177    /**178     * Switches between the numeric decimal and hexadecimal states.179     *180     * Equivalent to the `Numeric character reference state` in the HTML spec.181     *182     * @param input The string containing the entity (or a continuation of the entity).183     * @param offset The current offset.184     * @returns The number of characters that were consumed, or -1 if the entity is incomplete.185     */186    private stateNumericStart(input: string, offset: number): number {187        if (offset >= input.length) {188            return -1;189        }190 191        if ((input.charCodeAt(offset) | TO_LOWER_BIT) === CharCodes.LOWER_X) {192            this.state = EntityDecoderState.NumericHex;193            this.consumed += 1;194            return this.stateNumericHex(input, offset + 1);195        }196 197        this.state = EntityDecoderState.NumericDecimal;198        return this.stateNumericDecimal(input, offset);199    }200 201    private addToNumericResult(202        input: string,203        start: number,204        end: number,205        base: number,206    ): void {207        if (start !== end) {208            const digitCount = end - start;209            this.result =210                this.result * Math.pow(base, digitCount) +211                Number.parseInt(input.substr(start, digitCount), base);212            this.consumed += digitCount;213        }214    }215 216    /**217     * Parses a hexadecimal numeric entity.218     *219     * Equivalent to the `Hexademical character reference state` in the HTML spec.220     *221     * @param input The string containing the entity (or a continuation of the entity).222     * @param offset The current offset.223     * @returns The number of characters that were consumed, or -1 if the entity is incomplete.224     */225    private stateNumericHex(input: string, offset: number): number {226        const startIndex = offset;227 228        while (offset < input.length) {229            const char = input.charCodeAt(offset);230            if (isNumber(char) || isHexadecimalCharacter(char)) {231                offset += 1;232            } else {233                this.addToNumericResult(input, startIndex, offset, 16);234                return this.emitNumericEntity(char, 3);235            }236        }237 238        this.addToNumericResult(input, startIndex, offset, 16);239 240        return -1;241    }242 243    /**244     * Parses a decimal numeric entity.245     *246     * Equivalent to the `Decimal character reference state` in the HTML spec.247     *248     * @param input The string containing the entity (or a continuation of the entity).249     * @param offset The current offset.250     * @returns The number of characters that were consumed, or -1 if the entity is incomplete.251     */252    private stateNumericDecimal(input: string, offset: number): number {253        const startIndex = offset;254 255        while (offset < input.length) {256            const char = input.charCodeAt(offset);257            if (isNumber(char)) {258                offset += 1;259            } else {260                this.addToNumericResult(input, startIndex, offset, 10);261                return this.emitNumericEntity(char, 2);262            }263        }264 265        this.addToNumericResult(input, startIndex, offset, 10);266 267        return -1;268    }269 270    /**271     * Validate and emit a numeric entity.272     *273     * Implements the logic from the `Hexademical character reference start274     * state` and `Numeric character reference end state` in the HTML spec.275     *276     * @param lastCp The last code point of the entity. Used to see if the277     *               entity was terminated with a semicolon.278     * @param expectedLength The minimum number of characters that should be279     *                       consumed. Used to validate that at least one digit280     *                       was consumed.281     * @returns The number of characters that were consumed.282     */283    private emitNumericEntity(lastCp: number, expectedLength: number): number {284        // Ensure we consumed at least one digit.285        if (this.consumed <= expectedLength) {286            this.errors?.absenceOfDigitsInNumericCharacterReference(287                this.consumed,288            );289            return 0;290        }291 292        // Figure out if this is a legit end of the entity293        if (lastCp === CharCodes.SEMI) {294            this.consumed += 1;295        } else if (this.decodeMode === DecodingMode.Strict) {296            return 0;297        }298 299        this.emitCodePoint(replaceCodePoint(this.result), this.consumed);300 301        if (this.errors) {302            if (lastCp !== CharCodes.SEMI) {303                this.errors.missingSemicolonAfterCharacterReference();304            }305 306            this.errors.validateNumericCharacterReference(this.result);307        }308 309        return this.consumed;310    }311 312    /**313     * Parses a named entity.314     *315     * Equivalent to the `Named character reference state` in the HTML spec.316     *317     * @param input The string containing the entity (or a continuation of the entity).318     * @param offset The current offset.319     * @returns The number of characters that were consumed, or -1 if the entity is incomplete.320     */321    private stateNamedEntity(input: string, offset: number): number {322        const { decodeTree } = this;323        let current = decodeTree[this.treeIndex];324        // The mask is the number of bytes of the value, including the current byte.325        let valueLength = (current & BinTrieFlags.VALUE_LENGTH) >> 14;326 327        for (; offset < input.length; offset++, this.excess++) {328            const char = input.charCodeAt(offset);329 330            this.treeIndex = determineBranch(331                decodeTree,332                current,333                this.treeIndex + Math.max(1, valueLength),334                char,335            );336 337            if (this.treeIndex < 0) {338                return this.result === 0 ||339                    // If we are parsing an attribute340                    (this.decodeMode === DecodingMode.Attribute &&341                        // We shouldn't have consumed any characters after the entity,342                        (valueLength === 0 ||343                            // And there should be no invalid characters.344                            isEntityInAttributeInvalidEnd(char)))345                    ? 0346                    : this.emitNotTerminatedNamedEntity();347            }348 349            current = decodeTree[this.treeIndex];350            valueLength = (current & BinTrieFlags.VALUE_LENGTH) >> 14;351 352            // If the branch is a value, store it and continue353            if (valueLength !== 0) {354                // If the entity is terminated by a semicolon, we are done.355                if (char === CharCodes.SEMI) {356                    return this.emitNamedEntityData(357                        this.treeIndex,358                        valueLength,359                        this.consumed + this.excess,360                    );361                }362 363                // If we encounter a non-terminated (legacy) entity while parsing strictly, then ignore it.364                if (this.decodeMode !== DecodingMode.Strict) {365                    this.result = this.treeIndex;366                    this.consumed += this.excess;367                    this.excess = 0;368                }369            }370        }371 372        return -1;373    }374 375    /**376     * Emit a named entity that was not terminated with a semicolon.377     *378     * @returns The number of characters consumed.379     */380    private emitNotTerminatedNamedEntity(): number {381        const { result, decodeTree } = this;382 383        const valueLength =384            (decodeTree[result] & BinTrieFlags.VALUE_LENGTH) >> 14;385 386        this.emitNamedEntityData(result, valueLength, this.consumed);387        this.errors?.missingSemicolonAfterCharacterReference();388 389        return this.consumed;390    }391 392    /**393     * Emit a named entity.394     *395     * @param result The index of the entity in the decode tree.396     * @param valueLength The number of bytes in the entity.397     * @param consumed The number of characters consumed.398     *399     * @returns The number of characters consumed.400     */401    private emitNamedEntityData(402        result: number,403        valueLength: number,404        consumed: number,405    ): number {406        const { decodeTree } = this;407 408        this.emitCodePoint(409            valueLength === 1410                ? decodeTree[result] & ~BinTrieFlags.VALUE_LENGTH411                : decodeTree[result + 1],412            consumed,413        );414        if (valueLength === 3) {415            // For multi-byte values, we need to emit the second byte.416            this.emitCodePoint(decodeTree[result + 2], consumed);417        }418 419        return consumed;420    }421 422    /**423     * Signal to the parser that the end of the input was reached.424     *425     * Remaining data will be emitted and relevant errors will be produced.426     *427     * @returns The number of characters consumed.428     */429    end(): number {430        switch (this.state) {431            case EntityDecoderState.NamedEntity: {432                // Emit a named entity if we have one.433                return this.result !== 0 &&434                    (this.decodeMode !== DecodingMode.Attribute ||435                        this.result === this.treeIndex)436                    ? this.emitNotTerminatedNamedEntity()437                    : 0;438            }439            // Otherwise, emit a numeric entity if we have one.440            case EntityDecoderState.NumericDecimal: {441                return this.emitNumericEntity(0, 2);442            }443            case EntityDecoderState.NumericHex: {444                return this.emitNumericEntity(0, 3);445            }446            case EntityDecoderState.NumericStart: {447                this.errors?.absenceOfDigitsInNumericCharacterReference(448                    this.consumed,449                );450                return 0;451            }452            case EntityDecoderState.EntityStart: {453                // Return 0 if we have no entity.454                return 0;455            }456        }457    }458}459 460/**461 * Creates a function that decodes entities in a string.462 *463 * @param decodeTree The decode tree.464 * @returns A function that decodes entities in a string.465 */466function getDecoder(decodeTree: Uint16Array) {467    let returnValue = "";468    const decoder = new EntityDecoder(469        decodeTree,470        (data) => (returnValue += fromCodePoint(data)),471    );472 473    return function decodeWithTrie(474        input: string,475        decodeMode: DecodingMode,476    ): string {477        let lastIndex = 0;478        let offset = 0;479 480        while ((offset = input.indexOf("&", offset)) >= 0) {481            returnValue += input.slice(lastIndex, offset);482 483            decoder.startEntity(decodeMode);484 485            const length = decoder.write(486                input,487                // Skip the "&"488                offset + 1,489            );490 491            if (length < 0) {492                lastIndex = offset + decoder.end();493                break;494            }495 496            lastIndex = offset + length;497            // If `length` is 0, skip the current `&` and continue.498            offset = length === 0 ? lastIndex + 1 : lastIndex;499        }500 501        const result = returnValue + input.slice(lastIndex);502 503        // Make sure we don't keep a reference to the final string.504        returnValue = "";505 506        return result;507    };508}509 510/**511 * Determines the branch of the current node that is taken given the current512 * character. This function is used to traverse the trie.513 *514 * @param decodeTree The trie.515 * @param current The current node.516 * @param nodeIdx The index right after the current node and its value.517 * @param char The current character.518 * @returns The index of the next node, or -1 if no branch is taken.519 */520export function determineBranch(521    decodeTree: Uint16Array,522    current: number,523    nodeIndex: number,524    char: number,525): number {526    const branchCount = (current & BinTrieFlags.BRANCH_LENGTH) >> 7;527    const jumpOffset = current & BinTrieFlags.JUMP_TABLE;528 529    // Case 1: Single branch encoded in jump offset530    if (branchCount === 0) {531        return jumpOffset !== 0 && char === jumpOffset ? nodeIndex : -1;532    }533 534    // Case 2: Multiple branches encoded in jump table535    if (jumpOffset) {536        const value = char - jumpOffset;537 538        return value < 0 || value >= branchCount539            ? -1540            : decodeTree[nodeIndex + value] - 1;541    }542 543    // Case 3: Multiple branches encoded in dictionary544 545    // Binary search for the character.546    let lo = nodeIndex;547    let hi = lo + branchCount - 1;548 549    while (lo <= hi) {550        const mid = (lo + hi) >>> 1;551        const midValue = decodeTree[mid];552 553        if (midValue < char) {554            lo = mid + 1;555        } else if (midValue > char) {556            hi = mid - 1;557        } else {558            return decodeTree[mid + branchCount];559        }560    }561 562    return -1;563}564 565const htmlDecoder = /* #__PURE__ */ getDecoder(htmlDecodeTree);566const xmlDecoder = /* #__PURE__ */ getDecoder(xmlDecodeTree);567 568/**569 * Decodes an HTML string.570 *571 * @param htmlString The string to decode.572 * @param mode The decoding mode.573 * @returns The decoded string.574 */575export function decodeHTML(576    htmlString: string,577    mode: DecodingMode = DecodingMode.Legacy,578): string {579    return htmlDecoder(htmlString, mode);580}581 582/**583 * Decodes an HTML string in an attribute.584 *585 * @param htmlAttribute The string to decode.586 * @returns The decoded string.587 */588export function decodeHTMLAttribute(htmlAttribute: string): string {589    return htmlDecoder(htmlAttribute, DecodingMode.Attribute);590}591 592/**593 * Decodes an HTML string, requiring all entities to be terminated by a semicolon.594 *595 * @param htmlString The string to decode.596 * @returns The decoded string.597 */598export function decodeHTMLStrict(htmlString: string): string {599    return htmlDecoder(htmlString, DecodingMode.Strict);600}601 602/**603 * Decodes an XML string, requiring all entities to be terminated by a semicolon.604 *605 * @param xmlString The string to decode.606 * @returns The decoded string.607 */608export function decodeXML(xmlString: string): string {609    return xmlDecoder(xmlString, DecodingMode.Strict);610}611 612// Re-export for use by eg. htmlparser2613export { htmlDecodeTree } from "./generated/decode-data-html.js";614export { xmlDecodeTree } from "./generated/decode-data-xml.js";615 616export {617    decodeCodePoint,618    replaceCodePoint,619    fromCodePoint,620} from "./decode-codepoint.js";621 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai