Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1import { createRequire } from "node:module";2import { TOKENS_OFFSET_POS_32, TOKENS_LEN_POS_32 } from "../generated/constants.js";3import { isJsAst, parseAsyncRawImpl, parseSyncRawImpl, returnBufferToCache } from "./common.js";4 5const require = createRequire(import.meta.url);6 7/**8 * Parse JS/TS source synchronously on current thread, using raw transfer to speed up deserialization.9 *10 * @param {string} filename - Filename11 * @param {string} sourceText - Source text of file12 * @param {Object} options - Parsing options13 * @returns {Object} - Object with property getters for `program`, `module`, `comments`, and `errors`14 */15export function parseSyncRaw(filename, sourceText, options) {16 return parseSyncRawImpl(filename, sourceText, options, deserialize);17}18 19/**20 * Parse JS/TS source asynchronously, using raw transfer to speed up deserialization.21 *22 * Note that not all of the workload can happen on a separate thread.23 * Parsing on Rust side does happen in a separate thread, but deserialization of the AST to JS objects24 * has to happen on current thread. This synchronous deserialization work typically outweighs25 * the asynchronous parsing by a factor of around 3.26 *27 * i.e. the majority of the workload cannot be parallelized by using this method.28 *29 * Generally `parseSyncRaw` is preferable to use as it does not have the overhead of spawning a thread.30 * If you need to parallelize parsing multiple files, it is recommended to use worker threads.31 *32 * @param {string} filename - Filename33 * @param {string} sourceText - Source text of file34 * @param {Object} options - Parsing options35 * @returns {Object} - Object with property getters for `program`, `module`, `comments`, and `errors`36 */37export function parse(filename, sourceText, options) {38 return parseAsyncRawImpl(filename, sourceText, options, deserialize);39}40 41// Deserializers are large files, so lazy-loaded.42// `deserialize` functions are stored in this array once loaded.43// Index into these arrays is `isJs * 1 + range * 2 + experimentalParent * 4`.44const deserializers = [null, null, null, null, null, null, null, null];45const deserializerNames = [46 "ts",47 "js",48 "ts_range",49 "js_range",50 "ts_parent",51 "js_parent",52 "ts_range_parent",53 "js_range_parent",54];55 56/**57 * Deserialize whole AST from buffer.58 *59 * @param {Uint8Array} buffer - Buffer containing AST in raw form60 * @param {string} sourceText - Source for the file61 * @param {number} sourceByteLen - Length of source text in UTF-8 bytes62 * @param {Object} options - Parsing options63 * @returns {Object} - Object with property getters for `program`, `module`, `comments`, and `errors`64 */65function deserialize(buffer, sourceText, sourceByteLen, options) {66 const isJs = isJsAst(buffer),67 range = !!options.range,68 parent = !!options.experimentalParent;69 70 // Lazy load deserializer, and deserialize buffer to JS objects71 const deserializerIndex = +isJs | (+range << 1) | (+parent << 2);72 let deserializeThis = deserializers[deserializerIndex];73 if (deserializeThis === null) {74 deserializeThis = deserializers[deserializerIndex] = require(75 `../generated/deserialize/${deserializerNames[deserializerIndex]}.js`,76 ).deserialize;77 }78 79 const data = deserializeThis(buffer, sourceText, sourceByteLen);80 81 // Add a line comment for hashbang if JS.82 // Do not add comment if TS, to match `@typescript-eslint/parser`.83 // See https://github.com/oxc-project/oxc/blob/ea784f5f082e4c53c98afde9bf983afd0b95e44e/napi/parser/src/lib.rs#L106-L13084 if (isJs) {85 const { hashbang } = data.program;86 if (hashbang !== null) {87 data.comments.unshift(88 range89 ? {90 type: "Line",91 value: hashbang.value,92 start: hashbang.start,93 end: hashbang.end,94 range: hashbang.range,95 }96 : { type: "Line", value: hashbang.value, start: hashbang.start, end: hashbang.end },97 );98 }99 }100 101 // Deserialize tokens102 const tokens = options.experimentalTokens ? deserializeTokens(buffer, sourceText, isJs) : null;103 104 // Return buffer to cache, to be reused105 returnBufferToCache(buffer);106 107 // We cannot lazily deserialize in the getters, because the buffer might be re-used to parse108 // another file before the getter is called109 if (tokens !== null) {110 return {111 get program() {112 return data.program;113 },114 get module() {115 return data.module;116 },117 get comments() {118 return data.comments;119 },120 get tokens() {121 return tokens;122 },123 get errors() {124 return data.errors;125 },126 };127 }128 129 return {130 get program() {131 return data.program;132 },133 get module() {134 return data.module;135 },136 get comments() {137 return data.comments;138 },139 get errors() {140 return data.errors;141 },142 };143}144 145// `ESTreeKind` discriminants (set by Rust side)146const PRIVATE_IDENTIFIER_KIND = 2;147const REGEXP_KIND = 8;148 149// Indexed by `ESTreeKind` discriminant (matches `ESTreeKind` enum in `estree_kind.rs`)150const TOKEN_TYPES = [151 "Identifier",152 "Keyword",153 "PrivateIdentifier",154 "Punctuator",155 "Numeric",156 "String",157 "Boolean",158 "Null",159 "RegularExpression",160 "Template",161 "JSXText",162 "JSXIdentifier",163];164 165// Mask for active bits in `ESTreeKind` discriminants166const TOKEN_KIND_MASK = 15;167 168// Details of Rust `Token` type169const TOKEN_SIZE = 16;170 171/**172 * Deserialize tokens from buffer.173 * @param {Uint8Array} buffer - Buffer containing AST in raw form174 * @param {string} sourceText - Source for the file175 * @param {boolean} isJs - `true` if parsing in JS mode176 * @returns {Object[]} - Array of token objects177 */178function deserializeTokens(buffer, sourceText, isJs) {179 const { int32 } = buffer;180 181 let pos = int32[TOKENS_OFFSET_POS_32];182 const len = int32[TOKENS_LEN_POS_32];183 const endPos = pos + len * TOKEN_SIZE;184 185 const tokens = [];186 while (pos < endPos) {187 tokens.push(deserializeToken(pos, int32, sourceText, isJs));188 pos += TOKEN_SIZE;189 }190 return tokens;191}192 193/**194 * Deserialize a token from buffer at position `pos`.195 * @param {number} pos - Position in buffer containing Rust `Token` type196 * @param {Int32Array} int32 - Buffer containing AST in raw form as an `Int32Array`197 * @param {string} sourceText - Source for the file198 * @param {boolean} isJs - `true` if parsing in JS mode199 * @returns {Object} - Token object200 */201function deserializeToken(pos, int32, sourceText, isJs) {202 const pos32 = pos >> 2,203 start = int32[pos32],204 end = int32[pos32 + 1],205 kindAndFlags = int32[pos32 + 2];206 207 let value = sourceText.slice(start, end);208 209 // `Kind` is byte at index 8 in `Token`.210 // `Kind` has 12 variants numbered from 0 to 11.211 // We have to mask the bottom byte (`& 0xFF`), so may as well mask off bits which can't be set in `Kind` at same time.212 // This may allow V8 to generate more efficient code for `TOKEN_TYPES[kind]`.213 const kind = kindAndFlags & TOKEN_KIND_MASK;214 215 if (kind === REGEXP_KIND) {216 const patternEnd = value.lastIndexOf("/");217 return {218 type: "RegularExpression",219 value,220 regex: {221 pattern: value.slice(1, patternEnd),222 flags: value.slice(patternEnd + 1),223 },224 start,225 end,226 };227 }228 229 // Strip leading `#` from private identifiers230 if (kind === PRIVATE_IDENTIFIER_KIND) value = value.slice(1);231 232 // Unescape identifiers, keywords, and private identifiers in JS mode.233 // `is_escaped` flag is in byte 10 of `Token`, and is a `bool`.234 if (isJs && kind <= PRIVATE_IDENTIFIER_KIND && (kindAndFlags & 0x10000) !== 0) {235 value = unescapeIdentifier(value);236 }237 238 return { type: TOKEN_TYPES[kind], value, start, end };239}240 241/**242 * Unescape an identifier.243 *244 * We do this on JS side, because escaped identifiers are so extremely rare that this function245 * is never called in practice anyway.246 *247 * @param {string} name - Identifier name to unescape248 * @returns {string} - Unescaped identifier name249 */250function unescapeIdentifier(name) {251 return name.replace(/\\u(?:\{([0-9a-fA-F]+)\}|([0-9a-fA-F]{4}))/g, (_, hex1, hex2) =>252 String.fromCodePoint(parseInt(hex1 ?? hex2, 16)),253 );254}255 