Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1import os from "node:os";2import { BUFFER_ALIGN, BUFFER_SIZE, IS_TS_FLAG_POS } from "../generated/constants.js";3import {4 getBufferOffset,5 parseRaw as parseRawBinding,6 parseRawSync as parseRawSyncBinding,7} from "../bindings.js";8import { rawTransferSupported } from "./supported.js";9 10// Throw an error if running on a platform which raw transfer doesn't support.11//12// Note: This module is lazy-loaded only when user calls `parseSync` or `parseAsync` with13// `experimentalRawTransfer` or `experimentalLazy` options, or calls `experimentalGetLazyVisitor`.14if (!rawTransferSupported()) {15 throw new Error(16 "`experimentalRawTransfer` and `experimentalLazy` options are not supported " +17 "on 32-bit or big-endian systems, versions of NodeJS prior to v22.0.0, " +18 "versions of Deno prior to v2.0.0, or other runtimes",19 );20}21 22/**23 * Parse JS/TS source synchronously on current thread using raw transfer.24 *25 * Convert the buffer returned by Rust to a JS object with provided `convert` function.26 *27 * This function contains logic shared by both `parseSyncRaw` and `parseSyncLazy`.28 *29 * @param {string} filename - Filename30 * @param {string} sourceText - Source text of file31 * @param {Object} options - Parsing options32 * @param {function} convert - Function to convert the buffer returned from Rust into a JS object33 * @returns {Object} - The return value of `convert`34 */35export function parseSyncRawImpl(filename, sourceText, options, convert) {36 const { buffer, sourceByteLen } = prepareRaw(sourceText);37 parseRawSyncBinding(filename, buffer, sourceByteLen, options);38 return convert(buffer, sourceText, sourceByteLen, options);39}40 41// User should not schedule more async tasks than there are available CPUs, as it hurts performance,42// but it's a common mistake in async JS code to do exactly that.43//44// That anti-pattern looks like this when applied to Oxc:45//46// ```js47// const asts = await Promise.all(48// files.map(49// async (filename) => {50// const sourceText = await fs.readFile(filename, 'utf8');51// const ast = await oxc.parseAsync(filename, sourceText);52// return ast;53// }54// )55// );56// ```57//58// In most cases, that'd just result in a bit of degraded performance, and higher memory use because59// of loading sources into memory prematurely.60//61// However, raw transfer uses a 6 GiB buffer for each parsing operation.62// Most of the memory pages in those buffers are never touched, so this does not consume a huge amount63// of physical memory, but it does still consume virtual memory.64//65// If we allowed creating a large number of 6 GiB buffers simultaneously, it would quickly consume66// virtual memory space and risk memory exhaustion. The code above would exhaust all of bottom half67// (heap) of 48-bit virtual memory space if `files.length >= 21_845`. This is not a number which68// is unrealistic in real world code.69//70// To guard against this possibility, we implement a simple queue.71// No more than `os.availableParallelism()` files can be parsed simultaneously, and any further calls to72// `parseAsyncRaw` will be put in a queue, to execute once other tasks complete.73//74// Fallback to `os.cpus().length` on versions of NodeJS prior to v18.14.0, which do not support75// `os.availableParallelism`.76let availableCores = os.availableParallelism ? os.availableParallelism() : os.cpus().length;77const queue = [];78 79/**80 * Parse JS/TS source asynchronously using raw transfer.81 *82 * Convert the buffer returned by Rust to a JS object with provided `convert` function.83 *84 * Queues up parsing operations if more calls than number of CPU cores (see above).85 *86 * This function contains logic shared by both `parseAsyncRaw` and `parseAsyncLazy`.87 *88 * @param {string} filename - Filename89 * @param {string} sourceText - Source text of file90 * @param {Object} options - Parsing options91 * @param {function} convert - Function to convert the buffer returned from Rust into a JS object92 * @returns {Object} - The return value of `convert`93 */94export async function parseAsyncRawImpl(filename, sourceText, options, convert) {95 // Wait for a free CPU core if all CPUs are currently busy.96 //97 // Note: `availableCores` is NOT decremented if have to wait in the queue first,98 // and NOT incremented when parsing completes and it runs next task in the queue.99 //100 // This is to avoid a race condition if `parseAsyncRaw` is called during the microtick in between101 // `resolve` being called below, and the promise resolving here. In that case the new task could102 // start running, and then the promise resolves, and the queued task also starts running.103 // We'd then have `availableParallelism() + 1` tasks running simultaneously. Potentially, this could104 // happen repeatedly, with the number of tasks running simultaneously ever-increasing.105 if (availableCores === 0) {106 // All CPU cores are busy. Put this task in queue and wait for capacity to become available.107 await new Promise((resolve, _) => {108 queue.push(resolve);109 });110 } else {111 // A CPU core is available. Mark core as busy, and run parsing now.112 availableCores--;113 }114 115 // Parse116 const { buffer, sourceByteLen } = prepareRaw(sourceText);117 await parseRawBinding(filename, buffer, sourceByteLen, options);118 const data = convert(buffer, sourceText, sourceByteLen, options);119 120 // Free the CPU core121 if (queue.length > 0) {122 // Some further tasks waiting in queue. Run the next one.123 // Do not increment `availableCores` (see above).124 const resolve = queue.shift();125 resolve();126 } else {127 // No tasks waiting in queue. This CPU is now free.128 availableCores++;129 }130 131 return data;132}133 134const ARRAY_BUFFER_SIZE = BUFFER_SIZE + BUFFER_ALIGN;135const ONE_GIB = 1 << 30;136 137// We keep a cache of buffers for raw transfer, so we can reuse them as much as possible.138//139// When processing multiple files, it's ideal if can reuse an existing buffer, as it's more likely to140// be warm in CPU cache, it avoids allocations, and it saves work for the garbage collector.141//142// However, we also don't want to keep a load of large buffers around indefinitely using up memory,143// if they're not going to be used again.144//145// We have no knowledge of what pattern over time user may process files in (could be lots in quick146// succession, or more occasionally in a long-running process). So we try to use flexible caching147// strategy which is adaptable to many usage patterns.148//149// We use a 2-tier cache.150// Tier 1 uses strong references, tier 2 uses weak references.151//152// When parsing is complete and the buffer is no longer in use, push it to `buffers` (tier 1 cache).153// Set a timer to clear the cache when no activity for 10 seconds.154//155// When the timer expires, move all the buffers from tier 1 cache into `oldBuffers` (tier 2).156// They are stored there as `WeakRef`s, so the garbage collector is free to reclaim them.157//158// On the next call to `parseSync` or `parseAsync`, promote any buffers in tier 2 cache which were not159// already garbage collected back into tier 1 cache. This is on assumption that parsing one file160// indicates parsing as a whole is an ongoing process, and there will likely be further calls to161// `parseSync` / `parseAsync` in future.162//163// The weak tier 2 cache is because V8 does not necessarily free memory as soon as it's able to be164// freed. We don't want to block it from freeing memory, but if it's not done that yet, there's no165// point creating a new buffer, when one already exists.166const CLEAR_BUFFERS_TIMEOUT = 10_000; // 10 seconds167const buffers = [],168 oldBuffers = [];169let clearBuffersTimeout = null;170 171const textEncoder = new TextEncoder();172 173/**174 * Get a buffer (from cache if possible), and copy source text into it.175 *176 * @param {string} sourceText - Source text of file177 * @returns {Object} - Object of form `{ buffer, sourceByteLen }`.178 * - `buffer`: `Uint8Array` containing the AST in raw form.179 * - `sourceByteLen`: Length of source text in UTF-8 bytes180 * (which may not be equal to `sourceText.length` if source contains non-ASCII characters).181 */182export function prepareRaw(sourceText) {183 // Cancel timeout for clearing buffers184 if (clearBuffersTimeout !== null) {185 clearTimeout(clearBuffersTimeout);186 clearBuffersTimeout = null;187 }188 189 // Revive any discarded buffers which have not yet been garbage collected190 if (oldBuffers.length > 0) {191 const revivedBuffers = [];192 for (let oldBuffer of oldBuffers) {193 oldBuffer = oldBuffer.deref();194 if (oldBuffer !== undefined) revivedBuffers.push(oldBuffer);195 }196 oldBuffers.length = 0;197 if (revivedBuffers.length > 0) buffers.unshift(...revivedBuffers);198 }199 200 // Reuse existing buffer, or create a new one201 const buffer = buffers.length > 0 ? buffers.pop() : createBuffer();202 203 // Write source into start of buffer.204 // `TextEncoder` cannot write into a `Uint8Array` larger than 1 GiB,205 // so create a view into buffer of this size to write into.206 const sourceBuffer = new Uint8Array(buffer.buffer, buffer.byteOffset, ONE_GIB);207 const { read, written: sourceByteLen } = textEncoder.encodeInto(sourceText, sourceBuffer);208 if (read !== sourceText.length) throw new Error("Failed to write source text into buffer");209 210 return { buffer, sourceByteLen };211}212 213/**214 * Get if AST should be parsed as JS or TS.215 * Rust side sets a `bool` in this position in buffer which is `true` if TS.216 *217 * @param {Uint8Array} buffer - Buffer containing AST in raw form218 * @returns {boolean} - `true` if AST is JS, `false` if TS219 */220export function isJsAst(buffer) {221 return buffer[IS_TS_FLAG_POS] === 0;222}223 224/**225 * Return buffer to cache, to be reused.226 * Set a timer to clear buffers.227 *228 * @param {Uint8Array} buffer - Buffer229 * @returns {undefined}230 */231export function returnBufferToCache(buffer) {232 buffers.push(buffer);233 234 if (clearBuffersTimeout !== null) clearTimeout(clearBuffersTimeout);235 clearBuffersTimeout = setTimeout(clearBuffersCache, CLEAR_BUFFERS_TIMEOUT);236 clearBuffersTimeout.unref();237}238 239/**240 * Downgrade buffers in tier 1 cache (`buffers`) to tier 2 (`oldBuffers`)241 * so they can be garbage collected.242 *243 * @returns {undefined}244 */245function clearBuffersCache() {246 clearBuffersTimeout = null;247 248 for (const buffer of buffers) {249 oldBuffers.push(new WeakRef(buffer));250 }251 buffers.length = 0;252}253 254/**255 * Create a `Uint8Array` which is 2 GiB in size, with its start aligned on 4 GiB.256 *257 * Achieve this by creating a 6 GiB `ArrayBuffer`, getting the offset within it that's aligned to 4 GiB,258 * chopping off that number of bytes from the start, and shortening to 2 GiB.259 *260 * It's always possible to obtain a 2 GiB slice aligned on 4 GiB within a 6 GiB buffer,261 * no matter how the 6 GiB buffer is aligned.262 *263 * Note: On systems with virtual memory, this only consumes 6 GiB of *virtual* memory.264 * It does not consume physical memory until data is actually written to the `Uint8Array`.265 * Physical memory consumed corresponds to the quantity of data actually written.266 *267 * @returns {Uint8Array} - Buffer268 */269function createBuffer() {270 const arrayBuffer = new ArrayBuffer(ARRAY_BUFFER_SIZE);271 const offset = getBufferOffset(new Uint8Array(arrayBuffer));272 const buffer = new Uint8Array(arrayBuffer, offset, BUFFER_SIZE);273 buffer.int32 = new Int32Array(arrayBuffer, offset, BUFFER_SIZE / 4);274 buffer.float64 = new Float64Array(arrayBuffer, offset, BUFFER_SIZE / 8);275 return buffer;276}277 