Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
parse.ts438 linesDownload Raw Back to src
1/**2 * EventSource/Server-Sent Events parser3 * @see https://html.spec.whatwg.org/multipage/server-sent-events.html4 */5import {ParseError} from './errors.ts'6import type {EventSourceParser, ParserConfig} from './types.ts'7 8// ASCII codes used in the hot parsing paths.9const LF = 1010const CR = 1311const SPACE = 3212 13// oxlint-disable-next-line no-unused-vars14function noop(_arg: unknown) {15  // intentional noop16}17 18/**19 * Creates a new EventSource parser.20 *21 * @param config - Parser configuration. Accepts callbacks (see {@link ParserCallbacks})22 *   and options like `maxBufferSize` (see {@link ParserConfig}).23 *24 * @returns A new EventSource parser, with `feed` and `reset` methods.25 * @public26 */27export function createParser(config: ParserConfig): EventSourceParser {28  if (typeof config === 'function') {29    throw new TypeError(30      '`config` must be an object, got a function instead. Did you mean `createParser({onEvent: fn})`?',31    )32  }33 34  const {onEvent = noop, onError = noop, onRetry = noop, onComment, maxBufferSize} = config35 36  // Trailing bytes from prior `feed()` calls that did not yet form a complete line.37  // Stored as an array of fragments and only joined when a line terminator arrives.38  // Concatenating per-feed (`prefix + chunk`) is O(N²) when a single SSE line spans39  // many chunks (e.g. a large `data:` payload streamed in tiny slices, or an MCP-style40  // server that emits one giant content block). Buffering as fragments + joining once41  // makes the same workload linear.42  const pendingFragments: string[] = []43 44  // Running total of `pendingFragments` lengths, kept in sync with the array so the45  // `maxBufferSize` check doesn't have to walk the fragment list on every feed.46  let pendingFragmentsLength = 047 48  let isFirstChunk = true49  let id: string | undefined50  let data = ''51  let dataLines = 052  let eventType: string | undefined53 54  // Set after a `maxBufferSize` overflow. Once tripped, `feed()` throws until55  // `reset()` is called — see the comment on `maxBufferSize` in `ParserConfig`.56  let terminated = false57 58  /**59   * Feeds a chunk of the SSE stream to the parser. Any trailing bytes that do60   * not yet form a complete line are held back and prepended to the next chunk,61   * so callers can pass arbitrary slices of the stream without worrying about62   * line boundaries.63   *64   * Per the SSE spec, a UTF-8 BOM (0xEF 0xBB 0xBF) at the start of the very65   * first chunk is stripped before parsing.66   *67   * @see https://html.spec.whatwg.org/multipage/server-sent-events.html#parsing-an-event-stream68   */69  function feed(chunk: string) {70    if (terminated) {71      throw new Error(72        'Cannot feed parser: it was terminated after exceeding the configured max buffer size. Call `reset()` to resume parsing.',73      )74    }75 76    if (isFirstChunk) {77      isFirstChunk = false78      // Match and strip UTF-8 BOM from the start of the stream, if present.79      // (Per the spec, this is only valid at the very start of the stream)80      if (81        chunk.charCodeAt(0) === 0xef &&82        chunk.charCodeAt(1) === 0xbb &&83        chunk.charCodeAt(2) === 0xbf84      ) {85        chunk = chunk.slice(3)86      }87    }88 89    // Hot path: no buffered prefix from a prior partial line. Hand the chunk90    // straight to `processLines`, exactly like the original implementation.91    // Zero new work in the common case (every chunk ends with `\n\n`).92    if (pendingFragments.length === 0) {93      const trailing = processLines(chunk)94      if (trailing !== '') {95        pendingFragments.push(trailing)96        pendingFragmentsLength = trailing.length97      }98      checkBufferSize()99      return100    }101 102    // We have a buffered prefix. If this chunk also has no terminator, append103    // to the buffer without concatenating — that's the O(N²) trap we're104    // avoiding (large single `data:` payload split across many tiny chunks).105    if (chunk.indexOf('\n') === -1 && chunk.indexOf('\r') === -1) {106      pendingFragments.push(chunk)107      pendingFragmentsLength += chunk.length108      checkBufferSize()109      return110    }111 112    // Terminator arrived. Join the accumulated fragments + this chunk once,113    // process, and buffer any new trailing partial line.114    pendingFragments.push(chunk)115    const input = pendingFragments.join('')116    pendingFragments.length = 0117    pendingFragmentsLength = 0118    const trailing = processLines(input)119    if (trailing !== '') {120      pendingFragments.push(trailing)121      pendingFragmentsLength = trailing.length122    }123    checkBufferSize()124  }125 126  function checkBufferSize() {127    if (maxBufferSize === undefined) return128    if (pendingFragmentsLength + data.length <= maxBufferSize) return129 130    terminated = true131    pendingFragments.length = 0132    pendingFragmentsLength = 0133    id = undefined134    data = ''135    dataLines = 0136    eventType = undefined137    onError(138      new ParseError(`Buffered data exceeded max buffer size of ${maxBufferSize} characters`, {139        type: 'max-buffer-size-exceeded',140      }),141    )142  }143 144  /**145   * Splits `chunk` into SSE lines and dispatches each to the appropriate handler.146   * Returns any trailing bytes that did not terminate with a line break, so the147   * caller can prepend them to the next chunk.148   *149   * The SSE spec permits three line terminators: `\n`, `\r`, and `\r\n`. Real-world150   * streams almost always use plain `\n`, so we take a fast path when no `\r` is151   * present in the chunk. The slow path is spec-correct but does more work per line.152   */153  function processLines(chunk: string): string {154    let searchIndex = 0155 156    // Fast path: LF-only chunk (the common case for typical SSE servers).157    // We can scan forward with a single `indexOf('\n')` per line and inline158    // the hot-path branches for `data:` and `event:` without the CR bookkeeping159    // the slow path needs.160    if (chunk.indexOf('\r') === -1) {161      let lfIndex = chunk.indexOf('\n', searchIndex)162      while (lfIndex !== -1) {163        // Blank line: end-of-event marker. Dispatch the accumulated event (if any)164        // and reset the buffered fields. This is hoisted out of `parseLine` because165        // it's the single most common line shape after `data:` lines.166        if (searchIndex === lfIndex) {167          if (dataLines > 0) {168            onEvent({id, event: eventType, data})169          }170          id = undefined171          data = ''172          dataLines = 0173          eventType = undefined174          searchIndex = lfIndex + 1175          lfIndex = chunk.indexOf('\n', searchIndex)176          continue177        }178        const firstCharCode = chunk.charCodeAt(searchIndex)179        if (isDataPrefix(chunk, searchIndex, firstCharCode)) {180          // `data:` line — append the value to the event's data buffer.181          // 'data:'.length === 5, 'data: '.length === 6182          const valueStart =183            chunk.charCodeAt(searchIndex + 5) === SPACE ? searchIndex + 6 : searchIndex + 5184          const value = chunk.slice(valueStart, lfIndex)185          // Fast path within a fast path: if this is the first data line AND the186          // next char is another LF (i.e. `data:foo\n\n`), dispatch immediately187          // without ever writing to the `data` buffer. This is the shape of a188          // typical single-line SSE event (ChatGPT-style streams, etc.) and is189          // hot enough to be worth the duplication.190          if (dataLines === 0 && chunk.charCodeAt(lfIndex + 1) === LF) {191            onEvent({id, event: eventType, data: value})192            id = undefined193            data = ''194            eventType = undefined195            searchIndex = lfIndex + 2196            lfIndex = chunk.indexOf('\n', searchIndex)197            continue198          }199          // Multi-line data: concatenate with newline separator per spec.200          data = dataLines === 0 ? value : `${data}\n${value}`201          dataLines++202        } else if (isEventPrefix(chunk, searchIndex, firstCharCode)) {203          // `event:` line — set the event type for the next dispatch. Per spec,204          // an empty value resets `event type` to its default (undefined here).205          // 'event:'.length === 6, 'event: '.length === 7206          eventType =207            chunk.slice(208              chunk.charCodeAt(searchIndex + 6) === SPACE ? searchIndex + 7 : searchIndex + 6,209              lfIndex,210            ) || undefined211        } else {212          // Everything else: `id:`, `retry:`, comment lines (`:` prefix), unknown213          // fields, or malformed lines. These are rarer and go through the full214          // per-line parser, which handles the SSE field grammar in detail.215          parseLine(chunk, searchIndex, lfIndex)216        }217        searchIndex = lfIndex + 1218        lfIndex = chunk.indexOf('\n', searchIndex)219      }220      return chunk.slice(searchIndex)221    }222 223    // Slow path: the chunk contains at least one `\r`, so lines may be terminated224    // by `\r`, `\n`, or `\r\n`. We locate the next terminator by looking at both225    // the nearest `\r` and `\n` and picking whichever comes first.226    while (searchIndex < chunk.length) {227      const crIndex = chunk.indexOf('\r', searchIndex)228      const lfIndex = chunk.indexOf('\n', searchIndex)229 230      let lineEnd = -1231      if (crIndex !== -1 && lfIndex !== -1) {232        lineEnd = crIndex < lfIndex ? crIndex : lfIndex233      } else if (crIndex !== -1) {234        // A trailing `\r` at the very end of the chunk is ambiguous: it could be235        // a bare-CR terminator, or the first half of a `\r\n` whose `\n` arrives236        // in the next chunk. Defer until we see more input.237        if (crIndex === chunk.length - 1) {238          lineEnd = -1239        } else {240          lineEnd = crIndex241        }242      } else if (lfIndex !== -1) {243        lineEnd = lfIndex244      }245 246      if (lineEnd === -1) {247        break248      }249 250      parseLine(chunk, searchIndex, lineEnd)251      searchIndex = lineEnd + 1252      // If we just consumed a `\r` and the next char is `\n`, skip it so the253      // pair is treated as a single terminator rather than an empty line.254      if (chunk.charCodeAt(searchIndex - 1) === CR && chunk.charCodeAt(searchIndex) === LF) {255        searchIndex++256      }257    }258 259    return chunk.slice(searchIndex)260  }261 262  function parseLine(chunk: string, start: number, end: number) {263    if (start === end) {264      dispatchEvent()265      return266    }267 268    const firstCharCode = chunk.charCodeAt(start)269 270    if (isDataPrefix(chunk, start, firstCharCode)) {271      // 'data:'.length === 5, 'data: '.length === 6272      const valueStart = chunk.charCodeAt(start + 5) === SPACE ? start + 6 : start + 5273      const value = chunk.slice(valueStart, end)274      data = dataLines === 0 ? value : `${data}\n${value}`275      dataLines++276      return277    }278 279    if (isEventPrefix(chunk, start, firstCharCode)) {280      // 'event:'.length === 6, 'event: '.length === 7281      eventType =282        chunk.slice(chunk.charCodeAt(start + 6) === SPACE ? start + 7 : start + 6, end) || undefined283      return284    }285 286    // Fast path for "id:" — 'i' = 105, 'd' = 100, ':' = 58287    if (288      firstCharCode === 105 &&289      chunk.charCodeAt(start + 1) === 100 &&290      chunk.charCodeAt(start + 2) === 58291    ) {292      // 'id:'.length === 3, 'id: '.length === 4293      const value = chunk.slice(chunk.charCodeAt(start + 3) === SPACE ? start + 4 : start + 3, end)294      id = value.includes('\0') ? undefined : value295      return296    }297 298    // Comment line — ':' = 58299    if (firstCharCode === 58) {300      if (onComment) {301        const line = chunk.slice(start, end)302        // skip ':' (+1), or ': ' (+2) when a space follows303        onComment(line.slice(chunk.charCodeAt(start + 1) === SPACE ? 2 : 1))304      }305      return306    }307 308    const line = chunk.slice(start, end)309    const fieldSeparatorIndex = line.indexOf(':')310    if (fieldSeparatorIndex === -1) {311      processField(line, '', line)312      return313    }314 315    const field = line.slice(0, fieldSeparatorIndex)316    // skip ':' (+1), or ': ' (+2) when a space follows317    const offset = line.charCodeAt(fieldSeparatorIndex + 1) === SPACE ? 2 : 1318    const value = line.slice(fieldSeparatorIndex + offset)319    processField(field, value, line)320  }321 322  function processField(field: string, value: string, line: string) {323    // Field names must be compared literally, with no case folding performed.324    switch (field) {325      case 'event':326        // Set the `event type` buffer to field value327        eventType = value || undefined328        break329      case 'data':330        data = dataLines === 0 ? value : `${data}\n${value}`331        dataLines++332        break333      case 'id':334        // If the field value does not contain U+0000 NULL, then set the `ID` buffer to335        // the field value. Otherwise, ignore the field.336        id = value.includes('\0') ? undefined : value337        break338      case 'retry':339        // If the field value consists of only ASCII digits, then interpret the field value as an340        // integer in base ten, and set the event stream's reconnection time to that integer.341        // Otherwise, ignore the field.342        if (/^\d+$/.test(value)) {343          onRetry(parseInt(value, 10))344        } else {345          onError(346            new ParseError(`Invalid \`retry\` value: "${value}"`, {347              type: 'invalid-retry',348              value,349              line,350            }),351          )352        }353        break354      default:355        // Otherwise, the field is ignored.356        onError(357          new ParseError(358            `Unknown field "${field.length > 20 ? `${field.slice(0, 20)}…` : field}"`,359            {type: 'unknown-field', field, value, line},360          ),361        )362        break363    }364  }365 366  function dispatchEvent() {367    if (dataLines > 0) {368      onEvent({369        id,370        event: eventType,371        data,372      })373    }374 375    id = undefined376    data = ''377    dataLines = 0378    eventType = undefined379  }380 381  function reset(options: {consume?: boolean} = {}) {382    if (options.consume && pendingFragments.length > 0) {383      const incompleteLine = pendingFragments.join('')384      parseLine(incompleteLine, 0, incompleteLine.length)385    }386 387    isFirstChunk = true388    id = undefined389    data = ''390    dataLines = 0391    eventType = undefined392    pendingFragments.length = 0393    pendingFragmentsLength = 0394    terminated = false395  }396 397  return {feed, reset}398}399 400/**401 * Checks if `chunk` starts with the literal `data:` at index `i`.402 *403 * Equivalent to `chunk.startsWith('data:', i)`, but benchmarks show this404 * hand-unrolled char-code comparison is ~20% faster on common event types.405 * The caller passes `firstCharCode` (the code at `i`) so it can be reused406 * across prefix checks.407 *408 * ASCII: 'd' = 100, 'a' = 97, 't' = 116, 'a' = 97, ':' = 58409 */410function isDataPrefix(chunk: string, i: number, firstCharCode: number): boolean {411  return (412    firstCharCode === 100 &&413    chunk.charCodeAt(i + 1) === 97 &&414    chunk.charCodeAt(i + 2) === 116 &&415    chunk.charCodeAt(i + 3) === 97 &&416    chunk.charCodeAt(i + 4) === 58417  )418}419 420/**421 * Checks if `chunk` starts with the literal `event:` at index `i`.422 *423 * See {@link isDataPrefix} for why this is hand-unrolled rather than using424 * `String.prototype.startsWith`.425 *426 * ASCII: 'e' = 101, 'v' = 118, 'e' = 101, 'n' = 110, 't' = 116, ':' = 58427 */428function isEventPrefix(chunk: string, i: number, firstCharCode: number): boolean {429  return (430    firstCharCode === 101 &&431    chunk.charCodeAt(i + 1) === 118 &&432    chunk.charCodeAt(i + 2) === 101 &&433    chunk.charCodeAt(i + 3) === 110 &&434    chunk.charCodeAt(i + 4) === 116 &&435    chunk.charCodeAt(i + 5) === 58436  )437}438 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai