Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
text.js245 linesDownload Raw Back to initialize
1/**2 * @import {3 *   Code,4 *   InitialConstruct,5 *   Initializer,6 *   Resolver,7 *   State,8 *   TokenizeContext9 * } from 'micromark-util-types'10 */11 12import {ok as assert} from 'devlop'13import {codes, constants, types} from 'micromark-util-symbol'14 15export const resolver = {resolveAll: createResolver()}16export const string = initializeFactory('string')17export const text = initializeFactory('text')18 19/**20 * @param {'string' | 'text'} field21 *   Field.22 * @returns {InitialConstruct}23 *   Construct.24 */25function initializeFactory(field) {26  return {27    resolveAll: createResolver(28      field === 'text' ? resolveAllLineSuffixes : undefined29    ),30    tokenize: initializeText31  }32 33  /**34   * @this {TokenizeContext}35   *   Context.36   * @type {Initializer}37   */38  function initializeText(effects) {39    const self = this40    const constructs = this.parser.constructs[field]41    const text = effects.attempt(constructs, start, notText)42 43    return start44 45    /** @type {State} */46    function start(code) {47      return atBreak(code) ? text(code) : notText(code)48    }49 50    /** @type {State} */51    function notText(code) {52      if (code === codes.eof) {53        effects.consume(code)54        return55      }56 57      effects.enter(types.data)58      effects.consume(code)59      return data60    }61 62    /** @type {State} */63    function data(code) {64      if (atBreak(code)) {65        effects.exit(types.data)66        return text(code)67      }68 69      // Data.70      effects.consume(code)71      return data72    }73 74    /**75     * @param {Code} code76     *   Code.77     * @returns {boolean}78     *   Whether the code is a break.79     */80    function atBreak(code) {81      if (code === codes.eof) {82        return true83      }84 85      const list = constructs[code]86      let index = -187 88      if (list) {89        // Always populated by defaults.90        assert(Array.isArray(list), 'expected `disable.null` to be populated')91 92        while (++index < list.length) {93          const item = list[index]94          if (!item.previous || item.previous.call(self, self.previous)) {95            return true96          }97        }98      }99 100      return false101    }102  }103}104 105/**106 * @param {Resolver | undefined} [extraResolver]107 *   Resolver.108 * @returns {Resolver}109 *   Resolver.110 */111function createResolver(extraResolver) {112  return resolveAllText113 114  /** @type {Resolver} */115  function resolveAllText(events, context) {116    let index = -1117    /** @type {number | undefined} */118    let enter119 120    // A rather boring computation (to merge adjacent `data` events) which121    // improves mm performance by 29%.122    while (++index <= events.length) {123      if (enter === undefined) {124        if (events[index] && events[index][1].type === types.data) {125          enter = index126          index++127        }128      } else if (!events[index] || events[index][1].type !== types.data) {129        // Don’t do anything if there is one data token.130        if (index !== enter + 2) {131          events[enter][1].end = events[index - 1][1].end132          events.splice(enter + 2, index - enter - 2)133          index = enter + 2134        }135 136        enter = undefined137      }138    }139 140    return extraResolver ? extraResolver(events, context) : events141  }142}143 144/**145 * A rather ugly set of instructions which again looks at chunks in the input146 * stream.147 * The reason to do this here is that it is *much* faster to parse in reverse.148 * And that we can’t hook into `null` to split the line suffix before an EOF.149 * To do: figure out if we can make this into a clean utility, or even in core.150 * As it will be useful for GFMs literal autolink extension (and maybe even151 * tables?)152 *153 * @type {Resolver}154 */155function resolveAllLineSuffixes(events, context) {156  let eventIndex = 0 // Skip first.157 158  while (++eventIndex <= events.length) {159    if (160      (eventIndex === events.length ||161        events[eventIndex][1].type === types.lineEnding) &&162      events[eventIndex - 1][1].type === types.data163    ) {164      const data = events[eventIndex - 1][1]165      const chunks = context.sliceStream(data)166      let index = chunks.length167      let bufferIndex = -1168      let size = 0169      /** @type {boolean | undefined} */170      let tabs171 172      while (index--) {173        const chunk = chunks[index]174 175        if (typeof chunk === 'string') {176          bufferIndex = chunk.length177 178          while (chunk.charCodeAt(bufferIndex - 1) === codes.space) {179            size++180            bufferIndex--181          }182 183          if (bufferIndex) break184          bufferIndex = -1185        }186        // Number187        else if (chunk === codes.horizontalTab) {188          tabs = true189          size++190        } else if (chunk === codes.virtualSpace) {191          // Empty192        } else {193          // Replacement character, exit.194          index++195          break196        }197      }198 199      // Allow final trailing whitespace.200      if (context._contentTypeTextTrailing && eventIndex === events.length) {201        size = 0202      }203 204      if (size) {205        const token = {206          type:207            eventIndex === events.length ||208            tabs ||209            size < constants.hardBreakPrefixSizeMin210              ? types.lineSuffix211              : types.hardBreakTrailing,212          start: {213            _bufferIndex: index214              ? bufferIndex215              : data.start._bufferIndex + bufferIndex,216            _index: data.start._index + index,217            line: data.end.line,218            column: data.end.column - size,219            offset: data.end.offset - size220          },221          end: {...data.end}222        }223 224        data.end = {...token.start}225 226        if (data.start.offset === data.end.offset) {227          Object.assign(data, token)228        } else {229          events.splice(230            eventIndex,231            0,232            ['enter', token, context],233            ['exit', token, context]234          )235          eventIndex += 2236        }237      }238 239      eventIndex++240    }241  }242 243  return events244}245 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai