Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1/**2 * @import {3 * Code,4 * InitialConstruct,5 * Initializer,6 * Resolver,7 * State,8 * TokenizeContext9 * } from 'micromark-util-types'10 */11 12import {ok as assert} from 'devlop'13import {codes, constants, types} from 'micromark-util-symbol'14 15export const resolver = {resolveAll: createResolver()}16export const string = initializeFactory('string')17export const text = initializeFactory('text')18 19/**20 * @param {'string' | 'text'} field21 * Field.22 * @returns {InitialConstruct}23 * Construct.24 */25function initializeFactory(field) {26 return {27 resolveAll: createResolver(28 field === 'text' ? resolveAllLineSuffixes : undefined29 ),30 tokenize: initializeText31 }32 33 /**34 * @this {TokenizeContext}35 * Context.36 * @type {Initializer}37 */38 function initializeText(effects) {39 const self = this40 const constructs = this.parser.constructs[field]41 const text = effects.attempt(constructs, start, notText)42 43 return start44 45 /** @type {State} */46 function start(code) {47 return atBreak(code) ? text(code) : notText(code)48 }49 50 /** @type {State} */51 function notText(code) {52 if (code === codes.eof) {53 effects.consume(code)54 return55 }56 57 effects.enter(types.data)58 effects.consume(code)59 return data60 }61 62 /** @type {State} */63 function data(code) {64 if (atBreak(code)) {65 effects.exit(types.data)66 return text(code)67 }68 69 // Data.70 effects.consume(code)71 return data72 }73 74 /**75 * @param {Code} code76 * Code.77 * @returns {boolean}78 * Whether the code is a break.79 */80 function atBreak(code) {81 if (code === codes.eof) {82 return true83 }84 85 const list = constructs[code]86 let index = -187 88 if (list) {89 // Always populated by defaults.90 assert(Array.isArray(list), 'expected `disable.null` to be populated')91 92 while (++index < list.length) {93 const item = list[index]94 if (!item.previous || item.previous.call(self, self.previous)) {95 return true96 }97 }98 }99 100 return false101 }102 }103}104 105/**106 * @param {Resolver | undefined} [extraResolver]107 * Resolver.108 * @returns {Resolver}109 * Resolver.110 */111function createResolver(extraResolver) {112 return resolveAllText113 114 /** @type {Resolver} */115 function resolveAllText(events, context) {116 let index = -1117 /** @type {number | undefined} */118 let enter119 120 // A rather boring computation (to merge adjacent `data` events) which121 // improves mm performance by 29%.122 while (++index <= events.length) {123 if (enter === undefined) {124 if (events[index] && events[index][1].type === types.data) {125 enter = index126 index++127 }128 } else if (!events[index] || events[index][1].type !== types.data) {129 // Don’t do anything if there is one data token.130 if (index !== enter + 2) {131 events[enter][1].end = events[index - 1][1].end132 events.splice(enter + 2, index - enter - 2)133 index = enter + 2134 }135 136 enter = undefined137 }138 }139 140 return extraResolver ? extraResolver(events, context) : events141 }142}143 144/**145 * A rather ugly set of instructions which again looks at chunks in the input146 * stream.147 * The reason to do this here is that it is *much* faster to parse in reverse.148 * And that we can’t hook into `null` to split the line suffix before an EOF.149 * To do: figure out if we can make this into a clean utility, or even in core.150 * As it will be useful for GFMs literal autolink extension (and maybe even151 * tables?)152 *153 * @type {Resolver}154 */155function resolveAllLineSuffixes(events, context) {156 let eventIndex = 0 // Skip first.157 158 while (++eventIndex <= events.length) {159 if (160 (eventIndex === events.length ||161 events[eventIndex][1].type === types.lineEnding) &&162 events[eventIndex - 1][1].type === types.data163 ) {164 const data = events[eventIndex - 1][1]165 const chunks = context.sliceStream(data)166 let index = chunks.length167 let bufferIndex = -1168 let size = 0169 /** @type {boolean | undefined} */170 let tabs171 172 while (index--) {173 const chunk = chunks[index]174 175 if (typeof chunk === 'string') {176 bufferIndex = chunk.length177 178 while (chunk.charCodeAt(bufferIndex - 1) === codes.space) {179 size++180 bufferIndex--181 }182 183 if (bufferIndex) break184 bufferIndex = -1185 }186 // Number187 else if (chunk === codes.horizontalTab) {188 tabs = true189 size++190 } else if (chunk === codes.virtualSpace) {191 // Empty192 } else {193 // Replacement character, exit.194 index++195 break196 }197 }198 199 // Allow final trailing whitespace.200 if (context._contentTypeTextTrailing && eventIndex === events.length) {201 size = 0202 }203 204 if (size) {205 const token = {206 type:207 eventIndex === events.length ||208 tabs ||209 size < constants.hardBreakPrefixSizeMin210 ? types.lineSuffix211 : types.hardBreakTrailing,212 start: {213 _bufferIndex: index214 ? bufferIndex215 : data.start._bufferIndex + bufferIndex,216 _index: data.start._index + index,217 line: data.end.line,218 column: data.end.column - size,219 offset: data.end.offset - size220 },221 end: {...data.end}222 }223 224 data.end = {...token.start}225 226 if (data.start.offset === data.end.offset) {227 Object.assign(data, token)228 } else {229 events.splice(230 eventIndex,231 0,232 ['enter', token, context],233 ['exit', token, context]234 )235 eventIndex += 2236 }237 }238 239 eventIndex++240 }241 }242 243 return events244}245 