Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
parse.js616 linesDownload Raw Back to esm
1// this[BUFFER] is the remainder of a chunk if we're waiting for
2// the full 512 bytes of a header to come in.  We will Buffer.concat()
3// it to the next write(), which is a mem copy, but a small one.
4//
5// this[QUEUE] is a list of entries that haven't been emitted
6// yet this can only get filled up if the user keeps write()ing after
7// a write() returns false, or does a write() with more than one entry
8//
9// We don't buffer chunks, we always parse them and either create an
10// entry, or push it into the active entry.  The ReadEntry class knows
11// to throw data away if .ignore=true
12//
13// Shift entry off the buffer when it emits 'end', and emit 'entry' for
14// the next one in the list.
15//
16// At any time, we're pushing body chunks into the entry at WRITEENTRY,
17// and waiting for 'end' on the entry at READENTRY
18//
19// ignored entries get .resume() called on them straight away
20import { EventEmitter as EE } from 'events';
21import { BrotliDecompress, Unzip, ZstdDecompress } from 'minizlib';
22import { Header } from './header.js';
23import { Pax } from './pax.js';
24import { ReadEntry } from './read-entry.js';
25import { warnMethod, } from './warn-method.js';
26const maxMetaEntrySize = 1024 * 1024;
27const gzipHeader = Buffer.from([0x1f, 0x8b]);
28const zstdHeader = Buffer.from([0x28, 0xb5, 0x2f, 0xfd]);
29const ZIP_HEADER_LEN = Math.max(gzipHeader.length, zstdHeader.length);
30const STATE = Symbol('state');
31const WRITEENTRY = Symbol('writeEntry');
32const READENTRY = Symbol('readEntry');
33const NEXTENTRY = Symbol('nextEntry');
34const PROCESSENTRY = Symbol('processEntry');
35const EX = Symbol('extendedHeader');
36const GEX = Symbol('globalExtendedHeader');
37const META = Symbol('meta');
38const EMITMETA = Symbol('emitMeta');
39const BUFFER = Symbol('buffer');
40const QUEUE = Symbol('queue');
41const ENDED = Symbol('ended');
42const EMITTEDEND = Symbol('emittedEnd');
43const EMIT = Symbol('emit');
44const UNZIP = Symbol('unzip');
45const CONSUMECHUNK = Symbol('consumeChunk');
46const CONSUMECHUNKSUB = Symbol('consumeChunkSub');
47const CONSUMEBODY = Symbol('consumeBody');
48const CONSUMEMETA = Symbol('consumeMeta');
49const CONSUMEHEADER = Symbol('consumeHeader');
50const CONSUMING = Symbol('consuming');
51const BUFFERCONCAT = Symbol('bufferConcat');
52const MAYBEEND = Symbol('maybeEnd');
53const WRITING = Symbol('writing');
54const ABORTED = Symbol('aborted');
55const DONE = Symbol('onDone');
56const SAW_VALID_ENTRY = Symbol('sawValidEntry');
57const SAW_NULL_BLOCK = Symbol('sawNullBlock');
58const SAW_EOF = Symbol('sawEOF');
59const CLOSESTREAM = Symbol('closeStream');
60const noop = () => true;
61export class Parser extends EE {
62    file;
63    strict;
64    maxMetaEntrySize;
65    filter;
66    brotli;
67    zstd;
68    writable = true;
69    readable = false;
70    [QUEUE] = [];
71    [BUFFER];
72    [READENTRY];
73    [WRITEENTRY];
74    [STATE] = 'begin';
75    [META] = '';
76    [EX];
77    [GEX];
78    [ENDED] = false;
79    [UNZIP];
80    [ABORTED] = false;
81    [SAW_VALID_ENTRY];
82    [SAW_NULL_BLOCK] = false;
83    [SAW_EOF] = false;
84    [WRITING] = false;
85    [CONSUMING] = false;
86    [EMITTEDEND] = false;
87    constructor(opt = {}) {
88        super();
89        this.file = opt.file || '';
90        // these BADARCHIVE errors can't be detected early. listen on DONE.
91        this.on(DONE, () => {
92            if (this[STATE] === 'begin' ||
93                this[SAW_VALID_ENTRY] === false) {
94                // either less than 1 block of data, or all entries were invalid.
95                // Either way, probably not even a tarball.
96                this.warn('TAR_BAD_ARCHIVE', 'Unrecognized archive format');
97            }
98        });
99        if (opt.ondone) {
100            this.on(DONE, opt.ondone);
101        }
102        else {
103            this.on(DONE, () => {
104                this.emit('prefinish');
105                this.emit('finish');
106                this.emit('end');
107            });
108        }
109        this.strict = !!opt.strict;
110        this.maxMetaEntrySize = opt.maxMetaEntrySize || maxMetaEntrySize;
111        this.filter = typeof opt.filter === 'function' ? opt.filter : noop;
112        // Unlike gzip, brotli doesn't have any magic bytes to identify it
113        // Users need to explicitly tell us they're extracting a brotli file
114        // Or we infer from the file extension
115        const isTBR = opt.file &&
116            (opt.file.endsWith('.tar.br') || opt.file.endsWith('.tbr'));
117        // if it's a tbr file it MIGHT be brotli, but we don't know until
118        // we look at it and verify it's not a valid tar file.
119        this.brotli =
120            !(opt.gzip || opt.zstd) && opt.brotli !== undefined ? opt.brotli
121                : isTBR ? undefined
122                    : false;
123        // zstd has magic bytes to identify it, but we also support explicit options
124        // and file extension detection
125        const isTZST = opt.file &&
126            (opt.file.endsWith('.tar.zst') || opt.file.endsWith('.tzst'));
127        this.zstd =
128            !(opt.gzip || opt.brotli) && opt.zstd !== undefined ? opt.zstd
129                : isTZST ? true
130                    : undefined;
131        // have to set this so that streams are ok piping into it
132        this.on('end', () => this[CLOSESTREAM]());
133        if (typeof opt.onwarn === 'function') {
134            this.on('warn', opt.onwarn);
135        }
136        if (typeof opt.onReadEntry === 'function') {
137            this.on('entry', opt.onReadEntry);
138        }
139    }
140    warn(code, message, data = {}) {
141        warnMethod(this, code, message, data);
142    }
143    [CONSUMEHEADER](chunk, position) {
144        if (this[SAW_VALID_ENTRY] === undefined) {
145            this[SAW_VALID_ENTRY] = false;
146        }
147        let header;
148        try {
149            header = new Header(chunk, position, this[EX], this[GEX]);
150        }
151        catch (er) {
152            return this.warn('TAR_ENTRY_INVALID', er);
153        }
154        if (header.nullBlock) {
155            if (this[SAW_NULL_BLOCK]) {
156                this[SAW_EOF] = true;
157                // ending an archive with no entries.  pointless, but legal.
158                if (this[STATE] === 'begin') {
159                    this[STATE] = 'header';
160                }
161                this[EMIT]('eof');
162            }
163            else {
164                this[SAW_NULL_BLOCK] = true;
165                this[EMIT]('nullBlock');
166            }
167        }
168        else {
169            this[SAW_NULL_BLOCK] = false;
170            if (!header.cksumValid) {
171                this.warn('TAR_ENTRY_INVALID', 'checksum failure', { header });
172            }
173            else if (!header.path) {
174                this.warn('TAR_ENTRY_INVALID', 'path is required', { header });
175            }
176            else {
177                const type = header.type;
178                if (/^(Symbolic)?Link$/.test(type) && !header.linkpath) {
179                    this.warn('TAR_ENTRY_INVALID', 'linkpath required', {
180                        header,
181                    });
182                }
183                else if (!/^(Symbolic)?Link$/.test(type) &&
184                    !/^(Global)?ExtendedHeader$/.test(type) &&
185                    header.linkpath) {
186                    this.warn('TAR_ENTRY_INVALID', 'linkpath forbidden', {
187                        header,
188                    });
189                }
190                else {
191                    const entry = (this[WRITEENTRY] = new ReadEntry(header, this[EX], this[GEX]));
192                    // we do this for meta & ignored entries as well, because they
193                    // are still valid tar, or else we wouldn't know to ignore them
194                    if (!this[SAW_VALID_ENTRY]) {
195                        if (entry.remain) {
196                            // this might be the one!
197                            const onend = () => {
198                                if (!entry.invalid) {
199                                    this[SAW_VALID_ENTRY] = true;
200                                }
201                            };
202                            entry.on('end', onend);
203                        }
204                        else {
205                            this[SAW_VALID_ENTRY] = true;
206                        }
207                    }
208                    if (entry.meta) {
209                        if (entry.size > this.maxMetaEntrySize) {
210                            entry.ignore = true;
211                            this[EMIT]('ignoredEntry', entry);
212                            this[STATE] = 'ignore';
213                            entry.resume();
214                        }
215                        else if (entry.size > 0) {
216                            this[META] = '';
217                            entry.on('data', c => (this[META] += c));
218                            this[STATE] = 'meta';
219                        }
220                    }
221                    else {
222                        this[EX] = undefined;
223                        entry.ignore =
224                            entry.ignore || !this.filter(entry.path, entry);
225                        if (entry.ignore) {
226                            // probably valid, just not something we care about
227                            this[EMIT]('ignoredEntry', entry);
228                            this[STATE] = entry.remain ? 'ignore' : 'header';
229                            entry.resume();
230                        }
231                        else {
232                            if (entry.remain) {
233                                this[STATE] = 'body';
234                            }
235                            else {
236                                this[STATE] = 'header';
237                                entry.end();
238                            }
239                            if (!this[READENTRY]) {
240                                this[QUEUE].push(entry);
241                                this[NEXTENTRY]();
242                            }
243                            else {
244                                this[QUEUE].push(entry);
245                            }
246                        }
247                    }
248                }
249            }
250        }
251    }
252    [CLOSESTREAM]() {
253        queueMicrotask(() => this.emit('close'));
254    }
255    [PROCESSENTRY](entry) {
256        let go = true;
257        if (!entry) {
258            this[READENTRY] = undefined;
259            go = false;
260        }
261        else if (Array.isArray(entry)) {
262            const [ev, ...args] = entry;
263            this.emit(ev, ...args);
264        }
265        else {
266            this[READENTRY] = entry;
267            this.emit('entry', entry);
268            if (!entry.emittedEnd) {
269                entry.on('end', () => this[NEXTENTRY]());
270                go = false;
271            }
272        }
273        return go;
274    }
275    [NEXTENTRY]() {
276        do { } while (this[PROCESSENTRY](this[QUEUE].shift()));
277        if (!this[QUEUE].length) {
278            // At this point, there's nothing in the queue, but we may have an
279            // entry which is being consumed (readEntry).
280            // If we don't, then we definitely can handle more data.
281            // If we do, and either it's flowing, or it has never had any data
282            // written to it, then it needs more.
283            // The only other possibility is that it has returned false from a
284            // write() call, so we wait for the next drain to continue.
285            const re = this[READENTRY];
286            const drainNow = !re || re.flowing || re.size === re.remain;
287            if (drainNow) {
288                if (!this[WRITING]) {
289                    this.emit('drain');
290                }
291            }
292            else {
293                re.once('drain', () => this.emit('drain'));
294            }
295        }
296    }
297    [CONSUMEBODY](chunk, position) {
298        // write up to but no  more than writeEntry.blockRemain
299        const entry = this[WRITEENTRY];
300        /* c8 ignore start */
301        if (!entry) {
302            throw new Error('attempt to consume body without entry??');
303        }
304        const br = entry.blockRemain ?? 0;
305        /* c8 ignore stop */
306        const c = br >= chunk.length && position === 0 ?
307            chunk
308            : chunk.subarray(position, position + br);
309        entry.write(c);
310        if (!entry.blockRemain) {
311            this[STATE] = 'header';
312            this[WRITEENTRY] = undefined;
313            entry.end();
314        }
315        return c.length;
316    }
317    [CONSUMEMETA](chunk, position) {
318        const entry = this[WRITEENTRY];
319        const ret = this[CONSUMEBODY](chunk, position);
320        // if we finished, then the entry is reset
321        if (!this[WRITEENTRY] && entry) {
322            this[EMITMETA](entry);
323        }
324        return ret;
325    }
326    [EMIT](ev, data, extra) {
327        if (!this[QUEUE].length && !this[READENTRY]) {
328            this.emit(ev, data, extra);
329        }
330        else {
331            this[QUEUE].push([ev, data, extra]);
332        }
333    }
334    [EMITMETA](entry) {
335        this[EMIT]('meta', this[META]);
336        switch (entry.type) {
337            case 'ExtendedHeader':
338            case 'OldExtendedHeader':
339                this[EX] = Pax.parse(this[META], this[EX], false);
340                break;
341            case 'GlobalExtendedHeader':
342                this[GEX] = Pax.parse(this[META], this[GEX], true);
343                break;
344            case 'NextFileHasLongPath':
345            case 'OldGnuLongPath': {
346                const ex = this[EX] ?? Object.create(null);
347                this[EX] = ex;
348                ex.path = this[META].replace(/\0.*/, '');
349                break;
350            }
351            case 'NextFileHasLongLinkpath': {
352                const ex = this[EX] || Object.create(null);
353                this[EX] = ex;
354                ex.linkpath = this[META].replace(/\0.*/, '');
355                break;
356            }
357            /* c8 ignore start */
358            default:
359                throw new Error('unknown meta: ' + entry.type);
360            /* c8 ignore stop */
361        }
362    }
363    abort(error) {
364        this[ABORTED] = true;
365        this.emit('abort', error);
366        // always throws, even in non-strict mode
367        this.warn('TAR_ABORT', error, { recoverable: false });
368    }
369    write(chunk, encoding, cb) {
370        if (typeof encoding === 'function') {
371            cb = encoding;
372            encoding = undefined;
373        }
374        if (typeof chunk === 'string') {
375            chunk = Buffer.from(chunk,
376            /* c8 ignore next */
377            typeof encoding === 'string' ? encoding : 'utf8');
378        }
379        if (this[ABORTED]) {
380            /* c8 ignore next */
381            cb?.();
382            return false;
383        }
384        // first write, might be gzipped, zstd, or brotli compressed
385        const needSniff = this[UNZIP] === undefined ||
386            (this.brotli === undefined && this[UNZIP] === false);
387        if (needSniff && chunk) {
388            if (this[BUFFER]) {
389                chunk = Buffer.concat([this[BUFFER], chunk]);
390                this[BUFFER] = undefined;
391            }
392            if (chunk.length < ZIP_HEADER_LEN) {
393                this[BUFFER] = chunk;
394                /* c8 ignore next */
395                cb?.();
396                return true;
397            }
398            // look for gzip header
399            for (let i = 0; this[UNZIP] === undefined && i < gzipHeader.length; i++) {
400                if (chunk[i] !== gzipHeader[i]) {
401                    this[UNZIP] = false;
402                }
403            }
404            // look for zstd header if gzip header not found
405            let isZstd = false;
406            if (this[UNZIP] === false && this.zstd !== false) {
407                isZstd = true;
408                for (let i = 0; i < zstdHeader.length; i++) {
409                    if (chunk[i] !== zstdHeader[i]) {
410                        isZstd = false;
411                        break;
412                    }
413                }
414            }
415            const maybeBrotli = this.brotli === undefined && !isZstd;
416            if (this[UNZIP] === false && maybeBrotli) {
417                // read the first header to see if it's a valid tar file. If so,
418                // we can safely assume that it's not actually brotli, despite the
419                // .tbr or .tar.br file extension.
420                // if we ended before getting a full chunk, yes, def brotli
421                if (chunk.length < 512) {
422                    if (this[ENDED]) {
423                        this.brotli = true;
424                    }
425                    else {
426                        this[BUFFER] = chunk;
427                        /* c8 ignore next */
428                        cb?.();
429                        return true;
430                    }
431                }
432                else {
433                    // if it's tar, it's pretty reliably not brotli, chances of
434                    // that happening are astronomical.
435                    try {
436                        new Header(chunk.subarray(0, 512));
437                        this.brotli = false;
438                    }
439                    catch (_) {
440                        this.brotli = true;
441                    }
442                }
443            }
444            if (this[UNZIP] === undefined ||
445                (this[UNZIP] === false && (this.brotli || isZstd))) {
446                const ended = this[ENDED];
447                this[ENDED] = false;
448                this[UNZIP] =
449                    this[UNZIP] === undefined ? new Unzip({})
450                        : isZstd ? new ZstdDecompress({})
451                            : new BrotliDecompress({});
452                this[UNZIP].on('data', chunk => this[CONSUMECHUNK](chunk));
453                this[UNZIP].on('error', er => this.abort(er));
454                this[UNZIP].on('end', () => {
455                    this[ENDED] = true;
456                    this[CONSUMECHUNK]();
457                });
458                this[WRITING] = true;
459                const ret = !!this[UNZIP][ended ? 'end' : 'write'](chunk);
460                this[WRITING] = false;
461                cb?.();
462                return ret;
463            }
464        }
465        this[WRITING] = true;
466        if (this[UNZIP]) {
467            this[UNZIP].write(chunk);
468        }
469        else {
470            this[CONSUMECHUNK](chunk);
471        }
472        this[WRITING] = false;
473        // return false if there's a queue, or if the current entry isn't flowing
474        const ret = this[QUEUE].length ? false
475            : this[READENTRY] ? this[READENTRY].flowing
476                : true;
477        // if we have no queue, then that means a clogged READENTRY
478        if (!ret && !this[QUEUE].length) {
479            this[READENTRY]?.once('drain', () => this.emit('drain'));
480        }
481        /* c8 ignore next */
482        cb?.();
483        return ret;
484    }
485    [BUFFERCONCAT](c) {
486        if (c && !this[ABORTED]) {
487            this[BUFFER] =
488                this[BUFFER] ? Buffer.concat([this[BUFFER], c]) : c;
489        }
490    }
491    [MAYBEEND]() {
492        if (this[ENDED] &&
493            !this[EMITTEDEND] &&
494            !this[ABORTED] &&
495            !this[CONSUMING]) {
496            this[EMITTEDEND] = true;
497            const entry = this[WRITEENTRY];
498            if (entry && entry.blockRemain) {
499                // truncated, likely a damaged file
500                const have = this[BUFFER] ? this[BUFFER].length : 0;
501                this.warn('TAR_BAD_ARCHIVE', `Truncated input (needed ${entry.blockRemain} more bytes, only ${have} available)`, { entry });
502                if (this[BUFFER]) {
503                    entry.write(this[BUFFER]);
504                }
505                entry.end();
506            }
507            this[EMIT](DONE);
508        }
509    }
510    [CONSUMECHUNK](chunk) {
511        if (this[CONSUMING] && chunk) {
512            this[BUFFERCONCAT](chunk);
513        }
514        else if (!chunk && !this[BUFFER]) {
515            this[MAYBEEND]();
516        }
517        else if (chunk) {
518            this[CONSUMING] = true;
519            if (this[BUFFER]) {
520                this[BUFFERCONCAT](chunk);
521                const c = this[BUFFER];
522                this[BUFFER] = undefined;
523                this[CONSUMECHUNKSUB](c);
524            }
525            else {
526                this[CONSUMECHUNKSUB](chunk);
527            }
528            while (this[BUFFER] &&
529                this[BUFFER]?.length >= 512 &&
530                !this[ABORTED] &&
531                !this[SAW_EOF]) {
532                const c = this[BUFFER];
533                this[BUFFER] = undefined;
534                this[CONSUMECHUNKSUB](c);
535            }
536            this[CONSUMING] = false;
537        }
538        if (!this[BUFFER] || this[ENDED]) {
539            this[MAYBEEND]();
540        }
541    }
542    [CONSUMECHUNKSUB](chunk) {
543        // we know that we are in CONSUMING mode, so anything written goes into
544        // the buffer.  Advance the position and put any remainder in the buffer.
545        let position = 0;
546        const length = chunk.length;
547        while (position + 512 <= length &&
548            !this[ABORTED] &&
549            !this[SAW_EOF]) {
550            switch (this[STATE]) {
551                case 'begin':
552                case 'header':
553                    this[CONSUMEHEADER](chunk, position);
554                    position += 512;
555                    break;
556                case 'ignore':
557                case 'body':
558                    position += this[CONSUMEBODY](chunk, position);
559                    break;
560                case 'meta':
561                    position += this[CONSUMEMETA](chunk, position);
562                    break;
563                /* c8 ignore start */
564                default:
565                    throw new Error('invalid state: ' + this[STATE]);
566                /* c8 ignore stop */
567            }
568        }
569        if (position < length) {
570            if (this[BUFFER]) {
571                this[BUFFER] = Buffer.concat([
572                    chunk.subarray(position),
573                    this[BUFFER],
574                ]);
575            }
576            else {
577                this[BUFFER] = chunk.subarray(position);
578            }
579        }
580    }
581    end(chunk, encoding, cb) {
582        if (typeof chunk === 'function') {
583            cb = chunk;
584            encoding = undefined;
585            chunk = undefined;
586        }
587        if (typeof encoding === 'function') {
588            cb = encoding;
589            encoding = undefined;
590        }
591        if (typeof chunk === 'string') {
592            chunk = Buffer.from(chunk, encoding);
593        }
594        if (cb)
595            this.once('finish', cb);
596        if (!this[ABORTED]) {
597            if (this[UNZIP]) {
598                /* c8 ignore start */
599                if (chunk)
600                    this[UNZIP].write(chunk);
601                /* c8 ignore stop */
602                this[UNZIP].end();
603            }
604            else {
605                this[ENDED] = true;
606                if (this.brotli === undefined || this.zstd === undefined)
607                    chunk = chunk || Buffer.alloc(0);
608                if (chunk)
609                    this.write(chunk);
610                this[MAYBEEND]();
611            }
612        }
613        return this;
614    }
615}
616//# sourceMappingURL=parse.js.map
codekingpro/portable-devtools · Team Ai