codekingpro/portable-devtools
115k
1// this[BUFFER] is the remainder of a chunk if we're waiting for
2// the full 512 bytes of a header to come in. We will Buffer.concat()
3// it to the next write(), which is a mem copy, but a small one.
4//
5// this[QUEUE] is a list of entries that haven't been emitted
6// yet this can only get filled up if the user keeps write()ing after
7// a write() returns false, or does a write() with more than one entry
8//
9// We don't buffer chunks, we always parse them and either create an
10// entry, or push it into the active entry. The ReadEntry class knows
11// to throw data away if .ignore=true
12//
13// Shift entry off the buffer when it emits 'end', and emit 'entry' for
14// the next one in the list.
15//
16// At any time, we're pushing body chunks into the entry at WRITEENTRY,
17// and waiting for 'end' on the entry at READENTRY
18//
19// ignored entries get .resume() called on them straight away
20import { EventEmitter as EE } from 'events';
21import { BrotliDecompress, Unzip, ZstdDecompress } from 'minizlib';
22import { Header } from './header.js';
23import { Pax } from './pax.js';
24import { ReadEntry } from './read-entry.js';
25import { warnMethod, } from './warn-method.js';
26const maxMetaEntrySize = 1024 * 1024;
27const gzipHeader = Buffer.from([0x1f, 0x8b]);
28const zstdHeader = Buffer.from([0x28, 0xb5, 0x2f, 0xfd]);
29const ZIP_HEADER_LEN = Math.max(gzipHeader.length, zstdHeader.length);
30const STATE = Symbol('state');
31const WRITEENTRY = Symbol('writeEntry');
32const READENTRY = Symbol('readEntry');
33const NEXTENTRY = Symbol('nextEntry');
34const PROCESSENTRY = Symbol('processEntry');
35const EX = Symbol('extendedHeader');
36const GEX = Symbol('globalExtendedHeader');
37const META = Symbol('meta');
38const EMITMETA = Symbol('emitMeta');
39const BUFFER = Symbol('buffer');
40const QUEUE = Symbol('queue');
41const ENDED = Symbol('ended');
42const EMITTEDEND = Symbol('emittedEnd');
43const EMIT = Symbol('emit');
44const UNZIP = Symbol('unzip');
45const CONSUMECHUNK = Symbol('consumeChunk');
46const CONSUMECHUNKSUB = Symbol('consumeChunkSub');
47const CONSUMEBODY = Symbol('consumeBody');
48const CONSUMEMETA = Symbol('consumeMeta');
49const CONSUMEHEADER = Symbol('consumeHeader');
50const CONSUMING = Symbol('consuming');
51const BUFFERCONCAT = Symbol('bufferConcat');
52const MAYBEEND = Symbol('maybeEnd');
53const WRITING = Symbol('writing');
54const ABORTED = Symbol('aborted');
55const DONE = Symbol('onDone');
56const SAW_VALID_ENTRY = Symbol('sawValidEntry');
57const SAW_NULL_BLOCK = Symbol('sawNullBlock');
58const SAW_EOF = Symbol('sawEOF');
59const CLOSESTREAM = Symbol('closeStream');
60const noop = () => true;
61export class Parser extends EE {
62 file;
63 strict;
64 maxMetaEntrySize;
65 filter;
66 brotli;
67 zstd;
68 writable = true;
69 readable = false;
70 [QUEUE] = [];
71 [BUFFER];
72 [READENTRY];
73 [WRITEENTRY];
74 [STATE] = 'begin';
75 [META] = '';
76 [EX];
77 [GEX];
78 [ENDED] = false;
79 [UNZIP];
80 [ABORTED] = false;
81 [SAW_VALID_ENTRY];
82 [SAW_NULL_BLOCK] = false;
83 [SAW_EOF] = false;
84 [WRITING] = false;
85 [CONSUMING] = false;
86 [EMITTEDEND] = false;
87 constructor(opt = {}) {
88 super();
89 this.file = opt.file || '';
90 // these BADARCHIVE errors can't be detected early. listen on DONE.
91 this.on(DONE, () => {
92 if (this[STATE] === 'begin' ||
93 this[SAW_VALID_ENTRY] === false) {
94 // either less than 1 block of data, or all entries were invalid.
95 // Either way, probably not even a tarball.
96 this.warn('TAR_BAD_ARCHIVE', 'Unrecognized archive format');
97 }
98 });
99 if (opt.ondone) {
100 this.on(DONE, opt.ondone);
101 }
102 else {
103 this.on(DONE, () => {
104 this.emit('prefinish');
105 this.emit('finish');
106 this.emit('end');
107 });
108 }
109 this.strict = !!opt.strict;
110 this.maxMetaEntrySize = opt.maxMetaEntrySize || maxMetaEntrySize;
111 this.filter = typeof opt.filter === 'function' ? opt.filter : noop;
112 // Unlike gzip, brotli doesn't have any magic bytes to identify it
113 // Users need to explicitly tell us they're extracting a brotli file
114 // Or we infer from the file extension
115 const isTBR = opt.file &&
116 (opt.file.endsWith('.tar.br') || opt.file.endsWith('.tbr'));
117 // if it's a tbr file it MIGHT be brotli, but we don't know until
118 // we look at it and verify it's not a valid tar file.
119 this.brotli =
120 !(opt.gzip || opt.zstd) && opt.brotli !== undefined ? opt.brotli
121 : isTBR ? undefined
122 : false;
123 // zstd has magic bytes to identify it, but we also support explicit options
124 // and file extension detection
125 const isTZST = opt.file &&
126 (opt.file.endsWith('.tar.zst') || opt.file.endsWith('.tzst'));
127 this.zstd =
128 !(opt.gzip || opt.brotli) && opt.zstd !== undefined ? opt.zstd
129 : isTZST ? true
130 : undefined;
131 // have to set this so that streams are ok piping into it
132 this.on('end', () => this[CLOSESTREAM]());
133 if (typeof opt.onwarn === 'function') {
134 this.on('warn', opt.onwarn);
135 }
136 if (typeof opt.onReadEntry === 'function') {
137 this.on('entry', opt.onReadEntry);
138 }
139 }
140 warn(code, message, data = {}) {
141 warnMethod(this, code, message, data);
142 }
143 [CONSUMEHEADER](chunk, position) {
144 if (this[SAW_VALID_ENTRY] === undefined) {
145 this[SAW_VALID_ENTRY] = false;
146 }
147 let header;
148 try {
149 header = new Header(chunk, position, this[EX], this[GEX]);
150 }
151 catch (er) {
152 return this.warn('TAR_ENTRY_INVALID', er);
153 }
154 if (header.nullBlock) {
155 if (this[SAW_NULL_BLOCK]) {
156 this[SAW_EOF] = true;
157 // ending an archive with no entries. pointless, but legal.
158 if (this[STATE] === 'begin') {
159 this[STATE] = 'header';
160 }
161 this[EMIT]('eof');
162 }
163 else {
164 this[SAW_NULL_BLOCK] = true;
165 this[EMIT]('nullBlock');
166 }
167 }
168 else {
169 this[SAW_NULL_BLOCK] = false;
170 if (!header.cksumValid) {
171 this.warn('TAR_ENTRY_INVALID', 'checksum failure', { header });
172 }
173 else if (!header.path) {
174 this.warn('TAR_ENTRY_INVALID', 'path is required', { header });
175 }
176 else {
177 const type = header.type;
178 if (/^(Symbolic)?Link$/.test(type) && !header.linkpath) {
179 this.warn('TAR_ENTRY_INVALID', 'linkpath required', {
180 header,
181 });
182 }
183 else if (!/^(Symbolic)?Link$/.test(type) &&
184 !/^(Global)?ExtendedHeader$/.test(type) &&
185 header.linkpath) {
186 this.warn('TAR_ENTRY_INVALID', 'linkpath forbidden', {
187 header,
188 });
189 }
190 else {
191 const entry = (this[WRITEENTRY] = new ReadEntry(header, this[EX], this[GEX]));
192 // we do this for meta & ignored entries as well, because they
193 // are still valid tar, or else we wouldn't know to ignore them
194 if (!this[SAW_VALID_ENTRY]) {
195 if (entry.remain) {
196 // this might be the one!
197 const onend = () => {
198 if (!entry.invalid) {
199 this[SAW_VALID_ENTRY] = true;
200 }
201 };
202 entry.on('end', onend);
203 }
204 else {
205 this[SAW_VALID_ENTRY] = true;
206 }
207 }
208 if (entry.meta) {
209 if (entry.size > this.maxMetaEntrySize) {
210 entry.ignore = true;
211 this[EMIT]('ignoredEntry', entry);
212 this[STATE] = 'ignore';
213 entry.resume();
214 }
215 else if (entry.size > 0) {
216 this[META] = '';
217 entry.on('data', c => (this[META] += c));
218 this[STATE] = 'meta';
219 }
220 }
221 else {
222 this[EX] = undefined;
223 entry.ignore =
224 entry.ignore || !this.filter(entry.path, entry);
225 if (entry.ignore) {
226 // probably valid, just not something we care about
227 this[EMIT]('ignoredEntry', entry);
228 this[STATE] = entry.remain ? 'ignore' : 'header';
229 entry.resume();
230 }
231 else {
232 if (entry.remain) {
233 this[STATE] = 'body';
234 }
235 else {
236 this[STATE] = 'header';
237 entry.end();
238 }
239 if (!this[READENTRY]) {
240 this[QUEUE].push(entry);
241 this[NEXTENTRY]();
242 }
243 else {
244 this[QUEUE].push(entry);
245 }
246 }
247 }
248 }
249 }
250 }
251 }
252 [CLOSESTREAM]() {
253 queueMicrotask(() => this.emit('close'));
254 }
255 [PROCESSENTRY](entry) {
256 let go = true;
257 if (!entry) {
258 this[READENTRY] = undefined;
259 go = false;
260 }
261 else if (Array.isArray(entry)) {
262 const [ev, ...args] = entry;
263 this.emit(ev, ...args);
264 }
265 else {
266 this[READENTRY] = entry;
267 this.emit('entry', entry);
268 if (!entry.emittedEnd) {
269 entry.on('end', () => this[NEXTENTRY]());
270 go = false;
271 }
272 }
273 return go;
274 }
275 [NEXTENTRY]() {
276 do { } while (this[PROCESSENTRY](this[QUEUE].shift()));
277 if (!this[QUEUE].length) {
278 // At this point, there's nothing in the queue, but we may have an
279 // entry which is being consumed (readEntry).
280 // If we don't, then we definitely can handle more data.
281 // If we do, and either it's flowing, or it has never had any data
282 // written to it, then it needs more.
283 // The only other possibility is that it has returned false from a
284 // write() call, so we wait for the next drain to continue.
285 const re = this[READENTRY];
286 const drainNow = !re || re.flowing || re.size === re.remain;
287 if (drainNow) {
288 if (!this[WRITING]) {
289 this.emit('drain');
290 }
291 }
292 else {
293 re.once('drain', () => this.emit('drain'));
294 }
295 }
296 }
297 [CONSUMEBODY](chunk, position) {
298 // write up to but no more than writeEntry.blockRemain
299 const entry = this[WRITEENTRY];
300 /* c8 ignore start */
301 if (!entry) {
302 throw new Error('attempt to consume body without entry??');
303 }
304 const br = entry.blockRemain ?? 0;
305 /* c8 ignore stop */
306 const c = br >= chunk.length && position === 0 ?
307 chunk
308 : chunk.subarray(position, position + br);
309 entry.write(c);
310 if (!entry.blockRemain) {
311 this[STATE] = 'header';
312 this[WRITEENTRY] = undefined;
313 entry.end();
314 }
315 return c.length;
316 }
317 [CONSUMEMETA](chunk, position) {
318 const entry = this[WRITEENTRY];
319 const ret = this[CONSUMEBODY](chunk, position);
320 // if we finished, then the entry is reset
321 if (!this[WRITEENTRY] && entry) {
322 this[EMITMETA](entry);
323 }
324 return ret;
325 }
326 [EMIT](ev, data, extra) {
327 if (!this[QUEUE].length && !this[READENTRY]) {
328 this.emit(ev, data, extra);
329 }
330 else {
331 this[QUEUE].push([ev, data, extra]);
332 }
333 }
334 [EMITMETA](entry) {
335 this[EMIT]('meta', this[META]);
336 switch (entry.type) {
337 case 'ExtendedHeader':
338 case 'OldExtendedHeader':
339 this[EX] = Pax.parse(this[META], this[EX], false);
340 break;
341 case 'GlobalExtendedHeader':
342 this[GEX] = Pax.parse(this[META], this[GEX], true);
343 break;
344 case 'NextFileHasLongPath':
345 case 'OldGnuLongPath': {
346 const ex = this[EX] ?? Object.create(null);
347 this[EX] = ex;
348 ex.path = this[META].replace(/\0.*/, '');
349 break;
350 }
351 case 'NextFileHasLongLinkpath': {
352 const ex = this[EX] || Object.create(null);
353 this[EX] = ex;
354 ex.linkpath = this[META].replace(/\0.*/, '');
355 break;
356 }
357 /* c8 ignore start */
358 default:
359 throw new Error('unknown meta: ' + entry.type);
360 /* c8 ignore stop */
361 }
362 }
363 abort(error) {
364 this[ABORTED] = true;
365 this.emit('abort', error);
366 // always throws, even in non-strict mode
367 this.warn('TAR_ABORT', error, { recoverable: false });
368 }
369 write(chunk, encoding, cb) {
370 if (typeof encoding === 'function') {
371 cb = encoding;
372 encoding = undefined;
373 }
374 if (typeof chunk === 'string') {
375 chunk = Buffer.from(chunk,
376 /* c8 ignore next */
377 typeof encoding === 'string' ? encoding : 'utf8');
378 }
379 if (this[ABORTED]) {
380 /* c8 ignore next */
381 cb?.();
382 return false;
383 }
384 // first write, might be gzipped, zstd, or brotli compressed
385 const needSniff = this[UNZIP] === undefined ||
386 (this.brotli === undefined && this[UNZIP] === false);
387 if (needSniff && chunk) {
388 if (this[BUFFER]) {
389 chunk = Buffer.concat([this[BUFFER], chunk]);
390 this[BUFFER] = undefined;
391 }
392 if (chunk.length < ZIP_HEADER_LEN) {
393 this[BUFFER] = chunk;
394 /* c8 ignore next */
395 cb?.();
396 return true;
397 }
398 // look for gzip header
399 for (let i = 0; this[UNZIP] === undefined && i < gzipHeader.length; i++) {
400 if (chunk[i] !== gzipHeader[i]) {
401 this[UNZIP] = false;
402 }
403 }
404 // look for zstd header if gzip header not found
405 let isZstd = false;
406 if (this[UNZIP] === false && this.zstd !== false) {
407 isZstd = true;
408 for (let i = 0; i < zstdHeader.length; i++) {
409 if (chunk[i] !== zstdHeader[i]) {
410 isZstd = false;
411 break;
412 }
413 }
414 }
415 const maybeBrotli = this.brotli === undefined && !isZstd;
416 if (this[UNZIP] === false && maybeBrotli) {
417 // read the first header to see if it's a valid tar file. If so,
418 // we can safely assume that it's not actually brotli, despite the
419 // .tbr or .tar.br file extension.
420 // if we ended before getting a full chunk, yes, def brotli
421 if (chunk.length < 512) {
422 if (this[ENDED]) {
423 this.brotli = true;
424 }
425 else {
426 this[BUFFER] = chunk;
427 /* c8 ignore next */
428 cb?.();
429 return true;
430 }
431 }
432 else {
433 // if it's tar, it's pretty reliably not brotli, chances of
434 // that happening are astronomical.
435 try {
436 new Header(chunk.subarray(0, 512));
437 this.brotli = false;
438 }
439 catch (_) {
440 this.brotli = true;
441 }
442 }
443 }
444 if (this[UNZIP] === undefined ||
445 (this[UNZIP] === false && (this.brotli || isZstd))) {
446 const ended = this[ENDED];
447 this[ENDED] = false;
448 this[UNZIP] =
449 this[UNZIP] === undefined ? new Unzip({})
450 : isZstd ? new ZstdDecompress({})
451 : new BrotliDecompress({});
452 this[UNZIP].on('data', chunk => this[CONSUMECHUNK](chunk));
453 this[UNZIP].on('error', er => this.abort(er));
454 this[UNZIP].on('end', () => {
455 this[ENDED] = true;
456 this[CONSUMECHUNK]();
457 });
458 this[WRITING] = true;
459 const ret = !!this[UNZIP][ended ? 'end' : 'write'](chunk);
460 this[WRITING] = false;
461 cb?.();
462 return ret;
463 }
464 }
465 this[WRITING] = true;
466 if (this[UNZIP]) {
467 this[UNZIP].write(chunk);
468 }
469 else {
470 this[CONSUMECHUNK](chunk);
471 }
472 this[WRITING] = false;
473 // return false if there's a queue, or if the current entry isn't flowing
474 const ret = this[QUEUE].length ? false
475 : this[READENTRY] ? this[READENTRY].flowing
476 : true;
477 // if we have no queue, then that means a clogged READENTRY
478 if (!ret && !this[QUEUE].length) {
479 this[READENTRY]?.once('drain', () => this.emit('drain'));
480 }
481 /* c8 ignore next */
482 cb?.();
483 return ret;
484 }
485 [BUFFERCONCAT](c) {
486 if (c && !this[ABORTED]) {
487 this[BUFFER] =
488 this[BUFFER] ? Buffer.concat([this[BUFFER], c]) : c;
489 }
490 }
491 [MAYBEEND]() {
492 if (this[ENDED] &&
493 !this[EMITTEDEND] &&
494 !this[ABORTED] &&
495 !this[CONSUMING]) {
496 this[EMITTEDEND] = true;
497 const entry = this[WRITEENTRY];
498 if (entry && entry.blockRemain) {
499 // truncated, likely a damaged file
500 const have = this[BUFFER] ? this[BUFFER].length : 0;
501 this.warn('TAR_BAD_ARCHIVE', `Truncated input (needed ${entry.blockRemain} more bytes, only ${have} available)`, { entry });
502 if (this[BUFFER]) {
503 entry.write(this[BUFFER]);
504 }
505 entry.end();
506 }
507 this[EMIT](DONE);
508 }
509 }
510 [CONSUMECHUNK](chunk) {
511 if (this[CONSUMING] && chunk) {
512 this[BUFFERCONCAT](chunk);
513 }
514 else if (!chunk && !this[BUFFER]) {
515 this[MAYBEEND]();
516 }
517 else if (chunk) {
518 this[CONSUMING] = true;
519 if (this[BUFFER]) {
520 this[BUFFERCONCAT](chunk);
521 const c = this[BUFFER];
522 this[BUFFER] = undefined;
523 this[CONSUMECHUNKSUB](c);
524 }
525 else {
526 this[CONSUMECHUNKSUB](chunk);
527 }
528 while (this[BUFFER] &&
529 this[BUFFER]?.length >= 512 &&
530 !this[ABORTED] &&
531 !this[SAW_EOF]) {
532 const c = this[BUFFER];
533 this[BUFFER] = undefined;
534 this[CONSUMECHUNKSUB](c);
535 }
536 this[CONSUMING] = false;
537 }
538 if (!this[BUFFER] || this[ENDED]) {
539 this[MAYBEEND]();
540 }
541 }
542 [CONSUMECHUNKSUB](chunk) {
543 // we know that we are in CONSUMING mode, so anything written goes into
544 // the buffer. Advance the position and put any remainder in the buffer.
545 let position = 0;
546 const length = chunk.length;
547 while (position + 512 <= length &&
548 !this[ABORTED] &&
549 !this[SAW_EOF]) {
550 switch (this[STATE]) {
551 case 'begin':
552 case 'header':
553 this[CONSUMEHEADER](chunk, position);
554 position += 512;
555 break;
556 case 'ignore':
557 case 'body':
558 position += this[CONSUMEBODY](chunk, position);
559 break;
560 case 'meta':
561 position += this[CONSUMEMETA](chunk, position);
562 break;
563 /* c8 ignore start */
564 default:
565 throw new Error('invalid state: ' + this[STATE]);
566 /* c8 ignore stop */
567 }
568 }
569 if (position < length) {
570 if (this[BUFFER]) {
571 this[BUFFER] = Buffer.concat([
572 chunk.subarray(position),
573 this[BUFFER],
574 ]);
575 }
576 else {
577 this[BUFFER] = chunk.subarray(position);
578 }
579 }
580 }
581 end(chunk, encoding, cb) {
582 if (typeof chunk === 'function') {
583 cb = chunk;
584 encoding = undefined;
585 chunk = undefined;
586 }
587 if (typeof encoding === 'function') {
588 cb = encoding;
589 encoding = undefined;
590 }
591 if (typeof chunk === 'string') {
592 chunk = Buffer.from(chunk, encoding);
593 }
594 if (cb)
595 this.once('finish', cb);
596 if (!this[ABORTED]) {
597 if (this[UNZIP]) {
598 /* c8 ignore start */
599 if (chunk)
600 this[UNZIP].write(chunk);
601 /* c8 ignore stop */
602 this[UNZIP].end();
603 }
604 else {
605 this[ENDED] = true;
606 if (this.brotli === undefined || this.zstd === undefined)
607 chunk = chunk || Buffer.alloc(0);
608 if (chunk)
609 this.write(chunk);
610 this[MAYBEEND]();
611 }
612 }
613 return this;
614 }
615}
616//# sourceMappingURL=parse.js.map