codekingpro/portable-devtools
115k
1"use strict";
2// this[BUFFER] is the remainder of a chunk if we're waiting for
3// the full 512 bytes of a header to come in. We will Buffer.concat()
4// it to the next write(), which is a mem copy, but a small one.
5//
6// this[QUEUE] is a list of entries that haven't been emitted
7// yet this can only get filled up if the user keeps write()ing after
8// a write() returns false, or does a write() with more than one entry
9//
10// We don't buffer chunks, we always parse them and either create an
11// entry, or push it into the active entry. The ReadEntry class knows
12// to throw data away if .ignore=true
13//
14// Shift entry off the buffer when it emits 'end', and emit 'entry' for
15// the next one in the list.
16//
17// At any time, we're pushing body chunks into the entry at WRITEENTRY,
18// and waiting for 'end' on the entry at READENTRY
19//
20// ignored entries get .resume() called on them straight away
21Object.defineProperty(exports, "__esModule", { value: true });
22exports.Parser = void 0;
23const events_1 = require("events");
24const minizlib_1 = require("minizlib");
25const header_js_1 = require("./header.js");
26const pax_js_1 = require("./pax.js");
27const read_entry_js_1 = require("./read-entry.js");
28const warn_method_js_1 = require("./warn-method.js");
29const maxMetaEntrySize = 1024 * 1024;
30const gzipHeader = Buffer.from([0x1f, 0x8b]);
31const zstdHeader = Buffer.from([0x28, 0xb5, 0x2f, 0xfd]);
32const ZIP_HEADER_LEN = Math.max(gzipHeader.length, zstdHeader.length);
33const STATE = Symbol('state');
34const WRITEENTRY = Symbol('writeEntry');
35const READENTRY = Symbol('readEntry');
36const NEXTENTRY = Symbol('nextEntry');
37const PROCESSENTRY = Symbol('processEntry');
38const EX = Symbol('extendedHeader');
39const GEX = Symbol('globalExtendedHeader');
40const META = Symbol('meta');
41const EMITMETA = Symbol('emitMeta');
42const BUFFER = Symbol('buffer');
43const QUEUE = Symbol('queue');
44const ENDED = Symbol('ended');
45const EMITTEDEND = Symbol('emittedEnd');
46const EMIT = Symbol('emit');
47const UNZIP = Symbol('unzip');
48const CONSUMECHUNK = Symbol('consumeChunk');
49const CONSUMECHUNKSUB = Symbol('consumeChunkSub');
50const CONSUMEBODY = Symbol('consumeBody');
51const CONSUMEMETA = Symbol('consumeMeta');
52const CONSUMEHEADER = Symbol('consumeHeader');
53const CONSUMING = Symbol('consuming');
54const BUFFERCONCAT = Symbol('bufferConcat');
55const MAYBEEND = Symbol('maybeEnd');
56const WRITING = Symbol('writing');
57const ABORTED = Symbol('aborted');
58const DONE = Symbol('onDone');
59const SAW_VALID_ENTRY = Symbol('sawValidEntry');
60const SAW_NULL_BLOCK = Symbol('sawNullBlock');
61const SAW_EOF = Symbol('sawEOF');
62const CLOSESTREAM = Symbol('closeStream');
63const noop = () => true;
64class Parser extends events_1.EventEmitter {
65 file;
66 strict;
67 maxMetaEntrySize;
68 filter;
69 brotli;
70 zstd;
71 writable = true;
72 readable = false;
73 [QUEUE] = [];
74 [BUFFER];
75 [READENTRY];
76 [WRITEENTRY];
77 [STATE] = 'begin';
78 [META] = '';
79 [EX];
80 [GEX];
81 [ENDED] = false;
82 [UNZIP];
83 [ABORTED] = false;
84 [SAW_VALID_ENTRY];
85 [SAW_NULL_BLOCK] = false;
86 [SAW_EOF] = false;
87 [WRITING] = false;
88 [CONSUMING] = false;
89 [EMITTEDEND] = false;
90 constructor(opt = {}) {
91 super();
92 this.file = opt.file || '';
93 // these BADARCHIVE errors can't be detected early. listen on DONE.
94 this.on(DONE, () => {
95 if (this[STATE] === 'begin' ||
96 this[SAW_VALID_ENTRY] === false) {
97 // either less than 1 block of data, or all entries were invalid.
98 // Either way, probably not even a tarball.
99 this.warn('TAR_BAD_ARCHIVE', 'Unrecognized archive format');
100 }
101 });
102 if (opt.ondone) {
103 this.on(DONE, opt.ondone);
104 }
105 else {
106 this.on(DONE, () => {
107 this.emit('prefinish');
108 this.emit('finish');
109 this.emit('end');
110 });
111 }
112 this.strict = !!opt.strict;
113 this.maxMetaEntrySize = opt.maxMetaEntrySize || maxMetaEntrySize;
114 this.filter = typeof opt.filter === 'function' ? opt.filter : noop;
115 // Unlike gzip, brotli doesn't have any magic bytes to identify it
116 // Users need to explicitly tell us they're extracting a brotli file
117 // Or we infer from the file extension
118 const isTBR = opt.file &&
119 (opt.file.endsWith('.tar.br') || opt.file.endsWith('.tbr'));
120 // if it's a tbr file it MIGHT be brotli, but we don't know until
121 // we look at it and verify it's not a valid tar file.
122 this.brotli =
123 !(opt.gzip || opt.zstd) && opt.brotli !== undefined ? opt.brotli
124 : isTBR ? undefined
125 : false;
126 // zstd has magic bytes to identify it, but we also support explicit options
127 // and file extension detection
128 const isTZST = opt.file &&
129 (opt.file.endsWith('.tar.zst') || opt.file.endsWith('.tzst'));
130 this.zstd =
131 !(opt.gzip || opt.brotli) && opt.zstd !== undefined ? opt.zstd
132 : isTZST ? true
133 : undefined;
134 // have to set this so that streams are ok piping into it
135 this.on('end', () => this[CLOSESTREAM]());
136 if (typeof opt.onwarn === 'function') {
137 this.on('warn', opt.onwarn);
138 }
139 if (typeof opt.onReadEntry === 'function') {
140 this.on('entry', opt.onReadEntry);
141 }
142 }
143 warn(code, message, data = {}) {
144 (0, warn_method_js_1.warnMethod)(this, code, message, data);
145 }
146 [CONSUMEHEADER](chunk, position) {
147 if (this[SAW_VALID_ENTRY] === undefined) {
148 this[SAW_VALID_ENTRY] = false;
149 }
150 let header;
151 try {
152 header = new header_js_1.Header(chunk, position, this[EX], this[GEX]);
153 }
154 catch (er) {
155 return this.warn('TAR_ENTRY_INVALID', er);
156 }
157 if (header.nullBlock) {
158 if (this[SAW_NULL_BLOCK]) {
159 this[SAW_EOF] = true;
160 // ending an archive with no entries. pointless, but legal.
161 if (this[STATE] === 'begin') {
162 this[STATE] = 'header';
163 }
164 this[EMIT]('eof');
165 }
166 else {
167 this[SAW_NULL_BLOCK] = true;
168 this[EMIT]('nullBlock');
169 }
170 }
171 else {
172 this[SAW_NULL_BLOCK] = false;
173 if (!header.cksumValid) {
174 this.warn('TAR_ENTRY_INVALID', 'checksum failure', { header });
175 }
176 else if (!header.path) {
177 this.warn('TAR_ENTRY_INVALID', 'path is required', { header });
178 }
179 else {
180 const type = header.type;
181 if (/^(Symbolic)?Link$/.test(type) && !header.linkpath) {
182 this.warn('TAR_ENTRY_INVALID', 'linkpath required', {
183 header,
184 });
185 }
186 else if (!/^(Symbolic)?Link$/.test(type) &&
187 !/^(Global)?ExtendedHeader$/.test(type) &&
188 header.linkpath) {
189 this.warn('TAR_ENTRY_INVALID', 'linkpath forbidden', {
190 header,
191 });
192 }
193 else {
194 const entry = (this[WRITEENTRY] = new read_entry_js_1.ReadEntry(header, this[EX], this[GEX]));
195 // we do this for meta & ignored entries as well, because they
196 // are still valid tar, or else we wouldn't know to ignore them
197 if (!this[SAW_VALID_ENTRY]) {
198 if (entry.remain) {
199 // this might be the one!
200 const onend = () => {
201 if (!entry.invalid) {
202 this[SAW_VALID_ENTRY] = true;
203 }
204 };
205 entry.on('end', onend);
206 }
207 else {
208 this[SAW_VALID_ENTRY] = true;
209 }
210 }
211 if (entry.meta) {
212 if (entry.size > this.maxMetaEntrySize) {
213 entry.ignore = true;
214 this[EMIT]('ignoredEntry', entry);
215 this[STATE] = 'ignore';
216 entry.resume();
217 }
218 else if (entry.size > 0) {
219 this[META] = '';
220 entry.on('data', c => (this[META] += c));
221 this[STATE] = 'meta';
222 }
223 }
224 else {
225 this[EX] = undefined;
226 entry.ignore =
227 entry.ignore || !this.filter(entry.path, entry);
228 if (entry.ignore) {
229 // probably valid, just not something we care about
230 this[EMIT]('ignoredEntry', entry);
231 this[STATE] = entry.remain ? 'ignore' : 'header';
232 entry.resume();
233 }
234 else {
235 if (entry.remain) {
236 this[STATE] = 'body';
237 }
238 else {
239 this[STATE] = 'header';
240 entry.end();
241 }
242 if (!this[READENTRY]) {
243 this[QUEUE].push(entry);
244 this[NEXTENTRY]();
245 }
246 else {
247 this[QUEUE].push(entry);
248 }
249 }
250 }
251 }
252 }
253 }
254 }
255 [CLOSESTREAM]() {
256 queueMicrotask(() => this.emit('close'));
257 }
258 [PROCESSENTRY](entry) {
259 let go = true;
260 if (!entry) {
261 this[READENTRY] = undefined;
262 go = false;
263 }
264 else if (Array.isArray(entry)) {
265 const [ev, ...args] = entry;
266 this.emit(ev, ...args);
267 }
268 else {
269 this[READENTRY] = entry;
270 this.emit('entry', entry);
271 if (!entry.emittedEnd) {
272 entry.on('end', () => this[NEXTENTRY]());
273 go = false;
274 }
275 }
276 return go;
277 }
278 [NEXTENTRY]() {
279 do { } while (this[PROCESSENTRY](this[QUEUE].shift()));
280 if (!this[QUEUE].length) {
281 // At this point, there's nothing in the queue, but we may have an
282 // entry which is being consumed (readEntry).
283 // If we don't, then we definitely can handle more data.
284 // If we do, and either it's flowing, or it has never had any data
285 // written to it, then it needs more.
286 // The only other possibility is that it has returned false from a
287 // write() call, so we wait for the next drain to continue.
288 const re = this[READENTRY];
289 const drainNow = !re || re.flowing || re.size === re.remain;
290 if (drainNow) {
291 if (!this[WRITING]) {
292 this.emit('drain');
293 }
294 }
295 else {
296 re.once('drain', () => this.emit('drain'));
297 }
298 }
299 }
300 [CONSUMEBODY](chunk, position) {
301 // write up to but no more than writeEntry.blockRemain
302 const entry = this[WRITEENTRY];
303 /* c8 ignore start */
304 if (!entry) {
305 throw new Error('attempt to consume body without entry??');
306 }
307 const br = entry.blockRemain ?? 0;
308 /* c8 ignore stop */
309 const c = br >= chunk.length && position === 0 ?
310 chunk
311 : chunk.subarray(position, position + br);
312 entry.write(c);
313 if (!entry.blockRemain) {
314 this[STATE] = 'header';
315 this[WRITEENTRY] = undefined;
316 entry.end();
317 }
318 return c.length;
319 }
320 [CONSUMEMETA](chunk, position) {
321 const entry = this[WRITEENTRY];
322 const ret = this[CONSUMEBODY](chunk, position);
323 // if we finished, then the entry is reset
324 if (!this[WRITEENTRY] && entry) {
325 this[EMITMETA](entry);
326 }
327 return ret;
328 }
329 [EMIT](ev, data, extra) {
330 if (!this[QUEUE].length && !this[READENTRY]) {
331 this.emit(ev, data, extra);
332 }
333 else {
334 this[QUEUE].push([ev, data, extra]);
335 }
336 }
337 [EMITMETA](entry) {
338 this[EMIT]('meta', this[META]);
339 switch (entry.type) {
340 case 'ExtendedHeader':
341 case 'OldExtendedHeader':
342 this[EX] = pax_js_1.Pax.parse(this[META], this[EX], false);
343 break;
344 case 'GlobalExtendedHeader':
345 this[GEX] = pax_js_1.Pax.parse(this[META], this[GEX], true);
346 break;
347 case 'NextFileHasLongPath':
348 case 'OldGnuLongPath': {
349 const ex = this[EX] ?? Object.create(null);
350 this[EX] = ex;
351 ex.path = this[META].replace(/\0.*/, '');
352 break;
353 }
354 case 'NextFileHasLongLinkpath': {
355 const ex = this[EX] || Object.create(null);
356 this[EX] = ex;
357 ex.linkpath = this[META].replace(/\0.*/, '');
358 break;
359 }
360 /* c8 ignore start */
361 default:
362 throw new Error('unknown meta: ' + entry.type);
363 /* c8 ignore stop */
364 }
365 }
366 abort(error) {
367 this[ABORTED] = true;
368 this.emit('abort', error);
369 // always throws, even in non-strict mode
370 this.warn('TAR_ABORT', error, { recoverable: false });
371 }
372 write(chunk, encoding, cb) {
373 if (typeof encoding === 'function') {
374 cb = encoding;
375 encoding = undefined;
376 }
377 if (typeof chunk === 'string') {
378 chunk = Buffer.from(chunk,
379 /* c8 ignore next */
380 typeof encoding === 'string' ? encoding : 'utf8');
381 }
382 if (this[ABORTED]) {
383 /* c8 ignore next */
384 cb?.();
385 return false;
386 }
387 // first write, might be gzipped, zstd, or brotli compressed
388 const needSniff = this[UNZIP] === undefined ||
389 (this.brotli === undefined && this[UNZIP] === false);
390 if (needSniff && chunk) {
391 if (this[BUFFER]) {
392 chunk = Buffer.concat([this[BUFFER], chunk]);
393 this[BUFFER] = undefined;
394 }
395 if (chunk.length < ZIP_HEADER_LEN) {
396 this[BUFFER] = chunk;
397 /* c8 ignore next */
398 cb?.();
399 return true;
400 }
401 // look for gzip header
402 for (let i = 0; this[UNZIP] === undefined && i < gzipHeader.length; i++) {
403 if (chunk[i] !== gzipHeader[i]) {
404 this[UNZIP] = false;
405 }
406 }
407 // look for zstd header if gzip header not found
408 let isZstd = false;
409 if (this[UNZIP] === false && this.zstd !== false) {
410 isZstd = true;
411 for (let i = 0; i < zstdHeader.length; i++) {
412 if (chunk[i] !== zstdHeader[i]) {
413 isZstd = false;
414 break;
415 }
416 }
417 }
418 const maybeBrotli = this.brotli === undefined && !isZstd;
419 if (this[UNZIP] === false && maybeBrotli) {
420 // read the first header to see if it's a valid tar file. If so,
421 // we can safely assume that it's not actually brotli, despite the
422 // .tbr or .tar.br file extension.
423 // if we ended before getting a full chunk, yes, def brotli
424 if (chunk.length < 512) {
425 if (this[ENDED]) {
426 this.brotli = true;
427 }
428 else {
429 this[BUFFER] = chunk;
430 /* c8 ignore next */
431 cb?.();
432 return true;
433 }
434 }
435 else {
436 // if it's tar, it's pretty reliably not brotli, chances of
437 // that happening are astronomical.
438 try {
439 new header_js_1.Header(chunk.subarray(0, 512));
440 this.brotli = false;
441 }
442 catch (_) {
443 this.brotli = true;
444 }
445 }
446 }
447 if (this[UNZIP] === undefined ||
448 (this[UNZIP] === false && (this.brotli || isZstd))) {
449 const ended = this[ENDED];
450 this[ENDED] = false;
451 this[UNZIP] =
452 this[UNZIP] === undefined ? new minizlib_1.Unzip({})
453 : isZstd ? new minizlib_1.ZstdDecompress({})
454 : new minizlib_1.BrotliDecompress({});
455 this[UNZIP].on('data', chunk => this[CONSUMECHUNK](chunk));
456 this[UNZIP].on('error', er => this.abort(er));
457 this[UNZIP].on('end', () => {
458 this[ENDED] = true;
459 this[CONSUMECHUNK]();
460 });
461 this[WRITING] = true;
462 const ret = !!this[UNZIP][ended ? 'end' : 'write'](chunk);
463 this[WRITING] = false;
464 cb?.();
465 return ret;
466 }
467 }
468 this[WRITING] = true;
469 if (this[UNZIP]) {
470 this[UNZIP].write(chunk);
471 }
472 else {
473 this[CONSUMECHUNK](chunk);
474 }
475 this[WRITING] = false;
476 // return false if there's a queue, or if the current entry isn't flowing
477 const ret = this[QUEUE].length ? false
478 : this[READENTRY] ? this[READENTRY].flowing
479 : true;
480 // if we have no queue, then that means a clogged READENTRY
481 if (!ret && !this[QUEUE].length) {
482 this[READENTRY]?.once('drain', () => this.emit('drain'));
483 }
484 /* c8 ignore next */
485 cb?.();
486 return ret;
487 }
488 [BUFFERCONCAT](c) {
489 if (c && !this[ABORTED]) {
490 this[BUFFER] =
491 this[BUFFER] ? Buffer.concat([this[BUFFER], c]) : c;
492 }
493 }
494 [MAYBEEND]() {
495 if (this[ENDED] &&
496 !this[EMITTEDEND] &&
497 !this[ABORTED] &&
498 !this[CONSUMING]) {
499 this[EMITTEDEND] = true;
500 const entry = this[WRITEENTRY];
501 if (entry && entry.blockRemain) {
502 // truncated, likely a damaged file
503 const have = this[BUFFER] ? this[BUFFER].length : 0;
504 this.warn('TAR_BAD_ARCHIVE', `Truncated input (needed ${entry.blockRemain} more bytes, only ${have} available)`, { entry });
505 if (this[BUFFER]) {
506 entry.write(this[BUFFER]);
507 }
508 entry.end();
509 }
510 this[EMIT](DONE);
511 }
512 }
513 [CONSUMECHUNK](chunk) {
514 if (this[CONSUMING] && chunk) {
515 this[BUFFERCONCAT](chunk);
516 }
517 else if (!chunk && !this[BUFFER]) {
518 this[MAYBEEND]();
519 }
520 else if (chunk) {
521 this[CONSUMING] = true;
522 if (this[BUFFER]) {
523 this[BUFFERCONCAT](chunk);
524 const c = this[BUFFER];
525 this[BUFFER] = undefined;
526 this[CONSUMECHUNKSUB](c);
527 }
528 else {
529 this[CONSUMECHUNKSUB](chunk);
530 }
531 while (this[BUFFER] &&
532 this[BUFFER]?.length >= 512 &&
533 !this[ABORTED] &&
534 !this[SAW_EOF]) {
535 const c = this[BUFFER];
536 this[BUFFER] = undefined;
537 this[CONSUMECHUNKSUB](c);
538 }
539 this[CONSUMING] = false;
540 }
541 if (!this[BUFFER] || this[ENDED]) {
542 this[MAYBEEND]();
543 }
544 }
545 [CONSUMECHUNKSUB](chunk) {
546 // we know that we are in CONSUMING mode, so anything written goes into
547 // the buffer. Advance the position and put any remainder in the buffer.
548 let position = 0;
549 const length = chunk.length;
550 while (position + 512 <= length &&
551 !this[ABORTED] &&
552 !this[SAW_EOF]) {
553 switch (this[STATE]) {
554 case 'begin':
555 case 'header':
556 this[CONSUMEHEADER](chunk, position);
557 position += 512;
558 break;
559 case 'ignore':
560 case 'body':
561 position += this[CONSUMEBODY](chunk, position);
562 break;
563 case 'meta':
564 position += this[CONSUMEMETA](chunk, position);
565 break;
566 /* c8 ignore start */
567 default:
568 throw new Error('invalid state: ' + this[STATE]);
569 /* c8 ignore stop */
570 }
571 }
572 if (position < length) {
573 if (this[BUFFER]) {
574 this[BUFFER] = Buffer.concat([
575 chunk.subarray(position),
576 this[BUFFER],
577 ]);
578 }
579 else {
580 this[BUFFER] = chunk.subarray(position);
581 }
582 }
583 }
584 end(chunk, encoding, cb) {
585 if (typeof chunk === 'function') {
586 cb = chunk;
587 encoding = undefined;
588 chunk = undefined;
589 }
590 if (typeof encoding === 'function') {
591 cb = encoding;
592 encoding = undefined;
593 }
594 if (typeof chunk === 'string') {
595 chunk = Buffer.from(chunk, encoding);
596 }
597 if (cb)
598 this.once('finish', cb);
599 if (!this[ABORTED]) {
600 if (this[UNZIP]) {
601 /* c8 ignore start */
602 if (chunk)
603 this[UNZIP].write(chunk);
604 /* c8 ignore stop */
605 this[UNZIP].end();
606 }
607 else {
608 this[ENDED] = true;
609 if (this.brotli === undefined || this.zstd === undefined)
610 chunk = chunk || Buffer.alloc(0);
611 if (chunk)
612 this.write(chunk);
613 this[MAYBEEND]();
614 }
615 }
616 return this;
617 }
618}
619exports.Parser = Parser;
620//# sourceMappingURL=parse.js.map