codekingpro/portable-devtools
115k
1"use strict";
2// this[BUFFER] is the remainder of a chunk if we're waiting for
3// the full 512 bytes of a header to come in. We will Buffer.concat()
4// it to the next write(), which is a mem copy, but a small one.
5//
6// this[QUEUE] is a list of entries that haven't been emitted
7// yet this can only get filled up if the user keeps write()ing after
8// a write() returns false, or does a write() with more than one entry
9//
10// We don't buffer chunks, we always parse them and either create an
11// entry, or push it into the active entry. The ReadEntry class knows
12// to throw data away if .ignore=true
13//
14// Shift entry off the buffer when it emits 'end', and emit 'entry' for
15// the next one in the list.
16//
17// At any time, we're pushing body chunks into the entry at WRITEENTRY,
18// and waiting for 'end' on the entry at READENTRY
19//
20// ignored entries get .resume() called on them straight away
21Object.defineProperty(exports, "__esModule", { value: true });
22exports.Parser = void 0;
23const events_1 = require("events");
24const minizlib_1 = require("minizlib");
25const header_js_1 = require("./header.js");
26const pax_js_1 = require("./pax.js");
27const read_entry_js_1 = require("./read-entry.js");
28const warn_method_js_1 = require("./warn-method.js");
29const maxMetaEntrySize = 1024 * 1024;
30const gzipHeader = Buffer.from([0x1f, 0x8b]);
31const zstdHeader = Buffer.from([0x28, 0xb5, 0x2f, 0xfd]);
32const ZIP_HEADER_LEN = Math.max(gzipHeader.length, zstdHeader.length);
33const STATE = Symbol('state');
34const WRITEENTRY = Symbol('writeEntry');
35const READENTRY = Symbol('readEntry');
36const NEXTENTRY = Symbol('nextEntry');
37const PROCESSENTRY = Symbol('processEntry');
38const EX = Symbol('extendedHeader');
39const GEX = Symbol('globalExtendedHeader');
40const META = Symbol('meta');
41const EMITMETA = Symbol('emitMeta');
42const BUFFER = Symbol('buffer');
43const QUEUE = Symbol('queue');
44const ENDED = Symbol('ended');
45const EMITTEDEND = Symbol('emittedEnd');
46const EMIT = Symbol('emit');
47const UNZIP = Symbol('unzip');
48const CONSUMECHUNK = Symbol('consumeChunk');
49const CONSUMECHUNKSUB = Symbol('consumeChunkSub');
50const CONSUMEBODY = Symbol('consumeBody');
51const CONSUMEMETA = Symbol('consumeMeta');
52const CONSUMEHEADER = Symbol('consumeHeader');
53const CONSUMING = Symbol('consuming');
54const BUFFERCONCAT = Symbol('bufferConcat');
55const MAYBEEND = Symbol('maybeEnd');
56const WRITING = Symbol('writing');
57const ABORTED = Symbol('aborted');
58const DONE = Symbol('onDone');
59const SAW_VALID_ENTRY = Symbol('sawValidEntry');
60const SAW_NULL_BLOCK = Symbol('sawNullBlock');
61const SAW_EOF = Symbol('sawEOF');
62const CLOSESTREAM = Symbol('closeStream');
63const noop = () => true;
64class Parser extends events_1.EventEmitter {
65 file;
66 strict;
67 maxMetaEntrySize;
68 filter;
69 brotli;
70 zstd;
71 writable = true;
72 readable = false;
73 [QUEUE] = [];
74 [BUFFER];
75 [READENTRY];
76 [WRITEENTRY];
77 [STATE] = 'begin';
78 [META] = '';
79 [EX];
80 [GEX];
81 [ENDED] = false;
82 [UNZIP];
83 [ABORTED] = false;
84 [SAW_VALID_ENTRY];
85 [SAW_NULL_BLOCK] = false;
86 [SAW_EOF] = false;
87 [WRITING] = false;
88 [CONSUMING] = false;
89 [EMITTEDEND] = false;
90 constructor(opt = {}) {
91 super();
92 this.file = opt.file || '';
93 // these BADARCHIVE errors can't be detected early. listen on DONE.
94 this.on(DONE, () => {
95 if (this[STATE] === 'begin' || this[SAW_VALID_ENTRY] === false) {
96 // either less than 1 block of data, or all entries were invalid.
97 // Either way, probably not even a tarball.
98 this.warn('TAR_BAD_ARCHIVE', 'Unrecognized archive format');
99 }
100 });
101 if (opt.ondone) {
102 this.on(DONE, opt.ondone);
103 }
104 else {
105 this.on(DONE, () => {
106 this.emit('prefinish');
107 this.emit('finish');
108 this.emit('end');
109 });
110 }
111 this.strict = !!opt.strict;
112 this.maxMetaEntrySize = opt.maxMetaEntrySize || maxMetaEntrySize;
113 this.filter = typeof opt.filter === 'function' ? opt.filter : noop;
114 // Unlike gzip, brotli doesn't have any magic bytes to identify it
115 // Users need to explicitly tell us they're extracting a brotli file
116 // Or we infer from the file extension
117 const isTBR = opt.file &&
118 (opt.file.endsWith('.tar.br') || opt.file.endsWith('.tbr'));
119 // if it's a tbr file it MIGHT be brotli, but we don't know until
120 // we look at it and verify it's not a valid tar file.
121 this.brotli =
122 !(opt.gzip || opt.zstd) && opt.brotli !== undefined ? opt.brotli
123 : isTBR ? undefined
124 : false;
125 // zstd has magic bytes to identify it, but we also support explicit options
126 // and file extension detection
127 const isTZST = opt.file &&
128 (opt.file.endsWith('.tar.zst') || opt.file.endsWith('.tzst'));
129 this.zstd =
130 !(opt.gzip || opt.brotli) && opt.zstd !== undefined ? opt.zstd
131 : isTZST ? true
132 : undefined;
133 // have to set this so that streams are ok piping into it
134 this.on('end', () => this[CLOSESTREAM]());
135 if (typeof opt.onwarn === 'function') {
136 this.on('warn', opt.onwarn);
137 }
138 if (typeof opt.onReadEntry === 'function') {
139 this.on('entry', opt.onReadEntry);
140 }
141 }
142 warn(code, message, data = {}) {
143 (0, warn_method_js_1.warnMethod)(this, code, message, data);
144 }
145 [CONSUMEHEADER](chunk, position) {
146 if (this[SAW_VALID_ENTRY] === undefined) {
147 this[SAW_VALID_ENTRY] = false;
148 }
149 let header;
150 try {
151 header = new header_js_1.Header(chunk, position, this[EX], this[GEX]);
152 }
153 catch (er) {
154 return this.warn('TAR_ENTRY_INVALID', er);
155 }
156 if (header.nullBlock) {
157 if (this[SAW_NULL_BLOCK]) {
158 this[SAW_EOF] = true;
159 // ending an archive with no entries. pointless, but legal.
160 if (this[STATE] === 'begin') {
161 this[STATE] = 'header';
162 }
163 this[EMIT]('eof');
164 }
165 else {
166 this[SAW_NULL_BLOCK] = true;
167 this[EMIT]('nullBlock');
168 }
169 }
170 else {
171 this[SAW_NULL_BLOCK] = false;
172 if (!header.cksumValid) {
173 this.warn('TAR_ENTRY_INVALID', 'checksum failure', { header });
174 }
175 else if (!header.path) {
176 this.warn('TAR_ENTRY_INVALID', 'path is required', { header });
177 }
178 else {
179 const type = header.type;
180 if (/^(Symbolic)?Link$/.test(type) && !header.linkpath) {
181 this.warn('TAR_ENTRY_INVALID', 'linkpath required', {
182 header,
183 });
184 }
185 else if (!/^(Symbolic)?Link$/.test(type) &&
186 !/^(Global)?ExtendedHeader$/.test(type) &&
187 header.linkpath) {
188 this.warn('TAR_ENTRY_INVALID', 'linkpath forbidden', {
189 header,
190 });
191 }
192 else {
193 const entry = (this[WRITEENTRY] = new read_entry_js_1.ReadEntry(header, this[EX], this[GEX]));
194 // we do this for meta & ignored entries as well, because they
195 // are still valid tar, or else we wouldn't know to ignore them
196 if (!this[SAW_VALID_ENTRY]) {
197 if (entry.remain) {
198 // this might be the one!
199 const onend = () => {
200 if (!entry.invalid) {
201 this[SAW_VALID_ENTRY] = true;
202 }
203 };
204 entry.on('end', onend);
205 }
206 else {
207 this[SAW_VALID_ENTRY] = true;
208 }
209 }
210 if (entry.meta) {
211 if (entry.size > this.maxMetaEntrySize) {
212 entry.ignore = true;
213 this[EMIT]('ignoredEntry', entry);
214 this[STATE] = 'ignore';
215 entry.resume();
216 }
217 else if (entry.size > 0) {
218 this[META] = '';
219 entry.on('data', c => (this[META] += c));
220 this[STATE] = 'meta';
221 }
222 }
223 else {
224 this[EX] = undefined;
225 entry.ignore = entry.ignore || !this.filter(entry.path, entry);
226 if (entry.ignore) {
227 // probably valid, just not something we care about
228 this[EMIT]('ignoredEntry', entry);
229 this[STATE] = entry.remain ? 'ignore' : 'header';
230 entry.resume();
231 }
232 else {
233 if (entry.remain) {
234 this[STATE] = 'body';
235 }
236 else {
237 this[STATE] = 'header';
238 entry.end();
239 }
240 if (!this[READENTRY]) {
241 this[QUEUE].push(entry);
242 this[NEXTENTRY]();
243 }
244 else {
245 this[QUEUE].push(entry);
246 }
247 }
248 }
249 }
250 }
251 }
252 }
253 [CLOSESTREAM]() {
254 queueMicrotask(() => this.emit('close'));
255 }
256 [PROCESSENTRY](entry) {
257 let go = true;
258 if (!entry) {
259 this[READENTRY] = undefined;
260 go = false;
261 }
262 else if (Array.isArray(entry)) {
263 const [ev, ...args] = entry;
264 this.emit(ev, ...args);
265 }
266 else {
267 this[READENTRY] = entry;
268 this.emit('entry', entry);
269 if (!entry.emittedEnd) {
270 entry.on('end', () => this[NEXTENTRY]());
271 go = false;
272 }
273 }
274 return go;
275 }
276 [NEXTENTRY]() {
277 do { } while (this[PROCESSENTRY](this[QUEUE].shift()));
278 if (this[QUEUE].length === 0) {
279 // At this point, there's nothing in the queue, but we may have an
280 // entry which is being consumed (readEntry).
281 // If we don't, then we definitely can handle more data.
282 // If we do, and either it's flowing, or it has never had any data
283 // written to it, then it needs more.
284 // The only other possibility is that it has returned false from a
285 // write() call, so we wait for the next drain to continue.
286 const re = this[READENTRY];
287 const drainNow = !re || re.flowing || re.size === re.remain;
288 if (drainNow) {
289 if (!this[WRITING]) {
290 this.emit('drain');
291 }
292 }
293 else {
294 re.once('drain', () => this.emit('drain'));
295 }
296 }
297 }
298 [CONSUMEBODY](chunk, position) {
299 // write up to but no more than writeEntry.blockRemain
300 const entry = this[WRITEENTRY];
301 /* c8 ignore start */
302 if (!entry) {
303 throw new Error('attempt to consume body without entry??');
304 }
305 const br = entry.blockRemain ?? 0;
306 /* c8 ignore stop */
307 const c = br >= chunk.length && position === 0 ?
308 chunk
309 : chunk.subarray(position, position + br);
310 entry.write(c);
311 if (!entry.blockRemain) {
312 this[STATE] = 'header';
313 this[WRITEENTRY] = undefined;
314 entry.end();
315 }
316 return c.length;
317 }
318 [CONSUMEMETA](chunk, position) {
319 const entry = this[WRITEENTRY];
320 const ret = this[CONSUMEBODY](chunk, position);
321 // if we finished, then the entry is reset
322 if (!this[WRITEENTRY] && entry) {
323 this[EMITMETA](entry);
324 }
325 return ret;
326 }
327 [EMIT](ev, data, extra) {
328 if (this[QUEUE].length === 0 && !this[READENTRY]) {
329 this.emit(ev, data, extra);
330 }
331 else {
332 this[QUEUE].push([ev, data, extra]);
333 }
334 }
335 [EMITMETA](entry) {
336 this[EMIT]('meta', this[META]);
337 switch (entry.type) {
338 case 'ExtendedHeader':
339 case 'OldExtendedHeader':
340 this[EX] = pax_js_1.Pax.parse(this[META], this[EX], false);
341 break;
342 case 'GlobalExtendedHeader':
343 this[GEX] = pax_js_1.Pax.parse(this[META], this[GEX], true);
344 break;
345 case 'NextFileHasLongPath':
346 case 'OldGnuLongPath': {
347 const ex = this[EX] ?? Object.create(null);
348 this[EX] = ex;
349 ex.path = this[META].replace(/\0.*/, '');
350 break;
351 }
352 case 'NextFileHasLongLinkpath': {
353 const ex = this[EX] || Object.create(null);
354 this[EX] = ex;
355 ex.linkpath = this[META].replace(/\0.*/, '');
356 break;
357 }
358 /* c8 ignore start */
359 default:
360 throw new Error('unknown meta: ' + entry.type);
361 /* c8 ignore stop */
362 }
363 }
364 abort(error) {
365 this[ABORTED] = true;
366 this.emit('abort', error);
367 // always throws, even in non-strict mode
368 this.warn('TAR_ABORT', error, { recoverable: false });
369 }
370 write(chunk, encoding, cb) {
371 if (typeof encoding === 'function') {
372 cb = encoding;
373 encoding = undefined;
374 }
375 if (typeof chunk === 'string') {
376 chunk = Buffer.from(chunk,
377 /* c8 ignore next */
378 typeof encoding === 'string' ? encoding : 'utf8');
379 }
380 if (this[ABORTED]) {
381 /* c8 ignore next */
382 cb?.();
383 return false;
384 }
385 // first write, might be gzipped, zstd, or brotli compressed
386 const needSniff = this[UNZIP] === undefined ||
387 (this.brotli === undefined && this[UNZIP] === false);
388 if (needSniff && chunk) {
389 if (this[BUFFER]) {
390 chunk = Buffer.concat([this[BUFFER], chunk]);
391 this[BUFFER] = undefined;
392 }
393 if (chunk.length < ZIP_HEADER_LEN) {
394 this[BUFFER] = chunk;
395 /* c8 ignore next */
396 cb?.();
397 return true;
398 }
399 // look for gzip header
400 for (let i = 0; this[UNZIP] === undefined && i < gzipHeader.length; i++) {
401 if (chunk[i] !== gzipHeader[i]) {
402 this[UNZIP] = false;
403 }
404 }
405 // look for zstd header if gzip header not found
406 let isZstd = false;
407 if (this[UNZIP] === false && this.zstd !== false) {
408 isZstd = true;
409 for (let i = 0; i < zstdHeader.length; i++) {
410 if (chunk[i] !== zstdHeader[i]) {
411 isZstd = false;
412 break;
413 }
414 }
415 }
416 const maybeBrotli = this.brotli === undefined && !isZstd;
417 if (this[UNZIP] === false && maybeBrotli) {
418 // read the first header to see if it's a valid tar file. If so,
419 // we can safely assume that it's not actually brotli, despite the
420 // .tbr or .tar.br file extension.
421 // if we ended before getting a full chunk, yes, def brotli
422 if (chunk.length < 512) {
423 if (this[ENDED]) {
424 this.brotli = true;
425 }
426 else {
427 this[BUFFER] = chunk;
428 /* c8 ignore next */
429 cb?.();
430 return true;
431 }
432 }
433 else {
434 // if it's tar, it's pretty reliably not brotli, chances of
435 // that happening are astronomical.
436 try {
437 new header_js_1.Header(chunk.subarray(0, 512));
438 this.brotli = false;
439 }
440 catch (_) {
441 this.brotli = true;
442 }
443 }
444 }
445 if (this[UNZIP] === undefined ||
446 (this[UNZIP] === false && (this.brotli || isZstd))) {
447 const ended = this[ENDED];
448 this[ENDED] = false;
449 this[UNZIP] =
450 this[UNZIP] === undefined ? new minizlib_1.Unzip({})
451 : isZstd ? new minizlib_1.ZstdDecompress({})
452 : new minizlib_1.BrotliDecompress({});
453 this[UNZIP].on('data', chunk => this[CONSUMECHUNK](chunk));
454 this[UNZIP].on('error', er => this.abort(er));
455 this[UNZIP].on('end', () => {
456 this[ENDED] = true;
457 this[CONSUMECHUNK]();
458 });
459 this[WRITING] = true;
460 const ret = !!this[UNZIP][ended ? 'end' : 'write'](chunk);
461 this[WRITING] = false;
462 cb?.();
463 return ret;
464 }
465 }
466 this[WRITING] = true;
467 if (this[UNZIP]) {
468 this[UNZIP].write(chunk);
469 }
470 else {
471 this[CONSUMECHUNK](chunk);
472 }
473 this[WRITING] = false;
474 // return false if there's a queue, or if the current entry isn't flowing
475 const ret = this[QUEUE].length > 0 ? false
476 : this[READENTRY] ? this[READENTRY].flowing
477 : true;
478 // if we have no queue, then that means a clogged READENTRY
479 if (!ret && this[QUEUE].length === 0) {
480 this[READENTRY]?.once('drain', () => this.emit('drain'));
481 }
482 /* c8 ignore next */
483 cb?.();
484 return ret;
485 }
486 [BUFFERCONCAT](c) {
487 if (c && !this[ABORTED]) {
488 this[BUFFER] = this[BUFFER] ? Buffer.concat([this[BUFFER], c]) : c;
489 }
490 }
491 [MAYBEEND]() {
492 if (this[ENDED] &&
493 !this[EMITTEDEND] &&
494 !this[ABORTED] &&
495 !this[CONSUMING]) {
496 this[EMITTEDEND] = true;
497 const entry = this[WRITEENTRY];
498 if (entry && entry.blockRemain) {
499 // truncated, likely a damaged file
500 const have = this[BUFFER] ? this[BUFFER].length : 0;
501 this.warn('TAR_BAD_ARCHIVE', `Truncated input (needed ${entry.blockRemain} more bytes, only ${have} available)`, { entry });
502 if (this[BUFFER]) {
503 entry.write(this[BUFFER]);
504 }
505 entry.end();
506 }
507 this[EMIT](DONE);
508 }
509 }
510 [CONSUMECHUNK](chunk) {
511 if (this[CONSUMING] && chunk) {
512 this[BUFFERCONCAT](chunk);
513 }
514 else if (!chunk && !this[BUFFER]) {
515 this[MAYBEEND]();
516 }
517 else if (chunk) {
518 this[CONSUMING] = true;
519 if (this[BUFFER]) {
520 this[BUFFERCONCAT](chunk);
521 const c = this[BUFFER];
522 this[BUFFER] = undefined;
523 this[CONSUMECHUNKSUB](c);
524 }
525 else {
526 this[CONSUMECHUNKSUB](chunk);
527 }
528 while (this[BUFFER] &&
529 this[BUFFER]?.length >= 512 &&
530 !this[ABORTED] &&
531 !this[SAW_EOF]) {
532 const c = this[BUFFER];
533 this[BUFFER] = undefined;
534 this[CONSUMECHUNKSUB](c);
535 }
536 this[CONSUMING] = false;
537 }
538 if (!this[BUFFER] || this[ENDED]) {
539 this[MAYBEEND]();
540 }
541 }
542 [CONSUMECHUNKSUB](chunk) {
543 // we know that we are in CONSUMING mode, so anything written goes into
544 // the buffer. Advance the position and put any remainder in the buffer.
545 let position = 0;
546 const length = chunk.length;
547 while (position + 512 <= length && !this[ABORTED] && !this[SAW_EOF]) {
548 switch (this[STATE]) {
549 case 'begin':
550 case 'header':
551 this[CONSUMEHEADER](chunk, position);
552 position += 512;
553 break;
554 case 'ignore':
555 case 'body':
556 position += this[CONSUMEBODY](chunk, position);
557 break;
558 case 'meta':
559 position += this[CONSUMEMETA](chunk, position);
560 break;
561 /* c8 ignore start */
562 default:
563 throw new Error('invalid state: ' + this[STATE]);
564 /* c8 ignore stop */
565 }
566 }
567 if (position < length) {
568 this[BUFFER] =
569 this[BUFFER] ?
570 Buffer.concat([chunk.subarray(position), this[BUFFER]])
571 : chunk.subarray(position);
572 }
573 }
574 end(chunk, encoding, cb) {
575 if (typeof chunk === 'function') {
576 cb = chunk;
577 encoding = undefined;
578 chunk = undefined;
579 }
580 if (typeof encoding === 'function') {
581 cb = encoding;
582 encoding = undefined;
583 }
584 if (typeof chunk === 'string') {
585 chunk = Buffer.from(chunk, encoding);
586 }
587 if (cb)
588 this.once('finish', cb);
589 if (!this[ABORTED]) {
590 if (this[UNZIP]) {
591 /* c8 ignore start */
592 if (chunk)
593 this[UNZIP].write(chunk);
594 /* c8 ignore stop */
595 this[UNZIP].end();
596 }
597 else {
598 this[ENDED] = true;
599 if (this.brotli === undefined || this.zstd === undefined)
600 chunk = chunk || Buffer.alloc(0);
601 if (chunk)
602 this.write(chunk);
603 this[MAYBEEND]();
604 }
605 }
606 return this;
607 }
608}
609exports.Parser = Parser;
610//# sourceMappingURL=parse.js.map