Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
data-url.js745 linesDownload Raw Back to fetch
1'use strict'
2
3const assert = require('node:assert')
4
5const encoder = new TextEncoder()
6
7/**
8 * @see https://mimesniff.spec.whatwg.org/#http-token-code-point
9 */
10const HTTP_TOKEN_CODEPOINTS = /^[!#$%&'*+\-.^_|~A-Za-z0-9]+$/
11const HTTP_WHITESPACE_REGEX = /[\u000A\u000D\u0009\u0020]/ // eslint-disable-line
12const ASCII_WHITESPACE_REPLACE_REGEX = /[\u0009\u000A\u000C\u000D\u0020]/g // eslint-disable-line
13/**
14 * @see https://mimesniff.spec.whatwg.org/#http-quoted-string-token-code-point
15 */
16const HTTP_QUOTED_STRING_TOKENS = /^[\u0009\u0020-\u007E\u0080-\u00FF]+$/ // eslint-disable-line
17
18// https://fetch.spec.whatwg.org/#data-url-processor
19/** @param {URL} dataURL */
20function dataURLProcessor (dataURL) {
21  // 1. Assert: dataURL’s scheme is "data".
22  assert(dataURL.protocol === 'data:')
23
24  // 2. Let input be the result of running the URL
25  // serializer on dataURL with exclude fragment
26  // set to true.
27  let input = URLSerializer(dataURL, true)
28
29  // 3. Remove the leading "data:" string from input.
30  input = input.slice(5)
31
32  // 4. Let position point at the start of input.
33  const position = { position: 0 }
34
35  // 5. Let mimeType be the result of collecting a
36  // sequence of code points that are not equal
37  // to U+002C (,), given position.
38  let mimeType = collectASequenceOfCodePointsFast(
39    ',',
40    input,
41    position
42  )
43
44  // 6. Strip leading and trailing ASCII whitespace
45  // from mimeType.
46  // Undici implementation note: we need to store the
47  // length because if the mimetype has spaces removed,
48  // the wrong amount will be sliced from the input in
49  // step #9
50  const mimeTypeLength = mimeType.length
51  mimeType = removeASCIIWhitespace(mimeType, true, true)
52
53  // 7. If position is past the end of input, then
54  // return failure
55  if (position.position >= input.length) {
56    return 'failure'
57  }
58
59  // 8. Advance position by 1.
60  position.position++
61
62  // 9. Let encodedBody be the remainder of input.
63  const encodedBody = input.slice(mimeTypeLength + 1)
64
65  // 10. Let body be the percent-decoding of encodedBody.
66  let body = stringPercentDecode(encodedBody)
67
68  // 11. If mimeType ends with U+003B (;), followed by
69  // zero or more U+0020 SPACE, followed by an ASCII
70  // case-insensitive match for "base64", then:
71  if (/;(\u0020){0,}base64$/i.test(mimeType)) {
72    // 1. Let stringBody be the isomorphic decode of body.
73    const stringBody = isomorphicDecode(body)
74
75    // 2. Set body to the forgiving-base64 decode of
76    // stringBody.
77    body = forgivingBase64(stringBody)
78
79    // 3. If body is failure, then return failure.
80    if (body === 'failure') {
81      return 'failure'
82    }
83
84    // 4. Remove the last 6 code points from mimeType.
85    mimeType = mimeType.slice(0, -6)
86
87    // 5. Remove trailing U+0020 SPACE code points from mimeType,
88    // if any.
89    mimeType = mimeType.replace(/(\u0020)+$/, '')
90
91    // 6. Remove the last U+003B (;) code point from mimeType.
92    mimeType = mimeType.slice(0, -1)
93  }
94
95  // 12. If mimeType starts with U+003B (;), then prepend
96  // "text/plain" to mimeType.
97  if (mimeType.startsWith(';')) {
98    mimeType = 'text/plain' + mimeType
99  }
100
101  // 13. Let mimeTypeRecord be the result of parsing
102  // mimeType.
103  let mimeTypeRecord = parseMIMEType(mimeType)
104
105  // 14. If mimeTypeRecord is failure, then set
106  // mimeTypeRecord to text/plain;charset=US-ASCII.
107  if (mimeTypeRecord === 'failure') {
108    mimeTypeRecord = parseMIMEType('text/plain;charset=US-ASCII')
109  }
110
111  // 15. Return a new data: URL struct whose MIME
112  // type is mimeTypeRecord and body is body.
113  // https://fetch.spec.whatwg.org/#data-url-struct
114  return { mimeType: mimeTypeRecord, body }
115}
116
117// https://url.spec.whatwg.org/#concept-url-serializer
118/**
119 * @param {URL} url
120 * @param {boolean} excludeFragment
121 */
122function URLSerializer (url, excludeFragment = false) {
123  if (!excludeFragment) {
124    return url.href
125  }
126
127  const href = url.href
128  const hashLength = url.hash.length
129
130  const serialized = hashLength === 0 ? href : href.substring(0, href.length - hashLength)
131
132  if (!hashLength && href.endsWith('#')) {
133    return serialized.slice(0, -1)
134  }
135
136  return serialized
137}
138
139// https://infra.spec.whatwg.org/#collect-a-sequence-of-code-points
140/**
141 * @param {(char: string) => boolean} condition
142 * @param {string} input
143 * @param {{ position: number }} position
144 */
145function collectASequenceOfCodePoints (condition, input, position) {
146  // 1. Let result be the empty string.
147  let result = ''
148
149  // 2. While position doesn’t point past the end of input and the
150  // code point at position within input meets the condition condition:
151  while (position.position < input.length && condition(input[position.position])) {
152    // 1. Append that code point to the end of result.
153    result += input[position.position]
154
155    // 2. Advance position by 1.
156    position.position++
157  }
158
159  // 3. Return result.
160  return result
161}
162
163/**
164 * A faster collectASequenceOfCodePoints that only works when comparing a single character.
165 * @param {string} char
166 * @param {string} input
167 * @param {{ position: number }} position
168 */
169function collectASequenceOfCodePointsFast (char, input, position) {
170  const idx = input.indexOf(char, position.position)
171  const start = position.position
172
173  if (idx === -1) {
174    position.position = input.length
175    return input.slice(start)
176  }
177
178  position.position = idx
179  return input.slice(start, position.position)
180}
181
182// https://url.spec.whatwg.org/#string-percent-decode
183/** @param {string} input */
184function stringPercentDecode (input) {
185  // 1. Let bytes be the UTF-8 encoding of input.
186  const bytes = encoder.encode(input)
187
188  // 2. Return the percent-decoding of bytes.
189  return percentDecode(bytes)
190}
191
192/**
193 * @param {number} byte
194 */
195function isHexCharByte (byte) {
196  // 0-9 A-F a-f
197  return (byte >= 0x30 && byte <= 0x39) || (byte >= 0x41 && byte <= 0x46) || (byte >= 0x61 && byte <= 0x66)
198}
199
200/**
201 * @param {number} byte
202 */
203function hexByteToNumber (byte) {
204  return (
205    // 0-9
206    byte >= 0x30 && byte <= 0x39
207      ? (byte - 48)
208    // Convert to uppercase
209    // ((byte & 0xDF) - 65) + 10
210      : ((byte & 0xDF) - 55)
211  )
212}
213
214// https://url.spec.whatwg.org/#percent-decode
215/** @param {Uint8Array} input */
216function percentDecode (input) {
217  const length = input.length
218  // 1. Let output be an empty byte sequence.
219  /** @type {Uint8Array} */
220  const output = new Uint8Array(length)
221  let j = 0
222  // 2. For each byte byte in input:
223  for (let i = 0; i < length; ++i) {
224    const byte = input[i]
225
226    // 1. If byte is not 0x25 (%), then append byte to output.
227    if (byte !== 0x25) {
228      output[j++] = byte
229
230    // 2. Otherwise, if byte is 0x25 (%) and the next two bytes
231    // after byte in input are not in the ranges
232    // 0x30 (0) to 0x39 (9), 0x41 (A) to 0x46 (F),
233    // and 0x61 (a) to 0x66 (f), all inclusive, append byte
234    // to output.
235    } else if (
236      byte === 0x25 &&
237      !(isHexCharByte(input[i + 1]) && isHexCharByte(input[i + 2]))
238    ) {
239      output[j++] = 0x25
240
241    // 3. Otherwise:
242    } else {
243      // 1. Let bytePoint be the two bytes after byte in input,
244      // decoded, and then interpreted as hexadecimal number.
245      // 2. Append a byte whose value is bytePoint to output.
246      output[j++] = (hexByteToNumber(input[i + 1]) << 4) | hexByteToNumber(input[i + 2])
247
248      // 3. Skip the next two bytes in input.
249      i += 2
250    }
251  }
252
253  // 3. Return output.
254  return length === j ? output : output.subarray(0, j)
255}
256
257// https://mimesniff.spec.whatwg.org/#parse-a-mime-type
258/** @param {string} input */
259function parseMIMEType (input) {
260  // 1. Remove any leading and trailing HTTP whitespace
261  // from input.
262  input = removeHTTPWhitespace(input, true, true)
263
264  // 2. Let position be a position variable for input,
265  // initially pointing at the start of input.
266  const position = { position: 0 }
267
268  // 3. Let type be the result of collecting a sequence
269  // of code points that are not U+002F (/) from
270  // input, given position.
271  const type = collectASequenceOfCodePointsFast(
272    '/',
273    input,
274    position
275  )
276
277  // 4. If type is the empty string or does not solely
278  // contain HTTP token code points, then return failure.
279  // https://mimesniff.spec.whatwg.org/#http-token-code-point
280  if (type.length === 0 || !HTTP_TOKEN_CODEPOINTS.test(type)) {
281    return 'failure'
282  }
283
284  // 5. If position is past the end of input, then return
285  // failure
286  if (position.position > input.length) {
287    return 'failure'
288  }
289
290  // 6. Advance position by 1. (This skips past U+002F (/).)
291  position.position++
292
293  // 7. Let subtype be the result of collecting a sequence of
294  // code points that are not U+003B (;) from input, given
295  // position.
296  let subtype = collectASequenceOfCodePointsFast(
297    ';',
298    input,
299    position
300  )
301
302  // 8. Remove any trailing HTTP whitespace from subtype.
303  subtype = removeHTTPWhitespace(subtype, false, true)
304
305  // 9. If subtype is the empty string or does not solely
306  // contain HTTP token code points, then return failure.
307  if (subtype.length === 0 || !HTTP_TOKEN_CODEPOINTS.test(subtype)) {
308    return 'failure'
309  }
310
311  const typeLowercase = type.toLowerCase()
312  const subtypeLowercase = subtype.toLowerCase()
313
314  // 10. Let mimeType be a new MIME type record whose type
315  // is type, in ASCII lowercase, and subtype is subtype,
316  // in ASCII lowercase.
317  // https://mimesniff.spec.whatwg.org/#mime-type
318  const mimeType = {
319    type: typeLowercase,
320    subtype: subtypeLowercase,
321    /** @type {Map<string, string>} */
322    parameters: new Map(),
323    // https://mimesniff.spec.whatwg.org/#mime-type-essence
324    essence: `${typeLowercase}/${subtypeLowercase}`
325  }
326
327  // 11. While position is not past the end of input:
328  while (position.position < input.length) {
329    // 1. Advance position by 1. (This skips past U+003B (;).)
330    position.position++
331
332    // 2. Collect a sequence of code points that are HTTP
333    // whitespace from input given position.
334    collectASequenceOfCodePoints(
335      // https://fetch.spec.whatwg.org/#http-whitespace
336      char => HTTP_WHITESPACE_REGEX.test(char),
337      input,
338      position
339    )
340
341    // 3. Let parameterName be the result of collecting a
342    // sequence of code points that are not U+003B (;)
343    // or U+003D (=) from input, given position.
344    let parameterName = collectASequenceOfCodePoints(
345      (char) => char !== ';' && char !== '=',
346      input,
347      position
348    )
349
350    // 4. Set parameterName to parameterName, in ASCII
351    // lowercase.
352    parameterName = parameterName.toLowerCase()
353
354    // 5. If position is not past the end of input, then:
355    if (position.position < input.length) {
356      // 1. If the code point at position within input is
357      // U+003B (;), then continue.
358      if (input[position.position] === ';') {
359        continue
360      }
361
362      // 2. Advance position by 1. (This skips past U+003D (=).)
363      position.position++
364    }
365
366    // 6. If position is past the end of input, then break.
367    if (position.position > input.length) {
368      break
369    }
370
371    // 7. Let parameterValue be null.
372    let parameterValue = null
373
374    // 8. If the code point at position within input is
375    // U+0022 ("), then:
376    if (input[position.position] === '"') {
377      // 1. Set parameterValue to the result of collecting
378      // an HTTP quoted string from input, given position
379      // and the extract-value flag.
380      parameterValue = collectAnHTTPQuotedString(input, position, true)
381
382      // 2. Collect a sequence of code points that are not
383      // U+003B (;) from input, given position.
384      collectASequenceOfCodePointsFast(
385        ';',
386        input,
387        position
388      )
389
390    // 9. Otherwise:
391    } else {
392      // 1. Set parameterValue to the result of collecting
393      // a sequence of code points that are not U+003B (;)
394      // from input, given position.
395      parameterValue = collectASequenceOfCodePointsFast(
396        ';',
397        input,
398        position
399      )
400
401      // 2. Remove any trailing HTTP whitespace from parameterValue.
402      parameterValue = removeHTTPWhitespace(parameterValue, false, true)
403
404      // 3. If parameterValue is the empty string, then continue.
405      if (parameterValue.length === 0) {
406        continue
407      }
408    }
409
410    // 10. If all of the following are true
411    // - parameterName is not the empty string
412    // - parameterName solely contains HTTP token code points
413    // - parameterValue solely contains HTTP quoted-string token code points
414    // - mimeType’s parameters[parameterName] does not exist
415    // then set mimeType’s parameters[parameterName] to parameterValue.
416    if (
417      parameterName.length !== 0 &&
418      HTTP_TOKEN_CODEPOINTS.test(parameterName) &&
419      (parameterValue.length === 0 || HTTP_QUOTED_STRING_TOKENS.test(parameterValue)) &&
420      !mimeType.parameters.has(parameterName)
421    ) {
422      mimeType.parameters.set(parameterName, parameterValue)
423    }
424  }
425
426  // 12. Return mimeType.
427  return mimeType
428}
429
430// https://infra.spec.whatwg.org/#forgiving-base64-decode
431/** @param {string} data */
432function forgivingBase64 (data) {
433  // 1. Remove all ASCII whitespace from data.
434  data = data.replace(ASCII_WHITESPACE_REPLACE_REGEX, '')  // eslint-disable-line
435
436  let dataLength = data.length
437  // 2. If data’s code point length divides by 4 leaving
438  // no remainder, then:
439  if (dataLength % 4 === 0) {
440    // 1. If data ends with one or two U+003D (=) code points,
441    // then remove them from data.
442    if (data.charCodeAt(dataLength - 1) === 0x003D) {
443      --dataLength
444      if (data.charCodeAt(dataLength - 1) === 0x003D) {
445        --dataLength
446      }
447    }
448  }
449
450  // 3. If data’s code point length divides by 4 leaving
451  // a remainder of 1, then return failure.
452  if (dataLength % 4 === 1) {
453    return 'failure'
454  }
455
456  // 4. If data contains a code point that is not one of
457  //  U+002B (+)
458  //  U+002F (/)
459  //  ASCII alphanumeric
460  // then return failure.
461  if (/[^+/0-9A-Za-z]/.test(data.length === dataLength ? data : data.substring(0, dataLength))) {
462    return 'failure'
463  }
464
465  const buffer = Buffer.from(data, 'base64')
466  return new Uint8Array(buffer.buffer, buffer.byteOffset, buffer.byteLength)
467}
468
469// https://fetch.spec.whatwg.org/#collect-an-http-quoted-string
470// tests: https://fetch.spec.whatwg.org/#example-http-quoted-string
471/**
472 * @param {string} input
473 * @param {{ position: number }} position
474 * @param {boolean?} extractValue
475 */
476function collectAnHTTPQuotedString (input, position, extractValue) {
477  // 1. Let positionStart be position.
478  const positionStart = position.position
479
480  // 2. Let value be the empty string.
481  let value = ''
482
483  // 3. Assert: the code point at position within input
484  // is U+0022 (").
485  assert(input[position.position] === '"')
486
487  // 4. Advance position by 1.
488  position.position++
489
490  // 5. While true:
491  while (true) {
492    // 1. Append the result of collecting a sequence of code points
493    // that are not U+0022 (") or U+005C (\) from input, given
494    // position, to value.
495    value += collectASequenceOfCodePoints(
496      (char) => char !== '"' && char !== '\\',
497      input,
498      position
499    )
500
501    // 2. If position is past the end of input, then break.
502    if (position.position >= input.length) {
503      break
504    }
505
506    // 3. Let quoteOrBackslash be the code point at position within
507    // input.
508    const quoteOrBackslash = input[position.position]
509
510    // 4. Advance position by 1.
511    position.position++
512
513    // 5. If quoteOrBackslash is U+005C (\), then:
514    if (quoteOrBackslash === '\\') {
515      // 1. If position is past the end of input, then append
516      // U+005C (\) to value and break.
517      if (position.position >= input.length) {
518        value += '\\'
519        break
520      }
521
522      // 2. Append the code point at position within input to value.
523      value += input[position.position]
524
525      // 3. Advance position by 1.
526      position.position++
527
528    // 6. Otherwise:
529    } else {
530      // 1. Assert: quoteOrBackslash is U+0022 (").
531      assert(quoteOrBackslash === '"')
532
533      // 2. Break.
534      break
535    }
536  }
537
538  // 6. If the extract-value flag is set, then return value.
539  if (extractValue) {
540    return value
541  }
542
543  // 7. Return the code points from positionStart to position,
544  // inclusive, within input.
545  return input.slice(positionStart, position.position)
546}
547
548/**
549 * @see https://mimesniff.spec.whatwg.org/#serialize-a-mime-type
550 */
551function serializeAMimeType (mimeType) {
552  assert(mimeType !== 'failure')
553  const { parameters, essence } = mimeType
554
555  // 1. Let serialization be the concatenation of mimeType’s
556  //    type, U+002F (/), and mimeType’s subtype.
557  let serialization = essence
558
559  // 2. For each name → value of mimeType’s parameters:
560  for (let [name, value] of parameters.entries()) {
561    // 1. Append U+003B (;) to serialization.
562    serialization += ';'
563
564    // 2. Append name to serialization.
565    serialization += name
566
567    // 3. Append U+003D (=) to serialization.
568    serialization += '='
569
570    // 4. If value does not solely contain HTTP token code
571    //    points or value is the empty string, then:
572    if (!HTTP_TOKEN_CODEPOINTS.test(value)) {
573      // 1. Precede each occurrence of U+0022 (") or
574      //    U+005C (\) in value with U+005C (\).
575      value = value.replace(/(\\|")/g, '\\$1')
576
577      // 2. Prepend U+0022 (") to value.
578      value = '"' + value
579
580      // 3. Append U+0022 (") to value.
581      value += '"'
582    }
583
584    // 5. Append value to serialization.
585    serialization += value
586  }
587
588  // 3. Return serialization.
589  return serialization
590}
591
592/**
593 * @see https://fetch.spec.whatwg.org/#http-whitespace
594 * @param {number} char
595 */
596function isHTTPWhiteSpace (char) {
597  // "\r\n\t "
598  return char === 0x00d || char === 0x00a || char === 0x009 || char === 0x020
599}
600
601/**
602 * @see https://fetch.spec.whatwg.org/#http-whitespace
603 * @param {string} str
604 * @param {boolean} [leading=true]
605 * @param {boolean} [trailing=true]
606 */
607function removeHTTPWhitespace (str, leading = true, trailing = true) {
608  return removeChars(str, leading, trailing, isHTTPWhiteSpace)
609}
610
611/**
612 * @see https://infra.spec.whatwg.org/#ascii-whitespace
613 * @param {number} char
614 */
615function isASCIIWhitespace (char) {
616  // "\r\n\t\f "
617  return char === 0x00d || char === 0x00a || char === 0x009 || char === 0x00c || char === 0x020
618}
619
620/**
621 * @see https://infra.spec.whatwg.org/#strip-leading-and-trailing-ascii-whitespace
622 * @param {string} str
623 * @param {boolean} [leading=true]
624 * @param {boolean} [trailing=true]
625 */
626function removeASCIIWhitespace (str, leading = true, trailing = true) {
627  return removeChars(str, leading, trailing, isASCIIWhitespace)
628}
629
630/**
631 * @param {string} str
632 * @param {boolean} leading
633 * @param {boolean} trailing
634 * @param {(charCode: number) => boolean} predicate
635 * @returns
636 */
637function removeChars (str, leading, trailing, predicate) {
638  let lead = 0
639  let trail = str.length - 1
640
641  if (leading) {
642    while (lead < str.length && predicate(str.charCodeAt(lead))) lead++
643  }
644
645  if (trailing) {
646    while (trail > 0 && predicate(str.charCodeAt(trail))) trail--
647  }
648
649  return lead === 0 && trail === str.length - 1 ? str : str.slice(lead, trail + 1)
650}
651
652/**
653 * @see https://infra.spec.whatwg.org/#isomorphic-decode
654 * @param {Uint8Array} input
655 * @returns {string}
656 */
657function isomorphicDecode (input) {
658  // 1. To isomorphic decode a byte sequence input, return a string whose code point
659  //    length is equal to input’s length and whose code points have the same values
660  //    as the values of input’s bytes, in the same order.
661  const length = input.length
662  if ((2 << 15) - 1 > length) {
663    return String.fromCharCode.apply(null, input)
664  }
665  let result = ''; let i = 0
666  let addition = (2 << 15) - 1
667  while (i < length) {
668    if (i + addition > length) {
669      addition = length - i
670    }
671    result += String.fromCharCode.apply(null, input.subarray(i, i += addition))
672  }
673  return result
674}
675
676/**
677 * @see https://mimesniff.spec.whatwg.org/#minimize-a-supported-mime-type
678 * @param {Exclude<ReturnType<typeof parseMIMEType>, 'failure'>} mimeType
679 */
680function minimizeSupportedMimeType (mimeType) {
681  switch (mimeType.essence) {
682    case 'application/ecmascript':
683    case 'application/javascript':
684    case 'application/x-ecmascript':
685    case 'application/x-javascript':
686    case 'text/ecmascript':
687    case 'text/javascript':
688    case 'text/javascript1.0':
689    case 'text/javascript1.1':
690    case 'text/javascript1.2':
691    case 'text/javascript1.3':
692    case 'text/javascript1.4':
693    case 'text/javascript1.5':
694    case 'text/jscript':
695    case 'text/livescript':
696    case 'text/x-ecmascript':
697    case 'text/x-javascript':
698      // 1. If mimeType is a JavaScript MIME type, then return "text/javascript".
699      return 'text/javascript'
700    case 'application/json':
701    case 'text/json':
702      // 2. If mimeType is a JSON MIME type, then return "application/json".
703      return 'application/json'
704    case 'image/svg+xml':
705      // 3. If mimeType’s essence is "image/svg+xml", then return "image/svg+xml".
706      return 'image/svg+xml'
707    case 'text/xml':
708    case 'application/xml':
709      // 4. If mimeType is an XML MIME type, then return "application/xml".
710      return 'application/xml'
711  }
712
713  // 2. If mimeType is a JSON MIME type, then return "application/json".
714  if (mimeType.subtype.endsWith('+json')) {
715    return 'application/json'
716  }
717
718  // 4. If mimeType is an XML MIME type, then return "application/xml".
719  if (mimeType.subtype.endsWith('+xml')) {
720    return 'application/xml'
721  }
722
723  // 5. If mimeType is supported by the user agent, then return mimeType’s essence.
724  // Technically, node doesn't support any mimetypes.
725
726  // 6. Return the empty string.
727  return ''
728}
729
730module.exports = {
731  dataURLProcessor,
732  URLSerializer,
733  collectASequenceOfCodePoints,
734  collectASequenceOfCodePointsFast,
735  stringPercentDecode,
736  parseMIMEType,
737  collectAnHTTPQuotedString,
738  serializeAMimeType,
739  removeChars,
740  removeHTTPWhitespace,
741  minimizeSupportedMimeType,
742  HTTP_TOKEN_CODEPOINTS,
743  isomorphicDecode
744}
745 
codekingpro/portable-devtools · Team Ai