@depup/file-type 21.3.4-depup.0 → 22.0.0-depup.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,338 +4,91 @@ Primary entry point, Node.js specific entry point is index.js
4
4
 
5
5
  import * as Token from 'token-types';
6
6
  import * as strtok3 from 'strtok3/core';
7
- import {ZipHandler, GzipHandler} from '@tokenizer/inflate';
8
- import {getUintBE} from 'uint8array-extras';
7
+ import {GzipHandler} from '@tokenizer/inflate';
8
+ import {concatUint8Arrays} from 'uint8array-extras';
9
9
  import {
10
10
  stringToBytes,
11
11
  tarHeaderChecksumMatches,
12
12
  uint32SyncSafeToken,
13
- } from './util.js';
13
+ } from './tokens.js';
14
14
  import {extensions, mimeTypes} from './supported.js';
15
+ import {
16
+ maximumUntrustedSkipSizeInBytes,
17
+ ParserHardLimitError,
18
+ safeIgnore,
19
+ checkBytes,
20
+ hasUnknownFileSize,
21
+ } from './parser.js';
22
+ import {detectZip} from './detectors/zip.js';
23
+ import {detectEbml} from './detectors/ebml.js';
24
+ import {detectPng} from './detectors/png.js';
25
+ import {detectAsf} from './detectors/asf.js';
15
26
 
16
27
  export const reasonableDetectionSizeInBytes = 4100; // A fair amount of file-types are detectable within this range.
17
- // Keep defensive limits small enough to avoid accidental memory spikes from untrusted inputs.
18
28
  const maximumMpegOffsetTolerance = reasonableDetectionSizeInBytes - 2;
19
- const maximumZipEntrySizeInBytes = 1024 * 1024;
20
- const maximumZipEntryCount = 1024;
21
- const maximumZipBufferedReadSizeInBytes = (2 ** 31) - 1;
22
- const maximumUntrustedSkipSizeInBytes = 16 * 1024 * 1024;
23
- const maximumUnknownSizePayloadProbeSizeInBytes = maximumZipEntrySizeInBytes;
24
- const maximumZipTextEntrySizeInBytes = maximumZipEntrySizeInBytes;
25
29
  const maximumNestedGzipDetectionSizeInBytes = maximumUntrustedSkipSizeInBytes;
26
30
  const maximumNestedGzipProbeDepth = 1;
27
31
  const unknownSizeGzipProbeTimeoutInMilliseconds = 100;
28
32
  const maximumId3HeaderSizeInBytes = maximumUntrustedSkipSizeInBytes;
29
- const maximumEbmlDocumentTypeSizeInBytes = 64;
30
- const maximumEbmlElementPayloadSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
31
- const maximumEbmlElementCount = 256;
32
- const maximumPngChunkCount = 512;
33
- const maximumPngStreamScanBudgetInBytes = maximumUntrustedSkipSizeInBytes;
34
- const maximumAsfHeaderObjectCount = 512;
35
33
  const maximumTiffTagCount = 512;
36
34
  const maximumDetectionReentryCount = 256;
37
- const maximumPngChunkSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
38
- const maximumAsfHeaderPayloadSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
39
- const maximumTiffStreamIfdOffsetInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
35
+ const maximumTiffStreamIfdOffsetInBytes = 1024 * 1024;
40
36
  const maximumTiffIfdOffsetInBytes = maximumUntrustedSkipSizeInBytes;
41
- const recoverableZipErrorMessages = new Set([
42
- 'Unexpected signature',
43
- 'Encrypted ZIP',
44
- 'Expected Central-File-Header signature',
45
- ]);
46
- const recoverableZipErrorMessagePrefixes = [
47
- 'ZIP entry count exceeds ',
48
- 'Unsupported ZIP compression method:',
49
- 'ZIP entry compressed data exceeds ',
50
- 'ZIP entry decompressed data exceeds ',
51
- 'Expected data-descriptor-signature at position ',
52
- ];
53
- const recoverableZipErrorCodes = new Set([
54
- 'Z_BUF_ERROR',
55
- 'Z_DATA_ERROR',
56
- 'ERR_INVALID_STATE',
57
- ]);
58
-
59
- class ParserHardLimitError extends Error {}
60
-
61
- function patchWebByobTokenizerClose(tokenizer) {
62
- const streamReader = tokenizer?.streamReader;
63
- if (streamReader?.constructor?.name !== 'WebStreamByobReader') {
64
- return tokenizer;
65
- }
66
-
67
- const {reader} = streamReader;
68
- const cancelAndRelease = async () => {
69
- await reader.cancel();
70
- reader.releaseLock();
71
- };
72
-
73
- streamReader.close = cancelAndRelease;
74
- streamReader.abort = async () => {
75
- streamReader.interrupted = true;
76
- await cancelAndRelease();
77
- };
78
-
79
- return tokenizer;
80
- }
81
-
82
- function getSafeBound(value, maximum, reason) {
83
- if (
84
- !Number.isFinite(value)
85
- || value < 0
86
- || value > maximum
87
- ) {
88
- throw new ParserHardLimitError(`${reason} has invalid size ${value} (maximum ${maximum} bytes)`);
89
- }
90
-
91
- return value;
92
- }
93
-
94
- async function safeIgnore(tokenizer, length, {maximumLength = maximumUntrustedSkipSizeInBytes, reason = 'skip'} = {}) {
95
- const safeLength = getSafeBound(length, maximumLength, reason);
96
- await tokenizer.ignore(safeLength);
97
- }
98
-
99
- async function safeReadBuffer(tokenizer, buffer, options, {maximumLength = buffer.length, reason = 'read'} = {}) {
100
- const length = options?.length ?? buffer.length;
101
- const safeLength = getSafeBound(length, maximumLength, reason);
102
- return tokenizer.readBuffer(buffer, {
103
- ...options,
104
- length: safeLength,
105
- });
106
- }
107
-
108
- async function decompressDeflateRawWithLimit(data, {maximumLength = maximumZipEntrySizeInBytes} = {}) {
109
- const input = new ReadableStream({
110
- start(controller) {
111
- controller.enqueue(data);
112
- controller.close();
113
- },
114
- });
115
- const output = input.pipeThrough(new DecompressionStream('deflate-raw'));
116
- const reader = output.getReader();
117
- const chunks = [];
118
- let totalLength = 0;
119
-
120
- try {
121
- for (;;) {
122
- const {done, value} = await reader.read();
123
- if (done) {
124
- break;
125
- }
126
-
127
- totalLength += value.length;
128
- if (totalLength > maximumLength) {
129
- await reader.cancel();
130
- throw new Error(`ZIP entry decompressed data exceeds ${maximumLength} bytes`);
131
- }
132
-
133
- chunks.push(value);
134
- }
135
- } finally {
136
- reader.releaseLock();
137
- }
138
37
 
139
- const uncompressedData = new Uint8Array(totalLength);
140
- let offset = 0;
141
- for (const chunk of chunks) {
142
- uncompressedData.set(chunk, offset);
143
- offset += chunk.length;
144
- }
145
-
146
- return uncompressedData;
147
- }
148
-
149
- const zipDataDescriptorSignature = 0x08_07_4B_50;
150
- const zipDataDescriptorLengthInBytes = 16;
151
- const zipDataDescriptorOverlapLengthInBytes = zipDataDescriptorLengthInBytes - 1;
152
-
153
- function findZipDataDescriptorOffset(buffer, bytesConsumed) {
154
- if (buffer.length < zipDataDescriptorLengthInBytes) {
155
- return -1;
156
- }
157
-
158
- const lastPossibleDescriptorOffset = buffer.length - zipDataDescriptorLengthInBytes;
159
- for (let index = 0; index <= lastPossibleDescriptorOffset; index++) {
160
- if (
161
- Token.UINT32_LE.get(buffer, index) === zipDataDescriptorSignature
162
- && Token.UINT32_LE.get(buffer, index + 8) === bytesConsumed + index
163
- ) {
164
- return index;
165
- }
38
+ export function normalizeSampleSize(sampleSize) {
39
+ // `sampleSize` is an explicit caller-controlled tuning knob, not untrusted file input.
40
+ // Preserve valid caller-requested probe depth here; applications must bound attacker-derived option values themselves.
41
+ if (!Number.isFinite(sampleSize)) {
42
+ return reasonableDetectionSizeInBytes;
166
43
  }
167
44
 
168
- return -1;
169
- }
170
-
171
- function isPngAncillaryChunk(type) {
172
- return (type.codePointAt(0) & 0x20) !== 0;
45
+ return Math.max(1, Math.trunc(sampleSize));
173
46
  }
174
47
 
175
- function mergeByteChunks(chunks, totalLength) {
176
- const merged = new Uint8Array(totalLength);
177
- let offset = 0;
178
-
179
- for (const chunk of chunks) {
180
- merged.set(chunk, offset);
181
- offset += chunk.length;
48
+ function normalizeMpegOffsetTolerance(mpegOffsetTolerance) {
49
+ // This value controls scan depth and therefore worst-case CPU work.
50
+ if (!Number.isFinite(mpegOffsetTolerance)) {
51
+ return 0;
182
52
  }
183
53
 
184
- return merged;
54
+ return Math.max(0, Math.min(maximumMpegOffsetTolerance, Math.trunc(mpegOffsetTolerance)));
185
55
  }
186
56
 
187
- async function readZipDataDescriptorEntryWithLimit(zipHandler, {shouldBuffer, maximumLength = maximumZipEntrySizeInBytes} = {}) {
188
- const {syncBuffer} = zipHandler;
189
- const {length: syncBufferLength} = syncBuffer;
190
- const chunks = [];
191
- let bytesConsumed = 0;
192
-
193
- for (;;) {
194
- const length = await zipHandler.tokenizer.peekBuffer(syncBuffer, {mayBeLess: true});
195
- const dataDescriptorOffset = findZipDataDescriptorOffset(syncBuffer.subarray(0, length), bytesConsumed);
196
- const retainedLength = dataDescriptorOffset >= 0
197
- ? 0
198
- : (
199
- length === syncBufferLength
200
- ? Math.min(zipDataDescriptorOverlapLengthInBytes, length - 1)
201
- : 0
202
- );
203
- const chunkLength = dataDescriptorOffset >= 0 ? dataDescriptorOffset : length - retainedLength;
204
-
205
- if (chunkLength === 0) {
206
- break;
207
- }
208
-
209
- bytesConsumed += chunkLength;
210
- if (bytesConsumed > maximumLength) {
211
- throw new Error(`ZIP entry compressed data exceeds ${maximumLength} bytes`);
212
- }
213
-
214
- if (shouldBuffer) {
215
- const data = new Uint8Array(chunkLength);
216
- await zipHandler.tokenizer.readBuffer(data);
217
- chunks.push(data);
218
- } else {
219
- await zipHandler.tokenizer.ignore(chunkLength);
220
- }
221
-
222
- if (dataDescriptorOffset >= 0) {
223
- break;
224
- }
225
- }
226
-
227
- if (!hasUnknownFileSize(zipHandler.tokenizer)) {
228
- zipHandler.knownSizeDescriptorScannedBytes += bytesConsumed;
229
- }
230
-
231
- if (!shouldBuffer) {
232
- return;
57
+ function getKnownFileSizeOrMaximum(fileSize) {
58
+ if (!Number.isFinite(fileSize)) {
59
+ return Number.MAX_SAFE_INTEGER;
233
60
  }
234
61
 
235
- return mergeByteChunks(chunks, bytesConsumed);
62
+ return Math.max(0, fileSize);
236
63
  }
237
64
 
238
- function getRemainingZipScanBudget(zipHandler, startOffset) {
239
- if (hasUnknownFileSize(zipHandler.tokenizer)) {
240
- return Math.max(0, maximumUntrustedSkipSizeInBytes - (zipHandler.tokenizer.position - startOffset));
241
- }
242
-
243
- return Math.max(0, maximumZipEntrySizeInBytes - zipHandler.knownSizeDescriptorScannedBytes);
65
+ // Wrap stream in an identity TransformStream to avoid BYOB readers.
66
+ // Node.js has a bug where calling controller.close() inside a BYOB stream's
67
+ // pull() callback does not resolve pending reader.read() calls, causing
68
+ // permanent hangs on streams shorter than the requested read size.
69
+ // Using a default (non-BYOB) reader via TransformStream avoids this.
70
+ function toDefaultStream(stream) {
71
+ return stream.pipeThrough(new TransformStream());
244
72
  }
245
73
 
246
- async function readZipEntryData(zipHandler, zipHeader, {shouldBuffer, maximumDescriptorLength = maximumZipEntrySizeInBytes} = {}) {
247
- if (
248
- zipHeader.dataDescriptor
249
- && zipHeader.compressedSize === 0
250
- ) {
251
- return readZipDataDescriptorEntryWithLimit(zipHandler, {
252
- shouldBuffer,
253
- maximumLength: maximumDescriptorLength,
254
- });
255
- }
256
-
257
- if (!shouldBuffer) {
258
- await safeIgnore(zipHandler.tokenizer, zipHeader.compressedSize, {
259
- maximumLength: hasUnknownFileSize(zipHandler.tokenizer) ? maximumZipEntrySizeInBytes : zipHandler.tokenizer.fileInfo.size,
260
- reason: 'ZIP entry compressed data',
261
- });
262
- return;
74
+ function readWithSignal(reader, signal) {
75
+ if (signal === undefined) {
76
+ return reader.read();
263
77
  }
264
78
 
265
- const maximumLength = getMaximumZipBufferedReadLength(zipHandler.tokenizer);
266
- if (
267
- !Number.isFinite(zipHeader.compressedSize)
268
- || zipHeader.compressedSize < 0
269
- || zipHeader.compressedSize > maximumLength
270
- ) {
271
- throw new Error(`ZIP entry compressed data exceeds ${maximumLength} bytes`);
272
- }
79
+ signal.throwIfAborted();
273
80
 
274
- const fileData = new Uint8Array(zipHeader.compressedSize);
275
- await zipHandler.tokenizer.readBuffer(fileData);
276
- return fileData;
81
+ return Promise.race([
82
+ reader.read(),
83
+ new Promise((_resolve, reject) => {
84
+ signal.addEventListener('abort', () => {
85
+ reject(signal.reason);
86
+ reader.cancel(signal.reason).catch(() => {});
87
+ }, {once: true});
88
+ }),
89
+ ]);
277
90
  }
278
91
 
279
- // Override the default inflate to enforce decompression size limits, since @tokenizer/inflate does not expose a configuration hook for this.
280
- ZipHandler.prototype.inflate = async function (zipHeader, fileData, callback) {
281
- if (zipHeader.compressedMethod === 0) {
282
- return callback(fileData);
283
- }
284
-
285
- if (zipHeader.compressedMethod !== 8) {
286
- throw new Error(`Unsupported ZIP compression method: ${zipHeader.compressedMethod}`);
287
- }
288
-
289
- const uncompressedData = await decompressDeflateRawWithLimit(fileData, {maximumLength: maximumZipEntrySizeInBytes});
290
- return callback(uncompressedData);
291
- };
292
-
293
- ZipHandler.prototype.unzip = async function (fileCallback) {
294
- let stop = false;
295
- let zipEntryCount = 0;
296
- const zipScanStart = this.tokenizer.position;
297
- this.knownSizeDescriptorScannedBytes = 0;
298
- do {
299
- if (hasExceededUnknownSizeScanBudget(this.tokenizer, zipScanStart, maximumUntrustedSkipSizeInBytes)) {
300
- throw new ParserHardLimitError(`ZIP stream probing exceeds ${maximumUntrustedSkipSizeInBytes} bytes`);
301
- }
302
-
303
- const zipHeader = await this.readLocalFileHeader();
304
- if (!zipHeader) {
305
- break;
306
- }
307
-
308
- zipEntryCount++;
309
- if (zipEntryCount > maximumZipEntryCount) {
310
- throw new Error(`ZIP entry count exceeds ${maximumZipEntryCount}`);
311
- }
312
-
313
- const next = fileCallback(zipHeader);
314
- stop = Boolean(next.stop);
315
- await this.tokenizer.ignore(zipHeader.extraFieldLength);
316
- const fileData = await readZipEntryData(this, zipHeader, {
317
- shouldBuffer: Boolean(next.handler),
318
- maximumDescriptorLength: Math.min(maximumZipEntrySizeInBytes, getRemainingZipScanBudget(this, zipScanStart)),
319
- });
320
-
321
- if (next.handler) {
322
- await this.inflate(zipHeader, fileData, next.handler);
323
- }
324
-
325
- if (zipHeader.dataDescriptor) {
326
- const dataDescriptor = new Uint8Array(zipDataDescriptorLengthInBytes);
327
- await this.tokenizer.readBuffer(dataDescriptor);
328
- if (Token.UINT32_LE.get(dataDescriptor, 0) !== zipDataDescriptorSignature) {
329
- throw new Error(`Expected data-descriptor-signature at position ${this.tokenizer.position - dataDescriptor.length}`);
330
- }
331
- }
332
-
333
- if (hasExceededUnknownSizeScanBudget(this.tokenizer, zipScanStart, maximumUntrustedSkipSizeInBytes)) {
334
- throw new ParserHardLimitError(`ZIP stream probing exceeds ${maximumUntrustedSkipSizeInBytes} bytes`);
335
- }
336
- } while (!stop);
337
- };
338
-
339
92
  function createByteLimitedReadableStream(stream, maximumBytes) {
340
93
  const reader = stream.getReader();
341
94
  let emittedBytes = 0;
@@ -402,388 +155,6 @@ export async function fileTypeFromBlob(blob, options) {
402
155
  return new FileTypeParser(options).fromBlob(blob);
403
156
  }
404
157
 
405
- function getFileTypeFromMimeType(mimeType) {
406
- mimeType = mimeType.toLowerCase();
407
- switch (mimeType) {
408
- case 'application/epub+zip':
409
- return {
410
- ext: 'epub',
411
- mime: mimeType,
412
- };
413
- case 'application/vnd.oasis.opendocument.text':
414
- return {
415
- ext: 'odt',
416
- mime: mimeType,
417
- };
418
- case 'application/vnd.oasis.opendocument.text-template':
419
- return {
420
- ext: 'ott',
421
- mime: mimeType,
422
- };
423
- case 'application/vnd.oasis.opendocument.spreadsheet':
424
- return {
425
- ext: 'ods',
426
- mime: mimeType,
427
- };
428
- case 'application/vnd.oasis.opendocument.spreadsheet-template':
429
- return {
430
- ext: 'ots',
431
- mime: mimeType,
432
- };
433
- case 'application/vnd.oasis.opendocument.presentation':
434
- return {
435
- ext: 'odp',
436
- mime: mimeType,
437
- };
438
- case 'application/vnd.oasis.opendocument.presentation-template':
439
- return {
440
- ext: 'otp',
441
- mime: mimeType,
442
- };
443
- case 'application/vnd.oasis.opendocument.graphics':
444
- return {
445
- ext: 'odg',
446
- mime: mimeType,
447
- };
448
- case 'application/vnd.oasis.opendocument.graphics-template':
449
- return {
450
- ext: 'otg',
451
- mime: mimeType,
452
- };
453
- case 'application/vnd.openxmlformats-officedocument.presentationml.slideshow':
454
- return {
455
- ext: 'ppsx',
456
- mime: mimeType,
457
- };
458
- case 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet':
459
- return {
460
- ext: 'xlsx',
461
- mime: mimeType,
462
- };
463
- case 'application/vnd.ms-excel.sheet.macroenabled':
464
- return {
465
- ext: 'xlsm',
466
- mime: 'application/vnd.ms-excel.sheet.macroenabled.12',
467
- };
468
- case 'application/vnd.openxmlformats-officedocument.spreadsheetml.template':
469
- return {
470
- ext: 'xltx',
471
- mime: mimeType,
472
- };
473
- case 'application/vnd.ms-excel.template.macroenabled':
474
- return {
475
- ext: 'xltm',
476
- mime: 'application/vnd.ms-excel.template.macroenabled.12',
477
- };
478
- case 'application/vnd.ms-powerpoint.slideshow.macroenabled':
479
- return {
480
- ext: 'ppsm',
481
- mime: 'application/vnd.ms-powerpoint.slideshow.macroenabled.12',
482
- };
483
- case 'application/vnd.openxmlformats-officedocument.wordprocessingml.document':
484
- return {
485
- ext: 'docx',
486
- mime: mimeType,
487
- };
488
- case 'application/vnd.ms-word.document.macroenabled':
489
- return {
490
- ext: 'docm',
491
- mime: 'application/vnd.ms-word.document.macroenabled.12',
492
- };
493
- case 'application/vnd.openxmlformats-officedocument.wordprocessingml.template':
494
- return {
495
- ext: 'dotx',
496
- mime: mimeType,
497
- };
498
- case 'application/vnd.ms-word.template.macroenabledtemplate':
499
- return {
500
- ext: 'dotm',
501
- mime: 'application/vnd.ms-word.template.macroenabled.12',
502
- };
503
- case 'application/vnd.openxmlformats-officedocument.presentationml.template':
504
- return {
505
- ext: 'potx',
506
- mime: mimeType,
507
- };
508
- case 'application/vnd.ms-powerpoint.template.macroenabled':
509
- return {
510
- ext: 'potm',
511
- mime: 'application/vnd.ms-powerpoint.template.macroenabled.12',
512
- };
513
- case 'application/vnd.openxmlformats-officedocument.presentationml.presentation':
514
- return {
515
- ext: 'pptx',
516
- mime: mimeType,
517
- };
518
- case 'application/vnd.ms-powerpoint.presentation.macroenabled':
519
- return {
520
- ext: 'pptm',
521
- mime: 'application/vnd.ms-powerpoint.presentation.macroenabled.12',
522
- };
523
- case 'application/vnd.ms-visio.drawing':
524
- return {
525
- ext: 'vsdx',
526
- mime: 'application/vnd.visio',
527
- };
528
- case 'application/vnd.ms-package.3dmanufacturing-3dmodel+xml':
529
- return {
530
- ext: '3mf',
531
- mime: 'model/3mf',
532
- };
533
- default:
534
- }
535
- }
536
-
537
- function _check(buffer, headers, options) {
538
- options = {
539
- offset: 0,
540
- ...options,
541
- };
542
-
543
- for (const [index, header] of headers.entries()) {
544
- // If a bitmask is set
545
- if (options.mask) {
546
- // If header doesn't equal `buf` with bits masked off
547
- if (header !== (options.mask[index] & buffer[index + options.offset])) {
548
- return false;
549
- }
550
- } else if (header !== buffer[index + options.offset]) {
551
- return false;
552
- }
553
- }
554
-
555
- return true;
556
- }
557
-
558
- export function normalizeSampleSize(sampleSize) {
559
- // `sampleSize` is an explicit caller-controlled tuning knob, not untrusted file input.
560
- // Preserve valid caller-requested probe depth here; applications must bound attacker-derived option values themselves.
561
- if (!Number.isFinite(sampleSize)) {
562
- return reasonableDetectionSizeInBytes;
563
- }
564
-
565
- return Math.max(1, Math.trunc(sampleSize));
566
- }
567
-
568
- function readByobReaderWithSignal(reader, buffer, signal) {
569
- if (signal === undefined) {
570
- return reader.read(buffer);
571
- }
572
-
573
- signal.throwIfAborted();
574
-
575
- return new Promise((resolve, reject) => {
576
- const cleanup = () => {
577
- signal.removeEventListener('abort', onAbort);
578
- };
579
-
580
- const onAbort = () => {
581
- const abortReason = signal.reason;
582
- cleanup();
583
-
584
- (async () => {
585
- try {
586
- await reader.cancel(abortReason);
587
- } catch {}
588
- })();
589
-
590
- reject(abortReason);
591
- };
592
-
593
- signal.addEventListener('abort', onAbort, {once: true});
594
- (async () => {
595
- try {
596
- const result = await reader.read(buffer);
597
- cleanup();
598
- resolve(result);
599
- } catch (error) {
600
- cleanup();
601
- reject(error);
602
- }
603
- })();
604
- });
605
- }
606
-
607
- function normalizeMpegOffsetTolerance(mpegOffsetTolerance) {
608
- // This value controls scan depth and therefore worst-case CPU work.
609
- if (!Number.isFinite(mpegOffsetTolerance)) {
610
- return 0;
611
- }
612
-
613
- return Math.max(0, Math.min(maximumMpegOffsetTolerance, Math.trunc(mpegOffsetTolerance)));
614
- }
615
-
616
- function getKnownFileSizeOrMaximum(fileSize) {
617
- if (!Number.isFinite(fileSize)) {
618
- return Number.MAX_SAFE_INTEGER;
619
- }
620
-
621
- return Math.max(0, fileSize);
622
- }
623
-
624
- function hasUnknownFileSize(tokenizer) {
625
- const fileSize = tokenizer.fileInfo.size;
626
- return (
627
- !Number.isFinite(fileSize)
628
- || fileSize === Number.MAX_SAFE_INTEGER
629
- );
630
- }
631
-
632
- function hasExceededUnknownSizeScanBudget(tokenizer, startOffset, maximumBytes) {
633
- return (
634
- hasUnknownFileSize(tokenizer)
635
- && tokenizer.position - startOffset > maximumBytes
636
- );
637
- }
638
-
639
- function getMaximumZipBufferedReadLength(tokenizer) {
640
- const fileSize = tokenizer.fileInfo.size;
641
- const remainingBytes = Number.isFinite(fileSize)
642
- ? Math.max(0, fileSize - tokenizer.position)
643
- : Number.MAX_SAFE_INTEGER;
644
-
645
- return Math.min(remainingBytes, maximumZipBufferedReadSizeInBytes);
646
- }
647
-
648
- function isRecoverableZipError(error) {
649
- if (error instanceof strtok3.EndOfStreamError) {
650
- return true;
651
- }
652
-
653
- if (error instanceof ParserHardLimitError) {
654
- return true;
655
- }
656
-
657
- if (!(error instanceof Error)) {
658
- return false;
659
- }
660
-
661
- if (recoverableZipErrorMessages.has(error.message)) {
662
- return true;
663
- }
664
-
665
- if (recoverableZipErrorCodes.has(error.code)) {
666
- return true;
667
- }
668
-
669
- for (const prefix of recoverableZipErrorMessagePrefixes) {
670
- if (error.message.startsWith(prefix)) {
671
- return true;
672
- }
673
- }
674
-
675
- return false;
676
- }
677
-
678
- function canReadZipEntryForDetection(zipHeader, maximumSize = maximumZipEntrySizeInBytes) {
679
- const sizes = [zipHeader.compressedSize, zipHeader.uncompressedSize];
680
- for (const size of sizes) {
681
- if (
682
- !Number.isFinite(size)
683
- || size < 0
684
- || size > maximumSize
685
- ) {
686
- return false;
687
- }
688
- }
689
-
690
- return true;
691
- }
692
-
693
- function createOpenXmlZipDetectionState() {
694
- return {
695
- hasContentTypesEntry: false,
696
- hasParsedContentTypesEntry: false,
697
- isParsingContentTypes: false,
698
- hasUnparseableContentTypes: false,
699
- hasWordDirectory: false,
700
- hasPresentationDirectory: false,
701
- hasSpreadsheetDirectory: false,
702
- hasThreeDimensionalModelEntry: false,
703
- };
704
- }
705
-
706
- function updateOpenXmlZipDetectionStateFromFilename(openXmlState, filename) {
707
- if (filename.startsWith('word/')) {
708
- openXmlState.hasWordDirectory = true;
709
- }
710
-
711
- if (filename.startsWith('ppt/')) {
712
- openXmlState.hasPresentationDirectory = true;
713
- }
714
-
715
- if (filename.startsWith('xl/')) {
716
- openXmlState.hasSpreadsheetDirectory = true;
717
- }
718
-
719
- if (
720
- filename.startsWith('3D/')
721
- && filename.endsWith('.model')
722
- ) {
723
- openXmlState.hasThreeDimensionalModelEntry = true;
724
- }
725
- }
726
-
727
- function getOpenXmlFileTypeFromZipEntries(openXmlState) {
728
- // Only use directory-name heuristic when [Content_Types].xml was present in the archive
729
- // but its handler was skipped (not invoked, not currently running, and not already resolved).
730
- // This avoids guessing from directory names when content-type parsing already gave a definitive answer or failed.
731
- if (
732
- !openXmlState.hasContentTypesEntry
733
- || openXmlState.hasUnparseableContentTypes
734
- || openXmlState.isParsingContentTypes
735
- || openXmlState.hasParsedContentTypesEntry
736
- ) {
737
- return;
738
- }
739
-
740
- if (openXmlState.hasWordDirectory) {
741
- return {
742
- ext: 'docx',
743
- mime: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
744
- };
745
- }
746
-
747
- if (openXmlState.hasPresentationDirectory) {
748
- return {
749
- ext: 'pptx',
750
- mime: 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
751
- };
752
- }
753
-
754
- if (openXmlState.hasSpreadsheetDirectory) {
755
- return {
756
- ext: 'xlsx',
757
- mime: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
758
- };
759
- }
760
-
761
- if (openXmlState.hasThreeDimensionalModelEntry) {
762
- return {
763
- ext: '3mf',
764
- mime: 'model/3mf',
765
- };
766
- }
767
- }
768
-
769
- function getOpenXmlMimeTypeFromContentTypesXml(xmlContent) {
770
- // We only need the `ContentType="...main+xml"` value, so a small string scan is enough and avoids full XML parsing.
771
- const endPosition = xmlContent.indexOf('.main+xml"');
772
- if (endPosition === -1) {
773
- const mimeType = 'application/vnd.ms-package.3dmanufacturing-3dmodel+xml';
774
- if (xmlContent.includes(`ContentType="${mimeType}"`)) {
775
- return mimeType;
776
- }
777
-
778
- return;
779
- }
780
-
781
- const truncatedContent = xmlContent.slice(0, endPosition);
782
- const firstQuotePosition = truncatedContent.lastIndexOf('"');
783
- // If no quote is found, `lastIndexOf` returns -1 and this intentionally falls back to the full truncated prefix.
784
- return truncatedContent.slice(firstQuotePosition + 1);
785
- }
786
-
787
158
  export async function fileTypeFromTokenizer(tokenizer, options) {
788
159
  return new FileTypeParser(options).fromTokenizer(tokenizer);
789
160
  }
@@ -816,7 +187,7 @@ export class FileTypeParser {
816
187
  }
817
188
 
818
189
  createTokenizerFromWebStream(stream) {
819
- return patchWebByobTokenizerClose(strtok3.fromWebStream(stream, this.getTokenizerOptions()));
190
+ return strtok3.fromWebStream(toDefaultStream(stream), this.getTokenizerOptions());
820
191
  }
821
192
 
822
193
  async parseTokenizer(tokenizer, detectionReentryCount = 0) {
@@ -883,41 +254,96 @@ export class FileTypeParser {
883
254
  return this.fromTokenizer(tokenizer);
884
255
  }
885
256
 
257
+ async fromFile(path) {
258
+ this.options.signal?.throwIfAborted();
259
+ // TODO: Remove this when `strtok3.fromFile()` safely rejects non-regular filesystem objects without a pathname race.
260
+ const [{default: fsPromises}, {FileTokenizer}] = await Promise.all([
261
+ import('node:fs/promises'),
262
+ import('strtok3'),
263
+ ]);
264
+ const fileHandle = await fsPromises.open(path, fsPromises.constants.O_RDONLY | fsPromises.constants.O_NONBLOCK);
265
+ const fileStat = await fileHandle.stat();
266
+ if (!fileStat.isFile()) {
267
+ await fileHandle.close();
268
+ return;
269
+ }
270
+
271
+ const tokenizer = new FileTokenizer(fileHandle, {
272
+ ...this.getTokenizerOptions(),
273
+ fileInfo: {path, size: fileStat.size},
274
+ });
275
+ return this.fromTokenizer(tokenizer);
276
+ }
277
+
886
278
  async toDetectionStream(stream, options) {
279
+ this.options.signal?.throwIfAborted();
887
280
  const sampleSize = normalizeSampleSize(options?.sampleSize ?? reasonableDetectionSizeInBytes);
888
281
  let detectedFileType;
889
- let firstChunk;
282
+ let streamEnded = false;
890
283
 
891
- const reader = stream.getReader({mode: 'byob'});
892
- try {
893
- // Read the first chunk from the stream
894
- const {value: chunk, done} = await readByobReaderWithSignal(reader, new Uint8Array(sampleSize), this.options.signal);
895
- firstChunk = chunk;
896
- if (!done && chunk) {
897
- try {
898
- // Attempt to detect the file type from the chunk
899
- detectedFileType = await this.fromBuffer(chunk.subarray(0, sampleSize));
900
- } catch (error) {
901
- if (!(error instanceof strtok3.EndOfStreamError)) {
902
- throw error; // Re-throw non-EndOfStreamError
903
- }
284
+ const reader = stream.getReader();
285
+ const chunks = [];
286
+ let totalSize = 0;
904
287
 
905
- detectedFileType = undefined;
288
+ try {
289
+ while (totalSize < sampleSize) {
290
+ const {value, done} = await readWithSignal(reader, this.options.signal);
291
+ if (done || !value) {
292
+ streamEnded = true;
293
+ break;
906
294
  }
295
+
296
+ chunks.push(value);
297
+ totalSize += value.length;
907
298
  }
908
299
 
909
- firstChunk = chunk;
300
+ if (
301
+ !streamEnded
302
+ && totalSize === sampleSize
303
+ ) {
304
+ const {value, done} = await readWithSignal(reader, this.options.signal);
305
+ if (done || !value) {
306
+ streamEnded = true;
307
+ } else {
308
+ chunks.push(value);
309
+ totalSize += value.length;
310
+ }
311
+ }
910
312
  } finally {
911
- reader.releaseLock(); // Ensure the reader is released
313
+ reader.releaseLock();
912
314
  }
913
315
 
914
- // Create a new ReadableStream to manage locking issues
316
+ if (totalSize > 0) {
317
+ const sample = chunks.length === 1 ? chunks[0] : concatUint8Arrays(chunks);
318
+ try {
319
+ detectedFileType = await this.fromBuffer(sample.subarray(0, sampleSize));
320
+ } catch (error) {
321
+ if (!(error instanceof strtok3.EndOfStreamError)) {
322
+ throw error;
323
+ }
324
+
325
+ detectedFileType = undefined;
326
+ }
327
+
328
+ if (
329
+ !streamEnded
330
+ && detectedFileType?.ext === 'pages'
331
+ ) {
332
+ detectedFileType = {
333
+ ext: 'zip',
334
+ mime: 'application/zip',
335
+ };
336
+ }
337
+ }
338
+
339
+ // Prepend collected chunks and pipe the rest through
915
340
  const transformStream = new TransformStream({
916
- async start(controller) {
917
- controller.enqueue(firstChunk); // Enqueue the initial chunk
341
+ start(controller) {
342
+ for (const chunk of chunks) {
343
+ controller.enqueue(chunk);
344
+ }
918
345
  },
919
346
  transform(chunk, controller) {
920
- // Pass through the chunks without modification
921
347
  controller.enqueue(chunk);
922
348
  },
923
349
  });
@@ -951,7 +377,6 @@ export class FileTypeParser {
951
377
  }, unknownSizeGzipProbeTimeoutInMilliseconds);
952
378
  probeSignal = this.options.signal === undefined
953
379
  ? timeoutController.signal
954
- // eslint-disable-next-line n/no-unsupported-features/node-builtins
955
380
  : AbortSignal.any([this.options.signal, timeoutController.signal]);
956
381
  probeParser = new FileTypeParser({
957
382
  ...this.options,
@@ -994,7 +419,7 @@ export class FileTypeParser {
994
419
  }
995
420
 
996
421
  check(header, options) {
997
- return _check(this.buffer, header, options);
422
+ return checkBytes(this.buffer, header, options);
998
423
  }
999
424
 
1000
425
  checkString(header, options) {
@@ -1141,7 +566,7 @@ export class FileTypeParser {
1141
566
  const isUnknownFileSize = hasUnknownFileSize(tokenizer);
1142
567
  if (
1143
568
  !Number.isFinite(id3HeaderLength)
1144
- || id3HeaderLength < 0
569
+ || id3HeaderLength < 0
1145
570
  // Keep ID3 probing bounded for unknown-size streams to avoid attacker-controlled large skips.
1146
571
  || (
1147
572
  isUnknownFileSize
@@ -1267,108 +692,7 @@ export class FileTypeParser {
1267
692
  // Zip-based file formats
1268
693
  // Need to be before the `zip` check
1269
694
  if (this.check([0x50, 0x4B, 0x3, 0x4])) { // Local file header signature
1270
- let fileType;
1271
- const openXmlState = createOpenXmlZipDetectionState();
1272
-
1273
- try {
1274
- await new ZipHandler(tokenizer).unzip(zipHeader => {
1275
- updateOpenXmlZipDetectionStateFromFilename(openXmlState, zipHeader.filename);
1276
-
1277
- const isOpenXmlContentTypesEntry = zipHeader.filename === '[Content_Types].xml';
1278
- const openXmlFileTypeFromEntries = getOpenXmlFileTypeFromZipEntries(openXmlState);
1279
- if (
1280
- !isOpenXmlContentTypesEntry
1281
- && openXmlFileTypeFromEntries
1282
- ) {
1283
- fileType = openXmlFileTypeFromEntries;
1284
- return {
1285
- stop: true,
1286
- };
1287
- }
1288
-
1289
- switch (zipHeader.filename) {
1290
- case 'META-INF/mozilla.rsa':
1291
- fileType = {
1292
- ext: 'xpi',
1293
- mime: 'application/x-xpinstall',
1294
- };
1295
- return {
1296
- stop: true,
1297
- };
1298
- case 'META-INF/MANIFEST.MF':
1299
- fileType = {
1300
- ext: 'jar',
1301
- mime: 'application/java-archive',
1302
- };
1303
- return {
1304
- stop: true,
1305
- };
1306
- case 'mimetype':
1307
- if (!canReadZipEntryForDetection(zipHeader, maximumZipTextEntrySizeInBytes)) {
1308
- return {};
1309
- }
1310
-
1311
- return {
1312
- async handler(fileData) {
1313
- // Use TextDecoder to decode the UTF-8 encoded data
1314
- const mimeType = new TextDecoder('utf-8').decode(fileData).trim();
1315
- fileType = getFileTypeFromMimeType(mimeType);
1316
- },
1317
- stop: true,
1318
- };
1319
-
1320
- case '[Content_Types].xml': {
1321
- openXmlState.hasContentTypesEntry = true;
1322
-
1323
- if (!canReadZipEntryForDetection(zipHeader, maximumZipTextEntrySizeInBytes)) {
1324
- openXmlState.hasUnparseableContentTypes = true;
1325
- return {};
1326
- }
1327
-
1328
- openXmlState.isParsingContentTypes = true;
1329
- return {
1330
- async handler(fileData) {
1331
- // Use TextDecoder to decode the UTF-8 encoded data
1332
- const xmlContent = new TextDecoder('utf-8').decode(fileData);
1333
- const mimeType = getOpenXmlMimeTypeFromContentTypesXml(xmlContent);
1334
- if (mimeType) {
1335
- fileType = getFileTypeFromMimeType(mimeType);
1336
- }
1337
-
1338
- openXmlState.hasParsedContentTypesEntry = true;
1339
- openXmlState.isParsingContentTypes = false;
1340
- },
1341
- stop: true,
1342
- };
1343
- }
1344
-
1345
- default:
1346
- if (/classes\d*\.dex/.test(zipHeader.filename)) {
1347
- fileType = {
1348
- ext: 'apk',
1349
- mime: 'application/vnd.android.package-archive',
1350
- };
1351
- return {stop: true};
1352
- }
1353
-
1354
- return {};
1355
- }
1356
- });
1357
- } catch (error) {
1358
- if (!isRecoverableZipError(error)) {
1359
- throw error;
1360
- }
1361
-
1362
- if (openXmlState.isParsingContentTypes) {
1363
- openXmlState.isParsingContentTypes = false;
1364
- openXmlState.hasUnparseableContentTypes = true;
1365
- }
1366
- }
1367
-
1368
- return fileType ?? getOpenXmlFileTypeFromZipEntries(openXmlState) ?? {
1369
- ext: 'zip',
1370
- mime: 'application/zip',
1371
- };
695
+ return detectZip(tokenizer);
1372
696
  }
1373
697
 
1374
698
  if (this.checkString('OggS')) {
@@ -1378,7 +702,7 @@ export class FileTypeParser {
1378
702
  await tokenizer.readBuffer(type);
1379
703
 
1380
704
  // Needs to be before `ogg` check
1381
- if (_check(type, [0x4F, 0x70, 0x75, 0x73, 0x48, 0x65, 0x61, 0x64])) {
705
+ if (checkBytes(type, [0x4F, 0x70, 0x75, 0x73, 0x48, 0x65, 0x61, 0x64])) {
1382
706
  return {
1383
707
  ext: 'opus',
1384
708
  mime: 'audio/ogg; codecs=opus',
@@ -1386,7 +710,7 @@ export class FileTypeParser {
1386
710
  }
1387
711
 
1388
712
  // If ' theora' in header.
1389
- if (_check(type, [0x80, 0x74, 0x68, 0x65, 0x6F, 0x72, 0x61])) {
713
+ if (checkBytes(type, [0x80, 0x74, 0x68, 0x65, 0x6F, 0x72, 0x61])) {
1390
714
  return {
1391
715
  ext: 'ogv',
1392
716
  mime: 'video/ogg',
@@ -1394,7 +718,7 @@ export class FileTypeParser {
1394
718
  }
1395
719
 
1396
720
  // If '\x01video' in header.
1397
- if (_check(type, [0x01, 0x76, 0x69, 0x64, 0x65, 0x6F, 0x00])) {
721
+ if (checkBytes(type, [0x01, 0x76, 0x69, 0x64, 0x65, 0x6F, 0x00])) {
1398
722
  return {
1399
723
  ext: 'ogm',
1400
724
  mime: 'video/ogg',
@@ -1402,7 +726,7 @@ export class FileTypeParser {
1402
726
  }
1403
727
 
1404
728
  // If ' FLAC' in header https://xiph.org/flac/faq.html
1405
- if (_check(type, [0x7F, 0x46, 0x4C, 0x41, 0x43])) {
729
+ if (checkBytes(type, [0x7F, 0x46, 0x4C, 0x41, 0x43])) {
1406
730
  return {
1407
731
  ext: 'oga',
1408
732
  mime: 'audio/ogg',
@@ -1410,7 +734,7 @@ export class FileTypeParser {
1410
734
  }
1411
735
 
1412
736
  // 'Speex ' in header https://en.wikipedia.org/wiki/Speex
1413
- if (_check(type, [0x53, 0x70, 0x65, 0x65, 0x78, 0x20, 0x20])) {
737
+ if (checkBytes(type, [0x53, 0x70, 0x65, 0x65, 0x78, 0x20, 0x20])) {
1414
738
  return {
1415
739
  ext: 'spx',
1416
740
  mime: 'audio/ogg',
@@ -1418,7 +742,7 @@ export class FileTypeParser {
1418
742
  }
1419
743
 
1420
744
  // If '\x01vorbis' in header
1421
- if (_check(type, [0x01, 0x76, 0x6F, 0x72, 0x62, 0x69, 0x73])) {
745
+ if (checkBytes(type, [0x01, 0x76, 0x6F, 0x72, 0x62, 0x69, 0x73])) {
1422
746
  return {
1423
747
  ext: 'ogg',
1424
748
  mime: 'audio/ogg',
@@ -1494,7 +818,7 @@ export class FileTypeParser {
1494
818
  if (this.checkString('LZIP')) {
1495
819
  return {
1496
820
  ext: 'lz',
1497
- mime: 'application/x-lzip',
821
+ mime: 'application/lzip',
1498
822
  };
1499
823
  }
1500
824
 
@@ -1559,110 +883,7 @@ export class FileTypeParser {
1559
883
 
1560
884
  // https://github.com/file/file/blob/master/magic/Magdir/matroska
1561
885
  if (this.check([0x1A, 0x45, 0xDF, 0xA3])) { // Root element: EBML
1562
- async function readField() {
1563
- const msb = await tokenizer.peekNumber(Token.UINT8);
1564
- let mask = 0x80;
1565
- let ic = 0; // 0 = A, 1 = B, 2 = C, 3 = D
1566
-
1567
- while ((msb & mask) === 0 && mask !== 0) {
1568
- ++ic;
1569
- mask >>= 1;
1570
- }
1571
-
1572
- const id = new Uint8Array(ic + 1);
1573
- await safeReadBuffer(tokenizer, id, undefined, {
1574
- maximumLength: id.length,
1575
- reason: 'EBML field',
1576
- });
1577
- return id;
1578
- }
1579
-
1580
- async function readElement() {
1581
- const idField = await readField();
1582
- const lengthField = await readField();
1583
-
1584
- lengthField[0] ^= 0x80 >> (lengthField.length - 1);
1585
- const nrLength = Math.min(6, lengthField.length); // JavaScript can max read 6 bytes integer
1586
-
1587
- const idView = new DataView(idField.buffer);
1588
- const lengthView = new DataView(lengthField.buffer, lengthField.length - nrLength, nrLength);
1589
-
1590
- return {
1591
- id: getUintBE(idView),
1592
- len: getUintBE(lengthView),
1593
- };
1594
- }
1595
-
1596
- async function readChildren(children) {
1597
- let ebmlElementCount = 0;
1598
- while (children > 0) {
1599
- ebmlElementCount++;
1600
- if (ebmlElementCount > maximumEbmlElementCount) {
1601
- return;
1602
- }
1603
-
1604
- if (hasExceededUnknownSizeScanBudget(tokenizer, ebmlScanStart, maximumUntrustedSkipSizeInBytes)) {
1605
- return;
1606
- }
1607
-
1608
- const previousPosition = tokenizer.position;
1609
- const element = await readElement();
1610
-
1611
- if (element.id === 0x42_82) {
1612
- // `DocType` is a short string ("webm", "matroska", ...), reject implausible lengths to avoid large allocations.
1613
- if (element.len > maximumEbmlDocumentTypeSizeInBytes) {
1614
- return;
1615
- }
1616
-
1617
- const documentTypeLength = getSafeBound(element.len, maximumEbmlDocumentTypeSizeInBytes, 'EBML DocType');
1618
- const rawValue = await tokenizer.readToken(new Token.StringType(documentTypeLength));
1619
- return rawValue.replaceAll(/\00.*$/g, ''); // Return DocType
1620
- }
1621
-
1622
- if (
1623
- hasUnknownFileSize(tokenizer)
1624
- && (
1625
- !Number.isFinite(element.len)
1626
- || element.len < 0
1627
- || element.len > maximumEbmlElementPayloadSizeInBytes
1628
- )
1629
- ) {
1630
- return;
1631
- }
1632
-
1633
- await safeIgnore(tokenizer, element.len, {
1634
- maximumLength: hasUnknownFileSize(tokenizer) ? maximumEbmlElementPayloadSizeInBytes : tokenizer.fileInfo.size,
1635
- reason: 'EBML payload',
1636
- }); // ignore payload
1637
- --children;
1638
-
1639
- // Safeguard against malformed files: bail if the position did not advance.
1640
- if (tokenizer.position <= previousPosition) {
1641
- return;
1642
- }
1643
- }
1644
- }
1645
-
1646
- const rootElement = await readElement();
1647
- const ebmlScanStart = tokenizer.position;
1648
- const documentType = await readChildren(rootElement.len);
1649
-
1650
- switch (documentType) {
1651
- case 'webm':
1652
- return {
1653
- ext: 'webm',
1654
- mime: 'video/webm',
1655
- };
1656
-
1657
- case 'matroska':
1658
- return {
1659
- ext: 'mkv',
1660
- mime: 'video/matroska',
1661
- };
1662
-
1663
- default:
1664
- return;
1665
- }
886
+ return detectEbml(tokenizer);
1666
887
  }
1667
888
 
1668
889
  if (this.checkString('SQLi')) {
@@ -1760,7 +981,7 @@ export class FileTypeParser {
1760
981
  if (this.check([0x04, 0x22, 0x4D, 0x18])) {
1761
982
  return {
1762
983
  ext: 'lz4',
1763
- mime: 'application/x-lz4', // Invented by us
984
+ mime: 'application/x-lz4', // Informal, used by freedesktop.org shared-mime-info
1764
985
  };
1765
986
  }
1766
987
 
@@ -1795,7 +1016,7 @@ export class FileTypeParser {
1795
1016
  };
1796
1017
  }
1797
1018
 
1798
- if (this.checkString('{\\rtf')) {
1019
+ if (this.checkString(String.raw`{\rtf`)) {
1799
1020
  return {
1800
1021
  ext: 'rtf',
1801
1022
  mime: 'application/rtf',
@@ -1896,7 +1117,7 @@ export class FileTypeParser {
1896
1117
  if (this.checkString('DRACO')) {
1897
1118
  return {
1898
1119
  ext: 'drc',
1899
- mime: 'application/vnd.google.draco', // Invented by us
1120
+ mime: 'application/x-ft-draco',
1900
1121
  };
1901
1122
  }
1902
1123
 
@@ -1942,7 +1163,7 @@ export class FileTypeParser {
1942
1163
 
1943
1164
  if (this.checkString('AC')) {
1944
1165
  const version = new Token.StringType(4, 'latin1').get(this.buffer, 2);
1945
- if (version.match('^d*') && version >= 1000 && version <= 1050) {
1166
+ if (/^\d+$/v.test(version) && version >= 1000 && version <= 1050) {
1946
1167
  return {
1947
1168
  ext: 'dwg',
1948
1169
  mime: 'image/vnd.dwg',
@@ -1997,110 +1218,7 @@ export class FileTypeParser {
1997
1218
  // -- 8-byte signatures --
1998
1219
 
1999
1220
  if (this.check([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A])) {
2000
- const pngFileType = {
2001
- ext: 'png',
2002
- mime: 'image/png',
2003
- };
2004
-
2005
- const apngFileType = {
2006
- ext: 'apng',
2007
- mime: 'image/apng',
2008
- };
2009
-
2010
- // APNG format (https://wiki.mozilla.org/APNG_Specification)
2011
- // 1. Find the first IDAT (image data) chunk (49 44 41 54)
2012
- // 2. Check if there is an "acTL" chunk before the IDAT one (61 63 54 4C)
2013
-
2014
- // Offset calculated as follows:
2015
- // - 8 bytes: PNG signature
2016
- // - 4 (length) + 4 (chunk type) + 13 (chunk data) + 4 (CRC): IHDR chunk
2017
-
2018
- await tokenizer.ignore(8); // ignore PNG signature
2019
-
2020
- async function readChunkHeader() {
2021
- return {
2022
- length: await tokenizer.readToken(Token.INT32_BE),
2023
- type: await tokenizer.readToken(new Token.StringType(4, 'latin1')),
2024
- };
2025
- }
2026
-
2027
- const isUnknownPngStream = hasUnknownFileSize(tokenizer);
2028
- const pngScanStart = tokenizer.position;
2029
- let pngChunkCount = 0;
2030
- let hasSeenImageHeader = false;
2031
- do {
2032
- pngChunkCount++;
2033
- if (pngChunkCount > maximumPngChunkCount) {
2034
- break;
2035
- }
2036
-
2037
- if (hasExceededUnknownSizeScanBudget(tokenizer, pngScanStart, maximumPngStreamScanBudgetInBytes)) {
2038
- break;
2039
- }
2040
-
2041
- const previousPosition = tokenizer.position;
2042
- const chunk = await readChunkHeader();
2043
- if (chunk.length < 0) {
2044
- return; // Invalid chunk length
2045
- }
2046
-
2047
- if (chunk.type === 'IHDR') {
2048
- // PNG requires the first real image header to be a 13-byte IHDR chunk.
2049
- if (chunk.length !== 13) {
2050
- return;
2051
- }
2052
-
2053
- hasSeenImageHeader = true;
2054
- }
2055
-
2056
- switch (chunk.type) {
2057
- case 'IDAT':
2058
- return pngFileType;
2059
- case 'acTL':
2060
- return apngFileType;
2061
- default:
2062
- if (
2063
- !hasSeenImageHeader
2064
- && chunk.type !== 'CgBI'
2065
- ) {
2066
- return;
2067
- }
2068
-
2069
- if (
2070
- isUnknownPngStream
2071
- && chunk.length > maximumPngChunkSizeInBytes
2072
- ) {
2073
- // Avoid huge attacker-controlled skips when probing unknown-size streams.
2074
- return hasSeenImageHeader && isPngAncillaryChunk(chunk.type) ? pngFileType : undefined;
2075
- }
2076
-
2077
- try {
2078
- await safeIgnore(tokenizer, chunk.length + 4, {
2079
- maximumLength: isUnknownPngStream ? maximumPngChunkSizeInBytes + 4 : tokenizer.fileInfo.size,
2080
- reason: 'PNG chunk payload',
2081
- }); // Ignore chunk-data + CRC
2082
- } catch (error) {
2083
- if (
2084
- !isUnknownPngStream
2085
- && (
2086
- error instanceof ParserHardLimitError
2087
- || error instanceof strtok3.EndOfStreamError
2088
- )
2089
- ) {
2090
- return pngFileType;
2091
- }
2092
-
2093
- throw error;
2094
- }
2095
- }
2096
-
2097
- // Safeguard against malformed files: bail if the position did not advance.
2098
- if (tokenizer.position <= previousPosition) {
2099
- break;
2100
- }
2101
- } while (tokenizer.position + 8 < tokenizer.fileInfo.size);
2102
-
2103
- return pngFileType;
1221
+ return detectPng(tokenizer);
2104
1222
  }
2105
1223
 
2106
1224
  if (this.check([0x41, 0x52, 0x52, 0x4F, 0x57, 0x31, 0x00, 0x00])) {
@@ -2258,116 +1376,7 @@ export class FileTypeParser {
2258
1376
 
2259
1377
  // ASF_Header_Object first 80 bytes
2260
1378
  if (this.check([0x30, 0x26, 0xB2, 0x75, 0x8E, 0x66, 0xCF, 0x11, 0xA6, 0xD9])) {
2261
- let isMalformedAsf = false;
2262
- try {
2263
- async function readHeader() {
2264
- const guid = new Uint8Array(16);
2265
- await safeReadBuffer(tokenizer, guid, undefined, {
2266
- maximumLength: guid.length,
2267
- reason: 'ASF header GUID',
2268
- });
2269
- return {
2270
- id: guid,
2271
- size: Number(await tokenizer.readToken(Token.UINT64_LE)),
2272
- };
2273
- }
2274
-
2275
- await safeIgnore(tokenizer, 30, {
2276
- maximumLength: 30,
2277
- reason: 'ASF header prelude',
2278
- });
2279
- const isUnknownFileSize = hasUnknownFileSize(tokenizer);
2280
- const asfHeaderScanStart = tokenizer.position;
2281
- let asfHeaderObjectCount = 0;
2282
- while (tokenizer.position + 24 < tokenizer.fileInfo.size) {
2283
- asfHeaderObjectCount++;
2284
- if (asfHeaderObjectCount > maximumAsfHeaderObjectCount) {
2285
- break;
2286
- }
2287
-
2288
- if (hasExceededUnknownSizeScanBudget(tokenizer, asfHeaderScanStart, maximumUntrustedSkipSizeInBytes)) {
2289
- break;
2290
- }
2291
-
2292
- const previousPosition = tokenizer.position;
2293
- const header = await readHeader();
2294
- let payload = header.size - 24;
2295
- if (
2296
- !Number.isFinite(payload)
2297
- || payload < 0
2298
- ) {
2299
- isMalformedAsf = true;
2300
- break;
2301
- }
2302
-
2303
- if (_check(header.id, [0x91, 0x07, 0xDC, 0xB7, 0xB7, 0xA9, 0xCF, 0x11, 0x8E, 0xE6, 0x00, 0xC0, 0x0C, 0x20, 0x53, 0x65])) {
2304
- // Sync on Stream-Properties-Object (B7DC0791-A9B7-11CF-8EE6-00C00C205365)
2305
- const typeId = new Uint8Array(16);
2306
- payload -= await safeReadBuffer(tokenizer, typeId, undefined, {
2307
- maximumLength: typeId.length,
2308
- reason: 'ASF stream type GUID',
2309
- });
2310
-
2311
- if (_check(typeId, [0x40, 0x9E, 0x69, 0xF8, 0x4D, 0x5B, 0xCF, 0x11, 0xA8, 0xFD, 0x00, 0x80, 0x5F, 0x5C, 0x44, 0x2B])) {
2312
- // Found audio:
2313
- return {
2314
- ext: 'asf',
2315
- mime: 'audio/x-ms-asf',
2316
- };
2317
- }
2318
-
2319
- if (_check(typeId, [0xC0, 0xEF, 0x19, 0xBC, 0x4D, 0x5B, 0xCF, 0x11, 0xA8, 0xFD, 0x00, 0x80, 0x5F, 0x5C, 0x44, 0x2B])) {
2320
- // Found video:
2321
- return {
2322
- ext: 'asf',
2323
- mime: 'video/x-ms-asf',
2324
- };
2325
- }
2326
-
2327
- break;
2328
- }
2329
-
2330
- if (
2331
- isUnknownFileSize
2332
- && payload > maximumAsfHeaderPayloadSizeInBytes
2333
- ) {
2334
- isMalformedAsf = true;
2335
- break;
2336
- }
2337
-
2338
- await safeIgnore(tokenizer, payload, {
2339
- maximumLength: isUnknownFileSize ? maximumAsfHeaderPayloadSizeInBytes : tokenizer.fileInfo.size,
2340
- reason: 'ASF header payload',
2341
- });
2342
-
2343
- // Safeguard against malformed files: break if the position did not advance.
2344
- if (tokenizer.position <= previousPosition) {
2345
- isMalformedAsf = true;
2346
- break;
2347
- }
2348
- }
2349
- } catch (error) {
2350
- if (
2351
- error instanceof strtok3.EndOfStreamError
2352
- || error instanceof ParserHardLimitError
2353
- ) {
2354
- if (hasUnknownFileSize(tokenizer)) {
2355
- isMalformedAsf = true;
2356
- }
2357
- } else {
2358
- throw error;
2359
- }
2360
- }
2361
-
2362
- if (isMalformedAsf) {
2363
- return;
2364
- }
2365
-
2366
- // Default to ASF generic extension
2367
- return {
2368
- ext: 'asf',
2369
- mime: 'application/vnd.ms-asf',
2370
- };
1379
+ return detectAsf(tokenizer);
2371
1380
  }
2372
1381
 
2373
1382
  if (this.check([0xAB, 0x4B, 0x54, 0x58, 0x20, 0x31, 0x31, 0xBB, 0x0D, 0x0A, 0x1A, 0x0A])) {
@@ -2581,21 +1590,21 @@ export class FileTypeParser {
2581
1590
  if (this.check([0x4C, 0x00, 0x00, 0x00, 0x01, 0x14, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0xC0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x46])) {
2582
1591
  return {
2583
1592
  ext: 'lnk',
2584
- mime: 'application/x.ms.shortcut', // Invented by us
1593
+ mime: 'application/x-ms-shortcut', // Informal, used by freedesktop.org shared-mime-info
2585
1594
  };
2586
1595
  }
2587
1596
 
2588
1597
  if (this.check([0x62, 0x6F, 0x6F, 0x6B, 0x00, 0x00, 0x00, 0x00, 0x6D, 0x61, 0x72, 0x6B, 0x00, 0x00, 0x00, 0x00])) {
2589
1598
  return {
2590
1599
  ext: 'alias',
2591
- mime: 'application/x.apple.alias', // Invented by us
1600
+ mime: 'application/x-ft-apple.alias',
2592
1601
  };
2593
1602
  }
2594
1603
 
2595
1604
  if (this.checkString('Kaydara FBX Binary \u0000')) {
2596
1605
  return {
2597
1606
  ext: 'fbx',
2598
- mime: 'application/x.autodesk.fbx', // Invented by us
1607
+ mime: 'application/x-ft-fbx',
2599
1608
  };
2600
1609
  }
2601
1610
 
@@ -2897,3 +1906,7 @@ export class FileTypeParser {
2897
1906
 
2898
1907
  export const supportedExtensions = new Set(extensions);
2899
1908
  export const supportedMimeTypes = new Set(mimeTypes);
1909
+
1910
+ export async function fileTypeFromFile(path, options) {
1911
+ return (new FileTypeParser(options)).fromFile(path);
1912
+ }