@depup/file-type 21.3.4-depup.0 → 22.0.1-depup.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,338 +4,96 @@ Primary entry point, Node.js specific entry point is index.js
4
4
 
5
5
  import * as Token from 'token-types';
6
6
  import * as strtok3 from 'strtok3/core';
7
- import {ZipHandler, GzipHandler} from '@tokenizer/inflate';
8
- import {getUintBE} from 'uint8array-extras';
7
+ import {GzipHandler} from '@tokenizer/inflate';
8
+ import {concatUint8Arrays} from 'uint8array-extras';
9
9
  import {
10
10
  stringToBytes,
11
11
  tarHeaderChecksumMatches,
12
12
  uint32SyncSafeToken,
13
- } from './util.js';
13
+ } from './tokens.js';
14
14
  import {extensions, mimeTypes} from './supported.js';
15
+ import {
16
+ maximumUntrustedSkipSizeInBytes,
17
+ ParserHardLimitError,
18
+ safeIgnore,
19
+ checkBytes,
20
+ hasUnknownFileSize,
21
+ } from './parser.js';
22
+ import {detectZip} from './detectors/zip.js';
23
+ import {detectEbml} from './detectors/ebml.js';
24
+ import {detectPng} from './detectors/png.js';
25
+ import {detectAsf} from './detectors/asf.js';
15
26
 
16
27
  export const reasonableDetectionSizeInBytes = 4100; // A fair amount of file-types are detectable within this range.
17
- // Keep defensive limits small enough to avoid accidental memory spikes from untrusted inputs.
18
28
  const maximumMpegOffsetTolerance = reasonableDetectionSizeInBytes - 2;
19
- const maximumZipEntrySizeInBytes = 1024 * 1024;
20
- const maximumZipEntryCount = 1024;
21
- const maximumZipBufferedReadSizeInBytes = (2 ** 31) - 1;
22
- const maximumUntrustedSkipSizeInBytes = 16 * 1024 * 1024;
23
- const maximumUnknownSizePayloadProbeSizeInBytes = maximumZipEntrySizeInBytes;
24
- const maximumZipTextEntrySizeInBytes = maximumZipEntrySizeInBytes;
25
29
  const maximumNestedGzipDetectionSizeInBytes = maximumUntrustedSkipSizeInBytes;
26
30
  const maximumNestedGzipProbeDepth = 1;
27
31
  const unknownSizeGzipProbeTimeoutInMilliseconds = 100;
28
32
  const maximumId3HeaderSizeInBytes = maximumUntrustedSkipSizeInBytes;
29
- const maximumEbmlDocumentTypeSizeInBytes = 64;
30
- const maximumEbmlElementPayloadSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
31
- const maximumEbmlElementCount = 256;
32
- const maximumPngChunkCount = 512;
33
- const maximumPngStreamScanBudgetInBytes = maximumUntrustedSkipSizeInBytes;
34
- const maximumAsfHeaderObjectCount = 512;
35
33
  const maximumTiffTagCount = 512;
36
34
  const maximumDetectionReentryCount = 256;
37
- const maximumPngChunkSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
38
- const maximumAsfHeaderPayloadSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
39
- const maximumTiffStreamIfdOffsetInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
35
+ const maximumTiffStreamIfdOffsetInBytes = 1024 * 1024;
40
36
  const maximumTiffIfdOffsetInBytes = maximumUntrustedSkipSizeInBytes;
41
- const recoverableZipErrorMessages = new Set([
42
- 'Unexpected signature',
43
- 'Encrypted ZIP',
44
- 'Expected Central-File-Header signature',
45
- ]);
46
- const recoverableZipErrorMessagePrefixes = [
47
- 'ZIP entry count exceeds ',
48
- 'Unsupported ZIP compression method:',
49
- 'ZIP entry compressed data exceeds ',
50
- 'ZIP entry decompressed data exceeds ',
51
- 'Expected data-descriptor-signature at position ',
52
- ];
53
- const recoverableZipErrorCodes = new Set([
54
- 'Z_BUF_ERROR',
55
- 'Z_DATA_ERROR',
56
- 'ERR_INVALID_STATE',
57
- ]);
58
-
59
- class ParserHardLimitError extends Error {}
60
-
61
- function patchWebByobTokenizerClose(tokenizer) {
62
- const streamReader = tokenizer?.streamReader;
63
- if (streamReader?.constructor?.name !== 'WebStreamByobReader') {
64
- return tokenizer;
65
- }
66
-
67
- const {reader} = streamReader;
68
- const cancelAndRelease = async () => {
69
- await reader.cancel();
70
- reader.releaseLock();
71
- };
72
-
73
- streamReader.close = cancelAndRelease;
74
- streamReader.abort = async () => {
75
- streamReader.interrupted = true;
76
- await cancelAndRelease();
77
- };
78
-
79
- return tokenizer;
80
- }
81
37
 
82
- function getSafeBound(value, maximum, reason) {
83
- if (
84
- !Number.isFinite(value)
85
- || value < 0
86
- || value > maximum
87
- ) {
88
- throw new ParserHardLimitError(`${reason} has invalid size ${value} (maximum ${maximum} bytes)`);
38
+ export function normalizeSampleSize(sampleSize) {
39
+ // `sampleSize` is an explicit caller-controlled tuning knob, not untrusted file input.
40
+ // Preserve valid caller-requested probe depth here; applications must bound attacker-derived option values themselves.
41
+ if (!Number.isFinite(sampleSize)) {
42
+ return reasonableDetectionSizeInBytes;
89
43
  }
90
44
 
91
- return value;
92
- }
93
-
94
- async function safeIgnore(tokenizer, length, {maximumLength = maximumUntrustedSkipSizeInBytes, reason = 'skip'} = {}) {
95
- const safeLength = getSafeBound(length, maximumLength, reason);
96
- await tokenizer.ignore(safeLength);
97
- }
98
-
99
- async function safeReadBuffer(tokenizer, buffer, options, {maximumLength = buffer.length, reason = 'read'} = {}) {
100
- const length = options?.length ?? buffer.length;
101
- const safeLength = getSafeBound(length, maximumLength, reason);
102
- return tokenizer.readBuffer(buffer, {
103
- ...options,
104
- length: safeLength,
105
- });
45
+ return Math.max(1, Math.trunc(sampleSize));
106
46
  }
107
47
 
108
- async function decompressDeflateRawWithLimit(data, {maximumLength = maximumZipEntrySizeInBytes} = {}) {
109
- const input = new ReadableStream({
110
- start(controller) {
111
- controller.enqueue(data);
112
- controller.close();
113
- },
114
- });
115
- const output = input.pipeThrough(new DecompressionStream('deflate-raw'));
116
- const reader = output.getReader();
117
- const chunks = [];
118
- let totalLength = 0;
119
-
120
- try {
121
- for (;;) {
122
- const {done, value} = await reader.read();
123
- if (done) {
124
- break;
125
- }
126
-
127
- totalLength += value.length;
128
- if (totalLength > maximumLength) {
129
- await reader.cancel();
130
- throw new Error(`ZIP entry decompressed data exceeds ${maximumLength} bytes`);
131
- }
132
-
133
- chunks.push(value);
134
- }
135
- } finally {
136
- reader.releaseLock();
137
- }
138
-
139
- const uncompressedData = new Uint8Array(totalLength);
140
- let offset = 0;
141
- for (const chunk of chunks) {
142
- uncompressedData.set(chunk, offset);
143
- offset += chunk.length;
48
+ function normalizeMpegOffsetTolerance(mpegOffsetTolerance) {
49
+ // This value controls scan depth and therefore worst-case CPU work.
50
+ if (!Number.isFinite(mpegOffsetTolerance)) {
51
+ return 0;
144
52
  }
145
53
 
146
- return uncompressedData;
54
+ return Math.max(0, Math.min(maximumMpegOffsetTolerance, Math.trunc(mpegOffsetTolerance)));
147
55
  }
148
56
 
149
- const zipDataDescriptorSignature = 0x08_07_4B_50;
150
- const zipDataDescriptorLengthInBytes = 16;
151
- const zipDataDescriptorOverlapLengthInBytes = zipDataDescriptorLengthInBytes - 1;
152
-
153
- function findZipDataDescriptorOffset(buffer, bytesConsumed) {
154
- if (buffer.length < zipDataDescriptorLengthInBytes) {
155
- return -1;
156
- }
157
-
158
- const lastPossibleDescriptorOffset = buffer.length - zipDataDescriptorLengthInBytes;
159
- for (let index = 0; index <= lastPossibleDescriptorOffset; index++) {
160
- if (
161
- Token.UINT32_LE.get(buffer, index) === zipDataDescriptorSignature
162
- && Token.UINT32_LE.get(buffer, index + 8) === bytesConsumed + index
163
- ) {
164
- return index;
165
- }
57
+ function getKnownFileSizeOrMaximum(fileSize) {
58
+ if (!Number.isFinite(fileSize)) {
59
+ return Number.MAX_SAFE_INTEGER;
166
60
  }
167
61
 
168
- return -1;
62
+ return Math.max(0, fileSize);
169
63
  }
170
64
 
171
- function isPngAncillaryChunk(type) {
172
- return (type.codePointAt(0) & 0x20) !== 0;
65
+ // Keep the specifier non-literal at the call site so browser bundlers do not try to resolve Node-only imports.
66
+ function importAtRuntime(specifier) {
67
+ return import(specifier);
173
68
  }
174
69
 
175
- function mergeByteChunks(chunks, totalLength) {
176
- const merged = new Uint8Array(totalLength);
177
- let offset = 0;
178
-
179
- for (const chunk of chunks) {
180
- merged.set(chunk, offset);
181
- offset += chunk.length;
182
- }
183
-
184
- return merged;
70
+ // Wrap stream in an identity TransformStream to avoid BYOB readers.
71
+ // Node.js has a bug where calling controller.close() inside a BYOB stream's
72
+ // pull() callback does not resolve pending reader.read() calls, causing
73
+ // permanent hangs on streams shorter than the requested read size.
74
+ // Using a default (non-BYOB) reader via TransformStream avoids this.
75
+ function toDefaultStream(stream) {
76
+ return stream.pipeThrough(new TransformStream());
185
77
  }
186
78
 
187
- async function readZipDataDescriptorEntryWithLimit(zipHandler, {shouldBuffer, maximumLength = maximumZipEntrySizeInBytes} = {}) {
188
- const {syncBuffer} = zipHandler;
189
- const {length: syncBufferLength} = syncBuffer;
190
- const chunks = [];
191
- let bytesConsumed = 0;
192
-
193
- for (;;) {
194
- const length = await zipHandler.tokenizer.peekBuffer(syncBuffer, {mayBeLess: true});
195
- const dataDescriptorOffset = findZipDataDescriptorOffset(syncBuffer.subarray(0, length), bytesConsumed);
196
- const retainedLength = dataDescriptorOffset >= 0
197
- ? 0
198
- : (
199
- length === syncBufferLength
200
- ? Math.min(zipDataDescriptorOverlapLengthInBytes, length - 1)
201
- : 0
202
- );
203
- const chunkLength = dataDescriptorOffset >= 0 ? dataDescriptorOffset : length - retainedLength;
204
-
205
- if (chunkLength === 0) {
206
- break;
207
- }
208
-
209
- bytesConsumed += chunkLength;
210
- if (bytesConsumed > maximumLength) {
211
- throw new Error(`ZIP entry compressed data exceeds ${maximumLength} bytes`);
212
- }
213
-
214
- if (shouldBuffer) {
215
- const data = new Uint8Array(chunkLength);
216
- await zipHandler.tokenizer.readBuffer(data);
217
- chunks.push(data);
218
- } else {
219
- await zipHandler.tokenizer.ignore(chunkLength);
220
- }
221
-
222
- if (dataDescriptorOffset >= 0) {
223
- break;
224
- }
225
- }
226
-
227
- if (!hasUnknownFileSize(zipHandler.tokenizer)) {
228
- zipHandler.knownSizeDescriptorScannedBytes += bytesConsumed;
229
- }
230
-
231
- if (!shouldBuffer) {
232
- return;
233
- }
234
-
235
- return mergeByteChunks(chunks, bytesConsumed);
236
- }
237
-
238
- function getRemainingZipScanBudget(zipHandler, startOffset) {
239
- if (hasUnknownFileSize(zipHandler.tokenizer)) {
240
- return Math.max(0, maximumUntrustedSkipSizeInBytes - (zipHandler.tokenizer.position - startOffset));
241
- }
242
-
243
- return Math.max(0, maximumZipEntrySizeInBytes - zipHandler.knownSizeDescriptorScannedBytes);
244
- }
245
-
246
- async function readZipEntryData(zipHandler, zipHeader, {shouldBuffer, maximumDescriptorLength = maximumZipEntrySizeInBytes} = {}) {
247
- if (
248
- zipHeader.dataDescriptor
249
- && zipHeader.compressedSize === 0
250
- ) {
251
- return readZipDataDescriptorEntryWithLimit(zipHandler, {
252
- shouldBuffer,
253
- maximumLength: maximumDescriptorLength,
254
- });
255
- }
256
-
257
- if (!shouldBuffer) {
258
- await safeIgnore(zipHandler.tokenizer, zipHeader.compressedSize, {
259
- maximumLength: hasUnknownFileSize(zipHandler.tokenizer) ? maximumZipEntrySizeInBytes : zipHandler.tokenizer.fileInfo.size,
260
- reason: 'ZIP entry compressed data',
261
- });
262
- return;
79
+ function readWithSignal(reader, signal) {
80
+ if (signal === undefined) {
81
+ return reader.read();
263
82
  }
264
83
 
265
- const maximumLength = getMaximumZipBufferedReadLength(zipHandler.tokenizer);
266
- if (
267
- !Number.isFinite(zipHeader.compressedSize)
268
- || zipHeader.compressedSize < 0
269
- || zipHeader.compressedSize > maximumLength
270
- ) {
271
- throw new Error(`ZIP entry compressed data exceeds ${maximumLength} bytes`);
272
- }
84
+ signal.throwIfAborted();
273
85
 
274
- const fileData = new Uint8Array(zipHeader.compressedSize);
275
- await zipHandler.tokenizer.readBuffer(fileData);
276
- return fileData;
86
+ return Promise.race([
87
+ reader.read(),
88
+ new Promise((_resolve, reject) => {
89
+ signal.addEventListener('abort', () => {
90
+ reject(signal.reason);
91
+ reader.cancel(signal.reason).catch(() => {});
92
+ }, {once: true});
93
+ }),
94
+ ]);
277
95
  }
278
96
 
279
- // Override the default inflate to enforce decompression size limits, since @tokenizer/inflate does not expose a configuration hook for this.
280
- ZipHandler.prototype.inflate = async function (zipHeader, fileData, callback) {
281
- if (zipHeader.compressedMethod === 0) {
282
- return callback(fileData);
283
- }
284
-
285
- if (zipHeader.compressedMethod !== 8) {
286
- throw new Error(`Unsupported ZIP compression method: ${zipHeader.compressedMethod}`);
287
- }
288
-
289
- const uncompressedData = await decompressDeflateRawWithLimit(fileData, {maximumLength: maximumZipEntrySizeInBytes});
290
- return callback(uncompressedData);
291
- };
292
-
293
- ZipHandler.prototype.unzip = async function (fileCallback) {
294
- let stop = false;
295
- let zipEntryCount = 0;
296
- const zipScanStart = this.tokenizer.position;
297
- this.knownSizeDescriptorScannedBytes = 0;
298
- do {
299
- if (hasExceededUnknownSizeScanBudget(this.tokenizer, zipScanStart, maximumUntrustedSkipSizeInBytes)) {
300
- throw new ParserHardLimitError(`ZIP stream probing exceeds ${maximumUntrustedSkipSizeInBytes} bytes`);
301
- }
302
-
303
- const zipHeader = await this.readLocalFileHeader();
304
- if (!zipHeader) {
305
- break;
306
- }
307
-
308
- zipEntryCount++;
309
- if (zipEntryCount > maximumZipEntryCount) {
310
- throw new Error(`ZIP entry count exceeds ${maximumZipEntryCount}`);
311
- }
312
-
313
- const next = fileCallback(zipHeader);
314
- stop = Boolean(next.stop);
315
- await this.tokenizer.ignore(zipHeader.extraFieldLength);
316
- const fileData = await readZipEntryData(this, zipHeader, {
317
- shouldBuffer: Boolean(next.handler),
318
- maximumDescriptorLength: Math.min(maximumZipEntrySizeInBytes, getRemainingZipScanBudget(this, zipScanStart)),
319
- });
320
-
321
- if (next.handler) {
322
- await this.inflate(zipHeader, fileData, next.handler);
323
- }
324
-
325
- if (zipHeader.dataDescriptor) {
326
- const dataDescriptor = new Uint8Array(zipDataDescriptorLengthInBytes);
327
- await this.tokenizer.readBuffer(dataDescriptor);
328
- if (Token.UINT32_LE.get(dataDescriptor, 0) !== zipDataDescriptorSignature) {
329
- throw new Error(`Expected data-descriptor-signature at position ${this.tokenizer.position - dataDescriptor.length}`);
330
- }
331
- }
332
-
333
- if (hasExceededUnknownSizeScanBudget(this.tokenizer, zipScanStart, maximumUntrustedSkipSizeInBytes)) {
334
- throw new ParserHardLimitError(`ZIP stream probing exceeds ${maximumUntrustedSkipSizeInBytes} bytes`);
335
- }
336
- } while (!stop);
337
- };
338
-
339
97
  function createByteLimitedReadableStream(stream, maximumBytes) {
340
98
  const reader = stream.getReader();
341
99
  let emittedBytes = 0;
@@ -402,388 +160,6 @@ export async function fileTypeFromBlob(blob, options) {
402
160
  return new FileTypeParser(options).fromBlob(blob);
403
161
  }
404
162
 
405
- function getFileTypeFromMimeType(mimeType) {
406
- mimeType = mimeType.toLowerCase();
407
- switch (mimeType) {
408
- case 'application/epub+zip':
409
- return {
410
- ext: 'epub',
411
- mime: mimeType,
412
- };
413
- case 'application/vnd.oasis.opendocument.text':
414
- return {
415
- ext: 'odt',
416
- mime: mimeType,
417
- };
418
- case 'application/vnd.oasis.opendocument.text-template':
419
- return {
420
- ext: 'ott',
421
- mime: mimeType,
422
- };
423
- case 'application/vnd.oasis.opendocument.spreadsheet':
424
- return {
425
- ext: 'ods',
426
- mime: mimeType,
427
- };
428
- case 'application/vnd.oasis.opendocument.spreadsheet-template':
429
- return {
430
- ext: 'ots',
431
- mime: mimeType,
432
- };
433
- case 'application/vnd.oasis.opendocument.presentation':
434
- return {
435
- ext: 'odp',
436
- mime: mimeType,
437
- };
438
- case 'application/vnd.oasis.opendocument.presentation-template':
439
- return {
440
- ext: 'otp',
441
- mime: mimeType,
442
- };
443
- case 'application/vnd.oasis.opendocument.graphics':
444
- return {
445
- ext: 'odg',
446
- mime: mimeType,
447
- };
448
- case 'application/vnd.oasis.opendocument.graphics-template':
449
- return {
450
- ext: 'otg',
451
- mime: mimeType,
452
- };
453
- case 'application/vnd.openxmlformats-officedocument.presentationml.slideshow':
454
- return {
455
- ext: 'ppsx',
456
- mime: mimeType,
457
- };
458
- case 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet':
459
- return {
460
- ext: 'xlsx',
461
- mime: mimeType,
462
- };
463
- case 'application/vnd.ms-excel.sheet.macroenabled':
464
- return {
465
- ext: 'xlsm',
466
- mime: 'application/vnd.ms-excel.sheet.macroenabled.12',
467
- };
468
- case 'application/vnd.openxmlformats-officedocument.spreadsheetml.template':
469
- return {
470
- ext: 'xltx',
471
- mime: mimeType,
472
- };
473
- case 'application/vnd.ms-excel.template.macroenabled':
474
- return {
475
- ext: 'xltm',
476
- mime: 'application/vnd.ms-excel.template.macroenabled.12',
477
- };
478
- case 'application/vnd.ms-powerpoint.slideshow.macroenabled':
479
- return {
480
- ext: 'ppsm',
481
- mime: 'application/vnd.ms-powerpoint.slideshow.macroenabled.12',
482
- };
483
- case 'application/vnd.openxmlformats-officedocument.wordprocessingml.document':
484
- return {
485
- ext: 'docx',
486
- mime: mimeType,
487
- };
488
- case 'application/vnd.ms-word.document.macroenabled':
489
- return {
490
- ext: 'docm',
491
- mime: 'application/vnd.ms-word.document.macroenabled.12',
492
- };
493
- case 'application/vnd.openxmlformats-officedocument.wordprocessingml.template':
494
- return {
495
- ext: 'dotx',
496
- mime: mimeType,
497
- };
498
- case 'application/vnd.ms-word.template.macroenabledtemplate':
499
- return {
500
- ext: 'dotm',
501
- mime: 'application/vnd.ms-word.template.macroenabled.12',
502
- };
503
- case 'application/vnd.openxmlformats-officedocument.presentationml.template':
504
- return {
505
- ext: 'potx',
506
- mime: mimeType,
507
- };
508
- case 'application/vnd.ms-powerpoint.template.macroenabled':
509
- return {
510
- ext: 'potm',
511
- mime: 'application/vnd.ms-powerpoint.template.macroenabled.12',
512
- };
513
- case 'application/vnd.openxmlformats-officedocument.presentationml.presentation':
514
- return {
515
- ext: 'pptx',
516
- mime: mimeType,
517
- };
518
- case 'application/vnd.ms-powerpoint.presentation.macroenabled':
519
- return {
520
- ext: 'pptm',
521
- mime: 'application/vnd.ms-powerpoint.presentation.macroenabled.12',
522
- };
523
- case 'application/vnd.ms-visio.drawing':
524
- return {
525
- ext: 'vsdx',
526
- mime: 'application/vnd.visio',
527
- };
528
- case 'application/vnd.ms-package.3dmanufacturing-3dmodel+xml':
529
- return {
530
- ext: '3mf',
531
- mime: 'model/3mf',
532
- };
533
- default:
534
- }
535
- }
536
-
537
- function _check(buffer, headers, options) {
538
- options = {
539
- offset: 0,
540
- ...options,
541
- };
542
-
543
- for (const [index, header] of headers.entries()) {
544
- // If a bitmask is set
545
- if (options.mask) {
546
- // If header doesn't equal `buf` with bits masked off
547
- if (header !== (options.mask[index] & buffer[index + options.offset])) {
548
- return false;
549
- }
550
- } else if (header !== buffer[index + options.offset]) {
551
- return false;
552
- }
553
- }
554
-
555
- return true;
556
- }
557
-
558
- export function normalizeSampleSize(sampleSize) {
559
- // `sampleSize` is an explicit caller-controlled tuning knob, not untrusted file input.
560
- // Preserve valid caller-requested probe depth here; applications must bound attacker-derived option values themselves.
561
- if (!Number.isFinite(sampleSize)) {
562
- return reasonableDetectionSizeInBytes;
563
- }
564
-
565
- return Math.max(1, Math.trunc(sampleSize));
566
- }
567
-
568
- function readByobReaderWithSignal(reader, buffer, signal) {
569
- if (signal === undefined) {
570
- return reader.read(buffer);
571
- }
572
-
573
- signal.throwIfAborted();
574
-
575
- return new Promise((resolve, reject) => {
576
- const cleanup = () => {
577
- signal.removeEventListener('abort', onAbort);
578
- };
579
-
580
- const onAbort = () => {
581
- const abortReason = signal.reason;
582
- cleanup();
583
-
584
- (async () => {
585
- try {
586
- await reader.cancel(abortReason);
587
- } catch {}
588
- })();
589
-
590
- reject(abortReason);
591
- };
592
-
593
- signal.addEventListener('abort', onAbort, {once: true});
594
- (async () => {
595
- try {
596
- const result = await reader.read(buffer);
597
- cleanup();
598
- resolve(result);
599
- } catch (error) {
600
- cleanup();
601
- reject(error);
602
- }
603
- })();
604
- });
605
- }
606
-
607
- function normalizeMpegOffsetTolerance(mpegOffsetTolerance) {
608
- // This value controls scan depth and therefore worst-case CPU work.
609
- if (!Number.isFinite(mpegOffsetTolerance)) {
610
- return 0;
611
- }
612
-
613
- return Math.max(0, Math.min(maximumMpegOffsetTolerance, Math.trunc(mpegOffsetTolerance)));
614
- }
615
-
616
- function getKnownFileSizeOrMaximum(fileSize) {
617
- if (!Number.isFinite(fileSize)) {
618
- return Number.MAX_SAFE_INTEGER;
619
- }
620
-
621
- return Math.max(0, fileSize);
622
- }
623
-
624
- function hasUnknownFileSize(tokenizer) {
625
- const fileSize = tokenizer.fileInfo.size;
626
- return (
627
- !Number.isFinite(fileSize)
628
- || fileSize === Number.MAX_SAFE_INTEGER
629
- );
630
- }
631
-
632
- function hasExceededUnknownSizeScanBudget(tokenizer, startOffset, maximumBytes) {
633
- return (
634
- hasUnknownFileSize(tokenizer)
635
- && tokenizer.position - startOffset > maximumBytes
636
- );
637
- }
638
-
639
- function getMaximumZipBufferedReadLength(tokenizer) {
640
- const fileSize = tokenizer.fileInfo.size;
641
- const remainingBytes = Number.isFinite(fileSize)
642
- ? Math.max(0, fileSize - tokenizer.position)
643
- : Number.MAX_SAFE_INTEGER;
644
-
645
- return Math.min(remainingBytes, maximumZipBufferedReadSizeInBytes);
646
- }
647
-
648
- function isRecoverableZipError(error) {
649
- if (error instanceof strtok3.EndOfStreamError) {
650
- return true;
651
- }
652
-
653
- if (error instanceof ParserHardLimitError) {
654
- return true;
655
- }
656
-
657
- if (!(error instanceof Error)) {
658
- return false;
659
- }
660
-
661
- if (recoverableZipErrorMessages.has(error.message)) {
662
- return true;
663
- }
664
-
665
- if (recoverableZipErrorCodes.has(error.code)) {
666
- return true;
667
- }
668
-
669
- for (const prefix of recoverableZipErrorMessagePrefixes) {
670
- if (error.message.startsWith(prefix)) {
671
- return true;
672
- }
673
- }
674
-
675
- return false;
676
- }
677
-
678
- function canReadZipEntryForDetection(zipHeader, maximumSize = maximumZipEntrySizeInBytes) {
679
- const sizes = [zipHeader.compressedSize, zipHeader.uncompressedSize];
680
- for (const size of sizes) {
681
- if (
682
- !Number.isFinite(size)
683
- || size < 0
684
- || size > maximumSize
685
- ) {
686
- return false;
687
- }
688
- }
689
-
690
- return true;
691
- }
692
-
693
- function createOpenXmlZipDetectionState() {
694
- return {
695
- hasContentTypesEntry: false,
696
- hasParsedContentTypesEntry: false,
697
- isParsingContentTypes: false,
698
- hasUnparseableContentTypes: false,
699
- hasWordDirectory: false,
700
- hasPresentationDirectory: false,
701
- hasSpreadsheetDirectory: false,
702
- hasThreeDimensionalModelEntry: false,
703
- };
704
- }
705
-
706
- function updateOpenXmlZipDetectionStateFromFilename(openXmlState, filename) {
707
- if (filename.startsWith('word/')) {
708
- openXmlState.hasWordDirectory = true;
709
- }
710
-
711
- if (filename.startsWith('ppt/')) {
712
- openXmlState.hasPresentationDirectory = true;
713
- }
714
-
715
- if (filename.startsWith('xl/')) {
716
- openXmlState.hasSpreadsheetDirectory = true;
717
- }
718
-
719
- if (
720
- filename.startsWith('3D/')
721
- && filename.endsWith('.model')
722
- ) {
723
- openXmlState.hasThreeDimensionalModelEntry = true;
724
- }
725
- }
726
-
727
- function getOpenXmlFileTypeFromZipEntries(openXmlState) {
728
- // Only use directory-name heuristic when [Content_Types].xml was present in the archive
729
- // but its handler was skipped (not invoked, not currently running, and not already resolved).
730
- // This avoids guessing from directory names when content-type parsing already gave a definitive answer or failed.
731
- if (
732
- !openXmlState.hasContentTypesEntry
733
- || openXmlState.hasUnparseableContentTypes
734
- || openXmlState.isParsingContentTypes
735
- || openXmlState.hasParsedContentTypesEntry
736
- ) {
737
- return;
738
- }
739
-
740
- if (openXmlState.hasWordDirectory) {
741
- return {
742
- ext: 'docx',
743
- mime: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
744
- };
745
- }
746
-
747
- if (openXmlState.hasPresentationDirectory) {
748
- return {
749
- ext: 'pptx',
750
- mime: 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
751
- };
752
- }
753
-
754
- if (openXmlState.hasSpreadsheetDirectory) {
755
- return {
756
- ext: 'xlsx',
757
- mime: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
758
- };
759
- }
760
-
761
- if (openXmlState.hasThreeDimensionalModelEntry) {
762
- return {
763
- ext: '3mf',
764
- mime: 'model/3mf',
765
- };
766
- }
767
- }
768
-
769
- function getOpenXmlMimeTypeFromContentTypesXml(xmlContent) {
770
- // We only need the `ContentType="...main+xml"` value, so a small string scan is enough and avoids full XML parsing.
771
- const endPosition = xmlContent.indexOf('.main+xml"');
772
- if (endPosition === -1) {
773
- const mimeType = 'application/vnd.ms-package.3dmanufacturing-3dmodel+xml';
774
- if (xmlContent.includes(`ContentType="${mimeType}"`)) {
775
- return mimeType;
776
- }
777
-
778
- return;
779
- }
780
-
781
- const truncatedContent = xmlContent.slice(0, endPosition);
782
- const firstQuotePosition = truncatedContent.lastIndexOf('"');
783
- // If no quote is found, `lastIndexOf` returns -1 and this intentionally falls back to the full truncated prefix.
784
- return truncatedContent.slice(firstQuotePosition + 1);
785
- }
786
-
787
163
  export async function fileTypeFromTokenizer(tokenizer, options) {
788
164
  return new FileTypeParser(options).fromTokenizer(tokenizer);
789
165
  }
@@ -816,7 +192,7 @@ export class FileTypeParser {
816
192
  }
817
193
 
818
194
  createTokenizerFromWebStream(stream) {
819
- return patchWebByobTokenizerClose(strtok3.fromWebStream(stream, this.getTokenizerOptions()));
195
+ return strtok3.fromWebStream(toDefaultStream(stream), this.getTokenizerOptions());
820
196
  }
821
197
 
822
198
  async parseTokenizer(tokenizer, detectionReentryCount = 0) {
@@ -883,41 +259,96 @@ export class FileTypeParser {
883
259
  return this.fromTokenizer(tokenizer);
884
260
  }
885
261
 
262
+ async fromFile(path) {
263
+ this.options.signal?.throwIfAborted();
264
+ // TODO: Remove this when `strtok3.fromFile()` safely rejects non-regular filesystem objects without a pathname race.
265
+ const [{default: fsPromises}, {FileTokenizer}] = await Promise.all([
266
+ importAtRuntime('node:fs/promises'),
267
+ importAtRuntime('strtok3'),
268
+ ]);
269
+ const fileHandle = await fsPromises.open(path, fsPromises.constants.O_RDONLY | fsPromises.constants.O_NONBLOCK);
270
+ const fileStat = await fileHandle.stat();
271
+ if (!fileStat.isFile()) {
272
+ await fileHandle.close();
273
+ return;
274
+ }
275
+
276
+ const tokenizer = new FileTokenizer(fileHandle, {
277
+ ...this.getTokenizerOptions(),
278
+ fileInfo: {path, size: fileStat.size},
279
+ });
280
+ return this.fromTokenizer(tokenizer);
281
+ }
282
+
886
283
  async toDetectionStream(stream, options) {
284
+ this.options.signal?.throwIfAborted();
887
285
  const sampleSize = normalizeSampleSize(options?.sampleSize ?? reasonableDetectionSizeInBytes);
888
286
  let detectedFileType;
889
- let firstChunk;
287
+ let streamEnded = false;
890
288
 
891
- const reader = stream.getReader({mode: 'byob'});
892
- try {
893
- // Read the first chunk from the stream
894
- const {value: chunk, done} = await readByobReaderWithSignal(reader, new Uint8Array(sampleSize), this.options.signal);
895
- firstChunk = chunk;
896
- if (!done && chunk) {
897
- try {
898
- // Attempt to detect the file type from the chunk
899
- detectedFileType = await this.fromBuffer(chunk.subarray(0, sampleSize));
900
- } catch (error) {
901
- if (!(error instanceof strtok3.EndOfStreamError)) {
902
- throw error; // Re-throw non-EndOfStreamError
903
- }
289
+ const reader = stream.getReader();
290
+ const chunks = [];
291
+ let totalSize = 0;
904
292
 
905
- detectedFileType = undefined;
293
+ try {
294
+ while (totalSize < sampleSize) {
295
+ const {value, done} = await readWithSignal(reader, this.options.signal);
296
+ if (done || !value) {
297
+ streamEnded = true;
298
+ break;
906
299
  }
300
+
301
+ chunks.push(value);
302
+ totalSize += value.length;
907
303
  }
908
304
 
909
- firstChunk = chunk;
305
+ if (
306
+ !streamEnded
307
+ && totalSize === sampleSize
308
+ ) {
309
+ const {value, done} = await readWithSignal(reader, this.options.signal);
310
+ if (done || !value) {
311
+ streamEnded = true;
312
+ } else {
313
+ chunks.push(value);
314
+ totalSize += value.length;
315
+ }
316
+ }
910
317
  } finally {
911
- reader.releaseLock(); // Ensure the reader is released
318
+ reader.releaseLock();
912
319
  }
913
320
 
914
- // Create a new ReadableStream to manage locking issues
321
+ if (totalSize > 0) {
322
+ const sample = chunks.length === 1 ? chunks[0] : concatUint8Arrays(chunks);
323
+ try {
324
+ detectedFileType = await this.fromBuffer(sample.subarray(0, sampleSize));
325
+ } catch (error) {
326
+ if (!(error instanceof strtok3.EndOfStreamError)) {
327
+ throw error;
328
+ }
329
+
330
+ detectedFileType = undefined;
331
+ }
332
+
333
+ if (
334
+ !streamEnded
335
+ && detectedFileType?.ext === 'pages'
336
+ ) {
337
+ detectedFileType = {
338
+ ext: 'zip',
339
+ mime: 'application/zip',
340
+ };
341
+ }
342
+ }
343
+
344
+ // Prepend collected chunks and pipe the rest through
915
345
  const transformStream = new TransformStream({
916
- async start(controller) {
917
- controller.enqueue(firstChunk); // Enqueue the initial chunk
346
+ start(controller) {
347
+ for (const chunk of chunks) {
348
+ controller.enqueue(chunk);
349
+ }
918
350
  },
919
351
  transform(chunk, controller) {
920
- // Pass through the chunks without modification
921
352
  controller.enqueue(chunk);
922
353
  },
923
354
  });
@@ -951,7 +382,6 @@ export class FileTypeParser {
951
382
  }, unknownSizeGzipProbeTimeoutInMilliseconds);
952
383
  probeSignal = this.options.signal === undefined
953
384
  ? timeoutController.signal
954
- // eslint-disable-next-line n/no-unsupported-features/node-builtins
955
385
  : AbortSignal.any([this.options.signal, timeoutController.signal]);
956
386
  probeParser = new FileTypeParser({
957
387
  ...this.options,
@@ -994,7 +424,7 @@ export class FileTypeParser {
994
424
  }
995
425
 
996
426
  check(header, options) {
997
- return _check(this.buffer, header, options);
427
+ return checkBytes(this.buffer, header, options);
998
428
  }
999
429
 
1000
430
  checkString(header, options) {
@@ -1141,7 +571,7 @@ export class FileTypeParser {
1141
571
  const isUnknownFileSize = hasUnknownFileSize(tokenizer);
1142
572
  if (
1143
573
  !Number.isFinite(id3HeaderLength)
1144
- || id3HeaderLength < 0
574
+ || id3HeaderLength < 0
1145
575
  // Keep ID3 probing bounded for unknown-size streams to avoid attacker-controlled large skips.
1146
576
  || (
1147
577
  isUnknownFileSize
@@ -1267,108 +697,7 @@ export class FileTypeParser {
1267
697
  // Zip-based file formats
1268
698
  // Need to be before the `zip` check
1269
699
  if (this.check([0x50, 0x4B, 0x3, 0x4])) { // Local file header signature
1270
- let fileType;
1271
- const openXmlState = createOpenXmlZipDetectionState();
1272
-
1273
- try {
1274
- await new ZipHandler(tokenizer).unzip(zipHeader => {
1275
- updateOpenXmlZipDetectionStateFromFilename(openXmlState, zipHeader.filename);
1276
-
1277
- const isOpenXmlContentTypesEntry = zipHeader.filename === '[Content_Types].xml';
1278
- const openXmlFileTypeFromEntries = getOpenXmlFileTypeFromZipEntries(openXmlState);
1279
- if (
1280
- !isOpenXmlContentTypesEntry
1281
- && openXmlFileTypeFromEntries
1282
- ) {
1283
- fileType = openXmlFileTypeFromEntries;
1284
- return {
1285
- stop: true,
1286
- };
1287
- }
1288
-
1289
- switch (zipHeader.filename) {
1290
- case 'META-INF/mozilla.rsa':
1291
- fileType = {
1292
- ext: 'xpi',
1293
- mime: 'application/x-xpinstall',
1294
- };
1295
- return {
1296
- stop: true,
1297
- };
1298
- case 'META-INF/MANIFEST.MF':
1299
- fileType = {
1300
- ext: 'jar',
1301
- mime: 'application/java-archive',
1302
- };
1303
- return {
1304
- stop: true,
1305
- };
1306
- case 'mimetype':
1307
- if (!canReadZipEntryForDetection(zipHeader, maximumZipTextEntrySizeInBytes)) {
1308
- return {};
1309
- }
1310
-
1311
- return {
1312
- async handler(fileData) {
1313
- // Use TextDecoder to decode the UTF-8 encoded data
1314
- const mimeType = new TextDecoder('utf-8').decode(fileData).trim();
1315
- fileType = getFileTypeFromMimeType(mimeType);
1316
- },
1317
- stop: true,
1318
- };
1319
-
1320
- case '[Content_Types].xml': {
1321
- openXmlState.hasContentTypesEntry = true;
1322
-
1323
- if (!canReadZipEntryForDetection(zipHeader, maximumZipTextEntrySizeInBytes)) {
1324
- openXmlState.hasUnparseableContentTypes = true;
1325
- return {};
1326
- }
1327
-
1328
- openXmlState.isParsingContentTypes = true;
1329
- return {
1330
- async handler(fileData) {
1331
- // Use TextDecoder to decode the UTF-8 encoded data
1332
- const xmlContent = new TextDecoder('utf-8').decode(fileData);
1333
- const mimeType = getOpenXmlMimeTypeFromContentTypesXml(xmlContent);
1334
- if (mimeType) {
1335
- fileType = getFileTypeFromMimeType(mimeType);
1336
- }
1337
-
1338
- openXmlState.hasParsedContentTypesEntry = true;
1339
- openXmlState.isParsingContentTypes = false;
1340
- },
1341
- stop: true,
1342
- };
1343
- }
1344
-
1345
- default:
1346
- if (/classes\d*\.dex/.test(zipHeader.filename)) {
1347
- fileType = {
1348
- ext: 'apk',
1349
- mime: 'application/vnd.android.package-archive',
1350
- };
1351
- return {stop: true};
1352
- }
1353
-
1354
- return {};
1355
- }
1356
- });
1357
- } catch (error) {
1358
- if (!isRecoverableZipError(error)) {
1359
- throw error;
1360
- }
1361
-
1362
- if (openXmlState.isParsingContentTypes) {
1363
- openXmlState.isParsingContentTypes = false;
1364
- openXmlState.hasUnparseableContentTypes = true;
1365
- }
1366
- }
1367
-
1368
- return fileType ?? getOpenXmlFileTypeFromZipEntries(openXmlState) ?? {
1369
- ext: 'zip',
1370
- mime: 'application/zip',
1371
- };
700
+ return detectZip(tokenizer);
1372
701
  }
1373
702
 
1374
703
  if (this.checkString('OggS')) {
@@ -1378,7 +707,7 @@ export class FileTypeParser {
1378
707
  await tokenizer.readBuffer(type);
1379
708
 
1380
709
  // Needs to be before `ogg` check
1381
- if (_check(type, [0x4F, 0x70, 0x75, 0x73, 0x48, 0x65, 0x61, 0x64])) {
710
+ if (checkBytes(type, [0x4F, 0x70, 0x75, 0x73, 0x48, 0x65, 0x61, 0x64])) {
1382
711
  return {
1383
712
  ext: 'opus',
1384
713
  mime: 'audio/ogg; codecs=opus',
@@ -1386,7 +715,7 @@ export class FileTypeParser {
1386
715
  }
1387
716
 
1388
717
  // If ' theora' in header.
1389
- if (_check(type, [0x80, 0x74, 0x68, 0x65, 0x6F, 0x72, 0x61])) {
718
+ if (checkBytes(type, [0x80, 0x74, 0x68, 0x65, 0x6F, 0x72, 0x61])) {
1390
719
  return {
1391
720
  ext: 'ogv',
1392
721
  mime: 'video/ogg',
@@ -1394,7 +723,7 @@ export class FileTypeParser {
1394
723
  }
1395
724
 
1396
725
  // If '\x01video' in header.
1397
- if (_check(type, [0x01, 0x76, 0x69, 0x64, 0x65, 0x6F, 0x00])) {
726
+ if (checkBytes(type, [0x01, 0x76, 0x69, 0x64, 0x65, 0x6F, 0x00])) {
1398
727
  return {
1399
728
  ext: 'ogm',
1400
729
  mime: 'video/ogg',
@@ -1402,7 +731,7 @@ export class FileTypeParser {
1402
731
  }
1403
732
 
1404
733
  // If ' FLAC' in header https://xiph.org/flac/faq.html
1405
- if (_check(type, [0x7F, 0x46, 0x4C, 0x41, 0x43])) {
734
+ if (checkBytes(type, [0x7F, 0x46, 0x4C, 0x41, 0x43])) {
1406
735
  return {
1407
736
  ext: 'oga',
1408
737
  mime: 'audio/ogg',
@@ -1410,7 +739,7 @@ export class FileTypeParser {
1410
739
  }
1411
740
 
1412
741
  // 'Speex ' in header https://en.wikipedia.org/wiki/Speex
1413
- if (_check(type, [0x53, 0x70, 0x65, 0x65, 0x78, 0x20, 0x20])) {
742
+ if (checkBytes(type, [0x53, 0x70, 0x65, 0x65, 0x78, 0x20, 0x20])) {
1414
743
  return {
1415
744
  ext: 'spx',
1416
745
  mime: 'audio/ogg',
@@ -1418,7 +747,7 @@ export class FileTypeParser {
1418
747
  }
1419
748
 
1420
749
  // If '\x01vorbis' in header
1421
- if (_check(type, [0x01, 0x76, 0x6F, 0x72, 0x62, 0x69, 0x73])) {
750
+ if (checkBytes(type, [0x01, 0x76, 0x6F, 0x72, 0x62, 0x69, 0x73])) {
1422
751
  return {
1423
752
  ext: 'ogg',
1424
753
  mime: 'audio/ogg',
@@ -1494,7 +823,7 @@ export class FileTypeParser {
1494
823
  if (this.checkString('LZIP')) {
1495
824
  return {
1496
825
  ext: 'lz',
1497
- mime: 'application/x-lzip',
826
+ mime: 'application/lzip',
1498
827
  };
1499
828
  }
1500
829
 
@@ -1559,110 +888,7 @@ export class FileTypeParser {
1559
888
 
1560
889
  // https://github.com/file/file/blob/master/magic/Magdir/matroska
1561
890
  if (this.check([0x1A, 0x45, 0xDF, 0xA3])) { // Root element: EBML
1562
- async function readField() {
1563
- const msb = await tokenizer.peekNumber(Token.UINT8);
1564
- let mask = 0x80;
1565
- let ic = 0; // 0 = A, 1 = B, 2 = C, 3 = D
1566
-
1567
- while ((msb & mask) === 0 && mask !== 0) {
1568
- ++ic;
1569
- mask >>= 1;
1570
- }
1571
-
1572
- const id = new Uint8Array(ic + 1);
1573
- await safeReadBuffer(tokenizer, id, undefined, {
1574
- maximumLength: id.length,
1575
- reason: 'EBML field',
1576
- });
1577
- return id;
1578
- }
1579
-
1580
- async function readElement() {
1581
- const idField = await readField();
1582
- const lengthField = await readField();
1583
-
1584
- lengthField[0] ^= 0x80 >> (lengthField.length - 1);
1585
- const nrLength = Math.min(6, lengthField.length); // JavaScript can max read 6 bytes integer
1586
-
1587
- const idView = new DataView(idField.buffer);
1588
- const lengthView = new DataView(lengthField.buffer, lengthField.length - nrLength, nrLength);
1589
-
1590
- return {
1591
- id: getUintBE(idView),
1592
- len: getUintBE(lengthView),
1593
- };
1594
- }
1595
-
1596
- async function readChildren(children) {
1597
- let ebmlElementCount = 0;
1598
- while (children > 0) {
1599
- ebmlElementCount++;
1600
- if (ebmlElementCount > maximumEbmlElementCount) {
1601
- return;
1602
- }
1603
-
1604
- if (hasExceededUnknownSizeScanBudget(tokenizer, ebmlScanStart, maximumUntrustedSkipSizeInBytes)) {
1605
- return;
1606
- }
1607
-
1608
- const previousPosition = tokenizer.position;
1609
- const element = await readElement();
1610
-
1611
- if (element.id === 0x42_82) {
1612
- // `DocType` is a short string ("webm", "matroska", ...), reject implausible lengths to avoid large allocations.
1613
- if (element.len > maximumEbmlDocumentTypeSizeInBytes) {
1614
- return;
1615
- }
1616
-
1617
- const documentTypeLength = getSafeBound(element.len, maximumEbmlDocumentTypeSizeInBytes, 'EBML DocType');
1618
- const rawValue = await tokenizer.readToken(new Token.StringType(documentTypeLength));
1619
- return rawValue.replaceAll(/\00.*$/g, ''); // Return DocType
1620
- }
1621
-
1622
- if (
1623
- hasUnknownFileSize(tokenizer)
1624
- && (
1625
- !Number.isFinite(element.len)
1626
- || element.len < 0
1627
- || element.len > maximumEbmlElementPayloadSizeInBytes
1628
- )
1629
- ) {
1630
- return;
1631
- }
1632
-
1633
- await safeIgnore(tokenizer, element.len, {
1634
- maximumLength: hasUnknownFileSize(tokenizer) ? maximumEbmlElementPayloadSizeInBytes : tokenizer.fileInfo.size,
1635
- reason: 'EBML payload',
1636
- }); // ignore payload
1637
- --children;
1638
-
1639
- // Safeguard against malformed files: bail if the position did not advance.
1640
- if (tokenizer.position <= previousPosition) {
1641
- return;
1642
- }
1643
- }
1644
- }
1645
-
1646
- const rootElement = await readElement();
1647
- const ebmlScanStart = tokenizer.position;
1648
- const documentType = await readChildren(rootElement.len);
1649
-
1650
- switch (documentType) {
1651
- case 'webm':
1652
- return {
1653
- ext: 'webm',
1654
- mime: 'video/webm',
1655
- };
1656
-
1657
- case 'matroska':
1658
- return {
1659
- ext: 'mkv',
1660
- mime: 'video/matroska',
1661
- };
1662
-
1663
- default:
1664
- return;
1665
- }
891
+ return detectEbml(tokenizer);
1666
892
  }
1667
893
 
1668
894
  if (this.checkString('SQLi')) {
@@ -1760,7 +986,7 @@ export class FileTypeParser {
1760
986
  if (this.check([0x04, 0x22, 0x4D, 0x18])) {
1761
987
  return {
1762
988
  ext: 'lz4',
1763
- mime: 'application/x-lz4', // Invented by us
989
+ mime: 'application/x-lz4', // Informal, used by freedesktop.org shared-mime-info
1764
990
  };
1765
991
  }
1766
992
 
@@ -1795,7 +1021,7 @@ export class FileTypeParser {
1795
1021
  };
1796
1022
  }
1797
1023
 
1798
- if (this.checkString('{\\rtf')) {
1024
+ if (this.checkString(String.raw`{\rtf`)) {
1799
1025
  return {
1800
1026
  ext: 'rtf',
1801
1027
  mime: 'application/rtf',
@@ -1896,7 +1122,7 @@ export class FileTypeParser {
1896
1122
  if (this.checkString('DRACO')) {
1897
1123
  return {
1898
1124
  ext: 'drc',
1899
- mime: 'application/vnd.google.draco', // Invented by us
1125
+ mime: 'application/x-ft-draco',
1900
1126
  };
1901
1127
  }
1902
1128
 
@@ -1942,7 +1168,7 @@ export class FileTypeParser {
1942
1168
 
1943
1169
  if (this.checkString('AC')) {
1944
1170
  const version = new Token.StringType(4, 'latin1').get(this.buffer, 2);
1945
- if (version.match('^d*') && version >= 1000 && version <= 1050) {
1171
+ if (/^\d+$/v.test(version) && version >= 1000 && version <= 1050) {
1946
1172
  return {
1947
1173
  ext: 'dwg',
1948
1174
  mime: 'image/vnd.dwg',
@@ -1997,110 +1223,7 @@ export class FileTypeParser {
1997
1223
  // -- 8-byte signatures --
1998
1224
 
1999
1225
  if (this.check([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A])) {
2000
- const pngFileType = {
2001
- ext: 'png',
2002
- mime: 'image/png',
2003
- };
2004
-
2005
- const apngFileType = {
2006
- ext: 'apng',
2007
- mime: 'image/apng',
2008
- };
2009
-
2010
- // APNG format (https://wiki.mozilla.org/APNG_Specification)
2011
- // 1. Find the first IDAT (image data) chunk (49 44 41 54)
2012
- // 2. Check if there is an "acTL" chunk before the IDAT one (61 63 54 4C)
2013
-
2014
- // Offset calculated as follows:
2015
- // - 8 bytes: PNG signature
2016
- // - 4 (length) + 4 (chunk type) + 13 (chunk data) + 4 (CRC): IHDR chunk
2017
-
2018
- await tokenizer.ignore(8); // ignore PNG signature
2019
-
2020
- async function readChunkHeader() {
2021
- return {
2022
- length: await tokenizer.readToken(Token.INT32_BE),
2023
- type: await tokenizer.readToken(new Token.StringType(4, 'latin1')),
2024
- };
2025
- }
2026
-
2027
- const isUnknownPngStream = hasUnknownFileSize(tokenizer);
2028
- const pngScanStart = tokenizer.position;
2029
- let pngChunkCount = 0;
2030
- let hasSeenImageHeader = false;
2031
- do {
2032
- pngChunkCount++;
2033
- if (pngChunkCount > maximumPngChunkCount) {
2034
- break;
2035
- }
2036
-
2037
- if (hasExceededUnknownSizeScanBudget(tokenizer, pngScanStart, maximumPngStreamScanBudgetInBytes)) {
2038
- break;
2039
- }
2040
-
2041
- const previousPosition = tokenizer.position;
2042
- const chunk = await readChunkHeader();
2043
- if (chunk.length < 0) {
2044
- return; // Invalid chunk length
2045
- }
2046
-
2047
- if (chunk.type === 'IHDR') {
2048
- // PNG requires the first real image header to be a 13-byte IHDR chunk.
2049
- if (chunk.length !== 13) {
2050
- return;
2051
- }
2052
-
2053
- hasSeenImageHeader = true;
2054
- }
2055
-
2056
- switch (chunk.type) {
2057
- case 'IDAT':
2058
- return pngFileType;
2059
- case 'acTL':
2060
- return apngFileType;
2061
- default:
2062
- if (
2063
- !hasSeenImageHeader
2064
- && chunk.type !== 'CgBI'
2065
- ) {
2066
- return;
2067
- }
2068
-
2069
- if (
2070
- isUnknownPngStream
2071
- && chunk.length > maximumPngChunkSizeInBytes
2072
- ) {
2073
- // Avoid huge attacker-controlled skips when probing unknown-size streams.
2074
- return hasSeenImageHeader && isPngAncillaryChunk(chunk.type) ? pngFileType : undefined;
2075
- }
2076
-
2077
- try {
2078
- await safeIgnore(tokenizer, chunk.length + 4, {
2079
- maximumLength: isUnknownPngStream ? maximumPngChunkSizeInBytes + 4 : tokenizer.fileInfo.size,
2080
- reason: 'PNG chunk payload',
2081
- }); // Ignore chunk-data + CRC
2082
- } catch (error) {
2083
- if (
2084
- !isUnknownPngStream
2085
- && (
2086
- error instanceof ParserHardLimitError
2087
- || error instanceof strtok3.EndOfStreamError
2088
- )
2089
- ) {
2090
- return pngFileType;
2091
- }
2092
-
2093
- throw error;
2094
- }
2095
- }
2096
-
2097
- // Safeguard against malformed files: bail if the position did not advance.
2098
- if (tokenizer.position <= previousPosition) {
2099
- break;
2100
- }
2101
- } while (tokenizer.position + 8 < tokenizer.fileInfo.size);
2102
-
2103
- return pngFileType;
1226
+ return detectPng(tokenizer);
2104
1227
  }
2105
1228
 
2106
1229
  if (this.check([0x41, 0x52, 0x52, 0x4F, 0x57, 0x31, 0x00, 0x00])) {
@@ -2258,116 +1381,7 @@ export class FileTypeParser {
2258
1381
 
2259
1382
  // ASF_Header_Object first 80 bytes
2260
1383
  if (this.check([0x30, 0x26, 0xB2, 0x75, 0x8E, 0x66, 0xCF, 0x11, 0xA6, 0xD9])) {
2261
- let isMalformedAsf = false;
2262
- try {
2263
- async function readHeader() {
2264
- const guid = new Uint8Array(16);
2265
- await safeReadBuffer(tokenizer, guid, undefined, {
2266
- maximumLength: guid.length,
2267
- reason: 'ASF header GUID',
2268
- });
2269
- return {
2270
- id: guid,
2271
- size: Number(await tokenizer.readToken(Token.UINT64_LE)),
2272
- };
2273
- }
2274
-
2275
- await safeIgnore(tokenizer, 30, {
2276
- maximumLength: 30,
2277
- reason: 'ASF header prelude',
2278
- });
2279
- const isUnknownFileSize = hasUnknownFileSize(tokenizer);
2280
- const asfHeaderScanStart = tokenizer.position;
2281
- let asfHeaderObjectCount = 0;
2282
- while (tokenizer.position + 24 < tokenizer.fileInfo.size) {
2283
- asfHeaderObjectCount++;
2284
- if (asfHeaderObjectCount > maximumAsfHeaderObjectCount) {
2285
- break;
2286
- }
2287
-
2288
- if (hasExceededUnknownSizeScanBudget(tokenizer, asfHeaderScanStart, maximumUntrustedSkipSizeInBytes)) {
2289
- break;
2290
- }
2291
-
2292
- const previousPosition = tokenizer.position;
2293
- const header = await readHeader();
2294
- let payload = header.size - 24;
2295
- if (
2296
- !Number.isFinite(payload)
2297
- || payload < 0
2298
- ) {
2299
- isMalformedAsf = true;
2300
- break;
2301
- }
2302
-
2303
- if (_check(header.id, [0x91, 0x07, 0xDC, 0xB7, 0xB7, 0xA9, 0xCF, 0x11, 0x8E, 0xE6, 0x00, 0xC0, 0x0C, 0x20, 0x53, 0x65])) {
2304
- // Sync on Stream-Properties-Object (B7DC0791-A9B7-11CF-8EE6-00C00C205365)
2305
- const typeId = new Uint8Array(16);
2306
- payload -= await safeReadBuffer(tokenizer, typeId, undefined, {
2307
- maximumLength: typeId.length,
2308
- reason: 'ASF stream type GUID',
2309
- });
2310
-
2311
- if (_check(typeId, [0x40, 0x9E, 0x69, 0xF8, 0x4D, 0x5B, 0xCF, 0x11, 0xA8, 0xFD, 0x00, 0x80, 0x5F, 0x5C, 0x44, 0x2B])) {
2312
- // Found audio:
2313
- return {
2314
- ext: 'asf',
2315
- mime: 'audio/x-ms-asf',
2316
- };
2317
- }
2318
-
2319
- if (_check(typeId, [0xC0, 0xEF, 0x19, 0xBC, 0x4D, 0x5B, 0xCF, 0x11, 0xA8, 0xFD, 0x00, 0x80, 0x5F, 0x5C, 0x44, 0x2B])) {
2320
- // Found video:
2321
- return {
2322
- ext: 'asf',
2323
- mime: 'video/x-ms-asf',
2324
- };
2325
- }
2326
-
2327
- break;
2328
- }
2329
-
2330
- if (
2331
- isUnknownFileSize
2332
- && payload > maximumAsfHeaderPayloadSizeInBytes
2333
- ) {
2334
- isMalformedAsf = true;
2335
- break;
2336
- }
2337
-
2338
- await safeIgnore(tokenizer, payload, {
2339
- maximumLength: isUnknownFileSize ? maximumAsfHeaderPayloadSizeInBytes : tokenizer.fileInfo.size,
2340
- reason: 'ASF header payload',
2341
- });
2342
-
2343
- // Safeguard against malformed files: break if the position did not advance.
2344
- if (tokenizer.position <= previousPosition) {
2345
- isMalformedAsf = true;
2346
- break;
2347
- }
2348
- }
2349
- } catch (error) {
2350
- if (
2351
- error instanceof strtok3.EndOfStreamError
2352
- || error instanceof ParserHardLimitError
2353
- ) {
2354
- if (hasUnknownFileSize(tokenizer)) {
2355
- isMalformedAsf = true;
2356
- }
2357
- } else {
2358
- throw error;
2359
- }
2360
- }
2361
-
2362
- if (isMalformedAsf) {
2363
- return;
2364
- }
2365
-
2366
- // Default to ASF generic extension
2367
- return {
2368
- ext: 'asf',
2369
- mime: 'application/vnd.ms-asf',
2370
- };
1384
+ return detectAsf(tokenizer);
2371
1385
  }
2372
1386
 
2373
1387
  if (this.check([0xAB, 0x4B, 0x54, 0x58, 0x20, 0x31, 0x31, 0xBB, 0x0D, 0x0A, 0x1A, 0x0A])) {
@@ -2581,21 +1595,21 @@ export class FileTypeParser {
2581
1595
  if (this.check([0x4C, 0x00, 0x00, 0x00, 0x01, 0x14, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0xC0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x46])) {
2582
1596
  return {
2583
1597
  ext: 'lnk',
2584
- mime: 'application/x.ms.shortcut', // Invented by us
1598
+ mime: 'application/x-ms-shortcut', // Informal, used by freedesktop.org shared-mime-info
2585
1599
  };
2586
1600
  }
2587
1601
 
2588
1602
  if (this.check([0x62, 0x6F, 0x6F, 0x6B, 0x00, 0x00, 0x00, 0x00, 0x6D, 0x61, 0x72, 0x6B, 0x00, 0x00, 0x00, 0x00])) {
2589
1603
  return {
2590
1604
  ext: 'alias',
2591
- mime: 'application/x.apple.alias', // Invented by us
1605
+ mime: 'application/x-ft-apple.alias',
2592
1606
  };
2593
1607
  }
2594
1608
 
2595
1609
  if (this.checkString('Kaydara FBX Binary \u0000')) {
2596
1610
  return {
2597
1611
  ext: 'fbx',
2598
- mime: 'application/x.autodesk.fbx', // Invented by us
1612
+ mime: 'application/x-ft-fbx',
2599
1613
  };
2600
1614
  }
2601
1615
 
@@ -2897,3 +1911,7 @@ export class FileTypeParser {
2897
1911
 
2898
1912
  export const supportedExtensions = new Set(extensions);
2899
1913
  export const supportedMimeTypes = new Set(mimeTypes);
1914
+
1915
+ export async function fileTypeFromFile(path, options) {
1916
+ return (new FileTypeParser(options)).fromFile(path);
1917
+ }