@depup/file-type 21.3.4-depup.0 → 22.0.1-depup.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -11
- package/changes.json +3 -16
- package/package.json +38 -67
- package/readme.md +34 -104
- package/source/detectors/asf.js +127 -0
- package/source/detectors/ebml.js +120 -0
- package/source/detectors/png.js +123 -0
- package/source/detectors/zip.js +643 -0
- package/{core.d.ts → source/index.d.ts} +49 -22
- package/{core.js → source/index.js} +154 -1136
- package/source/index.test-d.ts +53 -0
- package/source/parser.js +65 -0
- package/{supported.js → source/supported.js} +14 -6
- package/{util.js → source/tokens.js} +2 -2
- package/index.d.ts +0 -98
- package/index.js +0 -163
|
@@ -4,338 +4,96 @@ Primary entry point, Node.js specific entry point is index.js
|
|
|
4
4
|
|
|
5
5
|
import * as Token from 'token-types';
|
|
6
6
|
import * as strtok3 from 'strtok3/core';
|
|
7
|
-
import {
|
|
8
|
-
import {
|
|
7
|
+
import {GzipHandler} from '@tokenizer/inflate';
|
|
8
|
+
import {concatUint8Arrays} from 'uint8array-extras';
|
|
9
9
|
import {
|
|
10
10
|
stringToBytes,
|
|
11
11
|
tarHeaderChecksumMatches,
|
|
12
12
|
uint32SyncSafeToken,
|
|
13
|
-
} from './
|
|
13
|
+
} from './tokens.js';
|
|
14
14
|
import {extensions, mimeTypes} from './supported.js';
|
|
15
|
+
import {
|
|
16
|
+
maximumUntrustedSkipSizeInBytes,
|
|
17
|
+
ParserHardLimitError,
|
|
18
|
+
safeIgnore,
|
|
19
|
+
checkBytes,
|
|
20
|
+
hasUnknownFileSize,
|
|
21
|
+
} from './parser.js';
|
|
22
|
+
import {detectZip} from './detectors/zip.js';
|
|
23
|
+
import {detectEbml} from './detectors/ebml.js';
|
|
24
|
+
import {detectPng} from './detectors/png.js';
|
|
25
|
+
import {detectAsf} from './detectors/asf.js';
|
|
15
26
|
|
|
16
27
|
export const reasonableDetectionSizeInBytes = 4100; // A fair amount of file-types are detectable within this range.
|
|
17
|
-
// Keep defensive limits small enough to avoid accidental memory spikes from untrusted inputs.
|
|
18
28
|
const maximumMpegOffsetTolerance = reasonableDetectionSizeInBytes - 2;
|
|
19
|
-
const maximumZipEntrySizeInBytes = 1024 * 1024;
|
|
20
|
-
const maximumZipEntryCount = 1024;
|
|
21
|
-
const maximumZipBufferedReadSizeInBytes = (2 ** 31) - 1;
|
|
22
|
-
const maximumUntrustedSkipSizeInBytes = 16 * 1024 * 1024;
|
|
23
|
-
const maximumUnknownSizePayloadProbeSizeInBytes = maximumZipEntrySizeInBytes;
|
|
24
|
-
const maximumZipTextEntrySizeInBytes = maximumZipEntrySizeInBytes;
|
|
25
29
|
const maximumNestedGzipDetectionSizeInBytes = maximumUntrustedSkipSizeInBytes;
|
|
26
30
|
const maximumNestedGzipProbeDepth = 1;
|
|
27
31
|
const unknownSizeGzipProbeTimeoutInMilliseconds = 100;
|
|
28
32
|
const maximumId3HeaderSizeInBytes = maximumUntrustedSkipSizeInBytes;
|
|
29
|
-
const maximumEbmlDocumentTypeSizeInBytes = 64;
|
|
30
|
-
const maximumEbmlElementPayloadSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
|
|
31
|
-
const maximumEbmlElementCount = 256;
|
|
32
|
-
const maximumPngChunkCount = 512;
|
|
33
|
-
const maximumPngStreamScanBudgetInBytes = maximumUntrustedSkipSizeInBytes;
|
|
34
|
-
const maximumAsfHeaderObjectCount = 512;
|
|
35
33
|
const maximumTiffTagCount = 512;
|
|
36
34
|
const maximumDetectionReentryCount = 256;
|
|
37
|
-
const
|
|
38
|
-
const maximumAsfHeaderPayloadSizeInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
|
|
39
|
-
const maximumTiffStreamIfdOffsetInBytes = maximumUnknownSizePayloadProbeSizeInBytes;
|
|
35
|
+
const maximumTiffStreamIfdOffsetInBytes = 1024 * 1024;
|
|
40
36
|
const maximumTiffIfdOffsetInBytes = maximumUntrustedSkipSizeInBytes;
|
|
41
|
-
const recoverableZipErrorMessages = new Set([
|
|
42
|
-
'Unexpected signature',
|
|
43
|
-
'Encrypted ZIP',
|
|
44
|
-
'Expected Central-File-Header signature',
|
|
45
|
-
]);
|
|
46
|
-
const recoverableZipErrorMessagePrefixes = [
|
|
47
|
-
'ZIP entry count exceeds ',
|
|
48
|
-
'Unsupported ZIP compression method:',
|
|
49
|
-
'ZIP entry compressed data exceeds ',
|
|
50
|
-
'ZIP entry decompressed data exceeds ',
|
|
51
|
-
'Expected data-descriptor-signature at position ',
|
|
52
|
-
];
|
|
53
|
-
const recoverableZipErrorCodes = new Set([
|
|
54
|
-
'Z_BUF_ERROR',
|
|
55
|
-
'Z_DATA_ERROR',
|
|
56
|
-
'ERR_INVALID_STATE',
|
|
57
|
-
]);
|
|
58
|
-
|
|
59
|
-
class ParserHardLimitError extends Error {}
|
|
60
|
-
|
|
61
|
-
function patchWebByobTokenizerClose(tokenizer) {
|
|
62
|
-
const streamReader = tokenizer?.streamReader;
|
|
63
|
-
if (streamReader?.constructor?.name !== 'WebStreamByobReader') {
|
|
64
|
-
return tokenizer;
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
const {reader} = streamReader;
|
|
68
|
-
const cancelAndRelease = async () => {
|
|
69
|
-
await reader.cancel();
|
|
70
|
-
reader.releaseLock();
|
|
71
|
-
};
|
|
72
|
-
|
|
73
|
-
streamReader.close = cancelAndRelease;
|
|
74
|
-
streamReader.abort = async () => {
|
|
75
|
-
streamReader.interrupted = true;
|
|
76
|
-
await cancelAndRelease();
|
|
77
|
-
};
|
|
78
|
-
|
|
79
|
-
return tokenizer;
|
|
80
|
-
}
|
|
81
37
|
|
|
82
|
-
function
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
) {
|
|
88
|
-
throw new ParserHardLimitError(`${reason} has invalid size ${value} (maximum ${maximum} bytes)`);
|
|
38
|
+
export function normalizeSampleSize(sampleSize) {
|
|
39
|
+
// `sampleSize` is an explicit caller-controlled tuning knob, not untrusted file input.
|
|
40
|
+
// Preserve valid caller-requested probe depth here; applications must bound attacker-derived option values themselves.
|
|
41
|
+
if (!Number.isFinite(sampleSize)) {
|
|
42
|
+
return reasonableDetectionSizeInBytes;
|
|
89
43
|
}
|
|
90
44
|
|
|
91
|
-
return
|
|
92
|
-
}
|
|
93
|
-
|
|
94
|
-
async function safeIgnore(tokenizer, length, {maximumLength = maximumUntrustedSkipSizeInBytes, reason = 'skip'} = {}) {
|
|
95
|
-
const safeLength = getSafeBound(length, maximumLength, reason);
|
|
96
|
-
await tokenizer.ignore(safeLength);
|
|
97
|
-
}
|
|
98
|
-
|
|
99
|
-
async function safeReadBuffer(tokenizer, buffer, options, {maximumLength = buffer.length, reason = 'read'} = {}) {
|
|
100
|
-
const length = options?.length ?? buffer.length;
|
|
101
|
-
const safeLength = getSafeBound(length, maximumLength, reason);
|
|
102
|
-
return tokenizer.readBuffer(buffer, {
|
|
103
|
-
...options,
|
|
104
|
-
length: safeLength,
|
|
105
|
-
});
|
|
45
|
+
return Math.max(1, Math.trunc(sampleSize));
|
|
106
46
|
}
|
|
107
47
|
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
controller.close();
|
|
113
|
-
},
|
|
114
|
-
});
|
|
115
|
-
const output = input.pipeThrough(new DecompressionStream('deflate-raw'));
|
|
116
|
-
const reader = output.getReader();
|
|
117
|
-
const chunks = [];
|
|
118
|
-
let totalLength = 0;
|
|
119
|
-
|
|
120
|
-
try {
|
|
121
|
-
for (;;) {
|
|
122
|
-
const {done, value} = await reader.read();
|
|
123
|
-
if (done) {
|
|
124
|
-
break;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
totalLength += value.length;
|
|
128
|
-
if (totalLength > maximumLength) {
|
|
129
|
-
await reader.cancel();
|
|
130
|
-
throw new Error(`ZIP entry decompressed data exceeds ${maximumLength} bytes`);
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
chunks.push(value);
|
|
134
|
-
}
|
|
135
|
-
} finally {
|
|
136
|
-
reader.releaseLock();
|
|
137
|
-
}
|
|
138
|
-
|
|
139
|
-
const uncompressedData = new Uint8Array(totalLength);
|
|
140
|
-
let offset = 0;
|
|
141
|
-
for (const chunk of chunks) {
|
|
142
|
-
uncompressedData.set(chunk, offset);
|
|
143
|
-
offset += chunk.length;
|
|
48
|
+
function normalizeMpegOffsetTolerance(mpegOffsetTolerance) {
|
|
49
|
+
// This value controls scan depth and therefore worst-case CPU work.
|
|
50
|
+
if (!Number.isFinite(mpegOffsetTolerance)) {
|
|
51
|
+
return 0;
|
|
144
52
|
}
|
|
145
53
|
|
|
146
|
-
return
|
|
54
|
+
return Math.max(0, Math.min(maximumMpegOffsetTolerance, Math.trunc(mpegOffsetTolerance)));
|
|
147
55
|
}
|
|
148
56
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
function findZipDataDescriptorOffset(buffer, bytesConsumed) {
|
|
154
|
-
if (buffer.length < zipDataDescriptorLengthInBytes) {
|
|
155
|
-
return -1;
|
|
156
|
-
}
|
|
157
|
-
|
|
158
|
-
const lastPossibleDescriptorOffset = buffer.length - zipDataDescriptorLengthInBytes;
|
|
159
|
-
for (let index = 0; index <= lastPossibleDescriptorOffset; index++) {
|
|
160
|
-
if (
|
|
161
|
-
Token.UINT32_LE.get(buffer, index) === zipDataDescriptorSignature
|
|
162
|
-
&& Token.UINT32_LE.get(buffer, index + 8) === bytesConsumed + index
|
|
163
|
-
) {
|
|
164
|
-
return index;
|
|
165
|
-
}
|
|
57
|
+
function getKnownFileSizeOrMaximum(fileSize) {
|
|
58
|
+
if (!Number.isFinite(fileSize)) {
|
|
59
|
+
return Number.MAX_SAFE_INTEGER;
|
|
166
60
|
}
|
|
167
61
|
|
|
168
|
-
return
|
|
62
|
+
return Math.max(0, fileSize);
|
|
169
63
|
}
|
|
170
64
|
|
|
171
|
-
|
|
172
|
-
|
|
65
|
+
// Keep the specifier non-literal at the call site so browser bundlers do not try to resolve Node-only imports.
|
|
66
|
+
function importAtRuntime(specifier) {
|
|
67
|
+
return import(specifier);
|
|
173
68
|
}
|
|
174
69
|
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
}
|
|
183
|
-
|
|
184
|
-
return merged;
|
|
70
|
+
// Wrap stream in an identity TransformStream to avoid BYOB readers.
|
|
71
|
+
// Node.js has a bug where calling controller.close() inside a BYOB stream's
|
|
72
|
+
// pull() callback does not resolve pending reader.read() calls, causing
|
|
73
|
+
// permanent hangs on streams shorter than the requested read size.
|
|
74
|
+
// Using a default (non-BYOB) reader via TransformStream avoids this.
|
|
75
|
+
function toDefaultStream(stream) {
|
|
76
|
+
return stream.pipeThrough(new TransformStream());
|
|
185
77
|
}
|
|
186
78
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
const chunks = [];
|
|
191
|
-
let bytesConsumed = 0;
|
|
192
|
-
|
|
193
|
-
for (;;) {
|
|
194
|
-
const length = await zipHandler.tokenizer.peekBuffer(syncBuffer, {mayBeLess: true});
|
|
195
|
-
const dataDescriptorOffset = findZipDataDescriptorOffset(syncBuffer.subarray(0, length), bytesConsumed);
|
|
196
|
-
const retainedLength = dataDescriptorOffset >= 0
|
|
197
|
-
? 0
|
|
198
|
-
: (
|
|
199
|
-
length === syncBufferLength
|
|
200
|
-
? Math.min(zipDataDescriptorOverlapLengthInBytes, length - 1)
|
|
201
|
-
: 0
|
|
202
|
-
);
|
|
203
|
-
const chunkLength = dataDescriptorOffset >= 0 ? dataDescriptorOffset : length - retainedLength;
|
|
204
|
-
|
|
205
|
-
if (chunkLength === 0) {
|
|
206
|
-
break;
|
|
207
|
-
}
|
|
208
|
-
|
|
209
|
-
bytesConsumed += chunkLength;
|
|
210
|
-
if (bytesConsumed > maximumLength) {
|
|
211
|
-
throw new Error(`ZIP entry compressed data exceeds ${maximumLength} bytes`);
|
|
212
|
-
}
|
|
213
|
-
|
|
214
|
-
if (shouldBuffer) {
|
|
215
|
-
const data = new Uint8Array(chunkLength);
|
|
216
|
-
await zipHandler.tokenizer.readBuffer(data);
|
|
217
|
-
chunks.push(data);
|
|
218
|
-
} else {
|
|
219
|
-
await zipHandler.tokenizer.ignore(chunkLength);
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
if (dataDescriptorOffset >= 0) {
|
|
223
|
-
break;
|
|
224
|
-
}
|
|
225
|
-
}
|
|
226
|
-
|
|
227
|
-
if (!hasUnknownFileSize(zipHandler.tokenizer)) {
|
|
228
|
-
zipHandler.knownSizeDescriptorScannedBytes += bytesConsumed;
|
|
229
|
-
}
|
|
230
|
-
|
|
231
|
-
if (!shouldBuffer) {
|
|
232
|
-
return;
|
|
233
|
-
}
|
|
234
|
-
|
|
235
|
-
return mergeByteChunks(chunks, bytesConsumed);
|
|
236
|
-
}
|
|
237
|
-
|
|
238
|
-
function getRemainingZipScanBudget(zipHandler, startOffset) {
|
|
239
|
-
if (hasUnknownFileSize(zipHandler.tokenizer)) {
|
|
240
|
-
return Math.max(0, maximumUntrustedSkipSizeInBytes - (zipHandler.tokenizer.position - startOffset));
|
|
241
|
-
}
|
|
242
|
-
|
|
243
|
-
return Math.max(0, maximumZipEntrySizeInBytes - zipHandler.knownSizeDescriptorScannedBytes);
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
async function readZipEntryData(zipHandler, zipHeader, {shouldBuffer, maximumDescriptorLength = maximumZipEntrySizeInBytes} = {}) {
|
|
247
|
-
if (
|
|
248
|
-
zipHeader.dataDescriptor
|
|
249
|
-
&& zipHeader.compressedSize === 0
|
|
250
|
-
) {
|
|
251
|
-
return readZipDataDescriptorEntryWithLimit(zipHandler, {
|
|
252
|
-
shouldBuffer,
|
|
253
|
-
maximumLength: maximumDescriptorLength,
|
|
254
|
-
});
|
|
255
|
-
}
|
|
256
|
-
|
|
257
|
-
if (!shouldBuffer) {
|
|
258
|
-
await safeIgnore(zipHandler.tokenizer, zipHeader.compressedSize, {
|
|
259
|
-
maximumLength: hasUnknownFileSize(zipHandler.tokenizer) ? maximumZipEntrySizeInBytes : zipHandler.tokenizer.fileInfo.size,
|
|
260
|
-
reason: 'ZIP entry compressed data',
|
|
261
|
-
});
|
|
262
|
-
return;
|
|
79
|
+
function readWithSignal(reader, signal) {
|
|
80
|
+
if (signal === undefined) {
|
|
81
|
+
return reader.read();
|
|
263
82
|
}
|
|
264
83
|
|
|
265
|
-
|
|
266
|
-
if (
|
|
267
|
-
!Number.isFinite(zipHeader.compressedSize)
|
|
268
|
-
|| zipHeader.compressedSize < 0
|
|
269
|
-
|| zipHeader.compressedSize > maximumLength
|
|
270
|
-
) {
|
|
271
|
-
throw new Error(`ZIP entry compressed data exceeds ${maximumLength} bytes`);
|
|
272
|
-
}
|
|
84
|
+
signal.throwIfAborted();
|
|
273
85
|
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
86
|
+
return Promise.race([
|
|
87
|
+
reader.read(),
|
|
88
|
+
new Promise((_resolve, reject) => {
|
|
89
|
+
signal.addEventListener('abort', () => {
|
|
90
|
+
reject(signal.reason);
|
|
91
|
+
reader.cancel(signal.reason).catch(() => {});
|
|
92
|
+
}, {once: true});
|
|
93
|
+
}),
|
|
94
|
+
]);
|
|
277
95
|
}
|
|
278
96
|
|
|
279
|
-
// Override the default inflate to enforce decompression size limits, since @tokenizer/inflate does not expose a configuration hook for this.
|
|
280
|
-
ZipHandler.prototype.inflate = async function (zipHeader, fileData, callback) {
|
|
281
|
-
if (zipHeader.compressedMethod === 0) {
|
|
282
|
-
return callback(fileData);
|
|
283
|
-
}
|
|
284
|
-
|
|
285
|
-
if (zipHeader.compressedMethod !== 8) {
|
|
286
|
-
throw new Error(`Unsupported ZIP compression method: ${zipHeader.compressedMethod}`);
|
|
287
|
-
}
|
|
288
|
-
|
|
289
|
-
const uncompressedData = await decompressDeflateRawWithLimit(fileData, {maximumLength: maximumZipEntrySizeInBytes});
|
|
290
|
-
return callback(uncompressedData);
|
|
291
|
-
};
|
|
292
|
-
|
|
293
|
-
ZipHandler.prototype.unzip = async function (fileCallback) {
|
|
294
|
-
let stop = false;
|
|
295
|
-
let zipEntryCount = 0;
|
|
296
|
-
const zipScanStart = this.tokenizer.position;
|
|
297
|
-
this.knownSizeDescriptorScannedBytes = 0;
|
|
298
|
-
do {
|
|
299
|
-
if (hasExceededUnknownSizeScanBudget(this.tokenizer, zipScanStart, maximumUntrustedSkipSizeInBytes)) {
|
|
300
|
-
throw new ParserHardLimitError(`ZIP stream probing exceeds ${maximumUntrustedSkipSizeInBytes} bytes`);
|
|
301
|
-
}
|
|
302
|
-
|
|
303
|
-
const zipHeader = await this.readLocalFileHeader();
|
|
304
|
-
if (!zipHeader) {
|
|
305
|
-
break;
|
|
306
|
-
}
|
|
307
|
-
|
|
308
|
-
zipEntryCount++;
|
|
309
|
-
if (zipEntryCount > maximumZipEntryCount) {
|
|
310
|
-
throw new Error(`ZIP entry count exceeds ${maximumZipEntryCount}`);
|
|
311
|
-
}
|
|
312
|
-
|
|
313
|
-
const next = fileCallback(zipHeader);
|
|
314
|
-
stop = Boolean(next.stop);
|
|
315
|
-
await this.tokenizer.ignore(zipHeader.extraFieldLength);
|
|
316
|
-
const fileData = await readZipEntryData(this, zipHeader, {
|
|
317
|
-
shouldBuffer: Boolean(next.handler),
|
|
318
|
-
maximumDescriptorLength: Math.min(maximumZipEntrySizeInBytes, getRemainingZipScanBudget(this, zipScanStart)),
|
|
319
|
-
});
|
|
320
|
-
|
|
321
|
-
if (next.handler) {
|
|
322
|
-
await this.inflate(zipHeader, fileData, next.handler);
|
|
323
|
-
}
|
|
324
|
-
|
|
325
|
-
if (zipHeader.dataDescriptor) {
|
|
326
|
-
const dataDescriptor = new Uint8Array(zipDataDescriptorLengthInBytes);
|
|
327
|
-
await this.tokenizer.readBuffer(dataDescriptor);
|
|
328
|
-
if (Token.UINT32_LE.get(dataDescriptor, 0) !== zipDataDescriptorSignature) {
|
|
329
|
-
throw new Error(`Expected data-descriptor-signature at position ${this.tokenizer.position - dataDescriptor.length}`);
|
|
330
|
-
}
|
|
331
|
-
}
|
|
332
|
-
|
|
333
|
-
if (hasExceededUnknownSizeScanBudget(this.tokenizer, zipScanStart, maximumUntrustedSkipSizeInBytes)) {
|
|
334
|
-
throw new ParserHardLimitError(`ZIP stream probing exceeds ${maximumUntrustedSkipSizeInBytes} bytes`);
|
|
335
|
-
}
|
|
336
|
-
} while (!stop);
|
|
337
|
-
};
|
|
338
|
-
|
|
339
97
|
function createByteLimitedReadableStream(stream, maximumBytes) {
|
|
340
98
|
const reader = stream.getReader();
|
|
341
99
|
let emittedBytes = 0;
|
|
@@ -402,388 +160,6 @@ export async function fileTypeFromBlob(blob, options) {
|
|
|
402
160
|
return new FileTypeParser(options).fromBlob(blob);
|
|
403
161
|
}
|
|
404
162
|
|
|
405
|
-
function getFileTypeFromMimeType(mimeType) {
|
|
406
|
-
mimeType = mimeType.toLowerCase();
|
|
407
|
-
switch (mimeType) {
|
|
408
|
-
case 'application/epub+zip':
|
|
409
|
-
return {
|
|
410
|
-
ext: 'epub',
|
|
411
|
-
mime: mimeType,
|
|
412
|
-
};
|
|
413
|
-
case 'application/vnd.oasis.opendocument.text':
|
|
414
|
-
return {
|
|
415
|
-
ext: 'odt',
|
|
416
|
-
mime: mimeType,
|
|
417
|
-
};
|
|
418
|
-
case 'application/vnd.oasis.opendocument.text-template':
|
|
419
|
-
return {
|
|
420
|
-
ext: 'ott',
|
|
421
|
-
mime: mimeType,
|
|
422
|
-
};
|
|
423
|
-
case 'application/vnd.oasis.opendocument.spreadsheet':
|
|
424
|
-
return {
|
|
425
|
-
ext: 'ods',
|
|
426
|
-
mime: mimeType,
|
|
427
|
-
};
|
|
428
|
-
case 'application/vnd.oasis.opendocument.spreadsheet-template':
|
|
429
|
-
return {
|
|
430
|
-
ext: 'ots',
|
|
431
|
-
mime: mimeType,
|
|
432
|
-
};
|
|
433
|
-
case 'application/vnd.oasis.opendocument.presentation':
|
|
434
|
-
return {
|
|
435
|
-
ext: 'odp',
|
|
436
|
-
mime: mimeType,
|
|
437
|
-
};
|
|
438
|
-
case 'application/vnd.oasis.opendocument.presentation-template':
|
|
439
|
-
return {
|
|
440
|
-
ext: 'otp',
|
|
441
|
-
mime: mimeType,
|
|
442
|
-
};
|
|
443
|
-
case 'application/vnd.oasis.opendocument.graphics':
|
|
444
|
-
return {
|
|
445
|
-
ext: 'odg',
|
|
446
|
-
mime: mimeType,
|
|
447
|
-
};
|
|
448
|
-
case 'application/vnd.oasis.opendocument.graphics-template':
|
|
449
|
-
return {
|
|
450
|
-
ext: 'otg',
|
|
451
|
-
mime: mimeType,
|
|
452
|
-
};
|
|
453
|
-
case 'application/vnd.openxmlformats-officedocument.presentationml.slideshow':
|
|
454
|
-
return {
|
|
455
|
-
ext: 'ppsx',
|
|
456
|
-
mime: mimeType,
|
|
457
|
-
};
|
|
458
|
-
case 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet':
|
|
459
|
-
return {
|
|
460
|
-
ext: 'xlsx',
|
|
461
|
-
mime: mimeType,
|
|
462
|
-
};
|
|
463
|
-
case 'application/vnd.ms-excel.sheet.macroenabled':
|
|
464
|
-
return {
|
|
465
|
-
ext: 'xlsm',
|
|
466
|
-
mime: 'application/vnd.ms-excel.sheet.macroenabled.12',
|
|
467
|
-
};
|
|
468
|
-
case 'application/vnd.openxmlformats-officedocument.spreadsheetml.template':
|
|
469
|
-
return {
|
|
470
|
-
ext: 'xltx',
|
|
471
|
-
mime: mimeType,
|
|
472
|
-
};
|
|
473
|
-
case 'application/vnd.ms-excel.template.macroenabled':
|
|
474
|
-
return {
|
|
475
|
-
ext: 'xltm',
|
|
476
|
-
mime: 'application/vnd.ms-excel.template.macroenabled.12',
|
|
477
|
-
};
|
|
478
|
-
case 'application/vnd.ms-powerpoint.slideshow.macroenabled':
|
|
479
|
-
return {
|
|
480
|
-
ext: 'ppsm',
|
|
481
|
-
mime: 'application/vnd.ms-powerpoint.slideshow.macroenabled.12',
|
|
482
|
-
};
|
|
483
|
-
case 'application/vnd.openxmlformats-officedocument.wordprocessingml.document':
|
|
484
|
-
return {
|
|
485
|
-
ext: 'docx',
|
|
486
|
-
mime: mimeType,
|
|
487
|
-
};
|
|
488
|
-
case 'application/vnd.ms-word.document.macroenabled':
|
|
489
|
-
return {
|
|
490
|
-
ext: 'docm',
|
|
491
|
-
mime: 'application/vnd.ms-word.document.macroenabled.12',
|
|
492
|
-
};
|
|
493
|
-
case 'application/vnd.openxmlformats-officedocument.wordprocessingml.template':
|
|
494
|
-
return {
|
|
495
|
-
ext: 'dotx',
|
|
496
|
-
mime: mimeType,
|
|
497
|
-
};
|
|
498
|
-
case 'application/vnd.ms-word.template.macroenabledtemplate':
|
|
499
|
-
return {
|
|
500
|
-
ext: 'dotm',
|
|
501
|
-
mime: 'application/vnd.ms-word.template.macroenabled.12',
|
|
502
|
-
};
|
|
503
|
-
case 'application/vnd.openxmlformats-officedocument.presentationml.template':
|
|
504
|
-
return {
|
|
505
|
-
ext: 'potx',
|
|
506
|
-
mime: mimeType,
|
|
507
|
-
};
|
|
508
|
-
case 'application/vnd.ms-powerpoint.template.macroenabled':
|
|
509
|
-
return {
|
|
510
|
-
ext: 'potm',
|
|
511
|
-
mime: 'application/vnd.ms-powerpoint.template.macroenabled.12',
|
|
512
|
-
};
|
|
513
|
-
case 'application/vnd.openxmlformats-officedocument.presentationml.presentation':
|
|
514
|
-
return {
|
|
515
|
-
ext: 'pptx',
|
|
516
|
-
mime: mimeType,
|
|
517
|
-
};
|
|
518
|
-
case 'application/vnd.ms-powerpoint.presentation.macroenabled':
|
|
519
|
-
return {
|
|
520
|
-
ext: 'pptm',
|
|
521
|
-
mime: 'application/vnd.ms-powerpoint.presentation.macroenabled.12',
|
|
522
|
-
};
|
|
523
|
-
case 'application/vnd.ms-visio.drawing':
|
|
524
|
-
return {
|
|
525
|
-
ext: 'vsdx',
|
|
526
|
-
mime: 'application/vnd.visio',
|
|
527
|
-
};
|
|
528
|
-
case 'application/vnd.ms-package.3dmanufacturing-3dmodel+xml':
|
|
529
|
-
return {
|
|
530
|
-
ext: '3mf',
|
|
531
|
-
mime: 'model/3mf',
|
|
532
|
-
};
|
|
533
|
-
default:
|
|
534
|
-
}
|
|
535
|
-
}
|
|
536
|
-
|
|
537
|
-
function _check(buffer, headers, options) {
|
|
538
|
-
options = {
|
|
539
|
-
offset: 0,
|
|
540
|
-
...options,
|
|
541
|
-
};
|
|
542
|
-
|
|
543
|
-
for (const [index, header] of headers.entries()) {
|
|
544
|
-
// If a bitmask is set
|
|
545
|
-
if (options.mask) {
|
|
546
|
-
// If header doesn't equal `buf` with bits masked off
|
|
547
|
-
if (header !== (options.mask[index] & buffer[index + options.offset])) {
|
|
548
|
-
return false;
|
|
549
|
-
}
|
|
550
|
-
} else if (header !== buffer[index + options.offset]) {
|
|
551
|
-
return false;
|
|
552
|
-
}
|
|
553
|
-
}
|
|
554
|
-
|
|
555
|
-
return true;
|
|
556
|
-
}
|
|
557
|
-
|
|
558
|
-
export function normalizeSampleSize(sampleSize) {
|
|
559
|
-
// `sampleSize` is an explicit caller-controlled tuning knob, not untrusted file input.
|
|
560
|
-
// Preserve valid caller-requested probe depth here; applications must bound attacker-derived option values themselves.
|
|
561
|
-
if (!Number.isFinite(sampleSize)) {
|
|
562
|
-
return reasonableDetectionSizeInBytes;
|
|
563
|
-
}
|
|
564
|
-
|
|
565
|
-
return Math.max(1, Math.trunc(sampleSize));
|
|
566
|
-
}
|
|
567
|
-
|
|
568
|
-
function readByobReaderWithSignal(reader, buffer, signal) {
|
|
569
|
-
if (signal === undefined) {
|
|
570
|
-
return reader.read(buffer);
|
|
571
|
-
}
|
|
572
|
-
|
|
573
|
-
signal.throwIfAborted();
|
|
574
|
-
|
|
575
|
-
return new Promise((resolve, reject) => {
|
|
576
|
-
const cleanup = () => {
|
|
577
|
-
signal.removeEventListener('abort', onAbort);
|
|
578
|
-
};
|
|
579
|
-
|
|
580
|
-
const onAbort = () => {
|
|
581
|
-
const abortReason = signal.reason;
|
|
582
|
-
cleanup();
|
|
583
|
-
|
|
584
|
-
(async () => {
|
|
585
|
-
try {
|
|
586
|
-
await reader.cancel(abortReason);
|
|
587
|
-
} catch {}
|
|
588
|
-
})();
|
|
589
|
-
|
|
590
|
-
reject(abortReason);
|
|
591
|
-
};
|
|
592
|
-
|
|
593
|
-
signal.addEventListener('abort', onAbort, {once: true});
|
|
594
|
-
(async () => {
|
|
595
|
-
try {
|
|
596
|
-
const result = await reader.read(buffer);
|
|
597
|
-
cleanup();
|
|
598
|
-
resolve(result);
|
|
599
|
-
} catch (error) {
|
|
600
|
-
cleanup();
|
|
601
|
-
reject(error);
|
|
602
|
-
}
|
|
603
|
-
})();
|
|
604
|
-
});
|
|
605
|
-
}
|
|
606
|
-
|
|
607
|
-
function normalizeMpegOffsetTolerance(mpegOffsetTolerance) {
|
|
608
|
-
// This value controls scan depth and therefore worst-case CPU work.
|
|
609
|
-
if (!Number.isFinite(mpegOffsetTolerance)) {
|
|
610
|
-
return 0;
|
|
611
|
-
}
|
|
612
|
-
|
|
613
|
-
return Math.max(0, Math.min(maximumMpegOffsetTolerance, Math.trunc(mpegOffsetTolerance)));
|
|
614
|
-
}
|
|
615
|
-
|
|
616
|
-
function getKnownFileSizeOrMaximum(fileSize) {
|
|
617
|
-
if (!Number.isFinite(fileSize)) {
|
|
618
|
-
return Number.MAX_SAFE_INTEGER;
|
|
619
|
-
}
|
|
620
|
-
|
|
621
|
-
return Math.max(0, fileSize);
|
|
622
|
-
}
|
|
623
|
-
|
|
624
|
-
function hasUnknownFileSize(tokenizer) {
|
|
625
|
-
const fileSize = tokenizer.fileInfo.size;
|
|
626
|
-
return (
|
|
627
|
-
!Number.isFinite(fileSize)
|
|
628
|
-
|| fileSize === Number.MAX_SAFE_INTEGER
|
|
629
|
-
);
|
|
630
|
-
}
|
|
631
|
-
|
|
632
|
-
function hasExceededUnknownSizeScanBudget(tokenizer, startOffset, maximumBytes) {
|
|
633
|
-
return (
|
|
634
|
-
hasUnknownFileSize(tokenizer)
|
|
635
|
-
&& tokenizer.position - startOffset > maximumBytes
|
|
636
|
-
);
|
|
637
|
-
}
|
|
638
|
-
|
|
639
|
-
function getMaximumZipBufferedReadLength(tokenizer) {
|
|
640
|
-
const fileSize = tokenizer.fileInfo.size;
|
|
641
|
-
const remainingBytes = Number.isFinite(fileSize)
|
|
642
|
-
? Math.max(0, fileSize - tokenizer.position)
|
|
643
|
-
: Number.MAX_SAFE_INTEGER;
|
|
644
|
-
|
|
645
|
-
return Math.min(remainingBytes, maximumZipBufferedReadSizeInBytes);
|
|
646
|
-
}
|
|
647
|
-
|
|
648
|
-
function isRecoverableZipError(error) {
|
|
649
|
-
if (error instanceof strtok3.EndOfStreamError) {
|
|
650
|
-
return true;
|
|
651
|
-
}
|
|
652
|
-
|
|
653
|
-
if (error instanceof ParserHardLimitError) {
|
|
654
|
-
return true;
|
|
655
|
-
}
|
|
656
|
-
|
|
657
|
-
if (!(error instanceof Error)) {
|
|
658
|
-
return false;
|
|
659
|
-
}
|
|
660
|
-
|
|
661
|
-
if (recoverableZipErrorMessages.has(error.message)) {
|
|
662
|
-
return true;
|
|
663
|
-
}
|
|
664
|
-
|
|
665
|
-
if (recoverableZipErrorCodes.has(error.code)) {
|
|
666
|
-
return true;
|
|
667
|
-
}
|
|
668
|
-
|
|
669
|
-
for (const prefix of recoverableZipErrorMessagePrefixes) {
|
|
670
|
-
if (error.message.startsWith(prefix)) {
|
|
671
|
-
return true;
|
|
672
|
-
}
|
|
673
|
-
}
|
|
674
|
-
|
|
675
|
-
return false;
|
|
676
|
-
}
|
|
677
|
-
|
|
678
|
-
function canReadZipEntryForDetection(zipHeader, maximumSize = maximumZipEntrySizeInBytes) {
|
|
679
|
-
const sizes = [zipHeader.compressedSize, zipHeader.uncompressedSize];
|
|
680
|
-
for (const size of sizes) {
|
|
681
|
-
if (
|
|
682
|
-
!Number.isFinite(size)
|
|
683
|
-
|| size < 0
|
|
684
|
-
|| size > maximumSize
|
|
685
|
-
) {
|
|
686
|
-
return false;
|
|
687
|
-
}
|
|
688
|
-
}
|
|
689
|
-
|
|
690
|
-
return true;
|
|
691
|
-
}
|
|
692
|
-
|
|
693
|
-
function createOpenXmlZipDetectionState() {
|
|
694
|
-
return {
|
|
695
|
-
hasContentTypesEntry: false,
|
|
696
|
-
hasParsedContentTypesEntry: false,
|
|
697
|
-
isParsingContentTypes: false,
|
|
698
|
-
hasUnparseableContentTypes: false,
|
|
699
|
-
hasWordDirectory: false,
|
|
700
|
-
hasPresentationDirectory: false,
|
|
701
|
-
hasSpreadsheetDirectory: false,
|
|
702
|
-
hasThreeDimensionalModelEntry: false,
|
|
703
|
-
};
|
|
704
|
-
}
|
|
705
|
-
|
|
706
|
-
function updateOpenXmlZipDetectionStateFromFilename(openXmlState, filename) {
|
|
707
|
-
if (filename.startsWith('word/')) {
|
|
708
|
-
openXmlState.hasWordDirectory = true;
|
|
709
|
-
}
|
|
710
|
-
|
|
711
|
-
if (filename.startsWith('ppt/')) {
|
|
712
|
-
openXmlState.hasPresentationDirectory = true;
|
|
713
|
-
}
|
|
714
|
-
|
|
715
|
-
if (filename.startsWith('xl/')) {
|
|
716
|
-
openXmlState.hasSpreadsheetDirectory = true;
|
|
717
|
-
}
|
|
718
|
-
|
|
719
|
-
if (
|
|
720
|
-
filename.startsWith('3D/')
|
|
721
|
-
&& filename.endsWith('.model')
|
|
722
|
-
) {
|
|
723
|
-
openXmlState.hasThreeDimensionalModelEntry = true;
|
|
724
|
-
}
|
|
725
|
-
}
|
|
726
|
-
|
|
727
|
-
function getOpenXmlFileTypeFromZipEntries(openXmlState) {
|
|
728
|
-
// Only use directory-name heuristic when [Content_Types].xml was present in the archive
|
|
729
|
-
// but its handler was skipped (not invoked, not currently running, and not already resolved).
|
|
730
|
-
// This avoids guessing from directory names when content-type parsing already gave a definitive answer or failed.
|
|
731
|
-
if (
|
|
732
|
-
!openXmlState.hasContentTypesEntry
|
|
733
|
-
|| openXmlState.hasUnparseableContentTypes
|
|
734
|
-
|| openXmlState.isParsingContentTypes
|
|
735
|
-
|| openXmlState.hasParsedContentTypesEntry
|
|
736
|
-
) {
|
|
737
|
-
return;
|
|
738
|
-
}
|
|
739
|
-
|
|
740
|
-
if (openXmlState.hasWordDirectory) {
|
|
741
|
-
return {
|
|
742
|
-
ext: 'docx',
|
|
743
|
-
mime: 'application/vnd.openxmlformats-officedocument.wordprocessingml.document',
|
|
744
|
-
};
|
|
745
|
-
}
|
|
746
|
-
|
|
747
|
-
if (openXmlState.hasPresentationDirectory) {
|
|
748
|
-
return {
|
|
749
|
-
ext: 'pptx',
|
|
750
|
-
mime: 'application/vnd.openxmlformats-officedocument.presentationml.presentation',
|
|
751
|
-
};
|
|
752
|
-
}
|
|
753
|
-
|
|
754
|
-
if (openXmlState.hasSpreadsheetDirectory) {
|
|
755
|
-
return {
|
|
756
|
-
ext: 'xlsx',
|
|
757
|
-
mime: 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet',
|
|
758
|
-
};
|
|
759
|
-
}
|
|
760
|
-
|
|
761
|
-
if (openXmlState.hasThreeDimensionalModelEntry) {
|
|
762
|
-
return {
|
|
763
|
-
ext: '3mf',
|
|
764
|
-
mime: 'model/3mf',
|
|
765
|
-
};
|
|
766
|
-
}
|
|
767
|
-
}
|
|
768
|
-
|
|
769
|
-
function getOpenXmlMimeTypeFromContentTypesXml(xmlContent) {
|
|
770
|
-
// We only need the `ContentType="...main+xml"` value, so a small string scan is enough and avoids full XML parsing.
|
|
771
|
-
const endPosition = xmlContent.indexOf('.main+xml"');
|
|
772
|
-
if (endPosition === -1) {
|
|
773
|
-
const mimeType = 'application/vnd.ms-package.3dmanufacturing-3dmodel+xml';
|
|
774
|
-
if (xmlContent.includes(`ContentType="${mimeType}"`)) {
|
|
775
|
-
return mimeType;
|
|
776
|
-
}
|
|
777
|
-
|
|
778
|
-
return;
|
|
779
|
-
}
|
|
780
|
-
|
|
781
|
-
const truncatedContent = xmlContent.slice(0, endPosition);
|
|
782
|
-
const firstQuotePosition = truncatedContent.lastIndexOf('"');
|
|
783
|
-
// If no quote is found, `lastIndexOf` returns -1 and this intentionally falls back to the full truncated prefix.
|
|
784
|
-
return truncatedContent.slice(firstQuotePosition + 1);
|
|
785
|
-
}
|
|
786
|
-
|
|
787
163
|
export async function fileTypeFromTokenizer(tokenizer, options) {
|
|
788
164
|
return new FileTypeParser(options).fromTokenizer(tokenizer);
|
|
789
165
|
}
|
|
@@ -816,7 +192,7 @@ export class FileTypeParser {
|
|
|
816
192
|
}
|
|
817
193
|
|
|
818
194
|
createTokenizerFromWebStream(stream) {
|
|
819
|
-
return
|
|
195
|
+
return strtok3.fromWebStream(toDefaultStream(stream), this.getTokenizerOptions());
|
|
820
196
|
}
|
|
821
197
|
|
|
822
198
|
async parseTokenizer(tokenizer, detectionReentryCount = 0) {
|
|
@@ -883,41 +259,96 @@ export class FileTypeParser {
|
|
|
883
259
|
return this.fromTokenizer(tokenizer);
|
|
884
260
|
}
|
|
885
261
|
|
|
262
|
+
async fromFile(path) {
|
|
263
|
+
this.options.signal?.throwIfAborted();
|
|
264
|
+
// TODO: Remove this when `strtok3.fromFile()` safely rejects non-regular filesystem objects without a pathname race.
|
|
265
|
+
const [{default: fsPromises}, {FileTokenizer}] = await Promise.all([
|
|
266
|
+
importAtRuntime('node:fs/promises'),
|
|
267
|
+
importAtRuntime('strtok3'),
|
|
268
|
+
]);
|
|
269
|
+
const fileHandle = await fsPromises.open(path, fsPromises.constants.O_RDONLY | fsPromises.constants.O_NONBLOCK);
|
|
270
|
+
const fileStat = await fileHandle.stat();
|
|
271
|
+
if (!fileStat.isFile()) {
|
|
272
|
+
await fileHandle.close();
|
|
273
|
+
return;
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
const tokenizer = new FileTokenizer(fileHandle, {
|
|
277
|
+
...this.getTokenizerOptions(),
|
|
278
|
+
fileInfo: {path, size: fileStat.size},
|
|
279
|
+
});
|
|
280
|
+
return this.fromTokenizer(tokenizer);
|
|
281
|
+
}
|
|
282
|
+
|
|
886
283
|
async toDetectionStream(stream, options) {
|
|
284
|
+
this.options.signal?.throwIfAborted();
|
|
887
285
|
const sampleSize = normalizeSampleSize(options?.sampleSize ?? reasonableDetectionSizeInBytes);
|
|
888
286
|
let detectedFileType;
|
|
889
|
-
let
|
|
287
|
+
let streamEnded = false;
|
|
890
288
|
|
|
891
|
-
const reader = stream.getReader(
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
const {value: chunk, done} = await readByobReaderWithSignal(reader, new Uint8Array(sampleSize), this.options.signal);
|
|
895
|
-
firstChunk = chunk;
|
|
896
|
-
if (!done && chunk) {
|
|
897
|
-
try {
|
|
898
|
-
// Attempt to detect the file type from the chunk
|
|
899
|
-
detectedFileType = await this.fromBuffer(chunk.subarray(0, sampleSize));
|
|
900
|
-
} catch (error) {
|
|
901
|
-
if (!(error instanceof strtok3.EndOfStreamError)) {
|
|
902
|
-
throw error; // Re-throw non-EndOfStreamError
|
|
903
|
-
}
|
|
289
|
+
const reader = stream.getReader();
|
|
290
|
+
const chunks = [];
|
|
291
|
+
let totalSize = 0;
|
|
904
292
|
|
|
905
|
-
|
|
293
|
+
try {
|
|
294
|
+
while (totalSize < sampleSize) {
|
|
295
|
+
const {value, done} = await readWithSignal(reader, this.options.signal);
|
|
296
|
+
if (done || !value) {
|
|
297
|
+
streamEnded = true;
|
|
298
|
+
break;
|
|
906
299
|
}
|
|
300
|
+
|
|
301
|
+
chunks.push(value);
|
|
302
|
+
totalSize += value.length;
|
|
907
303
|
}
|
|
908
304
|
|
|
909
|
-
|
|
305
|
+
if (
|
|
306
|
+
!streamEnded
|
|
307
|
+
&& totalSize === sampleSize
|
|
308
|
+
) {
|
|
309
|
+
const {value, done} = await readWithSignal(reader, this.options.signal);
|
|
310
|
+
if (done || !value) {
|
|
311
|
+
streamEnded = true;
|
|
312
|
+
} else {
|
|
313
|
+
chunks.push(value);
|
|
314
|
+
totalSize += value.length;
|
|
315
|
+
}
|
|
316
|
+
}
|
|
910
317
|
} finally {
|
|
911
|
-
reader.releaseLock();
|
|
318
|
+
reader.releaseLock();
|
|
912
319
|
}
|
|
913
320
|
|
|
914
|
-
|
|
321
|
+
if (totalSize > 0) {
|
|
322
|
+
const sample = chunks.length === 1 ? chunks[0] : concatUint8Arrays(chunks);
|
|
323
|
+
try {
|
|
324
|
+
detectedFileType = await this.fromBuffer(sample.subarray(0, sampleSize));
|
|
325
|
+
} catch (error) {
|
|
326
|
+
if (!(error instanceof strtok3.EndOfStreamError)) {
|
|
327
|
+
throw error;
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
detectedFileType = undefined;
|
|
331
|
+
}
|
|
332
|
+
|
|
333
|
+
if (
|
|
334
|
+
!streamEnded
|
|
335
|
+
&& detectedFileType?.ext === 'pages'
|
|
336
|
+
) {
|
|
337
|
+
detectedFileType = {
|
|
338
|
+
ext: 'zip',
|
|
339
|
+
mime: 'application/zip',
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
// Prepend collected chunks and pipe the rest through
|
|
915
345
|
const transformStream = new TransformStream({
|
|
916
|
-
|
|
917
|
-
|
|
346
|
+
start(controller) {
|
|
347
|
+
for (const chunk of chunks) {
|
|
348
|
+
controller.enqueue(chunk);
|
|
349
|
+
}
|
|
918
350
|
},
|
|
919
351
|
transform(chunk, controller) {
|
|
920
|
-
// Pass through the chunks without modification
|
|
921
352
|
controller.enqueue(chunk);
|
|
922
353
|
},
|
|
923
354
|
});
|
|
@@ -951,7 +382,6 @@ export class FileTypeParser {
|
|
|
951
382
|
}, unknownSizeGzipProbeTimeoutInMilliseconds);
|
|
952
383
|
probeSignal = this.options.signal === undefined
|
|
953
384
|
? timeoutController.signal
|
|
954
|
-
// eslint-disable-next-line n/no-unsupported-features/node-builtins
|
|
955
385
|
: AbortSignal.any([this.options.signal, timeoutController.signal]);
|
|
956
386
|
probeParser = new FileTypeParser({
|
|
957
387
|
...this.options,
|
|
@@ -994,7 +424,7 @@ export class FileTypeParser {
|
|
|
994
424
|
}
|
|
995
425
|
|
|
996
426
|
check(header, options) {
|
|
997
|
-
return
|
|
427
|
+
return checkBytes(this.buffer, header, options);
|
|
998
428
|
}
|
|
999
429
|
|
|
1000
430
|
checkString(header, options) {
|
|
@@ -1141,7 +571,7 @@ export class FileTypeParser {
|
|
|
1141
571
|
const isUnknownFileSize = hasUnknownFileSize(tokenizer);
|
|
1142
572
|
if (
|
|
1143
573
|
!Number.isFinite(id3HeaderLength)
|
|
1144
|
-
|
|
574
|
+
|| id3HeaderLength < 0
|
|
1145
575
|
// Keep ID3 probing bounded for unknown-size streams to avoid attacker-controlled large skips.
|
|
1146
576
|
|| (
|
|
1147
577
|
isUnknownFileSize
|
|
@@ -1267,108 +697,7 @@ export class FileTypeParser {
|
|
|
1267
697
|
// Zip-based file formats
|
|
1268
698
|
// Need to be before the `zip` check
|
|
1269
699
|
if (this.check([0x50, 0x4B, 0x3, 0x4])) { // Local file header signature
|
|
1270
|
-
|
|
1271
|
-
const openXmlState = createOpenXmlZipDetectionState();
|
|
1272
|
-
|
|
1273
|
-
try {
|
|
1274
|
-
await new ZipHandler(tokenizer).unzip(zipHeader => {
|
|
1275
|
-
updateOpenXmlZipDetectionStateFromFilename(openXmlState, zipHeader.filename);
|
|
1276
|
-
|
|
1277
|
-
const isOpenXmlContentTypesEntry = zipHeader.filename === '[Content_Types].xml';
|
|
1278
|
-
const openXmlFileTypeFromEntries = getOpenXmlFileTypeFromZipEntries(openXmlState);
|
|
1279
|
-
if (
|
|
1280
|
-
!isOpenXmlContentTypesEntry
|
|
1281
|
-
&& openXmlFileTypeFromEntries
|
|
1282
|
-
) {
|
|
1283
|
-
fileType = openXmlFileTypeFromEntries;
|
|
1284
|
-
return {
|
|
1285
|
-
stop: true,
|
|
1286
|
-
};
|
|
1287
|
-
}
|
|
1288
|
-
|
|
1289
|
-
switch (zipHeader.filename) {
|
|
1290
|
-
case 'META-INF/mozilla.rsa':
|
|
1291
|
-
fileType = {
|
|
1292
|
-
ext: 'xpi',
|
|
1293
|
-
mime: 'application/x-xpinstall',
|
|
1294
|
-
};
|
|
1295
|
-
return {
|
|
1296
|
-
stop: true,
|
|
1297
|
-
};
|
|
1298
|
-
case 'META-INF/MANIFEST.MF':
|
|
1299
|
-
fileType = {
|
|
1300
|
-
ext: 'jar',
|
|
1301
|
-
mime: 'application/java-archive',
|
|
1302
|
-
};
|
|
1303
|
-
return {
|
|
1304
|
-
stop: true,
|
|
1305
|
-
};
|
|
1306
|
-
case 'mimetype':
|
|
1307
|
-
if (!canReadZipEntryForDetection(zipHeader, maximumZipTextEntrySizeInBytes)) {
|
|
1308
|
-
return {};
|
|
1309
|
-
}
|
|
1310
|
-
|
|
1311
|
-
return {
|
|
1312
|
-
async handler(fileData) {
|
|
1313
|
-
// Use TextDecoder to decode the UTF-8 encoded data
|
|
1314
|
-
const mimeType = new TextDecoder('utf-8').decode(fileData).trim();
|
|
1315
|
-
fileType = getFileTypeFromMimeType(mimeType);
|
|
1316
|
-
},
|
|
1317
|
-
stop: true,
|
|
1318
|
-
};
|
|
1319
|
-
|
|
1320
|
-
case '[Content_Types].xml': {
|
|
1321
|
-
openXmlState.hasContentTypesEntry = true;
|
|
1322
|
-
|
|
1323
|
-
if (!canReadZipEntryForDetection(zipHeader, maximumZipTextEntrySizeInBytes)) {
|
|
1324
|
-
openXmlState.hasUnparseableContentTypes = true;
|
|
1325
|
-
return {};
|
|
1326
|
-
}
|
|
1327
|
-
|
|
1328
|
-
openXmlState.isParsingContentTypes = true;
|
|
1329
|
-
return {
|
|
1330
|
-
async handler(fileData) {
|
|
1331
|
-
// Use TextDecoder to decode the UTF-8 encoded data
|
|
1332
|
-
const xmlContent = new TextDecoder('utf-8').decode(fileData);
|
|
1333
|
-
const mimeType = getOpenXmlMimeTypeFromContentTypesXml(xmlContent);
|
|
1334
|
-
if (mimeType) {
|
|
1335
|
-
fileType = getFileTypeFromMimeType(mimeType);
|
|
1336
|
-
}
|
|
1337
|
-
|
|
1338
|
-
openXmlState.hasParsedContentTypesEntry = true;
|
|
1339
|
-
openXmlState.isParsingContentTypes = false;
|
|
1340
|
-
},
|
|
1341
|
-
stop: true,
|
|
1342
|
-
};
|
|
1343
|
-
}
|
|
1344
|
-
|
|
1345
|
-
default:
|
|
1346
|
-
if (/classes\d*\.dex/.test(zipHeader.filename)) {
|
|
1347
|
-
fileType = {
|
|
1348
|
-
ext: 'apk',
|
|
1349
|
-
mime: 'application/vnd.android.package-archive',
|
|
1350
|
-
};
|
|
1351
|
-
return {stop: true};
|
|
1352
|
-
}
|
|
1353
|
-
|
|
1354
|
-
return {};
|
|
1355
|
-
}
|
|
1356
|
-
});
|
|
1357
|
-
} catch (error) {
|
|
1358
|
-
if (!isRecoverableZipError(error)) {
|
|
1359
|
-
throw error;
|
|
1360
|
-
}
|
|
1361
|
-
|
|
1362
|
-
if (openXmlState.isParsingContentTypes) {
|
|
1363
|
-
openXmlState.isParsingContentTypes = false;
|
|
1364
|
-
openXmlState.hasUnparseableContentTypes = true;
|
|
1365
|
-
}
|
|
1366
|
-
}
|
|
1367
|
-
|
|
1368
|
-
return fileType ?? getOpenXmlFileTypeFromZipEntries(openXmlState) ?? {
|
|
1369
|
-
ext: 'zip',
|
|
1370
|
-
mime: 'application/zip',
|
|
1371
|
-
};
|
|
700
|
+
return detectZip(tokenizer);
|
|
1372
701
|
}
|
|
1373
702
|
|
|
1374
703
|
if (this.checkString('OggS')) {
|
|
@@ -1378,7 +707,7 @@ export class FileTypeParser {
|
|
|
1378
707
|
await tokenizer.readBuffer(type);
|
|
1379
708
|
|
|
1380
709
|
// Needs to be before `ogg` check
|
|
1381
|
-
if (
|
|
710
|
+
if (checkBytes(type, [0x4F, 0x70, 0x75, 0x73, 0x48, 0x65, 0x61, 0x64])) {
|
|
1382
711
|
return {
|
|
1383
712
|
ext: 'opus',
|
|
1384
713
|
mime: 'audio/ogg; codecs=opus',
|
|
@@ -1386,7 +715,7 @@ export class FileTypeParser {
|
|
|
1386
715
|
}
|
|
1387
716
|
|
|
1388
717
|
// If ' theora' in header.
|
|
1389
|
-
if (
|
|
718
|
+
if (checkBytes(type, [0x80, 0x74, 0x68, 0x65, 0x6F, 0x72, 0x61])) {
|
|
1390
719
|
return {
|
|
1391
720
|
ext: 'ogv',
|
|
1392
721
|
mime: 'video/ogg',
|
|
@@ -1394,7 +723,7 @@ export class FileTypeParser {
|
|
|
1394
723
|
}
|
|
1395
724
|
|
|
1396
725
|
// If '\x01video' in header.
|
|
1397
|
-
if (
|
|
726
|
+
if (checkBytes(type, [0x01, 0x76, 0x69, 0x64, 0x65, 0x6F, 0x00])) {
|
|
1398
727
|
return {
|
|
1399
728
|
ext: 'ogm',
|
|
1400
729
|
mime: 'video/ogg',
|
|
@@ -1402,7 +731,7 @@ export class FileTypeParser {
|
|
|
1402
731
|
}
|
|
1403
732
|
|
|
1404
733
|
// If ' FLAC' in header https://xiph.org/flac/faq.html
|
|
1405
|
-
if (
|
|
734
|
+
if (checkBytes(type, [0x7F, 0x46, 0x4C, 0x41, 0x43])) {
|
|
1406
735
|
return {
|
|
1407
736
|
ext: 'oga',
|
|
1408
737
|
mime: 'audio/ogg',
|
|
@@ -1410,7 +739,7 @@ export class FileTypeParser {
|
|
|
1410
739
|
}
|
|
1411
740
|
|
|
1412
741
|
// 'Speex ' in header https://en.wikipedia.org/wiki/Speex
|
|
1413
|
-
if (
|
|
742
|
+
if (checkBytes(type, [0x53, 0x70, 0x65, 0x65, 0x78, 0x20, 0x20])) {
|
|
1414
743
|
return {
|
|
1415
744
|
ext: 'spx',
|
|
1416
745
|
mime: 'audio/ogg',
|
|
@@ -1418,7 +747,7 @@ export class FileTypeParser {
|
|
|
1418
747
|
}
|
|
1419
748
|
|
|
1420
749
|
// If '\x01vorbis' in header
|
|
1421
|
-
if (
|
|
750
|
+
if (checkBytes(type, [0x01, 0x76, 0x6F, 0x72, 0x62, 0x69, 0x73])) {
|
|
1422
751
|
return {
|
|
1423
752
|
ext: 'ogg',
|
|
1424
753
|
mime: 'audio/ogg',
|
|
@@ -1494,7 +823,7 @@ export class FileTypeParser {
|
|
|
1494
823
|
if (this.checkString('LZIP')) {
|
|
1495
824
|
return {
|
|
1496
825
|
ext: 'lz',
|
|
1497
|
-
mime: 'application/
|
|
826
|
+
mime: 'application/lzip',
|
|
1498
827
|
};
|
|
1499
828
|
}
|
|
1500
829
|
|
|
@@ -1559,110 +888,7 @@ export class FileTypeParser {
|
|
|
1559
888
|
|
|
1560
889
|
// https://github.com/file/file/blob/master/magic/Magdir/matroska
|
|
1561
890
|
if (this.check([0x1A, 0x45, 0xDF, 0xA3])) { // Root element: EBML
|
|
1562
|
-
|
|
1563
|
-
const msb = await tokenizer.peekNumber(Token.UINT8);
|
|
1564
|
-
let mask = 0x80;
|
|
1565
|
-
let ic = 0; // 0 = A, 1 = B, 2 = C, 3 = D
|
|
1566
|
-
|
|
1567
|
-
while ((msb & mask) === 0 && mask !== 0) {
|
|
1568
|
-
++ic;
|
|
1569
|
-
mask >>= 1;
|
|
1570
|
-
}
|
|
1571
|
-
|
|
1572
|
-
const id = new Uint8Array(ic + 1);
|
|
1573
|
-
await safeReadBuffer(tokenizer, id, undefined, {
|
|
1574
|
-
maximumLength: id.length,
|
|
1575
|
-
reason: 'EBML field',
|
|
1576
|
-
});
|
|
1577
|
-
return id;
|
|
1578
|
-
}
|
|
1579
|
-
|
|
1580
|
-
async function readElement() {
|
|
1581
|
-
const idField = await readField();
|
|
1582
|
-
const lengthField = await readField();
|
|
1583
|
-
|
|
1584
|
-
lengthField[0] ^= 0x80 >> (lengthField.length - 1);
|
|
1585
|
-
const nrLength = Math.min(6, lengthField.length); // JavaScript can max read 6 bytes integer
|
|
1586
|
-
|
|
1587
|
-
const idView = new DataView(idField.buffer);
|
|
1588
|
-
const lengthView = new DataView(lengthField.buffer, lengthField.length - nrLength, nrLength);
|
|
1589
|
-
|
|
1590
|
-
return {
|
|
1591
|
-
id: getUintBE(idView),
|
|
1592
|
-
len: getUintBE(lengthView),
|
|
1593
|
-
};
|
|
1594
|
-
}
|
|
1595
|
-
|
|
1596
|
-
async function readChildren(children) {
|
|
1597
|
-
let ebmlElementCount = 0;
|
|
1598
|
-
while (children > 0) {
|
|
1599
|
-
ebmlElementCount++;
|
|
1600
|
-
if (ebmlElementCount > maximumEbmlElementCount) {
|
|
1601
|
-
return;
|
|
1602
|
-
}
|
|
1603
|
-
|
|
1604
|
-
if (hasExceededUnknownSizeScanBudget(tokenizer, ebmlScanStart, maximumUntrustedSkipSizeInBytes)) {
|
|
1605
|
-
return;
|
|
1606
|
-
}
|
|
1607
|
-
|
|
1608
|
-
const previousPosition = tokenizer.position;
|
|
1609
|
-
const element = await readElement();
|
|
1610
|
-
|
|
1611
|
-
if (element.id === 0x42_82) {
|
|
1612
|
-
// `DocType` is a short string ("webm", "matroska", ...), reject implausible lengths to avoid large allocations.
|
|
1613
|
-
if (element.len > maximumEbmlDocumentTypeSizeInBytes) {
|
|
1614
|
-
return;
|
|
1615
|
-
}
|
|
1616
|
-
|
|
1617
|
-
const documentTypeLength = getSafeBound(element.len, maximumEbmlDocumentTypeSizeInBytes, 'EBML DocType');
|
|
1618
|
-
const rawValue = await tokenizer.readToken(new Token.StringType(documentTypeLength));
|
|
1619
|
-
return rawValue.replaceAll(/\00.*$/g, ''); // Return DocType
|
|
1620
|
-
}
|
|
1621
|
-
|
|
1622
|
-
if (
|
|
1623
|
-
hasUnknownFileSize(tokenizer)
|
|
1624
|
-
&& (
|
|
1625
|
-
!Number.isFinite(element.len)
|
|
1626
|
-
|| element.len < 0
|
|
1627
|
-
|| element.len > maximumEbmlElementPayloadSizeInBytes
|
|
1628
|
-
)
|
|
1629
|
-
) {
|
|
1630
|
-
return;
|
|
1631
|
-
}
|
|
1632
|
-
|
|
1633
|
-
await safeIgnore(tokenizer, element.len, {
|
|
1634
|
-
maximumLength: hasUnknownFileSize(tokenizer) ? maximumEbmlElementPayloadSizeInBytes : tokenizer.fileInfo.size,
|
|
1635
|
-
reason: 'EBML payload',
|
|
1636
|
-
}); // ignore payload
|
|
1637
|
-
--children;
|
|
1638
|
-
|
|
1639
|
-
// Safeguard against malformed files: bail if the position did not advance.
|
|
1640
|
-
if (tokenizer.position <= previousPosition) {
|
|
1641
|
-
return;
|
|
1642
|
-
}
|
|
1643
|
-
}
|
|
1644
|
-
}
|
|
1645
|
-
|
|
1646
|
-
const rootElement = await readElement();
|
|
1647
|
-
const ebmlScanStart = tokenizer.position;
|
|
1648
|
-
const documentType = await readChildren(rootElement.len);
|
|
1649
|
-
|
|
1650
|
-
switch (documentType) {
|
|
1651
|
-
case 'webm':
|
|
1652
|
-
return {
|
|
1653
|
-
ext: 'webm',
|
|
1654
|
-
mime: 'video/webm',
|
|
1655
|
-
};
|
|
1656
|
-
|
|
1657
|
-
case 'matroska':
|
|
1658
|
-
return {
|
|
1659
|
-
ext: 'mkv',
|
|
1660
|
-
mime: 'video/matroska',
|
|
1661
|
-
};
|
|
1662
|
-
|
|
1663
|
-
default:
|
|
1664
|
-
return;
|
|
1665
|
-
}
|
|
891
|
+
return detectEbml(tokenizer);
|
|
1666
892
|
}
|
|
1667
893
|
|
|
1668
894
|
if (this.checkString('SQLi')) {
|
|
@@ -1760,7 +986,7 @@ export class FileTypeParser {
|
|
|
1760
986
|
if (this.check([0x04, 0x22, 0x4D, 0x18])) {
|
|
1761
987
|
return {
|
|
1762
988
|
ext: 'lz4',
|
|
1763
|
-
mime: 'application/x-lz4', //
|
|
989
|
+
mime: 'application/x-lz4', // Informal, used by freedesktop.org shared-mime-info
|
|
1764
990
|
};
|
|
1765
991
|
}
|
|
1766
992
|
|
|
@@ -1795,7 +1021,7 @@ export class FileTypeParser {
|
|
|
1795
1021
|
};
|
|
1796
1022
|
}
|
|
1797
1023
|
|
|
1798
|
-
if (this.checkString(
|
|
1024
|
+
if (this.checkString(String.raw`{\rtf`)) {
|
|
1799
1025
|
return {
|
|
1800
1026
|
ext: 'rtf',
|
|
1801
1027
|
mime: 'application/rtf',
|
|
@@ -1896,7 +1122,7 @@ export class FileTypeParser {
|
|
|
1896
1122
|
if (this.checkString('DRACO')) {
|
|
1897
1123
|
return {
|
|
1898
1124
|
ext: 'drc',
|
|
1899
|
-
mime: 'application/
|
|
1125
|
+
mime: 'application/x-ft-draco',
|
|
1900
1126
|
};
|
|
1901
1127
|
}
|
|
1902
1128
|
|
|
@@ -1942,7 +1168,7 @@ export class FileTypeParser {
|
|
|
1942
1168
|
|
|
1943
1169
|
if (this.checkString('AC')) {
|
|
1944
1170
|
const version = new Token.StringType(4, 'latin1').get(this.buffer, 2);
|
|
1945
|
-
if (
|
|
1171
|
+
if (/^\d+$/v.test(version) && version >= 1000 && version <= 1050) {
|
|
1946
1172
|
return {
|
|
1947
1173
|
ext: 'dwg',
|
|
1948
1174
|
mime: 'image/vnd.dwg',
|
|
@@ -1997,110 +1223,7 @@ export class FileTypeParser {
|
|
|
1997
1223
|
// -- 8-byte signatures --
|
|
1998
1224
|
|
|
1999
1225
|
if (this.check([0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A])) {
|
|
2000
|
-
|
|
2001
|
-
ext: 'png',
|
|
2002
|
-
mime: 'image/png',
|
|
2003
|
-
};
|
|
2004
|
-
|
|
2005
|
-
const apngFileType = {
|
|
2006
|
-
ext: 'apng',
|
|
2007
|
-
mime: 'image/apng',
|
|
2008
|
-
};
|
|
2009
|
-
|
|
2010
|
-
// APNG format (https://wiki.mozilla.org/APNG_Specification)
|
|
2011
|
-
// 1. Find the first IDAT (image data) chunk (49 44 41 54)
|
|
2012
|
-
// 2. Check if there is an "acTL" chunk before the IDAT one (61 63 54 4C)
|
|
2013
|
-
|
|
2014
|
-
// Offset calculated as follows:
|
|
2015
|
-
// - 8 bytes: PNG signature
|
|
2016
|
-
// - 4 (length) + 4 (chunk type) + 13 (chunk data) + 4 (CRC): IHDR chunk
|
|
2017
|
-
|
|
2018
|
-
await tokenizer.ignore(8); // ignore PNG signature
|
|
2019
|
-
|
|
2020
|
-
async function readChunkHeader() {
|
|
2021
|
-
return {
|
|
2022
|
-
length: await tokenizer.readToken(Token.INT32_BE),
|
|
2023
|
-
type: await tokenizer.readToken(new Token.StringType(4, 'latin1')),
|
|
2024
|
-
};
|
|
2025
|
-
}
|
|
2026
|
-
|
|
2027
|
-
const isUnknownPngStream = hasUnknownFileSize(tokenizer);
|
|
2028
|
-
const pngScanStart = tokenizer.position;
|
|
2029
|
-
let pngChunkCount = 0;
|
|
2030
|
-
let hasSeenImageHeader = false;
|
|
2031
|
-
do {
|
|
2032
|
-
pngChunkCount++;
|
|
2033
|
-
if (pngChunkCount > maximumPngChunkCount) {
|
|
2034
|
-
break;
|
|
2035
|
-
}
|
|
2036
|
-
|
|
2037
|
-
if (hasExceededUnknownSizeScanBudget(tokenizer, pngScanStart, maximumPngStreamScanBudgetInBytes)) {
|
|
2038
|
-
break;
|
|
2039
|
-
}
|
|
2040
|
-
|
|
2041
|
-
const previousPosition = tokenizer.position;
|
|
2042
|
-
const chunk = await readChunkHeader();
|
|
2043
|
-
if (chunk.length < 0) {
|
|
2044
|
-
return; // Invalid chunk length
|
|
2045
|
-
}
|
|
2046
|
-
|
|
2047
|
-
if (chunk.type === 'IHDR') {
|
|
2048
|
-
// PNG requires the first real image header to be a 13-byte IHDR chunk.
|
|
2049
|
-
if (chunk.length !== 13) {
|
|
2050
|
-
return;
|
|
2051
|
-
}
|
|
2052
|
-
|
|
2053
|
-
hasSeenImageHeader = true;
|
|
2054
|
-
}
|
|
2055
|
-
|
|
2056
|
-
switch (chunk.type) {
|
|
2057
|
-
case 'IDAT':
|
|
2058
|
-
return pngFileType;
|
|
2059
|
-
case 'acTL':
|
|
2060
|
-
return apngFileType;
|
|
2061
|
-
default:
|
|
2062
|
-
if (
|
|
2063
|
-
!hasSeenImageHeader
|
|
2064
|
-
&& chunk.type !== 'CgBI'
|
|
2065
|
-
) {
|
|
2066
|
-
return;
|
|
2067
|
-
}
|
|
2068
|
-
|
|
2069
|
-
if (
|
|
2070
|
-
isUnknownPngStream
|
|
2071
|
-
&& chunk.length > maximumPngChunkSizeInBytes
|
|
2072
|
-
) {
|
|
2073
|
-
// Avoid huge attacker-controlled skips when probing unknown-size streams.
|
|
2074
|
-
return hasSeenImageHeader && isPngAncillaryChunk(chunk.type) ? pngFileType : undefined;
|
|
2075
|
-
}
|
|
2076
|
-
|
|
2077
|
-
try {
|
|
2078
|
-
await safeIgnore(tokenizer, chunk.length + 4, {
|
|
2079
|
-
maximumLength: isUnknownPngStream ? maximumPngChunkSizeInBytes + 4 : tokenizer.fileInfo.size,
|
|
2080
|
-
reason: 'PNG chunk payload',
|
|
2081
|
-
}); // Ignore chunk-data + CRC
|
|
2082
|
-
} catch (error) {
|
|
2083
|
-
if (
|
|
2084
|
-
!isUnknownPngStream
|
|
2085
|
-
&& (
|
|
2086
|
-
error instanceof ParserHardLimitError
|
|
2087
|
-
|| error instanceof strtok3.EndOfStreamError
|
|
2088
|
-
)
|
|
2089
|
-
) {
|
|
2090
|
-
return pngFileType;
|
|
2091
|
-
}
|
|
2092
|
-
|
|
2093
|
-
throw error;
|
|
2094
|
-
}
|
|
2095
|
-
}
|
|
2096
|
-
|
|
2097
|
-
// Safeguard against malformed files: bail if the position did not advance.
|
|
2098
|
-
if (tokenizer.position <= previousPosition) {
|
|
2099
|
-
break;
|
|
2100
|
-
}
|
|
2101
|
-
} while (tokenizer.position + 8 < tokenizer.fileInfo.size);
|
|
2102
|
-
|
|
2103
|
-
return pngFileType;
|
|
1226
|
+
return detectPng(tokenizer);
|
|
2104
1227
|
}
|
|
2105
1228
|
|
|
2106
1229
|
if (this.check([0x41, 0x52, 0x52, 0x4F, 0x57, 0x31, 0x00, 0x00])) {
|
|
@@ -2258,116 +1381,7 @@ export class FileTypeParser {
|
|
|
2258
1381
|
|
|
2259
1382
|
// ASF_Header_Object first 80 bytes
|
|
2260
1383
|
if (this.check([0x30, 0x26, 0xB2, 0x75, 0x8E, 0x66, 0xCF, 0x11, 0xA6, 0xD9])) {
|
|
2261
|
-
|
|
2262
|
-
try {
|
|
2263
|
-
async function readHeader() {
|
|
2264
|
-
const guid = new Uint8Array(16);
|
|
2265
|
-
await safeReadBuffer(tokenizer, guid, undefined, {
|
|
2266
|
-
maximumLength: guid.length,
|
|
2267
|
-
reason: 'ASF header GUID',
|
|
2268
|
-
});
|
|
2269
|
-
return {
|
|
2270
|
-
id: guid,
|
|
2271
|
-
size: Number(await tokenizer.readToken(Token.UINT64_LE)),
|
|
2272
|
-
};
|
|
2273
|
-
}
|
|
2274
|
-
|
|
2275
|
-
await safeIgnore(tokenizer, 30, {
|
|
2276
|
-
maximumLength: 30,
|
|
2277
|
-
reason: 'ASF header prelude',
|
|
2278
|
-
});
|
|
2279
|
-
const isUnknownFileSize = hasUnknownFileSize(tokenizer);
|
|
2280
|
-
const asfHeaderScanStart = tokenizer.position;
|
|
2281
|
-
let asfHeaderObjectCount = 0;
|
|
2282
|
-
while (tokenizer.position + 24 < tokenizer.fileInfo.size) {
|
|
2283
|
-
asfHeaderObjectCount++;
|
|
2284
|
-
if (asfHeaderObjectCount > maximumAsfHeaderObjectCount) {
|
|
2285
|
-
break;
|
|
2286
|
-
}
|
|
2287
|
-
|
|
2288
|
-
if (hasExceededUnknownSizeScanBudget(tokenizer, asfHeaderScanStart, maximumUntrustedSkipSizeInBytes)) {
|
|
2289
|
-
break;
|
|
2290
|
-
}
|
|
2291
|
-
|
|
2292
|
-
const previousPosition = tokenizer.position;
|
|
2293
|
-
const header = await readHeader();
|
|
2294
|
-
let payload = header.size - 24;
|
|
2295
|
-
if (
|
|
2296
|
-
!Number.isFinite(payload)
|
|
2297
|
-
|| payload < 0
|
|
2298
|
-
) {
|
|
2299
|
-
isMalformedAsf = true;
|
|
2300
|
-
break;
|
|
2301
|
-
}
|
|
2302
|
-
|
|
2303
|
-
if (_check(header.id, [0x91, 0x07, 0xDC, 0xB7, 0xB7, 0xA9, 0xCF, 0x11, 0x8E, 0xE6, 0x00, 0xC0, 0x0C, 0x20, 0x53, 0x65])) {
|
|
2304
|
-
// Sync on Stream-Properties-Object (B7DC0791-A9B7-11CF-8EE6-00C00C205365)
|
|
2305
|
-
const typeId = new Uint8Array(16);
|
|
2306
|
-
payload -= await safeReadBuffer(tokenizer, typeId, undefined, {
|
|
2307
|
-
maximumLength: typeId.length,
|
|
2308
|
-
reason: 'ASF stream type GUID',
|
|
2309
|
-
});
|
|
2310
|
-
|
|
2311
|
-
if (_check(typeId, [0x40, 0x9E, 0x69, 0xF8, 0x4D, 0x5B, 0xCF, 0x11, 0xA8, 0xFD, 0x00, 0x80, 0x5F, 0x5C, 0x44, 0x2B])) {
|
|
2312
|
-
// Found audio:
|
|
2313
|
-
return {
|
|
2314
|
-
ext: 'asf',
|
|
2315
|
-
mime: 'audio/x-ms-asf',
|
|
2316
|
-
};
|
|
2317
|
-
}
|
|
2318
|
-
|
|
2319
|
-
if (_check(typeId, [0xC0, 0xEF, 0x19, 0xBC, 0x4D, 0x5B, 0xCF, 0x11, 0xA8, 0xFD, 0x00, 0x80, 0x5F, 0x5C, 0x44, 0x2B])) {
|
|
2320
|
-
// Found video:
|
|
2321
|
-
return {
|
|
2322
|
-
ext: 'asf',
|
|
2323
|
-
mime: 'video/x-ms-asf',
|
|
2324
|
-
};
|
|
2325
|
-
}
|
|
2326
|
-
|
|
2327
|
-
break;
|
|
2328
|
-
}
|
|
2329
|
-
|
|
2330
|
-
if (
|
|
2331
|
-
isUnknownFileSize
|
|
2332
|
-
&& payload > maximumAsfHeaderPayloadSizeInBytes
|
|
2333
|
-
) {
|
|
2334
|
-
isMalformedAsf = true;
|
|
2335
|
-
break;
|
|
2336
|
-
}
|
|
2337
|
-
|
|
2338
|
-
await safeIgnore(tokenizer, payload, {
|
|
2339
|
-
maximumLength: isUnknownFileSize ? maximumAsfHeaderPayloadSizeInBytes : tokenizer.fileInfo.size,
|
|
2340
|
-
reason: 'ASF header payload',
|
|
2341
|
-
});
|
|
2342
|
-
|
|
2343
|
-
// Safeguard against malformed files: break if the position did not advance.
|
|
2344
|
-
if (tokenizer.position <= previousPosition) {
|
|
2345
|
-
isMalformedAsf = true;
|
|
2346
|
-
break;
|
|
2347
|
-
}
|
|
2348
|
-
}
|
|
2349
|
-
} catch (error) {
|
|
2350
|
-
if (
|
|
2351
|
-
error instanceof strtok3.EndOfStreamError
|
|
2352
|
-
|| error instanceof ParserHardLimitError
|
|
2353
|
-
) {
|
|
2354
|
-
if (hasUnknownFileSize(tokenizer)) {
|
|
2355
|
-
isMalformedAsf = true;
|
|
2356
|
-
}
|
|
2357
|
-
} else {
|
|
2358
|
-
throw error;
|
|
2359
|
-
}
|
|
2360
|
-
}
|
|
2361
|
-
|
|
2362
|
-
if (isMalformedAsf) {
|
|
2363
|
-
return;
|
|
2364
|
-
}
|
|
2365
|
-
|
|
2366
|
-
// Default to ASF generic extension
|
|
2367
|
-
return {
|
|
2368
|
-
ext: 'asf',
|
|
2369
|
-
mime: 'application/vnd.ms-asf',
|
|
2370
|
-
};
|
|
1384
|
+
return detectAsf(tokenizer);
|
|
2371
1385
|
}
|
|
2372
1386
|
|
|
2373
1387
|
if (this.check([0xAB, 0x4B, 0x54, 0x58, 0x20, 0x31, 0x31, 0xBB, 0x0D, 0x0A, 0x1A, 0x0A])) {
|
|
@@ -2581,21 +1595,21 @@ export class FileTypeParser {
|
|
|
2581
1595
|
if (this.check([0x4C, 0x00, 0x00, 0x00, 0x01, 0x14, 0x02, 0x00, 0x00, 0x00, 0x00, 0x00, 0xC0, 0x00, 0x00, 0x00, 0x00, 0x00, 0x00, 0x46])) {
|
|
2582
1596
|
return {
|
|
2583
1597
|
ext: 'lnk',
|
|
2584
|
-
mime: 'application/x
|
|
1598
|
+
mime: 'application/x-ms-shortcut', // Informal, used by freedesktop.org shared-mime-info
|
|
2585
1599
|
};
|
|
2586
1600
|
}
|
|
2587
1601
|
|
|
2588
1602
|
if (this.check([0x62, 0x6F, 0x6F, 0x6B, 0x00, 0x00, 0x00, 0x00, 0x6D, 0x61, 0x72, 0x6B, 0x00, 0x00, 0x00, 0x00])) {
|
|
2589
1603
|
return {
|
|
2590
1604
|
ext: 'alias',
|
|
2591
|
-
mime: 'application/x
|
|
1605
|
+
mime: 'application/x-ft-apple.alias',
|
|
2592
1606
|
};
|
|
2593
1607
|
}
|
|
2594
1608
|
|
|
2595
1609
|
if (this.checkString('Kaydara FBX Binary \u0000')) {
|
|
2596
1610
|
return {
|
|
2597
1611
|
ext: 'fbx',
|
|
2598
|
-
mime: 'application/x
|
|
1612
|
+
mime: 'application/x-ft-fbx',
|
|
2599
1613
|
};
|
|
2600
1614
|
}
|
|
2601
1615
|
|
|
@@ -2897,3 +1911,7 @@ export class FileTypeParser {
|
|
|
2897
1911
|
|
|
2898
1912
|
export const supportedExtensions = new Set(extensions);
|
|
2899
1913
|
export const supportedMimeTypes = new Set(mimeTypes);
|
|
1914
|
+
|
|
1915
|
+
export async function fileTypeFromFile(path, options) {
|
|
1916
|
+
return (new FileTypeParser(options)).fromFile(path);
|
|
1917
|
+
}
|