officeparser 7.5.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -0
- package/dist/OfficeParser.js +54 -6
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +153 -153
- package/dist/officeparser.browser.mjs +147 -147
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +176 -176
- package/dist/officeparser.browser.slim.mjs +176 -176
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/OpenOfficeParser.js +17 -3
- package/dist/parsers/PowerPointParser.js +21 -5
- package/dist/parsers/WordParser.js +12 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +1 -1
|
@@ -83,19 +83,33 @@ function isFullParserConfig(config) {
|
|
|
83
83
|
/**
|
|
84
84
|
* Resolves a full parser configuration by merging defaults and user-provided overrides.
|
|
85
85
|
*
|
|
86
|
+
* The returned object always belongs solely to the caller of this function. That matters
|
|
87
|
+
* because a parse installs per-call state on the config it is handed, such as the collector
|
|
88
|
+
* that gathers warnings for one document's `ast.warnings`. Returning the caller's own object
|
|
89
|
+
* would attach that state to an object they may reuse, so a second parse would append its
|
|
90
|
+
* warnings to the first document's already-returned AST, and each parse would retain the
|
|
91
|
+
* previous one's state for as long as the config lived.
|
|
92
|
+
*
|
|
93
|
+
* Only the configuration containers are copied. Callbacks and `abortSignal` keep their
|
|
94
|
+
* identity, since a copy of an `AbortSignal` would no longer be tied to its controller.
|
|
95
|
+
*
|
|
86
96
|
* @param userConfig - Optional configuration provided by the user
|
|
87
|
-
* @returns A fully populated configuration object
|
|
97
|
+
* @returns A fully populated configuration object, owned by the caller
|
|
88
98
|
*/
|
|
89
99
|
function resolveParserConfig(userConfig) {
|
|
90
100
|
if (isFullParserConfig(userConfig)) {
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
101
|
+
const resolved = { ...userConfig };
|
|
102
|
+
resolved.ocrConfig = { ...userConfig.ocrConfig };
|
|
103
|
+
if (userConfig.ocrConfig?.timeout) {
|
|
104
|
+
resolved.ocrConfig.timeout = { ...userConfig.ocrConfig.timeout };
|
|
105
|
+
}
|
|
106
|
+
resolved.decompressionLimits = {
|
|
107
|
+
...(userConfig.decompressionLimits ?? defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.decompressionLimits)
|
|
108
|
+
};
|
|
109
|
+
if (userConfig.htmlParserConfig) {
|
|
110
|
+
resolved.htmlParserConfig = { ...userConfig.htmlParserConfig };
|
|
97
111
|
}
|
|
98
|
-
return
|
|
112
|
+
return resolved;
|
|
99
113
|
}
|
|
100
114
|
// 1. Start with full defaults (deep cloned)
|
|
101
115
|
const config = deepClone(defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG);
|
|
@@ -141,21 +155,52 @@ function resolveParserConfig(userConfig) {
|
|
|
141
155
|
}
|
|
142
156
|
return config;
|
|
143
157
|
}
|
|
158
|
+
/** The per-destination and metadata sub-objects a generator config groups its settings into. */
|
|
159
|
+
const GENERATOR_CONFIG_CONTAINERS = [
|
|
160
|
+
'metadataOverrides', 'htmlConfig', 'mdConfig', 'pdfConfig',
|
|
161
|
+
'csvConfig', 'textConfig', 'rtfConfig', 'chunksConfig',
|
|
162
|
+
];
|
|
163
|
+
/**
|
|
164
|
+
* Copies a generator config's containers so writes during generation cannot reach the caller.
|
|
165
|
+
*
|
|
166
|
+
* One level is enough: the containers are what generation writes to. Everything else is copied
|
|
167
|
+
* by reference on purpose, since callbacks, `styleMap` and `abortSignal` are values whose
|
|
168
|
+
* identity matters, and a duplicated `AbortSignal` would no longer be tied to its controller.
|
|
169
|
+
*
|
|
170
|
+
* @param source - The caller's configuration
|
|
171
|
+
* @returns An equivalent configuration owned by us
|
|
172
|
+
*/
|
|
173
|
+
function copyGeneratorConfigContainers(source) {
|
|
174
|
+
const copy = { ...source };
|
|
175
|
+
for (const key of GENERATOR_CONFIG_CONTAINERS) {
|
|
176
|
+
const container = source[key];
|
|
177
|
+
if (container && typeof container === 'object')
|
|
178
|
+
copy[key] = { ...container };
|
|
179
|
+
}
|
|
180
|
+
return copy;
|
|
181
|
+
}
|
|
144
182
|
/**
|
|
145
183
|
* Resolves a full, destination-specific configuration by merging defaults,
|
|
146
184
|
* AST-level settings, and user-provided overrides.
|
|
147
185
|
*
|
|
186
|
+
* As with {@link resolveParserConfig}, the returned object belongs solely to the caller of this
|
|
187
|
+
* function, so that per-run normalization cannot edit a config the caller still holds.
|
|
188
|
+
*
|
|
148
189
|
* @param destination - The target format
|
|
149
190
|
* @param userConfig - Optional configuration provided by the user
|
|
150
191
|
* @param astConfig - Optional configuration from the source AST (for inheritance)
|
|
151
|
-
* @returns A fully populated configuration object
|
|
192
|
+
* @returns A fully populated configuration object, owned by the caller
|
|
152
193
|
*/
|
|
153
194
|
function resolveGeneratorConfig(destination, astConfig, userConfig) {
|
|
154
|
-
//
|
|
155
|
-
//
|
|
195
|
+
// Already complete, so nothing to merge. Still copied rather than handed straight back, for
|
|
196
|
+
// the same reason as resolveParserConfig: generation writes to the config it is given. The
|
|
197
|
+
// width check below normalizes an invalid `containerWidth` to 'auto', and doing that to the
|
|
198
|
+
// caller's own object both edits a value they still hold and silences the warning on every
|
|
199
|
+
// later run, so the same config would report a problem once and then appear clean.
|
|
156
200
|
if (isFullGeneratorConfig(userConfig) && !astConfig) {
|
|
157
|
-
|
|
158
|
-
|
|
201
|
+
const resolved = copyGeneratorConfigContainers(userConfig);
|
|
202
|
+
validateHtmlConfigWidth(resolved.htmlConfig, resolved);
|
|
203
|
+
return resolved;
|
|
159
204
|
}
|
|
160
205
|
// 1. Start with full defaults (deep cloned to avoid reference sharing)
|
|
161
206
|
const config = deepClone(defaults_js_1.DEFAULT_GENERATOR_CONFIG);
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
* It defines standard error types, messages, and handling logic to ensure
|
|
6
6
|
* consistent error reporting across all parsers and the main entry point.
|
|
7
7
|
*/
|
|
8
|
-
import { OfficeErrorType, OfficeParserConfig, OfficeWarningType } from '../types.js';
|
|
8
|
+
import { OfficeError, OfficeErrorType, OfficeParserConfig, OfficeWarningType } from '../types.js';
|
|
9
9
|
/**
|
|
10
10
|
* Creates a formatted warning message for a specific warning type.
|
|
11
11
|
*
|
|
@@ -22,11 +22,17 @@ export declare const getWarningMessage: (type: OfficeWarningType, info?: any) =>
|
|
|
22
22
|
* @param info - Optional additional information
|
|
23
23
|
* @returns The Error object to be thrown
|
|
24
24
|
*/
|
|
25
|
-
export declare const getOfficeError: (type: OfficeErrorType, config?: OfficeParserConfig, info?: any) =>
|
|
25
|
+
export declare const getOfficeError: (type: OfficeErrorType, config?: OfficeParserConfig, info?: any) => OfficeError;
|
|
26
26
|
/**
|
|
27
27
|
* Wraps an existing error with OfficeParser context and performs corruption detection.
|
|
28
28
|
* Optionally logs the error to console.
|
|
29
29
|
*
|
|
30
|
+
* An error already built by {@link getOfficeError} is returned untouched: it carries an
|
|
31
|
+
* `officeIssue`, meaning it has been reported once and already bears the `[OfficeParser]: `
|
|
32
|
+
* header. Re-wrapping it would report the same issue a second time, prepend a second header,
|
|
33
|
+
* and flatten its specific error code to `FILE_CORRUPTED`. This is a marker check on the error
|
|
34
|
+
* object rather than a test against its message text, so it stays independent of wording.
|
|
35
|
+
*
|
|
30
36
|
* **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
|
|
31
37
|
* represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
|
|
32
38
|
* callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
|
package/dist/utils/errorUtils.js
CHANGED
|
@@ -11,6 +11,11 @@ exports.checkAbortSignal = exports.getAbortError = exports.logWarning = exports.
|
|
|
11
11
|
const types_js_1 = require("../types.js");
|
|
12
12
|
/** Error header prefix for all error messages */
|
|
13
13
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
14
|
+
// `OfficeError` (the public shape callers catch) lives in types.ts alongside `OfficeIssue`.
|
|
15
|
+
// Every error built by getOfficeError is branded with its issue, which serves two purposes:
|
|
16
|
+
// consumers branch on `err.officeIssue.code` instead of matching message text, and
|
|
17
|
+
// getWrappedError recognizes an error it has already reported and prefixed, so it neither
|
|
18
|
+
// reports it twice nor prepends a second header.
|
|
14
19
|
/**
|
|
15
20
|
* Lookup table for error messages.
|
|
16
21
|
* Some entries are functions that take parameters to build dynamic messages.
|
|
@@ -34,6 +39,9 @@ const ERROR_MESSAGES = {
|
|
|
34
39
|
[types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
|
|
35
40
|
[types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
|
|
36
41
|
[types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
|
|
42
|
+
[types_js_1.OfficeErrorType.ZIP_NO_ENTRIES_FOUND]: `No readable entries found in ZIP data. The input is corrupt, truncated, or not a ZIP archive: every ZIP-based document format requires at least one entry.`,
|
|
43
|
+
[types_js_1.OfficeErrorType.ZIP_TRUNCATED]: `Malformed ZIP data: no End of Central Directory record was found at the end of the input. Either the file was cut off during download or transfer, or extra data follows the archive; in both cases the entries recovered from it cannot be trusted to be the whole document.`,
|
|
44
|
+
[types_js_1.OfficeErrorType.REQUIRED_PART_MISSING]: (info) => `Your ${info.fileType} file is a readable ZIP archive but is missing its required '${info.part}' part, so it cannot be a valid ${info.fileType} document. The file is corrupt, incomplete, or mislabeled. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce the error.`,
|
|
37
45
|
[types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED]: `Document nesting depth exceeded the safe limit (possible denial-of-service input)`,
|
|
38
46
|
[types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
|
|
39
47
|
};
|
|
@@ -60,6 +68,8 @@ const WARNING_MESSAGES = {
|
|
|
60
68
|
[types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED]: (limit) => `Table cell limit (${limit}) reached while expanding repeated ODF cells/rows; the remaining cells were not materialized. A few hundred bytes of XML can request an unbounded number of cells via table:number-columns-repeated / table:number-rows-repeated, so this is capped. Raise decompressionLimits.maxTableCells if your documents legitimately exceed it.`,
|
|
61
69
|
[types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`,
|
|
62
70
|
[types_js_1.OfficeWarningType.METADATA_NOT_REPRESENTABLE]: (info) => `Custom metadata ${info.keys.map(k => `'${k}'`).join(', ')} could not be written to ${info.format} output: the format has a fixed metadata vocabulary with no place for caller-defined keys. The named metadata fields (title, author, etc.) were still applied.`,
|
|
71
|
+
[types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND]: `Workbook contains no worksheet parts (xl/worksheets/). If the workbook holds only chartsheets this is expected and there is simply no cell text to extract; otherwise the file may be incomplete.`,
|
|
72
|
+
[types_js_1.OfficeWarningType.NO_SLIDES_FOUND]: `Presentation contains no slides (ppt/slides/). A legitimately empty presentation produces this too, but if you expected content the file may be incomplete.`,
|
|
63
73
|
[types_js_1.OfficeWarningType.INVALID_STYLE_MAP_TAG]: (tag) => `styleMap output.tag ${JSON.stringify(tag)} is not an allowed element name and was ignored; the node's default tag was used instead. A tag name is written into both the opening and closing tag, so only a known-safe set of block, heading and inline elements is accepted.`
|
|
64
74
|
};
|
|
65
75
|
/**
|
|
@@ -122,13 +132,23 @@ const getOfficeError = (type, config, info) => {
|
|
|
122
132
|
details: info
|
|
123
133
|
};
|
|
124
134
|
reportIssue(issue, config);
|
|
125
|
-
|
|
135
|
+
const error = new Error(ERRORHEADER + message);
|
|
136
|
+
// Brand the error with the issue that produced it so getWrappedError can tell an
|
|
137
|
+
// already-reported, already-prefixed OfficeParser error from a raw third-party one.
|
|
138
|
+
error.officeIssue = issue;
|
|
139
|
+
return error;
|
|
126
140
|
};
|
|
127
141
|
exports.getOfficeError = getOfficeError;
|
|
128
142
|
/**
|
|
129
143
|
* Wraps an existing error with OfficeParser context and performs corruption detection.
|
|
130
144
|
* Optionally logs the error to console.
|
|
131
145
|
*
|
|
146
|
+
* An error already built by {@link getOfficeError} is returned untouched: it carries an
|
|
147
|
+
* `officeIssue`, meaning it has been reported once and already bears the `[OfficeParser]: `
|
|
148
|
+
* header. Re-wrapping it would report the same issue a second time, prepend a second header,
|
|
149
|
+
* and flatten its specific error code to `FILE_CORRUPTED`. This is a marker check on the error
|
|
150
|
+
* object rather than a test against its message text, so it stays independent of wording.
|
|
151
|
+
*
|
|
132
152
|
* **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
|
|
133
153
|
* represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
|
|
134
154
|
* callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
|
|
@@ -140,6 +160,8 @@ exports.getOfficeError = getOfficeError;
|
|
|
140
160
|
* @returns The wrapped Error object to be thrown
|
|
141
161
|
*/
|
|
142
162
|
const getWrappedError = (error, config, filePath) => {
|
|
163
|
+
if (error?.officeIssue)
|
|
164
|
+
return error;
|
|
143
165
|
let message = error.message || error;
|
|
144
166
|
let code = types_js_1.OfficeErrorType.FILE_CORRUPTED; // Default for wrapped errors
|
|
145
167
|
// Detect file corruption from common library error messages
|
package/dist/utils/zipUtils.d.ts
CHANGED
|
@@ -13,12 +13,12 @@
|
|
|
13
13
|
*
|
|
14
14
|
* @module zipUtils
|
|
15
15
|
*/
|
|
16
|
-
import { DecompressionLimits } from '../types.js';
|
|
16
|
+
import { DecompressionLimits, OfficeParserConfig, SupportedFileType } from '../types.js';
|
|
17
17
|
/**
|
|
18
18
|
* Represents a file extracted from a ZIP archive.
|
|
19
19
|
* Contains the file's path within the archive and its content as a Buffer.
|
|
20
20
|
*/
|
|
21
|
-
interface ZipFileContent {
|
|
21
|
+
export interface ZipFileContent {
|
|
22
22
|
/**
|
|
23
23
|
* The relative path of the file within the ZIP archive.
|
|
24
24
|
* @example "word/document.xml", "xl/worksheets/sheet1.xml", "ppt/slides/slide1.xml"
|
|
@@ -46,6 +46,9 @@ interface ZipFileContent {
|
|
|
46
46
|
* @param zipInput - The ZIP file as a Node.js Buffer
|
|
47
47
|
* @param filterFn - A predicate function to determine which files to extract.
|
|
48
48
|
* Receives the filename and returns true to extract, false to skip.
|
|
49
|
+
* @param limits - Decompression limits guarding against zip bombs
|
|
50
|
+
* @param config - Parser configuration, so extraction failures honour `onWarning` /
|
|
51
|
+
* `outputErrorToConsole` like every other reported issue
|
|
49
52
|
* @returns A promise resolving to an array of extracted files
|
|
50
53
|
* @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
|
|
51
54
|
*
|
|
@@ -70,5 +73,62 @@ interface ZipFileContent {
|
|
|
70
73
|
*
|
|
71
74
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
72
75
|
*/
|
|
73
|
-
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
|
|
74
|
-
|
|
76
|
+
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits, config?: OfficeParserConfig) => Promise<ZipFileContent[]>;
|
|
77
|
+
/**
|
|
78
|
+
* Finds the archive part that every document of a given format must contain, and fails loudly
|
|
79
|
+
* when it is absent.
|
|
80
|
+
*
|
|
81
|
+
* A readable ZIP archive is not by itself a document: an archive can decompress perfectly and
|
|
82
|
+
* still be a renamed photo bundle, a partial upload, or a file mislabeled with the wrong
|
|
83
|
+
* extension. Without this check a parser finds no content to walk and returns an empty AST,
|
|
84
|
+
* which a caller cannot distinguish from a document that genuinely has nothing in it. Every
|
|
85
|
+
* ZIP-backed format has one part it cannot be valid without, so its absence is a hard error.
|
|
86
|
+
*
|
|
87
|
+
* Pass the parser's own regex/predicate for the part rather than a fresh copy of the path, so
|
|
88
|
+
* this check and the code that later reads the part cannot drift apart.
|
|
89
|
+
*
|
|
90
|
+
* @param files - The entries extracted from the archive
|
|
91
|
+
* @param matcher - Predicate identifying the required part by its path within the archive
|
|
92
|
+
* @param config - Parser configuration, so the error is reported through the caller's handlers
|
|
93
|
+
* @param info - The document format and the human-readable part name, used in the message
|
|
94
|
+
* @returns The matching entry
|
|
95
|
+
* @throws {Error} A typed REQUIRED_PART_MISSING error when no entry matches
|
|
96
|
+
*
|
|
97
|
+
* @example
|
|
98
|
+
* ```typescript
|
|
99
|
+
* const document = findRequiredPart(files, p => !!p.match(documentFileRegex), config,
|
|
100
|
+
* { fileType: 'docx', part: 'word/document.xml' });
|
|
101
|
+
* ```
|
|
102
|
+
*/
|
|
103
|
+
export declare const findRequiredPart: (files: ZipFileContent[], matcher: (path: string) => boolean, config: OfficeParserConfig, info: {
|
|
104
|
+
fileType: string;
|
|
105
|
+
part: string;
|
|
106
|
+
}) => ZipFileContent;
|
|
107
|
+
/**
|
|
108
|
+
* Resolves which office format a ZIP archive actually holds, by reading the part that names it.
|
|
109
|
+
*
|
|
110
|
+
* This exists because magic-byte sniffing is a heuristic that gives up. `file-type` identifies an
|
|
111
|
+
* OOXML package by parsing `[Content_Types].xml`, but it walks the archive under fixed budgets:
|
|
112
|
+
* at most 1024 entries, and (for entries whose sizes are deferred to a trailing data descriptor,
|
|
113
|
+
* general-purpose flag bit 3) about 1 MiB of scanning to locate those descriptors. An archive
|
|
114
|
+
* that puts enough data before `[Content_Types].xml` to exhaust either budget is reported as a
|
|
115
|
+
* generic `zip`, which previously surfaced to the caller as "add support for zip files" for a
|
|
116
|
+
* perfectly valid document. Both layouts occur in the wild: streaming ZIP writers set bit 3, and
|
|
117
|
+
* a media-heavy deck can hold more than 1024 parts.
|
|
118
|
+
*
|
|
119
|
+
* We already ship a ZIP reader that has neither limitation, so rather than guessing from the
|
|
120
|
+
* first bytes this opens the archive and reads the declaration directly. It is deliberately the
|
|
121
|
+
* fallback rather than the primary check, since the byte-level sniff is far cheaper and settles
|
|
122
|
+
* every non-ZIP format.
|
|
123
|
+
*
|
|
124
|
+
* @param zipInput - The candidate archive
|
|
125
|
+
* @param limits - Decompression limits, so sniffing an untrusted file stays bounded
|
|
126
|
+
* @returns The format the archive declares, or `undefined` if it declares none or cannot be read
|
|
127
|
+
*
|
|
128
|
+
* @example
|
|
129
|
+
* ```typescript
|
|
130
|
+
* // A presentation whose [Content_Types].xml sits behind 2 MiB of streamed entries
|
|
131
|
+
* await detectOfficeTypeFromZip(buffer, limits); // -> 'pptx'
|
|
132
|
+
* ```
|
|
133
|
+
*/
|
|
134
|
+
export declare const detectOfficeTypeFromZip: (zipInput: Buffer, limits: DecompressionLimits) => Promise<SupportedFileType | undefined>;
|
package/dist/utils/zipUtils.js
CHANGED
|
@@ -15,10 +15,26 @@
|
|
|
15
15
|
* @module zipUtils
|
|
16
16
|
*/
|
|
17
17
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
18
|
-
exports.extractFiles = void 0;
|
|
18
|
+
exports.detectOfficeTypeFromZip = exports.findRequiredPart = exports.extractFiles = void 0;
|
|
19
19
|
const fflate_1 = require("fflate");
|
|
20
20
|
const types_js_1 = require("../types.js");
|
|
21
21
|
const errorUtils_js_1 = require("./errorUtils.js");
|
|
22
|
+
/**
|
|
23
|
+
* Signature of the End Of Central Directory record ("PK\x05\x06"), the trailer every ZIP
|
|
24
|
+
* archive ends with. Its presence is what distinguishes a complete archive from one that
|
|
25
|
+
* was cut off in transfer.
|
|
26
|
+
*/
|
|
27
|
+
const EOCD_SIGNATURE = Buffer.from([0x50, 0x4b, 0x05, 0x06]);
|
|
28
|
+
/** Size of the fixed portion of an End Of Central Directory record, in bytes. */
|
|
29
|
+
const EOCD_RECORD_MIN_BYTES = 22;
|
|
30
|
+
/** Maximum size of the optional archive comment that may trail the EOCD record, in bytes. */
|
|
31
|
+
const ZIP_MAX_COMMENT_BYTES = 65535;
|
|
32
|
+
/**
|
|
33
|
+
* How far back from the end of the input the EOCD record may start. The record is last in
|
|
34
|
+
* the file apart from its own variable-length comment, so searching this window is
|
|
35
|
+
* sufficient and bounded regardless of archive size.
|
|
36
|
+
*/
|
|
37
|
+
const EOCD_SEARCH_WINDOW_BYTES = EOCD_RECORD_MIN_BYTES + ZIP_MAX_COMMENT_BYTES;
|
|
22
38
|
/**
|
|
23
39
|
* Extracts files from a ZIP archive with optional filtering.
|
|
24
40
|
*
|
|
@@ -34,6 +50,9 @@ const errorUtils_js_1 = require("./errorUtils.js");
|
|
|
34
50
|
* @param zipInput - The ZIP file as a Node.js Buffer
|
|
35
51
|
* @param filterFn - A predicate function to determine which files to extract.
|
|
36
52
|
* Receives the filename and returns true to extract, false to skip.
|
|
53
|
+
* @param limits - Decompression limits guarding against zip bombs
|
|
54
|
+
* @param config - Parser configuration, so extraction failures honour `onWarning` /
|
|
55
|
+
* `outputErrorToConsole` like every other reported issue
|
|
37
56
|
* @returns A promise resolving to an array of extracted files
|
|
38
57
|
* @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
|
|
39
58
|
*
|
|
@@ -58,7 +77,7 @@ const errorUtils_js_1 = require("./errorUtils.js");
|
|
|
58
77
|
*
|
|
59
78
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
60
79
|
*/
|
|
61
|
-
const extractFiles = (zipInput, filterFn, limits) => {
|
|
80
|
+
const extractFiles = (zipInput, filterFn, limits, config) => {
|
|
62
81
|
const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
|
|
63
82
|
? limits.maxUncompressedBytes
|
|
64
83
|
: 512 * 1024 * 1024;
|
|
@@ -93,7 +112,7 @@ const extractFiles = (zipInput, filterFn, limits) => {
|
|
|
93
112
|
return;
|
|
94
113
|
totalEntryCount++;
|
|
95
114
|
if (totalEntryCount > maxZipEntries) {
|
|
96
|
-
fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED,
|
|
115
|
+
fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, config, maxZipEntries));
|
|
97
116
|
return;
|
|
98
117
|
}
|
|
99
118
|
if (!filterFn(file.name))
|
|
@@ -111,7 +130,7 @@ const extractFiles = (zipInput, filterFn, limits) => {
|
|
|
111
130
|
if (chunk && chunk.length) {
|
|
112
131
|
actualTotalBytes += chunk.length;
|
|
113
132
|
if (actualTotalBytes > maxUncompressedBytes) {
|
|
114
|
-
fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED,
|
|
133
|
+
fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, config, maxUncompressedBytes));
|
|
115
134
|
return;
|
|
116
135
|
}
|
|
117
136
|
chunks.push(Buffer.from(chunk));
|
|
@@ -147,7 +166,172 @@ const extractFiles = (zipInput, filterFn, limits) => {
|
|
|
147
166
|
return;
|
|
148
167
|
}
|
|
149
168
|
pushComplete = true;
|
|
169
|
+
// fflate's streaming Unzip emits nothing, and no error, when the input is not a
|
|
170
|
+
// ZIP archive, unlike the central-directory-based unzip() this replaced (which
|
|
171
|
+
// rejected with "invalid zip data"). Zero entries can never be a valid document
|
|
172
|
+
// here, since every ZIP-backed format requires at least one part, so treat it as
|
|
173
|
+
// corrupt input rather than resolving into an empty, successfully-parsed document.
|
|
174
|
+
//
|
|
175
|
+
// This is checked before the truncation check below because it produces the better
|
|
176
|
+
// message for input that is not an archive at all, and because garbage that happens
|
|
177
|
+
// to contain the EOCD signature would otherwise slip past.
|
|
178
|
+
if (totalEntryCount === 0) {
|
|
179
|
+
fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_NO_ENTRIES_FOUND, config));
|
|
180
|
+
return;
|
|
181
|
+
}
|
|
182
|
+
// The streaming reader recovers entries from local file headers alone, so an archive
|
|
183
|
+
// cut short still yields whatever entries preceded the cut - silently, and possibly
|
|
184
|
+
// missing parts that came after it. The central-directory-based reader used before
|
|
185
|
+
// 7.3.0 rejected such input outright. Requiring the trailer that terminates every
|
|
186
|
+
// complete archive restores that: absent it, the data is truncated and the entries
|
|
187
|
+
// recovered cannot be trusted to be the whole document.
|
|
188
|
+
//
|
|
189
|
+
// Deliberately not gated on pendingFiles: when a cut lands inside an entry's
|
|
190
|
+
// compressed data that entry's final callback never fires, so this is also what
|
|
191
|
+
// settles the promise instead of leaving the caller waiting forever.
|
|
192
|
+
const tail = zipInput.subarray(Math.max(0, zipInput.length - EOCD_SEARCH_WINDOW_BYTES));
|
|
193
|
+
const eocdIndex = tail.lastIndexOf(EOCD_SIGNATURE);
|
|
194
|
+
if (eocdIndex === -1 || eocdIndex + EOCD_RECORD_MIN_BYTES > tail.length) {
|
|
195
|
+
fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_TRUNCATED, config));
|
|
196
|
+
return;
|
|
197
|
+
}
|
|
150
198
|
maybeResolve();
|
|
151
199
|
});
|
|
152
200
|
};
|
|
153
201
|
exports.extractFiles = extractFiles;
|
|
202
|
+
/**
|
|
203
|
+
* Finds the archive part that every document of a given format must contain, and fails loudly
|
|
204
|
+
* when it is absent.
|
|
205
|
+
*
|
|
206
|
+
* A readable ZIP archive is not by itself a document: an archive can decompress perfectly and
|
|
207
|
+
* still be a renamed photo bundle, a partial upload, or a file mislabeled with the wrong
|
|
208
|
+
* extension. Without this check a parser finds no content to walk and returns an empty AST,
|
|
209
|
+
* which a caller cannot distinguish from a document that genuinely has nothing in it. Every
|
|
210
|
+
* ZIP-backed format has one part it cannot be valid without, so its absence is a hard error.
|
|
211
|
+
*
|
|
212
|
+
* Pass the parser's own regex/predicate for the part rather than a fresh copy of the path, so
|
|
213
|
+
* this check and the code that later reads the part cannot drift apart.
|
|
214
|
+
*
|
|
215
|
+
* @param files - The entries extracted from the archive
|
|
216
|
+
* @param matcher - Predicate identifying the required part by its path within the archive
|
|
217
|
+
* @param config - Parser configuration, so the error is reported through the caller's handlers
|
|
218
|
+
* @param info - The document format and the human-readable part name, used in the message
|
|
219
|
+
* @returns The matching entry
|
|
220
|
+
* @throws {Error} A typed REQUIRED_PART_MISSING error when no entry matches
|
|
221
|
+
*
|
|
222
|
+
* @example
|
|
223
|
+
* ```typescript
|
|
224
|
+
* const document = findRequiredPart(files, p => !!p.match(documentFileRegex), config,
|
|
225
|
+
* { fileType: 'docx', part: 'word/document.xml' });
|
|
226
|
+
* ```
|
|
227
|
+
*/
|
|
228
|
+
const findRequiredPart = (files, matcher, config, info) => {
|
|
229
|
+
const found = files.find(file => matcher(file.path));
|
|
230
|
+
if (!found)
|
|
231
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.REQUIRED_PART_MISSING, config, info);
|
|
232
|
+
return found;
|
|
233
|
+
};
|
|
234
|
+
exports.findRequiredPart = findRequiredPart;
|
|
235
|
+
/** The part naming an OOXML package's document type. */
|
|
236
|
+
const OOXML_CONTENT_TYPES_PATH = '[Content_Types].xml';
|
|
237
|
+
/** The part naming an ODF or EPUB package's document type. */
|
|
238
|
+
const ODF_MIMETYPE_PATH = 'mimetype';
|
|
239
|
+
/**
|
|
240
|
+
* Substrings of the main-part content type each OOXML format declares in
|
|
241
|
+
* `[Content_Types].xml`, matched as plain text because only this one value is needed.
|
|
242
|
+
*/
|
|
243
|
+
const OOXML_MAIN_CONTENT_TYPES = [
|
|
244
|
+
['wordprocessingml.document.main+xml', 'docx'],
|
|
245
|
+
['spreadsheetml.sheet.main+xml', 'xlsx'],
|
|
246
|
+
['presentationml.presentation.main+xml', 'pptx'],
|
|
247
|
+
];
|
|
248
|
+
/** Exact `mimetype` entry contents for the packages that carry one. */
|
|
249
|
+
const PACKAGE_MIMETYPES = {
|
|
250
|
+
'application/vnd.oasis.opendocument.text': 'odt',
|
|
251
|
+
'application/vnd.oasis.opendocument.spreadsheet': 'ods',
|
|
252
|
+
'application/vnd.oasis.opendocument.presentation': 'odp',
|
|
253
|
+
'application/epub+zip': 'epub',
|
|
254
|
+
};
|
|
255
|
+
/** First two bytes of every ZIP local file header ("PK"). */
|
|
256
|
+
const ZIP_MAGIC_BYTES = [0x50, 0x4b];
|
|
257
|
+
/**
|
|
258
|
+
* Reporting is suppressed while sniffing: the input is not yet known to be a document, so a
|
|
259
|
+
* failure here is an inconclusive guess rather than something the caller did wrong. Without a
|
|
260
|
+
* config, `getOfficeError` would write these to the console.
|
|
261
|
+
*/
|
|
262
|
+
const SILENT_DETECTION_CONFIG = { outputErrorToConsole: false };
|
|
263
|
+
/** Whether a buffer starts with the ZIP local file header signature. */
|
|
264
|
+
const looksLikeZip = (buffer) => buffer.length >= ZIP_MAGIC_BYTES.length && ZIP_MAGIC_BYTES.every((byte, i) => buffer[i] === byte);
|
|
265
|
+
/**
|
|
266
|
+
* How much a type sniff may inflate before giving up, in bytes.
|
|
267
|
+
*
|
|
268
|
+
* The two parts read here name the format and nothing else, so they are tiny in any real
|
|
269
|
+
* document: a few hundred bytes of `mimetype`, a few kilobytes of `[Content_Types].xml`. A
|
|
270
|
+
* crafted archive could declare them as hundreds of megabytes, and since the document is
|
|
271
|
+
* inflated again during the parse that follows, honouring the full decompression budget here
|
|
272
|
+
* would let a single call spend it twice. This cap keeps sniffing cheap; an archive that
|
|
273
|
+
* exceeds it is simply reported as unidentified.
|
|
274
|
+
*/
|
|
275
|
+
const MAX_DETECTION_INFLATED_BYTES = 4 * 1024 * 1024;
|
|
276
|
+
/**
|
|
277
|
+
* Resolves which office format a ZIP archive actually holds, by reading the part that names it.
|
|
278
|
+
*
|
|
279
|
+
* This exists because magic-byte sniffing is a heuristic that gives up. `file-type` identifies an
|
|
280
|
+
* OOXML package by parsing `[Content_Types].xml`, but it walks the archive under fixed budgets:
|
|
281
|
+
* at most 1024 entries, and (for entries whose sizes are deferred to a trailing data descriptor,
|
|
282
|
+
* general-purpose flag bit 3) about 1 MiB of scanning to locate those descriptors. An archive
|
|
283
|
+
* that puts enough data before `[Content_Types].xml` to exhaust either budget is reported as a
|
|
284
|
+
* generic `zip`, which previously surfaced to the caller as "add support for zip files" for a
|
|
285
|
+
* perfectly valid document. Both layouts occur in the wild: streaming ZIP writers set bit 3, and
|
|
286
|
+
* a media-heavy deck can hold more than 1024 parts.
|
|
287
|
+
*
|
|
288
|
+
* We already ship a ZIP reader that has neither limitation, so rather than guessing from the
|
|
289
|
+
* first bytes this opens the archive and reads the declaration directly. It is deliberately the
|
|
290
|
+
* fallback rather than the primary check, since the byte-level sniff is far cheaper and settles
|
|
291
|
+
* every non-ZIP format.
|
|
292
|
+
*
|
|
293
|
+
* @param zipInput - The candidate archive
|
|
294
|
+
* @param limits - Decompression limits, so sniffing an untrusted file stays bounded
|
|
295
|
+
* @returns The format the archive declares, or `undefined` if it declares none or cannot be read
|
|
296
|
+
*
|
|
297
|
+
* @example
|
|
298
|
+
* ```typescript
|
|
299
|
+
* // A presentation whose [Content_Types].xml sits behind 2 MiB of streamed entries
|
|
300
|
+
* await detectOfficeTypeFromZip(buffer, limits); // -> 'pptx'
|
|
301
|
+
* ```
|
|
302
|
+
*/
|
|
303
|
+
const detectOfficeTypeFromZip = async (zipInput, limits) => {
|
|
304
|
+
if (!looksLikeZip(zipInput))
|
|
305
|
+
return undefined;
|
|
306
|
+
let files;
|
|
307
|
+
try {
|
|
308
|
+
files = await (0, exports.extractFiles)(zipInput, name => name === OOXML_CONTENT_TYPES_PATH || name === ODF_MIMETYPE_PATH,
|
|
309
|
+
// Never inflate more for a sniff than the caller already allows for the parse, and
|
|
310
|
+
// never more than a sniff could legitimately need.
|
|
311
|
+
{
|
|
312
|
+
...limits,
|
|
313
|
+
maxUncompressedBytes: Math.min(limits?.maxUncompressedBytes ?? MAX_DETECTION_INFLATED_BYTES, MAX_DETECTION_INFLATED_BYTES),
|
|
314
|
+
}, SILENT_DETECTION_CONFIG);
|
|
315
|
+
}
|
|
316
|
+
catch {
|
|
317
|
+
// Unreadable, truncated, or not an archive at all. The caller keeps whatever the
|
|
318
|
+
// byte-level sniff decided, and the parser it dispatches to reports the real problem.
|
|
319
|
+
return undefined;
|
|
320
|
+
}
|
|
321
|
+
// ODF and EPUB state their type outright, so prefer that over inspecting OOXML parts.
|
|
322
|
+
const mimetypeEntry = files.find(file => file.path === ODF_MIMETYPE_PATH);
|
|
323
|
+
if (mimetypeEntry) {
|
|
324
|
+
const declared = PACKAGE_MIMETYPES[mimetypeEntry.content.toString().trim()];
|
|
325
|
+
if (declared)
|
|
326
|
+
return declared;
|
|
327
|
+
}
|
|
328
|
+
const contentTypesEntry = files.find(file => file.path === OOXML_CONTENT_TYPES_PATH);
|
|
329
|
+
if (contentTypesEntry) {
|
|
330
|
+
const contentTypes = contentTypesEntry.content.toString();
|
|
331
|
+
const match = OOXML_MAIN_CONTENT_TYPES.find(([marker]) => contentTypes.includes(marker));
|
|
332
|
+
if (match)
|
|
333
|
+
return match[1];
|
|
334
|
+
}
|
|
335
|
+
return undefined;
|
|
336
|
+
};
|
|
337
|
+
exports.detectOfficeTypeFromZip = detectOfficeTypeFromZip;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "7.5.
|
|
3
|
+
"version": "7.5.1",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html, .epub) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, PDF, EPUB, and RAG-focused chunks.",
|
|
5
5
|
"funding": "https://github.com/sponsors/harshankur",
|
|
6
6
|
"main": "dist/index.js",
|