officeparser 7.5.0 → 7.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -83,19 +83,33 @@ function isFullParserConfig(config) {
83
83
  /**
84
84
  * Resolves a full parser configuration by merging defaults and user-provided overrides.
85
85
  *
86
+ * The returned object always belongs solely to the caller of this function. That matters
87
+ * because a parse installs per-call state on the config it is handed, such as the collector
88
+ * that gathers warnings for one document's `ast.warnings`. Returning the caller's own object
89
+ * would attach that state to an object they may reuse, so a second parse would append its
90
+ * warnings to the first document's already-returned AST, and each parse would retain the
91
+ * previous one's state for as long as the config lived.
92
+ *
93
+ * Only the configuration containers are copied. Callbacks and `abortSignal` keep their
94
+ * identity, since a copy of an `AbortSignal` would no longer be tied to its controller.
95
+ *
86
96
  * @param userConfig - Optional configuration provided by the user
87
- * @returns A fully populated configuration object
97
+ * @returns A fully populated configuration object, owned by the caller
88
98
  */
89
99
  function resolveParserConfig(userConfig) {
90
100
  if (isFullParserConfig(userConfig)) {
91
- if (!userConfig.decompressionLimits) {
92
- userConfig.decompressionLimits = {
93
- maxUncompressedBytes: 512 * 1024 * 1024,
94
- maxZipEntries: 10000,
95
- maxTableCells: 1000000,
96
- };
101
+ const resolved = { ...userConfig };
102
+ resolved.ocrConfig = { ...userConfig.ocrConfig };
103
+ if (userConfig.ocrConfig?.timeout) {
104
+ resolved.ocrConfig.timeout = { ...userConfig.ocrConfig.timeout };
105
+ }
106
+ resolved.decompressionLimits = {
107
+ ...(userConfig.decompressionLimits ?? defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG.decompressionLimits)
108
+ };
109
+ if (userConfig.htmlParserConfig) {
110
+ resolved.htmlParserConfig = { ...userConfig.htmlParserConfig };
97
111
  }
98
- return userConfig;
112
+ return resolved;
99
113
  }
100
114
  // 1. Start with full defaults (deep cloned)
101
115
  const config = deepClone(defaults_js_1.DEFAULT_OFFICE_PARSER_CONFIG);
@@ -141,21 +155,52 @@ function resolveParserConfig(userConfig) {
141
155
  }
142
156
  return config;
143
157
  }
158
+ /** The per-destination and metadata sub-objects a generator config groups its settings into. */
159
+ const GENERATOR_CONFIG_CONTAINERS = [
160
+ 'metadataOverrides', 'htmlConfig', 'mdConfig', 'pdfConfig',
161
+ 'csvConfig', 'textConfig', 'rtfConfig', 'chunksConfig',
162
+ ];
163
+ /**
164
+ * Copies a generator config's containers so writes during generation cannot reach the caller.
165
+ *
166
+ * One level is enough: the containers are what generation writes to. Everything else is copied
167
+ * by reference on purpose, since callbacks, `styleMap` and `abortSignal` are values whose
168
+ * identity matters, and a duplicated `AbortSignal` would no longer be tied to its controller.
169
+ *
170
+ * @param source - The caller's configuration
171
+ * @returns An equivalent configuration owned by us
172
+ */
173
+ function copyGeneratorConfigContainers(source) {
174
+ const copy = { ...source };
175
+ for (const key of GENERATOR_CONFIG_CONTAINERS) {
176
+ const container = source[key];
177
+ if (container && typeof container === 'object')
178
+ copy[key] = { ...container };
179
+ }
180
+ return copy;
181
+ }
144
182
  /**
145
183
  * Resolves a full, destination-specific configuration by merging defaults,
146
184
  * AST-level settings, and user-provided overrides.
147
185
  *
186
+ * As with {@link resolveParserConfig}, the returned object belongs solely to the caller of this
187
+ * function, so that per-run normalization cannot edit a config the caller still holds.
188
+ *
148
189
  * @param destination - The target format
149
190
  * @param userConfig - Optional configuration provided by the user
150
191
  * @param astConfig - Optional configuration from the source AST (for inheritance)
151
- * @returns A fully populated configuration object
192
+ * @returns A fully populated configuration object, owned by the caller
152
193
  */
153
194
  function resolveGeneratorConfig(destination, astConfig, userConfig) {
154
- // If it's already a full config and we don't need to merge AST config, return it as is.
155
- // We assume FullGeneratorConfig is already "safe" (references resolved).
195
+ // Already complete, so nothing to merge. Still copied rather than handed straight back, for
196
+ // the same reason as resolveParserConfig: generation writes to the config it is given. The
197
+ // width check below normalizes an invalid `containerWidth` to 'auto', and doing that to the
198
+ // caller's own object both edits a value they still hold and silences the warning on every
199
+ // later run, so the same config would report a problem once and then appear clean.
156
200
  if (isFullGeneratorConfig(userConfig) && !astConfig) {
157
- validateHtmlConfigWidth(userConfig.htmlConfig, userConfig);
158
- return userConfig;
201
+ const resolved = copyGeneratorConfigContainers(userConfig);
202
+ validateHtmlConfigWidth(resolved.htmlConfig, resolved);
203
+ return resolved;
159
204
  }
160
205
  // 1. Start with full defaults (deep cloned to avoid reference sharing)
161
206
  const config = deepClone(defaults_js_1.DEFAULT_GENERATOR_CONFIG);
@@ -5,7 +5,7 @@
5
5
  * It defines standard error types, messages, and handling logic to ensure
6
6
  * consistent error reporting across all parsers and the main entry point.
7
7
  */
8
- import { OfficeErrorType, OfficeParserConfig, OfficeWarningType } from '../types.js';
8
+ import { OfficeError, OfficeErrorType, OfficeParserConfig, OfficeWarningType } from '../types.js';
9
9
  /**
10
10
  * Creates a formatted warning message for a specific warning type.
11
11
  *
@@ -22,11 +22,17 @@ export declare const getWarningMessage: (type: OfficeWarningType, info?: any) =>
22
22
  * @param info - Optional additional information
23
23
  * @returns The Error object to be thrown
24
24
  */
25
- export declare const getOfficeError: (type: OfficeErrorType, config?: OfficeParserConfig, info?: any) => Error;
25
+ export declare const getOfficeError: (type: OfficeErrorType, config?: OfficeParserConfig, info?: any) => OfficeError;
26
26
  /**
27
27
  * Wraps an existing error with OfficeParser context and performs corruption detection.
28
28
  * Optionally logs the error to console.
29
29
  *
30
+ * An error already built by {@link getOfficeError} is returned untouched: it carries an
31
+ * `officeIssue`, meaning it has been reported once and already bears the `[OfficeParser]: `
32
+ * header. Re-wrapping it would report the same issue a second time, prepend a second header,
33
+ * and flatten its specific error code to `FILE_CORRUPTED`. This is a marker check on the error
34
+ * object rather than a test against its message text, so it stays independent of wording.
35
+ *
30
36
  * **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
31
37
  * represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
32
38
  * callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
@@ -11,6 +11,11 @@ exports.checkAbortSignal = exports.getAbortError = exports.logWarning = exports.
11
11
  const types_js_1 = require("../types.js");
12
12
  /** Error header prefix for all error messages */
13
13
  const ERRORHEADER = "[OfficeParser]: ";
14
+ // `OfficeError` (the public shape callers catch) lives in types.ts alongside `OfficeIssue`.
15
+ // Every error built by getOfficeError is branded with its issue, which serves two purposes:
16
+ // consumers branch on `err.officeIssue.code` instead of matching message text, and
17
+ // getWrappedError recognizes an error it has already reported and prefixed, so it neither
18
+ // reports it twice nor prepends a second header.
14
19
  /**
15
20
  * Lookup table for error messages.
16
21
  * Some entries are functions that take parameters to build dynamic messages.
@@ -34,6 +39,9 @@ const ERROR_MESSAGES = {
34
39
  [types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
35
40
  [types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
36
41
  [types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
42
+ [types_js_1.OfficeErrorType.ZIP_NO_ENTRIES_FOUND]: `No readable entries found in ZIP data. The input is corrupt, truncated, or not a ZIP archive: every ZIP-based document format requires at least one entry.`,
43
+ [types_js_1.OfficeErrorType.ZIP_TRUNCATED]: `Malformed ZIP data: no End of Central Directory record was found at the end of the input. Either the file was cut off during download or transfer, or extra data follows the archive; in both cases the entries recovered from it cannot be trusted to be the whole document.`,
44
+ [types_js_1.OfficeErrorType.REQUIRED_PART_MISSING]: (info) => `Your ${info.fileType} file is a readable ZIP archive but is missing its required '${info.part}' part, so it cannot be a valid ${info.fileType} document. The file is corrupt, incomplete, or mislabeled. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce the error.`,
37
45
  [types_js_1.OfficeErrorType.MAX_NESTING_DEPTH_EXCEEDED]: `Document nesting depth exceeded the safe limit (possible denial-of-service input)`,
38
46
  [types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
39
47
  };
@@ -60,6 +68,8 @@ const WARNING_MESSAGES = {
60
68
  [types_js_1.OfficeWarningType.TABLE_CELL_LIMIT_EXCEEDED]: (limit) => `Table cell limit (${limit}) reached while expanding repeated ODF cells/rows; the remaining cells were not materialized. A few hundred bytes of XML can request an unbounded number of cells via table:number-columns-repeated / table:number-rows-repeated, so this is capped. Raise decompressionLimits.maxTableCells if your documents legitimately exceed it.`,
61
69
  [types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`,
62
70
  [types_js_1.OfficeWarningType.METADATA_NOT_REPRESENTABLE]: (info) => `Custom metadata ${info.keys.map(k => `'${k}'`).join(', ')} could not be written to ${info.format} output: the format has a fixed metadata vocabulary with no place for caller-defined keys. The named metadata fields (title, author, etc.) were still applied.`,
71
+ [types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND]: `Workbook contains no worksheet parts (xl/worksheets/). If the workbook holds only chartsheets this is expected and there is simply no cell text to extract; otherwise the file may be incomplete.`,
72
+ [types_js_1.OfficeWarningType.NO_SLIDES_FOUND]: `Presentation contains no slides (ppt/slides/). A legitimately empty presentation produces this too, but if you expected content the file may be incomplete.`,
63
73
  [types_js_1.OfficeWarningType.INVALID_STYLE_MAP_TAG]: (tag) => `styleMap output.tag ${JSON.stringify(tag)} is not an allowed element name and was ignored; the node's default tag was used instead. A tag name is written into both the opening and closing tag, so only a known-safe set of block, heading and inline elements is accepted.`
64
74
  };
65
75
  /**
@@ -122,13 +132,23 @@ const getOfficeError = (type, config, info) => {
122
132
  details: info
123
133
  };
124
134
  reportIssue(issue, config);
125
- return new Error(ERRORHEADER + message);
135
+ const error = new Error(ERRORHEADER + message);
136
+ // Brand the error with the issue that produced it so getWrappedError can tell an
137
+ // already-reported, already-prefixed OfficeParser error from a raw third-party one.
138
+ error.officeIssue = issue;
139
+ return error;
126
140
  };
127
141
  exports.getOfficeError = getOfficeError;
128
142
  /**
129
143
  * Wraps an existing error with OfficeParser context and performs corruption detection.
130
144
  * Optionally logs the error to console.
131
145
  *
146
+ * An error already built by {@link getOfficeError} is returned untouched: it carries an
147
+ * `officeIssue`, meaning it has been reported once and already bears the `[OfficeParser]: `
148
+ * header. Re-wrapping it would report the same issue a second time, prepend a second header,
149
+ * and flatten its specific error code to `FILE_CORRUPTED`. This is a marker check on the error
150
+ * object rather than a test against its message text, so it stays independent of wording.
151
+ *
132
152
  * **Important**: Do NOT pass AbortErrors to this function. AbortErrors (err.name === 'AbortError')
133
153
  * represent deliberate user cancellation and must be re-thrown as-is from the catch block so that
134
154
  * callers can reliably detect them via `err.name === 'AbortError'` or `err instanceof DOMException`.
@@ -140,6 +160,8 @@ exports.getOfficeError = getOfficeError;
140
160
  * @returns The wrapped Error object to be thrown
141
161
  */
142
162
  const getWrappedError = (error, config, filePath) => {
163
+ if (error?.officeIssue)
164
+ return error;
143
165
  let message = error.message || error;
144
166
  let code = types_js_1.OfficeErrorType.FILE_CORRUPTED; // Default for wrapped errors
145
167
  // Detect file corruption from common library error messages
@@ -13,12 +13,12 @@
13
13
  *
14
14
  * @module zipUtils
15
15
  */
16
- import { DecompressionLimits } from '../types.js';
16
+ import { DecompressionLimits, OfficeParserConfig, SupportedFileType } from '../types.js';
17
17
  /**
18
18
  * Represents a file extracted from a ZIP archive.
19
19
  * Contains the file's path within the archive and its content as a Buffer.
20
20
  */
21
- interface ZipFileContent {
21
+ export interface ZipFileContent {
22
22
  /**
23
23
  * The relative path of the file within the ZIP archive.
24
24
  * @example "word/document.xml", "xl/worksheets/sheet1.xml", "ppt/slides/slide1.xml"
@@ -46,6 +46,9 @@ interface ZipFileContent {
46
46
  * @param zipInput - The ZIP file as a Node.js Buffer
47
47
  * @param filterFn - A predicate function to determine which files to extract.
48
48
  * Receives the filename and returns true to extract, false to skip.
49
+ * @param limits - Decompression limits guarding against zip bombs
50
+ * @param config - Parser configuration, so extraction failures honour `onWarning` /
51
+ * `outputErrorToConsole` like every other reported issue
49
52
  * @returns A promise resolving to an array of extracted files
50
53
  * @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
51
54
  *
@@ -70,5 +73,62 @@ interface ZipFileContent {
70
73
  *
71
74
  * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
72
75
  */
73
- export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
74
- export {};
76
+ export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits, config?: OfficeParserConfig) => Promise<ZipFileContent[]>;
77
+ /**
78
+ * Finds the archive part that every document of a given format must contain, and fails loudly
79
+ * when it is absent.
80
+ *
81
+ * A readable ZIP archive is not by itself a document: an archive can decompress perfectly and
82
+ * still be a renamed photo bundle, a partial upload, or a file mislabeled with the wrong
83
+ * extension. Without this check a parser finds no content to walk and returns an empty AST,
84
+ * which a caller cannot distinguish from a document that genuinely has nothing in it. Every
85
+ * ZIP-backed format has one part it cannot be valid without, so its absence is a hard error.
86
+ *
87
+ * Pass the parser's own regex/predicate for the part rather than a fresh copy of the path, so
88
+ * this check and the code that later reads the part cannot drift apart.
89
+ *
90
+ * @param files - The entries extracted from the archive
91
+ * @param matcher - Predicate identifying the required part by its path within the archive
92
+ * @param config - Parser configuration, so the error is reported through the caller's handlers
93
+ * @param info - The document format and the human-readable part name, used in the message
94
+ * @returns The matching entry
95
+ * @throws {Error} A typed REQUIRED_PART_MISSING error when no entry matches
96
+ *
97
+ * @example
98
+ * ```typescript
99
+ * const document = findRequiredPart(files, p => !!p.match(documentFileRegex), config,
100
+ * { fileType: 'docx', part: 'word/document.xml' });
101
+ * ```
102
+ */
103
+ export declare const findRequiredPart: (files: ZipFileContent[], matcher: (path: string) => boolean, config: OfficeParserConfig, info: {
104
+ fileType: string;
105
+ part: string;
106
+ }) => ZipFileContent;
107
+ /**
108
+ * Resolves which office format a ZIP archive actually holds, by reading the part that names it.
109
+ *
110
+ * This exists because magic-byte sniffing is a heuristic that gives up. `file-type` identifies an
111
+ * OOXML package by parsing `[Content_Types].xml`, but it walks the archive under fixed budgets:
112
+ * at most 1024 entries, and (for entries whose sizes are deferred to a trailing data descriptor,
113
+ * general-purpose flag bit 3) about 1 MiB of scanning to locate those descriptors. An archive
114
+ * that puts enough data before `[Content_Types].xml` to exhaust either budget is reported as a
115
+ * generic `zip`, which previously surfaced to the caller as "add support for zip files" for a
116
+ * perfectly valid document. Both layouts occur in the wild: streaming ZIP writers set bit 3, and
117
+ * a media-heavy deck can hold more than 1024 parts.
118
+ *
119
+ * We already ship a ZIP reader that has neither limitation, so rather than guessing from the
120
+ * first bytes this opens the archive and reads the declaration directly. It is deliberately the
121
+ * fallback rather than the primary check, since the byte-level sniff is far cheaper and settles
122
+ * every non-ZIP format.
123
+ *
124
+ * @param zipInput - The candidate archive
125
+ * @param limits - Decompression limits, so sniffing an untrusted file stays bounded
126
+ * @returns The format the archive declares, or `undefined` if it declares none or cannot be read
127
+ *
128
+ * @example
129
+ * ```typescript
130
+ * // A presentation whose [Content_Types].xml sits behind 2 MiB of streamed entries
131
+ * await detectOfficeTypeFromZip(buffer, limits); // -> 'pptx'
132
+ * ```
133
+ */
134
+ export declare const detectOfficeTypeFromZip: (zipInput: Buffer, limits: DecompressionLimits) => Promise<SupportedFileType | undefined>;
@@ -15,10 +15,26 @@
15
15
  * @module zipUtils
16
16
  */
17
17
  Object.defineProperty(exports, "__esModule", { value: true });
18
- exports.extractFiles = void 0;
18
+ exports.detectOfficeTypeFromZip = exports.findRequiredPart = exports.extractFiles = void 0;
19
19
  const fflate_1 = require("fflate");
20
20
  const types_js_1 = require("../types.js");
21
21
  const errorUtils_js_1 = require("./errorUtils.js");
22
+ /**
23
+ * Signature of the End Of Central Directory record ("PK\x05\x06"), the trailer every ZIP
24
+ * archive ends with. Its presence is what distinguishes a complete archive from one that
25
+ * was cut off in transfer.
26
+ */
27
+ const EOCD_SIGNATURE = Buffer.from([0x50, 0x4b, 0x05, 0x06]);
28
+ /** Size of the fixed portion of an End Of Central Directory record, in bytes. */
29
+ const EOCD_RECORD_MIN_BYTES = 22;
30
+ /** Maximum size of the optional archive comment that may trail the EOCD record, in bytes. */
31
+ const ZIP_MAX_COMMENT_BYTES = 65535;
32
+ /**
33
+ * How far back from the end of the input the EOCD record may start. The record is last in
34
+ * the file apart from its own variable-length comment, so searching this window is
35
+ * sufficient and bounded regardless of archive size.
36
+ */
37
+ const EOCD_SEARCH_WINDOW_BYTES = EOCD_RECORD_MIN_BYTES + ZIP_MAX_COMMENT_BYTES;
22
38
  /**
23
39
  * Extracts files from a ZIP archive with optional filtering.
24
40
  *
@@ -34,6 +50,9 @@ const errorUtils_js_1 = require("./errorUtils.js");
34
50
  * @param zipInput - The ZIP file as a Node.js Buffer
35
51
  * @param filterFn - A predicate function to determine which files to extract.
36
52
  * Receives the filename and returns true to extract, false to skip.
53
+ * @param limits - Decompression limits guarding against zip bombs
54
+ * @param config - Parser configuration, so extraction failures honour `onWarning` /
55
+ * `outputErrorToConsole` like every other reported issue
37
56
  * @returns A promise resolving to an array of extracted files
38
57
  * @throws {Error} If the ZIP file cannot be opened or an entry cannot be read
39
58
  *
@@ -58,7 +77,7 @@ const errorUtils_js_1 = require("./errorUtils.js");
58
77
  *
59
78
  * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
60
79
  */
61
- const extractFiles = (zipInput, filterFn, limits) => {
80
+ const extractFiles = (zipInput, filterFn, limits, config) => {
62
81
  const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
63
82
  ? limits.maxUncompressedBytes
64
83
  : 512 * 1024 * 1024;
@@ -93,7 +112,7 @@ const extractFiles = (zipInput, filterFn, limits) => {
93
112
  return;
94
113
  totalEntryCount++;
95
114
  if (totalEntryCount > maxZipEntries) {
96
- fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, undefined, maxZipEntries));
115
+ fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, config, maxZipEntries));
97
116
  return;
98
117
  }
99
118
  if (!filterFn(file.name))
@@ -111,7 +130,7 @@ const extractFiles = (zipInput, filterFn, limits) => {
111
130
  if (chunk && chunk.length) {
112
131
  actualTotalBytes += chunk.length;
113
132
  if (actualTotalBytes > maxUncompressedBytes) {
114
- fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, undefined, maxUncompressedBytes));
133
+ fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, config, maxUncompressedBytes));
115
134
  return;
116
135
  }
117
136
  chunks.push(Buffer.from(chunk));
@@ -147,7 +166,172 @@ const extractFiles = (zipInput, filterFn, limits) => {
147
166
  return;
148
167
  }
149
168
  pushComplete = true;
169
+ // fflate's streaming Unzip emits nothing, and no error, when the input is not a
170
+ // ZIP archive, unlike the central-directory-based unzip() this replaced (which
171
+ // rejected with "invalid zip data"). Zero entries can never be a valid document
172
+ // here, since every ZIP-backed format requires at least one part, so treat it as
173
+ // corrupt input rather than resolving into an empty, successfully-parsed document.
174
+ //
175
+ // This is checked before the truncation check below because it produces the better
176
+ // message for input that is not an archive at all, and because garbage that happens
177
+ // to contain the EOCD signature would otherwise slip past.
178
+ if (totalEntryCount === 0) {
179
+ fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_NO_ENTRIES_FOUND, config));
180
+ return;
181
+ }
182
+ // The streaming reader recovers entries from local file headers alone, so an archive
183
+ // cut short still yields whatever entries preceded the cut - silently, and possibly
184
+ // missing parts that came after it. The central-directory-based reader used before
185
+ // 7.3.0 rejected such input outright. Requiring the trailer that terminates every
186
+ // complete archive restores that: absent it, the data is truncated and the entries
187
+ // recovered cannot be trusted to be the whole document.
188
+ //
189
+ // Deliberately not gated on pendingFiles: when a cut lands inside an entry's
190
+ // compressed data that entry's final callback never fires, so this is also what
191
+ // settles the promise instead of leaving the caller waiting forever.
192
+ const tail = zipInput.subarray(Math.max(0, zipInput.length - EOCD_SEARCH_WINDOW_BYTES));
193
+ const eocdIndex = tail.lastIndexOf(EOCD_SIGNATURE);
194
+ if (eocdIndex === -1 || eocdIndex + EOCD_RECORD_MIN_BYTES > tail.length) {
195
+ fail((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_TRUNCATED, config));
196
+ return;
197
+ }
150
198
  maybeResolve();
151
199
  });
152
200
  };
153
201
  exports.extractFiles = extractFiles;
202
+ /**
203
+ * Finds the archive part that every document of a given format must contain, and fails loudly
204
+ * when it is absent.
205
+ *
206
+ * A readable ZIP archive is not by itself a document: an archive can decompress perfectly and
207
+ * still be a renamed photo bundle, a partial upload, or a file mislabeled with the wrong
208
+ * extension. Without this check a parser finds no content to walk and returns an empty AST,
209
+ * which a caller cannot distinguish from a document that genuinely has nothing in it. Every
210
+ * ZIP-backed format has one part it cannot be valid without, so its absence is a hard error.
211
+ *
212
+ * Pass the parser's own regex/predicate for the part rather than a fresh copy of the path, so
213
+ * this check and the code that later reads the part cannot drift apart.
214
+ *
215
+ * @param files - The entries extracted from the archive
216
+ * @param matcher - Predicate identifying the required part by its path within the archive
217
+ * @param config - Parser configuration, so the error is reported through the caller's handlers
218
+ * @param info - The document format and the human-readable part name, used in the message
219
+ * @returns The matching entry
220
+ * @throws {Error} A typed REQUIRED_PART_MISSING error when no entry matches
221
+ *
222
+ * @example
223
+ * ```typescript
224
+ * const document = findRequiredPart(files, p => !!p.match(documentFileRegex), config,
225
+ * { fileType: 'docx', part: 'word/document.xml' });
226
+ * ```
227
+ */
228
+ const findRequiredPart = (files, matcher, config, info) => {
229
+ const found = files.find(file => matcher(file.path));
230
+ if (!found)
231
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.REQUIRED_PART_MISSING, config, info);
232
+ return found;
233
+ };
234
+ exports.findRequiredPart = findRequiredPart;
235
+ /** The part naming an OOXML package's document type. */
236
+ const OOXML_CONTENT_TYPES_PATH = '[Content_Types].xml';
237
+ /** The part naming an ODF or EPUB package's document type. */
238
+ const ODF_MIMETYPE_PATH = 'mimetype';
239
+ /**
240
+ * Substrings of the main-part content type each OOXML format declares in
241
+ * `[Content_Types].xml`, matched as plain text because only this one value is needed.
242
+ */
243
+ const OOXML_MAIN_CONTENT_TYPES = [
244
+ ['wordprocessingml.document.main+xml', 'docx'],
245
+ ['spreadsheetml.sheet.main+xml', 'xlsx'],
246
+ ['presentationml.presentation.main+xml', 'pptx'],
247
+ ];
248
+ /** Exact `mimetype` entry contents for the packages that carry one. */
249
+ const PACKAGE_MIMETYPES = {
250
+ 'application/vnd.oasis.opendocument.text': 'odt',
251
+ 'application/vnd.oasis.opendocument.spreadsheet': 'ods',
252
+ 'application/vnd.oasis.opendocument.presentation': 'odp',
253
+ 'application/epub+zip': 'epub',
254
+ };
255
+ /** First two bytes of every ZIP local file header ("PK"). */
256
+ const ZIP_MAGIC_BYTES = [0x50, 0x4b];
257
+ /**
258
+ * Reporting is suppressed while sniffing: the input is not yet known to be a document, so a
259
+ * failure here is an inconclusive guess rather than something the caller did wrong. Without a
260
+ * config, `getOfficeError` would write these to the console.
261
+ */
262
+ const SILENT_DETECTION_CONFIG = { outputErrorToConsole: false };
263
+ /** Whether a buffer starts with the ZIP local file header signature. */
264
+ const looksLikeZip = (buffer) => buffer.length >= ZIP_MAGIC_BYTES.length && ZIP_MAGIC_BYTES.every((byte, i) => buffer[i] === byte);
265
+ /**
266
+ * How much a type sniff may inflate before giving up, in bytes.
267
+ *
268
+ * The two parts read here name the format and nothing else, so they are tiny in any real
269
+ * document: a few hundred bytes of `mimetype`, a few kilobytes of `[Content_Types].xml`. A
270
+ * crafted archive could declare them as hundreds of megabytes, and since the document is
271
+ * inflated again during the parse that follows, honouring the full decompression budget here
272
+ * would let a single call spend it twice. This cap keeps sniffing cheap; an archive that
273
+ * exceeds it is simply reported as unidentified.
274
+ */
275
+ const MAX_DETECTION_INFLATED_BYTES = 4 * 1024 * 1024;
276
+ /**
277
+ * Resolves which office format a ZIP archive actually holds, by reading the part that names it.
278
+ *
279
+ * This exists because magic-byte sniffing is a heuristic that gives up. `file-type` identifies an
280
+ * OOXML package by parsing `[Content_Types].xml`, but it walks the archive under fixed budgets:
281
+ * at most 1024 entries, and (for entries whose sizes are deferred to a trailing data descriptor,
282
+ * general-purpose flag bit 3) about 1 MiB of scanning to locate those descriptors. An archive
283
+ * that puts enough data before `[Content_Types].xml` to exhaust either budget is reported as a
284
+ * generic `zip`, which previously surfaced to the caller as "add support for zip files" for a
285
+ * perfectly valid document. Both layouts occur in the wild: streaming ZIP writers set bit 3, and
286
+ * a media-heavy deck can hold more than 1024 parts.
287
+ *
288
+ * We already ship a ZIP reader that has neither limitation, so rather than guessing from the
289
+ * first bytes this opens the archive and reads the declaration directly. It is deliberately the
290
+ * fallback rather than the primary check, since the byte-level sniff is far cheaper and settles
291
+ * every non-ZIP format.
292
+ *
293
+ * @param zipInput - The candidate archive
294
+ * @param limits - Decompression limits, so sniffing an untrusted file stays bounded
295
+ * @returns The format the archive declares, or `undefined` if it declares none or cannot be read
296
+ *
297
+ * @example
298
+ * ```typescript
299
+ * // A presentation whose [Content_Types].xml sits behind 2 MiB of streamed entries
300
+ * await detectOfficeTypeFromZip(buffer, limits); // -> 'pptx'
301
+ * ```
302
+ */
303
+ const detectOfficeTypeFromZip = async (zipInput, limits) => {
304
+ if (!looksLikeZip(zipInput))
305
+ return undefined;
306
+ let files;
307
+ try {
308
+ files = await (0, exports.extractFiles)(zipInput, name => name === OOXML_CONTENT_TYPES_PATH || name === ODF_MIMETYPE_PATH,
309
+ // Never inflate more for a sniff than the caller already allows for the parse, and
310
+ // never more than a sniff could legitimately need.
311
+ {
312
+ ...limits,
313
+ maxUncompressedBytes: Math.min(limits?.maxUncompressedBytes ?? MAX_DETECTION_INFLATED_BYTES, MAX_DETECTION_INFLATED_BYTES),
314
+ }, SILENT_DETECTION_CONFIG);
315
+ }
316
+ catch {
317
+ // Unreadable, truncated, or not an archive at all. The caller keeps whatever the
318
+ // byte-level sniff decided, and the parser it dispatches to reports the real problem.
319
+ return undefined;
320
+ }
321
+ // ODF and EPUB state their type outright, so prefer that over inspecting OOXML parts.
322
+ const mimetypeEntry = files.find(file => file.path === ODF_MIMETYPE_PATH);
323
+ if (mimetypeEntry) {
324
+ const declared = PACKAGE_MIMETYPES[mimetypeEntry.content.toString().trim()];
325
+ if (declared)
326
+ return declared;
327
+ }
328
+ const contentTypesEntry = files.find(file => file.path === OOXML_CONTENT_TYPES_PATH);
329
+ if (contentTypesEntry) {
330
+ const contentTypes = contentTypesEntry.content.toString();
331
+ const match = OOXML_MAIN_CONTENT_TYPES.find(([marker]) => contentTypes.includes(marker));
332
+ if (match)
333
+ return match[1];
334
+ }
335
+ return undefined;
336
+ };
337
+ exports.detectOfficeTypeFromZip = detectOfficeTypeFromZip;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "7.5.0",
3
+ "version": "7.5.1",
4
4
  "description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html, .epub) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, PDF, EPUB, and RAG-focused chunks.",
5
5
  "funding": "https://github.com/sponsors/harshankur",
6
6
  "main": "dist/index.js",