officeparser 7.5.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -0
- package/dist/OfficeParser.js +54 -6
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +153 -153
- package/dist/officeparser.browser.mjs +147 -147
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +176 -176
- package/dist/officeparser.browser.slim.mjs +176 -176
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/OpenOfficeParser.js +17 -3
- package/dist/parsers/PowerPointParser.js +21 -5
- package/dist/parsers/WordParser.js +12 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -206,6 +206,15 @@ const ast = await officeParser.parseOffice(buffer);
|
|
|
206
206
|
> const ast = await officeParser.parseOffice(markdownBuffer, { fileType: 'md' });
|
|
207
207
|
> ```
|
|
208
208
|
|
|
209
|
+
> [!NOTE]
|
|
210
|
+
> **ZIP-backed formats are identified from inside the archive.** DOCX, XLSX, PPTX, ODT, ODS, ODP
|
|
211
|
+
> and EPUB are all ZIP files, and telling them apart from the first bytes alone is unreliable for
|
|
212
|
+
> archives written by streaming producers or holding very many parts. When the byte signature is
|
|
213
|
+
> inconclusive, the archive is opened and the format is read from its own declaration
|
|
214
|
+
> (`[Content_Types].xml`, or the `mimetype` entry), so these parse from a buffer without a hint.
|
|
215
|
+
> Supplying `fileType` remains the fastest and most certain route: it decides which parser runs,
|
|
216
|
+
> and for these formats no archive inspection is done at all.
|
|
217
|
+
|
|
209
218
|
### Cancellation with AbortSignal
|
|
210
219
|
|
|
211
220
|
You can pass a standard `AbortSignal` (e.g. from an `AbortController`) to cancel an active parse operation. This is especially useful for setting request-level timeouts or canceling long-running parses (like large PDFs with OCR).
|
|
@@ -533,6 +542,34 @@ interface OfficeIssue {
|
|
|
533
542
|
}
|
|
534
543
|
```
|
|
535
544
|
|
|
545
|
+
Thrown errors carry the same object on `error.officeIssue`, so a failed parse is identified by
|
|
546
|
+
the same stable `code` you would branch on for a warning, rather than by matching message text:
|
|
547
|
+
|
|
548
|
+
```js
|
|
549
|
+
try {
|
|
550
|
+
const ast = await officeParser.parseOffice(buffer, { fileType: 'docx' });
|
|
551
|
+
} catch (err) {
|
|
552
|
+
switch (err.officeIssue?.code) {
|
|
553
|
+
case 'ZIP_NO_ENTRIES_FOUND': // not a ZIP archive at all
|
|
554
|
+
case 'ZIP_TRUNCATED': // cut off in transfer, entries incomplete
|
|
555
|
+
case 'REQUIRED_PART_MISSING': // readable ZIP, but not the format it claims
|
|
556
|
+
console.error('Unusable file:', err.officeIssue.message);
|
|
557
|
+
break;
|
|
558
|
+
default:
|
|
559
|
+
throw err;
|
|
560
|
+
}
|
|
561
|
+
}
|
|
562
|
+
```
|
|
563
|
+
|
|
564
|
+
> [!IMPORTANT]
|
|
565
|
+
> **A corrupt file throws; it does not parse as an empty document.** If an archive is not
|
|
566
|
+
> readable, is truncated, or is missing the part its format requires (`word/document.xml`,
|
|
567
|
+
> `xl/workbook.xml`, `ppt/presentation.xml`, ODF `content.xml`, the EPUB OPF), parsing rejects
|
|
568
|
+
> with one of the codes above. An empty result therefore means the document really is empty.
|
|
569
|
+
> Files that are legitimately empty still parse, and say so through `onWarning` /
|
|
570
|
+
> `ast.warnings` (`NO_WORKSHEETS_FOUND` for a chartsheet-only workbook, `NO_SLIDES_FOUND` for a
|
|
571
|
+
> presentation with no slides).
|
|
572
|
+
|
|
536
573
|
---
|
|
537
574
|
|
|
538
575
|
## Deep Dive: Document Components
|
package/dist/OfficeParser.js
CHANGED
|
@@ -55,6 +55,32 @@ const envUtils_js_1 = require("./utils/envUtils.js");
|
|
|
55
55
|
const errorUtils_js_1 = require("./utils/errorUtils.js");
|
|
56
56
|
const moduleLoader_js_1 = require("./utils/moduleLoader.js");
|
|
57
57
|
const ocrUtils_js_1 = require("./utils/ocrUtils.js");
|
|
58
|
+
const zipUtils_js_1 = require("./utils/zipUtils.js");
|
|
59
|
+
/** What magic-byte sniffing reports for an archive it could not identify further. */
|
|
60
|
+
const GENERIC_ZIP_EXTENSION = 'zip';
|
|
61
|
+
/** The formats that are ZIP archives, and so cannot be contradicted by a bare `zip` result. */
|
|
62
|
+
const ZIP_BACKED_FILE_TYPES = new Set(['docx', 'xlsx', 'pptx', 'odt', 'ods', 'odp', 'epub']);
|
|
63
|
+
/**
|
|
64
|
+
* Upgrades a magic-byte result of `zip` (or none at all) into the specific office format the
|
|
65
|
+
* archive declares, by reading that declaration from inside the archive.
|
|
66
|
+
*
|
|
67
|
+
* Byte sniffing identifies an OOXML package by parsing `[Content_Types].xml`, but it walks the
|
|
68
|
+
* archive under fixed budgets and reports a plain `zip` when it runs out before finding that
|
|
69
|
+
* part. Since `zip` is not a format this library parses, a valid document then failed as an
|
|
70
|
+
* unsupported file type. Our own reader has no such budget, so it settles the question whenever
|
|
71
|
+
* sniffing is inconclusive.
|
|
72
|
+
*
|
|
73
|
+
* @param detected - What magic-byte sniffing reported, if anything
|
|
74
|
+
* @param buffer - The file content
|
|
75
|
+
* @param config - Resolved parser configuration, for its decompression limits
|
|
76
|
+
* @returns The resolved type, the original detection when nothing better is found, or undefined
|
|
77
|
+
*/
|
|
78
|
+
const resolveZipBackedType = async (detected, buffer, config) => {
|
|
79
|
+
if (detected && detected !== GENERIC_ZIP_EXTENSION)
|
|
80
|
+
return detected;
|
|
81
|
+
const resolved = await (0, zipUtils_js_1.detectOfficeTypeFromZip)(buffer, config.decompressionLimits ?? {});
|
|
82
|
+
return resolved ?? detected;
|
|
83
|
+
};
|
|
58
84
|
/**
|
|
59
85
|
* Main parser class providing office document parsing functionality.
|
|
60
86
|
*
|
|
@@ -168,15 +194,16 @@ class OfficeParser {
|
|
|
168
194
|
// This matches v6 behavior and prevents crashes in older Node environments
|
|
169
195
|
// where file-type 22.x might be incompatible.
|
|
170
196
|
if (buffer.length > 0 && !ext) {
|
|
197
|
+
let detected;
|
|
171
198
|
try {
|
|
172
199
|
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
173
200
|
const type = await fileTypeFromBuffer(buffer);
|
|
174
201
|
if (type) {
|
|
175
|
-
|
|
202
|
+
detected = type.ext;
|
|
176
203
|
}
|
|
177
204
|
else {
|
|
178
205
|
// If no extension could be detected and none was provided,
|
|
179
|
-
// it might be a text-based format (csv, md, html) which
|
|
206
|
+
// it might be a text-based format (csv, md, html) which
|
|
180
207
|
// lack magic bytes. We'll let the switch default handle it.
|
|
181
208
|
}
|
|
182
209
|
}
|
|
@@ -184,16 +211,28 @@ class OfficeParser {
|
|
|
184
211
|
// Log warning but don't crash; the switch below will handle unsupported/missing ext
|
|
185
212
|
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
|
|
186
213
|
}
|
|
214
|
+
ext = await resolveZipBackedType(detected, buffer, internalConfig) ?? '';
|
|
187
215
|
}
|
|
188
216
|
else if (buffer.length > 0 && ext) {
|
|
189
|
-
// If extension is known, we can optionally verify it, but we wrap it
|
|
217
|
+
// If extension is known, we can optionally verify it, but we wrap it
|
|
190
218
|
// in a try-catch to avoid breaking Node 18 if file-type fails to load.
|
|
191
219
|
try {
|
|
192
220
|
const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
|
|
193
221
|
const type = await fileTypeFromBuffer(buffer);
|
|
194
|
-
|
|
222
|
+
// A bare `zip` cannot contradict a caller who already said "this is a
|
|
223
|
+
// docx", so there is nothing a closer look could add. Skipping it keeps an
|
|
224
|
+
// explicit fileType the cheapest route, rather than making it pay for an
|
|
225
|
+
// archive scan that exists only to decide whether to warn.
|
|
226
|
+
const worthResolving = !(type?.ext === GENERIC_ZIP_EXTENSION && ZIP_BACKED_FILE_TYPES.has(ext.toLowerCase()));
|
|
227
|
+
const detected = worthResolving
|
|
228
|
+
? await resolveZipBackedType(type?.ext, buffer, internalConfig)
|
|
229
|
+
: type?.ext;
|
|
230
|
+
// A bare `zip` says only that the bytes are an archive, which every format
|
|
231
|
+
// on this path already is. Reporting it as a mismatch against the caller's
|
|
232
|
+
// own extension is noise, so only a resolved format is worth comparing.
|
|
233
|
+
if (detected && detected !== GENERIC_ZIP_EXTENSION && detected.toLowerCase() !== ext.toLowerCase()) {
|
|
195
234
|
// Mismatch found between authoritative extension and detected content
|
|
196
|
-
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected
|
|
235
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected, expected: ext });
|
|
197
236
|
}
|
|
198
237
|
}
|
|
199
238
|
catch (error) {
|
|
@@ -218,7 +257,16 @@ class OfficeParser {
|
|
|
218
257
|
case 'odt':
|
|
219
258
|
case 'odp':
|
|
220
259
|
case 'ods':
|
|
221
|
-
|
|
260
|
+
// The three ODF types share one parser, which needs to know which of them
|
|
261
|
+
// it is looking at. It normally reads that from the archive's mimetype
|
|
262
|
+
// entry; passing the resolved type along gives it something accurate to
|
|
263
|
+
// fall back on when that entry is missing.
|
|
264
|
+
//
|
|
265
|
+
// Overridden on a copy rather than on internalConfig: resolveParserConfig
|
|
266
|
+
// returns an already-complete config by reference, so writing to it would
|
|
267
|
+
// pin the caller's own object to this file's type and misroute every later
|
|
268
|
+
// parse that reused it.
|
|
269
|
+
result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, { ...internalConfig, fileType: ext.toLowerCase() });
|
|
222
270
|
break;
|
|
223
271
|
case 'pdf':
|
|
224
272
|
result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
|
package/dist/index.d.ts
CHANGED
|
@@ -51,10 +51,10 @@
|
|
|
51
51
|
import { OfficeParser } from './OfficeParser.js';
|
|
52
52
|
import { OfficeGenerator } from './OfficeGenerator.js';
|
|
53
53
|
import { OfficeConverter } from './OfficeConverter.js';
|
|
54
|
-
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, ConversionResult } from './types.js';
|
|
54
|
+
import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult } from './types.js';
|
|
55
55
|
declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
56
56
|
declare const terminateOcr: typeof OfficeParser.terminateOcr;
|
|
57
57
|
declare const convert: typeof OfficeConverter.convert;
|
|
58
58
|
declare const generate: typeof OfficeGenerator.generate;
|
|
59
|
-
export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, ConversionResult, };
|
|
59
|
+
export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, };
|
|
60
60
|
export default OfficeParser;
|
|
@@ -41,6 +41,12 @@ export declare enum OfficeErrorType {
|
|
|
41
41
|
ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
|
|
42
42
|
/** ZIP uncompressed size limit exceeded */
|
|
43
43
|
ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
|
|
44
|
+
/** ZIP data yielded no readable entries (corrupt, truncated, or not a ZIP archive) */
|
|
45
|
+
ZIP_NO_ENTRIES_FOUND = "ZIP_NO_ENTRIES_FOUND",
|
|
46
|
+
/** ZIP data is truncated: the End of Central Directory record is absent */
|
|
47
|
+
ZIP_TRUNCATED = "ZIP_TRUNCATED",
|
|
48
|
+
/** A readable ZIP archive is missing the part its document format requires */
|
|
49
|
+
REQUIRED_PART_MISSING = "REQUIRED_PART_MISSING",
|
|
44
50
|
/** Document element/structure nesting exceeded the safe recursion depth */
|
|
45
51
|
MAX_NESTING_DEPTH_EXCEEDED = "MAX_NESTING_DEPTH_EXCEEDED",
|
|
46
52
|
/** Embedding call timed out */
|
|
@@ -90,7 +96,11 @@ export declare enum OfficeWarningType {
|
|
|
90
96
|
/** A metadata override could not be represented in the destination format's vocabulary */
|
|
91
97
|
METADATA_NOT_REPRESENTABLE = "METADATA_NOT_REPRESENTABLE",
|
|
92
98
|
/** A styleMap output.tag was not an allowed element name and was ignored */
|
|
93
|
-
INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG"
|
|
99
|
+
INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG",
|
|
100
|
+
/** A workbook archive contains no worksheet parts (chartsheet-only workbooks are legitimate) */
|
|
101
|
+
NO_WORKSHEETS_FOUND = "NO_WORKSHEETS_FOUND",
|
|
102
|
+
/** A presentation archive contains no slides (a zero-slide presentation is legitimate) */
|
|
103
|
+
NO_SLIDES_FOUND = "NO_SLIDES_FOUND"
|
|
94
104
|
}
|
|
95
105
|
/**
|
|
96
106
|
* Consolidated timeout settings for OCR operations.
|
|
@@ -463,6 +473,29 @@ export interface OfficeIssue {
|
|
|
463
473
|
/** Optional additional context or original error object. */
|
|
464
474
|
details?: any;
|
|
465
475
|
}
|
|
476
|
+
/**
|
|
477
|
+
* An Error thrown by OfficeParser, carrying the structured issue that produced it.
|
|
478
|
+
*
|
|
479
|
+
* Catching code can branch on `error.officeIssue.code`, the same stable enum used for warnings,
|
|
480
|
+
* instead of matching against message text. Errors that originate outside the library (and
|
|
481
|
+
* `AbortError`, which is deliberately re-thrown untouched so cancellation stays detectable via
|
|
482
|
+
* `error.name`) do not carry this property, hence the optional marker.
|
|
483
|
+
*
|
|
484
|
+
* @example
|
|
485
|
+
* ```typescript
|
|
486
|
+
* try {
|
|
487
|
+
* await parseOffice(buffer, { fileType: 'docx' });
|
|
488
|
+
* } catch (err) {
|
|
489
|
+
* if ((err as OfficeError).officeIssue?.code === OfficeErrorType.REQUIRED_PART_MISSING) {
|
|
490
|
+
* // the archive is readable, but it is not a docx
|
|
491
|
+
* }
|
|
492
|
+
* }
|
|
493
|
+
* ```
|
|
494
|
+
*/
|
|
495
|
+
export interface OfficeError extends Error {
|
|
496
|
+
/** The structured issue this error was created from. */
|
|
497
|
+
officeIssue?: OfficeIssue;
|
|
498
|
+
}
|
|
466
499
|
/**
|
|
467
500
|
* The result of a document conversion operation.
|
|
468
501
|
*/
|