officeparser 7.2.1 → 7.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -751,6 +751,7 @@ Pass as the second argument to `parseOffice(file, config)`.
751
751
  | `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal cross-references from AST |
752
752
  | `fileType` | `SupportedFileType \| null` | `null` | **Required for text-based binary data** (`'md'`, `'html'`, `'csv'`) as these lack magic bytes. |
753
753
  | `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
754
+ | `decompressionLimits` | `DecompressionLimits` | `{ maxUncompressedBytes: 512MB, maxZipEntries: 10000 }` | **New**: Limits applied during ZIP extraction to protect against excessive memory and resource usage |
754
755
  | `pdfWorkerSrc` | `string` | CDN (jsDelivr) | Path/URL to `pdf.worker.min.mjs` (required in browser) |
755
756
  | `onWarning` | `(issue: OfficeIssue) => void` | — | Callback for non-fatal parsing issues |
756
757
  | `abortSignal` | `AbortSignal \| null` | `null` | Optional signal to cancel parsing (rejects with AbortError) |
@@ -1006,12 +1007,17 @@ await officeParser.terminateOcr(); // immediate exit
1006
1007
 
1007
1008
  ## Browser Usage
1008
1009
 
1009
- Two bundles are available in the `dist/` directory:
1010
+ Four bundles are available in the `dist/` directory:
1010
1011
 
1011
- | Bundle | Usage |
1012
- |--------|-------|
1013
- | `officeparser.browser.mjs` | ESM, use with `import` statements or modern bundlers (Vite, Webpack, Next.js) |
1014
- | `officeparser.browser.iife.js` | IIFE, use with a `<script>` tag; exposes the global `officeParser` object |
1012
+ | Bundle | Type | Description |
1013
+ |--------|------|-------------|
1014
+ | `officeparser.browser.mjs` | ESM | Standard ESM bundle for modern bundlers (Vite, Webpack, Next.js). |
1015
+ | `officeparser.browser.iife.js` | IIFE | Standard UMD bundle for direct `<script>` inclusion (exposes global `officeParser`). |
1016
+ | `officeparser.browser.slim.mjs` | ESM | Slim ESM bundle with Tesseract.js (OCR) stubbed out and remote CDN URLs removed. |
1017
+ | `officeparser.browser.slim.iife.js` | IIFE | Slim UMD bundle with Tesseract.js (OCR) stubbed out and remote CDN URLs removed. |
1018
+
1019
+ ### Manifest V3 & Extension Compliance (Slim Bundles)
1020
+ For strict browser environments like **Chrome/Edge Manifest V3 extensions**, remotely hosted code is forbidden. Use the **slim** bundles (`officeparser.browser.slim.mjs` or `officeparser.browser.slim.iife.js`) as they do not include default remote CDN urls or the Tesseract OCR engine.
1015
1021
 
1016
1022
  ### ESM (Vite / Webpack / Next.js)
1017
1023
 
@@ -1054,12 +1060,12 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer);
1054
1060
 
1055
1061
  // Or specify your own:
1056
1062
  const ast = await officeParser.parseOffice(pdfArrayBuffer, {
1057
- pdfWorkerSrc: 'https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs'
1063
+ pdfWorkerSrc: 'https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs'
1058
1064
  });
1059
1065
  ```
1060
1066
 
1061
1067
  > [!NOTE]
1062
- > The `pdfjs-dist` worker version must match the version bundled with `officeparser` (currently **5.6.205**).
1068
+ > The `pdfjs-dist` worker version must match the version bundled with `officeparser` (currently **6.1.200**).
1063
1069
 
1064
1070
  ---
1065
1071
 
@@ -1068,7 +1074,7 @@ const ast = await officeParser.parseOffice(pdfArrayBuffer, {
1068
1074
  | Symptom | Fix |
1069
1075
  |---------|-----|
1070
1076
  | Node.js process stays alive after finishing | Call `await officeParser.terminateOcr()` at end of script when OCR was used |
1071
- | `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `5.6.205` |
1077
+ | `"Worker not found"` in browser for PDF | Verify `pdfWorkerSrc` points to `pdf.worker.min.mjs` matching version `6.1.200` |
1072
1078
  | Low OCR accuracy | Verify `ocrConfig.language` matches the document language; quality depends on image resolution |
1073
1079
  | Out of memory on large Excel files | Call `ast.toText()` early and discard the AST object to allow garbage collection |
1074
1080
  | `md`/`html`/`csv` buffer not detected | Add `fileType: 'md'` (or `'html'`, `'csv'`) to config (these formats have no magic bytes) |
package/dist/defaults.js CHANGED
@@ -1,8 +1,8 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.DEFAULT_GENERATOR_CONFIG = exports.DEFAULT_SEMANTIC_CHUNKING_CONFIG = exports.DEFAULT_DOCUMENT_STRUCTURE_CHUNKING_CONFIG = exports.DEFAULT_FIXED_SIZE_CHUNKING_CONFIG = exports.DEFAULT_OFFICE_PARSER_CONFIG = exports.DEFAULT_ABBREVIATIONS = exports.DEFAULT_SENTENCE_BOUNDARY_REGEX = void 0;
4
- const PDFJS_VERSION = '5.6.205';
5
- const DEFAULT_PDF_WORKER_SRC = `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
4
+ const PDFJS_VERSION = '6.1.200';
5
+ const DEFAULT_PDF_WORKER_SRC = typeof __SLIM__ !== 'undefined' && __SLIM__ ? '' : `https://cdn.jsdelivr.net/npm/pdfjs-dist@${PDFJS_VERSION}/build/pdf.worker.min.mjs`;
6
6
  /**
7
7
  * The default regex used for identifying sentence boundaries.
8
8
  * When this default is used, the generator employs a high-fidelity "robust"
@@ -59,13 +59,17 @@ exports.DEFAULT_OFFICE_PARSER_CONFIG = {
59
59
  ignoreInternalLinks: false,
60
60
  fileType: null,
61
61
  csvDelimiter: ',',
62
+ decompressionLimits: {
63
+ maxUncompressedBytes: 512 * 1024 * 1024,
64
+ maxZipEntries: 10000,
65
+ },
62
66
  };
63
67
  /**
64
68
  * Default configuration for HTML generation.
65
69
  */
66
70
  const DEFAULT_HTML_GENERATOR_CONFIG = {
67
71
  standalone: true,
68
- chartJsSrc: 'https://cdn.jsdelivr.net/npm/chart.js',
72
+ chartJsSrc: typeof __SLIM__ !== 'undefined' && __SLIM__ ? '' : 'https://cdn.jsdelivr.net/npm/chart.js',
69
73
  containerWidth: 'auto',
70
74
  customCss: '',
71
75
  injections: {
@@ -644,7 +644,7 @@ class ChunkingGenerator extends BaseGenerator_js_1.BaseGenerator {
644
644
  let timerId;
645
645
  const timeoutPromise = new Promise((_, reject) => {
646
646
  timerId = setTimeout(() => {
647
- reject(new Error(`Embedding call timed out after ${timeoutMs}ms`));
647
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT, this.config, timeoutMs));
648
648
  }, timeoutMs);
649
649
  });
650
650
  return Promise.race([call, timeoutPromise]).finally(() => {
@@ -469,7 +469,7 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
469
469
  src = `data:${attachment.mimeType || 'image/png'};base64,${attachment.data}`;
470
470
  }
471
471
  }
472
- const img = `<img src="${src}" alt="${this.escape(node.text || meta?.altText || '')}"${className}${mappedAttrs}${styleAttr}>`;
472
+ const img = `<img src="${this.escape(src)}" alt="${this.escape(node.text || meta?.altText || '')}"${className}${mappedAttrs}${styleAttr}>`;
473
473
  const content = this.config.includeFormatting ? `<div class="image-container">${img}<div class="caption">${this.escape(attachmentName || '')}</div></div>` : img;
474
474
  return `${extraAnchors}<div${idAttr}>${content}</div>`;
475
475
  }
@@ -34,7 +34,15 @@ export declare enum OfficeErrorType {
34
34
  /** Semantic chunking strategy is selected but no embedding function is provided */
35
35
  MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
36
36
  /** The operation was aborted */
37
- OPERATION_ABORTED = "OPERATION_ABORTED"
37
+ OPERATION_ABORTED = "OPERATION_ABORTED",
38
+ /** ZIP entry count exceeds limit */
39
+ ZIP_ENTRY_COUNT_LIMIT_EXCEEDED = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED",
40
+ /** ZIP entry missing a valid declared size */
41
+ ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
42
+ /** ZIP uncompressed size limit exceeded */
43
+ ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
44
+ /** Embedding call timed out */
45
+ EMBEDDING_TIMEOUT = "EMBEDDING_TIMEOUT"
38
46
  }
39
47
  /**
40
48
  * Standard warning types for OfficeParser.
@@ -311,7 +319,7 @@ export interface OfficeParserConfig {
311
319
  * The URL/path to the PDF.js worker script.
312
320
  *
313
321
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
314
- * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
322
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@6.1.200/build/pdf.worker.min.mjs`.
315
323
  * You can override this with your own local path or a different CDN link.
316
324
  */
317
325
  pdfWorkerSrc?: string;
@@ -349,6 +357,28 @@ export interface OfficeParserConfig {
349
357
  * Defaults to ',' but can be overridden (e.g., ';', '\t').
350
358
  */
351
359
  csvDelimiter?: string;
360
+ /**
361
+ * Limits and checks applied during ZIP extraction to protect against excessive
362
+ * memory and resource usage.
363
+ */
364
+ decompressionLimits?: DecompressionLimits;
365
+ }
366
+ /**
367
+ * Limits applied to ZIP archive decompression.
368
+ */
369
+ export interface DecompressionLimits {
370
+ /**
371
+ * Maximum allowed total uncompressed size (in bytes) of files extracted from a ZIP archive.
372
+ * Applies to OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) formats.
373
+ * Default is 536870912 (512 MB).
374
+ */
375
+ maxUncompressedBytes?: number;
376
+ /**
377
+ * Maximum allowed number of entries (files and directories) in a ZIP archive.
378
+ * Applies to OOXML (DOCX, XLSX, PPTX) and ODF (ODT, ODP, ODS) formats.
379
+ * Default is 10000.
380
+ */
381
+ maxZipEntries?: number;
352
382
  }
353
383
  /**
354
384
  * Represents a single issue (warning, error, or info) generated during document processing.