officeparser 7.2.0 → 7.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/types.js CHANGED
@@ -9,6 +9,8 @@ var OfficeErrorType;
9
9
  (function (OfficeErrorType) {
10
10
  /** Unsupported file extension */
11
11
  OfficeErrorType["EXTENSION_UNSUPPORTED"] = "EXTENSION_UNSUPPORTED";
12
+ /** Unsupported output generator format */
13
+ OfficeErrorType["FORMAT_UNSUPPORTED"] = "FORMAT_UNSUPPORTED";
12
14
  /** File appears to be corrupted or malformed */
13
15
  OfficeErrorType["FILE_CORRUPTED"] = "FILE_CORRUPTED";
14
16
  /** File could not be found at the specified path */
@@ -35,6 +37,14 @@ var OfficeErrorType;
35
37
  OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
36
38
  /** The operation was aborted */
37
39
  OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
40
+ /** ZIP entry count exceeds limit */
41
+ OfficeErrorType["ZIP_ENTRY_COUNT_LIMIT_EXCEEDED"] = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED";
42
+ /** ZIP entry missing a valid declared size */
43
+ OfficeErrorType["ZIP_ENTRY_INVALID_SIZE"] = "ZIP_ENTRY_INVALID_SIZE";
44
+ /** ZIP uncompressed size limit exceeded */
45
+ OfficeErrorType["ZIP_SIZE_LIMIT_EXCEEDED"] = "ZIP_SIZE_LIMIT_EXCEEDED";
46
+ /** Embedding call timed out */
47
+ OfficeErrorType["EMBEDDING_TIMEOUT"] = "EMBEDDING_TIMEOUT";
38
48
  })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
39
49
  /**
40
50
  * Standard warning types for OfficeParser.
@@ -57,6 +57,12 @@ function isFullParserConfig(config) {
57
57
  */
58
58
  function resolveParserConfig(userConfig) {
59
59
  if (isFullParserConfig(userConfig)) {
60
+ if (!userConfig.decompressionLimits) {
61
+ userConfig.decompressionLimits = {
62
+ maxUncompressedBytes: 512 * 1024 * 1024,
63
+ maxZipEntries: 10000,
64
+ };
65
+ }
60
66
  return userConfig;
61
67
  }
62
68
  // 1. Start with full defaults (deep cloned)
@@ -65,9 +71,15 @@ function resolveParserConfig(userConfig) {
65
71
  return config;
66
72
  }
67
73
  // 2. Merge user config
68
- // We handle ocrConfig specially to avoid shallow-overwriting the whole object
69
- const { ocrConfig, ...rest } = userConfig;
74
+ // We handle ocrConfig and decompressionLimits specially to avoid shallow-overwriting the whole objects
75
+ const { ocrConfig, decompressionLimits, ...rest } = userConfig;
70
76
  Object.assign(config, rest);
77
+ if (decompressionLimits) {
78
+ config.decompressionLimits = {
79
+ ...config.decompressionLimits,
80
+ ...decompressionLimits,
81
+ };
82
+ }
71
83
  if (ocrConfig) {
72
84
  const { timeout, ...ocrRest } = ocrConfig;
73
85
  config.ocrConfig = {
@@ -17,6 +17,7 @@ const ERRORHEADER = "[OfficeParser]: ";
17
17
  */
18
18
  const ERROR_MESSAGES = {
19
19
  [types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
20
+ [types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks.`,
20
21
  [types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
21
22
  [types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
22
23
  [types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
@@ -29,7 +30,11 @@ const ERROR_MESSAGES = {
29
30
  [types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
30
31
  [types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
31
32
  [types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`,
32
- [types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`
33
+ [types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`,
34
+ [types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
35
+ [types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
36
+ [types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
37
+ [types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
33
38
  };
34
39
  /**
35
40
  * Lookup table for warning messages.
@@ -19,7 +19,7 @@ async function loadNodeEsmModule(specifier) {
19
19
  // This is especially important in Node 18 for sub-paths of packages.
20
20
  try {
21
21
  const { pathToFileURL } = await import('url');
22
- // @ts-ignore - require.resolve is available in Node.js
22
+ // @ts-ignore - require.resolve is available in Node.js CJS context
23
23
  const absolutePath = require.resolve(specifier);
24
24
  const fileUrl = pathToFileURL(absolutePath).href;
25
25
  return import(fileUrl);
@@ -29,6 +29,16 @@ async function loadNodeEsmModule(specifier) {
29
29
  return import(specifier);
30
30
  }
31
31
  }
32
+ /**
33
+ * Returns true if require.resolve is available in this runtime context.
34
+ * It is NOT available in native ESM (e.g. the .mjs wrapper) or browser environments.
35
+ * Checking this guards against a ReferenceError that would silently trigger
36
+ * the bundled fallback path for non-bundled ESM consumers.
37
+ */
38
+ function isRequireAvailable() {
39
+ // @ts-ignore - require may not exist in ESM context
40
+ return typeof require !== 'undefined' && typeof require.resolve === 'function';
41
+ }
32
42
  /**
33
43
  * Specialized loader for file-type
34
44
  */
@@ -36,10 +46,19 @@ async function loadFileType() {
36
46
  if (!envUtils_js_1.isBrowser) {
37
47
  // Ensure environment polyfills for Node.js 18 support
38
48
  (0, envUtils_js_1.ensureEnvPolyfills)();
39
- // Node.js path: Use dynamic import wrapper for CJS compatibility
40
- return loadNodeEsmModule('file-type');
49
+ if (isRequireAvailable()) {
50
+ // Check if node_modules is available at runtime.
51
+ // @ts-ignore - require.resolve is available in Node.js CJS context
52
+ require.resolve(String('file-type'));
53
+ // node_modules present: use the path-resolved loader for Node 18 ESM compatibility
54
+ return loadNodeEsmModule('file-type');
55
+ }
41
56
  }
42
- // Browser path: standard dynamic import() is handled by bundlers (e.g. esbuild/Vite)
57
+ // Covers three cases with one return:
58
+ // 1. Node.js standalone/bundled (SEA or bundled CJS): bundler inlines the module at build time.
59
+ // 2. Node.js native ESM (no require available): runtime resolves the bare specifier.
60
+ // 3. Browser: bundler (esbuild/Vite) handles the static-looking dynamic import().
61
+ // Note: bypasses the Node 18 sub-path ESM fix in loadNodeEsmModule, but bundlers handle it.
43
62
  return import('file-type');
44
63
  }
45
64
  /**
@@ -49,14 +68,39 @@ async function loadPdfJs() {
49
68
  if (!envUtils_js_1.isBrowser) {
50
69
  // Ensure environment polyfills for Node.js 18 support
51
70
  (0, envUtils_js_1.ensureEnvPolyfills)();
52
- // Node.js environment: require legacy build for stability with ESM-only main
53
- try {
54
- return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
55
- }
56
- catch {
57
- return await loadNodeEsmModule('pdfjs-dist');
71
+ if (isRequireAvailable()) {
72
+ // @ts-ignore - require.resolve is available in Node.js CJS context
73
+ const pkgExists = (() => { try {
74
+ require.resolve(String('pdfjs-dist'));
75
+ return true;
76
+ }
77
+ catch {
78
+ return false;
79
+ } })();
80
+ if (pkgExists) {
81
+ // node_modules present: try the legacy build path first for stability with
82
+ // ESM-only main, then fall back to the package root.
83
+ // Errors from these loaders are NOT swallowed into the bundled-fallback path.
84
+ try {
85
+ return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
86
+ }
87
+ catch {
88
+ return await loadNodeEsmModule('pdfjs-dist');
89
+ }
90
+ }
58
91
  }
92
+ // Standalone/bundled fallback for Node.js (SEA, bundled CJS without node_modules, or native ESM context):
93
+ // Load both PDF.js and its worker, and register the worker on globalThis to enable the offline fake-worker.
94
+ // @ts-ignore - Ignore type check for local .mjs files in node_modules/packaging
95
+ const [pdfjs, pdfjsWorker] = await Promise.all([
96
+ // @ts-ignore - mjs imports may not have types
97
+ import('pdfjs-dist/legacy/build/pdf.mjs'),
98
+ // @ts-ignore - mjs imports may not have types
99
+ import('pdfjs-dist/legacy/build/pdf.worker.mjs')
100
+ ]);
101
+ globalThis.pdfjsWorker = pdfjsWorker;
102
+ return pdfjs;
59
103
  }
60
- // Browser environment: esbuild handles standard static-looking dynamic import()
104
+ // Browser environment: bundler (esbuild/Vite) handles the standard dynamic import()
61
105
  return import('pdfjs-dist');
62
106
  }
@@ -13,6 +13,7 @@
13
13
  *
14
14
  * @module zipUtils
15
15
  */
16
+ import { DecompressionLimits } from '../types.js';
16
17
  /**
17
18
  * Represents a file extracted from a ZIP archive.
18
19
  * Contains the file's path within the archive and its content as a Buffer.
@@ -69,5 +70,5 @@ interface ZipFileContent {
69
70
  *
70
71
  * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
71
72
  */
72
- export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
73
+ export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
73
74
  export {};
@@ -17,6 +17,8 @@
17
17
  Object.defineProperty(exports, "__esModule", { value: true });
18
18
  exports.extractFiles = void 0;
19
19
  const fflate_1 = require("fflate");
20
+ const types_js_1 = require("../types.js");
21
+ const errorUtils_js_1 = require("./errorUtils.js");
20
22
  /**
21
23
  * Extracts files from a ZIP archive with optional filtering.
22
24
  *
@@ -56,9 +58,39 @@ const fflate_1 = require("fflate");
56
58
  *
57
59
  * @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
58
60
  */
59
- const extractFiles = (zipInput, filterFn) => {
61
+ const extractFiles = (zipInput, filterFn, limits) => {
62
+ const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
63
+ ? limits.maxUncompressedBytes
64
+ : 512 * 1024 * 1024;
65
+ const maxZipEntries = limits?.maxZipEntries !== undefined && Number.isFinite(limits.maxZipEntries) && limits.maxZipEntries >= 0
66
+ ? limits.maxZipEntries
67
+ : 10000;
60
68
  return new Promise((resolve, reject) => {
61
- (0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), { filter: (file) => filterFn(file.name) }, (err, decompressed) => {
69
+ let totalEntryCount = 0;
70
+ let entryCount = 0;
71
+ let declaredTotal = 0;
72
+ (0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), {
73
+ filter: (file) => {
74
+ totalEntryCount++;
75
+ if (totalEntryCount > maxZipEntries) {
76
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, undefined, maxZipEntries));
77
+ return false;
78
+ }
79
+ if (!filterFn(file.name))
80
+ return false;
81
+ if (typeof file.originalSize !== 'number' || !Number.isFinite(file.originalSize) || file.originalSize < 0) {
82
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE));
83
+ return false;
84
+ }
85
+ entryCount++;
86
+ declaredTotal += file.originalSize;
87
+ if (declaredTotal > maxUncompressedBytes) {
88
+ reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, undefined, maxUncompressedBytes));
89
+ return false;
90
+ }
91
+ return true;
92
+ }
93
+ }, (err, decompressed) => {
62
94
  if (err)
63
95
  return reject(err);
64
96
  resolve(Object.entries(decompressed).map(([path, data]) => ({
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "7.2.0",
3
+ "version": "7.2.2",
4
4
  "description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, and RAG-focused chunks.",
5
5
  "funding": "https://github.com/sponsors/harshankur",
6
6
  "main": "dist/index.js",
@@ -31,16 +31,18 @@
31
31
  "sync:versions": "node scripts/sync-pdfjs-versions.js",
32
32
  "sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && mkdir -p docs/test/files && cp test/files/* docs/test/files/",
33
33
  "lint": "eslint src",
34
- "test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator",
34
+ "test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator && npm run test:cli",
35
35
  "test:baseline": "npm run test:parser:baseline && npm run test:generator:baseline",
36
36
  "test:parser": "npx tsx test/parser/testOfficeParser.ts",
37
37
  "test:parser:baseline": "npx tsx test/parser/testOfficeParser.ts baseline",
38
38
  "test:generator": "npx tsx test/generator/testOfficeGenerator.ts",
39
39
  "test:generator:baseline": "npx tsx test/generator/testOfficeGenerator.ts baseline",
40
40
  "test:artifacts": "npx tsx test/testShippingArtifacts.ts",
41
+ "test:cli": "npx tsx test/cli/testCli.ts",
41
42
  "test:visualizer": "node test/testVisualizer.js",
43
+ "test:integration": "node test/testIntegration.js",
42
44
  "test:license": "npm run sbom && node scripts/validate-licenses.js",
43
- "test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output",
45
+ "test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output test/cli/results",
44
46
  "clean": "rm -rf dist && npm run test:clean",
45
47
  "sbom": "npx --yes @cyclonedx/cyclonedx-npm --output-format json --output-file dist/sbom.cdx.json --omit dev",
46
48
  "prepublishOnly": "npm run build",
@@ -123,6 +125,7 @@
123
125
  "esbuild-plugins-node-modules-polyfill": "^1.8.1",
124
126
  "eslint": "^10.3.0",
125
127
  "husky": "^9.1.7",
128
+ "postject": "^1.0.0-alpha.6",
126
129
  "process": "^0.11.10",
127
130
  "puppeteer": "^22.15.0",
128
131
  "tsx": "^4.21.0",