officeparser 7.2.0 → 7.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +66 -44
- package/dist/OfficeGenerator.d.ts +5 -1
- package/dist/OfficeGenerator.js +14 -2
- package/dist/OfficeParser.d.ts +1 -1
- package/dist/OfficeParser.js +1 -1
- package/dist/cli.d.ts +18 -16
- package/dist/cli.js +255 -90
- package/dist/defaults.js +5 -1
- package/dist/generators/BaseGenerator.d.ts +1 -0
- package/dist/generators/BaseGenerator.js +13 -1
- package/dist/generators/ChunkingGenerator.js +25 -5
- package/dist/generators/HtmlGenerator.js +20 -3
- package/dist/generators/MarkdownGenerator.js +32 -1
- package/dist/generators/RtfGenerator.js +6 -0
- package/dist/generators/TextGenerator.js +6 -0
- package/dist/officeparser.browser.d.ts +39 -3
- package/dist/officeparser.browser.iife.js +294 -169
- package/dist/officeparser.browser.mjs +294 -169
- package/dist/parsers/ExcelParser.js +1 -1
- package/dist/parsers/OpenOfficeParser.js +1 -1
- package/dist/parsers/PdfParser.js +5 -2
- package/dist/parsers/PowerPointParser.js +1 -1
- package/dist/parsers/WordParser.js +1 -1
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +34 -2
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.js +14 -2
- package/dist/utils/errorUtils.js +6 -1
- package/dist/utils/moduleLoader.js +55 -11
- package/dist/utils/zipUtils.d.ts +2 -1
- package/dist/utils/zipUtils.js +34 -2
- package/package.json +6 -3
package/dist/types.js
CHANGED
|
@@ -9,6 +9,8 @@ var OfficeErrorType;
|
|
|
9
9
|
(function (OfficeErrorType) {
|
|
10
10
|
/** Unsupported file extension */
|
|
11
11
|
OfficeErrorType["EXTENSION_UNSUPPORTED"] = "EXTENSION_UNSUPPORTED";
|
|
12
|
+
/** Unsupported output generator format */
|
|
13
|
+
OfficeErrorType["FORMAT_UNSUPPORTED"] = "FORMAT_UNSUPPORTED";
|
|
12
14
|
/** File appears to be corrupted or malformed */
|
|
13
15
|
OfficeErrorType["FILE_CORRUPTED"] = "FILE_CORRUPTED";
|
|
14
16
|
/** File could not be found at the specified path */
|
|
@@ -35,6 +37,14 @@ var OfficeErrorType;
|
|
|
35
37
|
OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
|
|
36
38
|
/** The operation was aborted */
|
|
37
39
|
OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
|
|
40
|
+
/** ZIP entry count exceeds limit */
|
|
41
|
+
OfficeErrorType["ZIP_ENTRY_COUNT_LIMIT_EXCEEDED"] = "ZIP_ENTRY_COUNT_LIMIT_EXCEEDED";
|
|
42
|
+
/** ZIP entry missing a valid declared size */
|
|
43
|
+
OfficeErrorType["ZIP_ENTRY_INVALID_SIZE"] = "ZIP_ENTRY_INVALID_SIZE";
|
|
44
|
+
/** ZIP uncompressed size limit exceeded */
|
|
45
|
+
OfficeErrorType["ZIP_SIZE_LIMIT_EXCEEDED"] = "ZIP_SIZE_LIMIT_EXCEEDED";
|
|
46
|
+
/** Embedding call timed out */
|
|
47
|
+
OfficeErrorType["EMBEDDING_TIMEOUT"] = "EMBEDDING_TIMEOUT";
|
|
38
48
|
})(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
|
|
39
49
|
/**
|
|
40
50
|
* Standard warning types for OfficeParser.
|
|
@@ -57,6 +57,12 @@ function isFullParserConfig(config) {
|
|
|
57
57
|
*/
|
|
58
58
|
function resolveParserConfig(userConfig) {
|
|
59
59
|
if (isFullParserConfig(userConfig)) {
|
|
60
|
+
if (!userConfig.decompressionLimits) {
|
|
61
|
+
userConfig.decompressionLimits = {
|
|
62
|
+
maxUncompressedBytes: 512 * 1024 * 1024,
|
|
63
|
+
maxZipEntries: 10000,
|
|
64
|
+
};
|
|
65
|
+
}
|
|
60
66
|
return userConfig;
|
|
61
67
|
}
|
|
62
68
|
// 1. Start with full defaults (deep cloned)
|
|
@@ -65,9 +71,15 @@ function resolveParserConfig(userConfig) {
|
|
|
65
71
|
return config;
|
|
66
72
|
}
|
|
67
73
|
// 2. Merge user config
|
|
68
|
-
// We handle ocrConfig specially to avoid shallow-overwriting the whole
|
|
69
|
-
const { ocrConfig, ...rest } = userConfig;
|
|
74
|
+
// We handle ocrConfig and decompressionLimits specially to avoid shallow-overwriting the whole objects
|
|
75
|
+
const { ocrConfig, decompressionLimits, ...rest } = userConfig;
|
|
70
76
|
Object.assign(config, rest);
|
|
77
|
+
if (decompressionLimits) {
|
|
78
|
+
config.decompressionLimits = {
|
|
79
|
+
...config.decompressionLimits,
|
|
80
|
+
...decompressionLimits,
|
|
81
|
+
};
|
|
82
|
+
}
|
|
71
83
|
if (ocrConfig) {
|
|
72
84
|
const { timeout, ...ocrRest } = ocrConfig;
|
|
73
85
|
config.ocrConfig = {
|
package/dist/utils/errorUtils.js
CHANGED
|
@@ -17,6 +17,7 @@ const ERRORHEADER = "[OfficeParser]: ";
|
|
|
17
17
|
*/
|
|
18
18
|
const ERROR_MESSAGES = {
|
|
19
19
|
[types_js_1.OfficeErrorType.EXTENSION_UNSUPPORTED]: (ext) => `Sorry, OfficeParser currently supports docx, pptx, xlsx, odt, odp, ods, pdf, rtf, md, html, csv files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
20
|
+
[types_js_1.OfficeErrorType.FORMAT_UNSUPPORTED]: (format) => `Sorry, OfficeGenerator does not support generating '${format}' files. Supported formats: json, text, md, html, csv, rtf, pdf, chunks.`,
|
|
20
21
|
[types_js_1.OfficeErrorType.FILE_CORRUPTED]: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
21
22
|
[types_js_1.OfficeErrorType.FILE_DOES_NOT_EXIST]: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
22
23
|
[types_js_1.OfficeErrorType.LOCATION_NOT_FOUND]: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
|
|
@@ -29,7 +30,11 @@ const ERROR_MESSAGES = {
|
|
|
29
30
|
[types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
|
|
30
31
|
[types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
|
|
31
32
|
[types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`,
|
|
32
|
-
[types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted
|
|
33
|
+
[types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`,
|
|
34
|
+
[types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
|
|
35
|
+
[types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
|
|
36
|
+
[types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
|
|
37
|
+
[types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
|
|
33
38
|
};
|
|
34
39
|
/**
|
|
35
40
|
* Lookup table for warning messages.
|
|
@@ -19,7 +19,7 @@ async function loadNodeEsmModule(specifier) {
|
|
|
19
19
|
// This is especially important in Node 18 for sub-paths of packages.
|
|
20
20
|
try {
|
|
21
21
|
const { pathToFileURL } = await import('url');
|
|
22
|
-
// @ts-ignore - require.resolve is available in Node.js
|
|
22
|
+
// @ts-ignore - require.resolve is available in Node.js CJS context
|
|
23
23
|
const absolutePath = require.resolve(specifier);
|
|
24
24
|
const fileUrl = pathToFileURL(absolutePath).href;
|
|
25
25
|
return import(fileUrl);
|
|
@@ -29,6 +29,16 @@ async function loadNodeEsmModule(specifier) {
|
|
|
29
29
|
return import(specifier);
|
|
30
30
|
}
|
|
31
31
|
}
|
|
32
|
+
/**
|
|
33
|
+
* Returns true if require.resolve is available in this runtime context.
|
|
34
|
+
* It is NOT available in native ESM (e.g. the .mjs wrapper) or browser environments.
|
|
35
|
+
* Checking this guards against a ReferenceError that would silently trigger
|
|
36
|
+
* the bundled fallback path for non-bundled ESM consumers.
|
|
37
|
+
*/
|
|
38
|
+
function isRequireAvailable() {
|
|
39
|
+
// @ts-ignore - require may not exist in ESM context
|
|
40
|
+
return typeof require !== 'undefined' && typeof require.resolve === 'function';
|
|
41
|
+
}
|
|
32
42
|
/**
|
|
33
43
|
* Specialized loader for file-type
|
|
34
44
|
*/
|
|
@@ -36,10 +46,19 @@ async function loadFileType() {
|
|
|
36
46
|
if (!envUtils_js_1.isBrowser) {
|
|
37
47
|
// Ensure environment polyfills for Node.js 18 support
|
|
38
48
|
(0, envUtils_js_1.ensureEnvPolyfills)();
|
|
39
|
-
|
|
40
|
-
|
|
49
|
+
if (isRequireAvailable()) {
|
|
50
|
+
// Check if node_modules is available at runtime.
|
|
51
|
+
// @ts-ignore - require.resolve is available in Node.js CJS context
|
|
52
|
+
require.resolve(String('file-type'));
|
|
53
|
+
// node_modules present: use the path-resolved loader for Node 18 ESM compatibility
|
|
54
|
+
return loadNodeEsmModule('file-type');
|
|
55
|
+
}
|
|
41
56
|
}
|
|
42
|
-
//
|
|
57
|
+
// Covers three cases with one return:
|
|
58
|
+
// 1. Node.js standalone/bundled (SEA or bundled CJS): bundler inlines the module at build time.
|
|
59
|
+
// 2. Node.js native ESM (no require available): runtime resolves the bare specifier.
|
|
60
|
+
// 3. Browser: bundler (esbuild/Vite) handles the static-looking dynamic import().
|
|
61
|
+
// Note: bypasses the Node 18 sub-path ESM fix in loadNodeEsmModule, but bundlers handle it.
|
|
43
62
|
return import('file-type');
|
|
44
63
|
}
|
|
45
64
|
/**
|
|
@@ -49,14 +68,39 @@ async function loadPdfJs() {
|
|
|
49
68
|
if (!envUtils_js_1.isBrowser) {
|
|
50
69
|
// Ensure environment polyfills for Node.js 18 support
|
|
51
70
|
(0, envUtils_js_1.ensureEnvPolyfills)();
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
71
|
+
if (isRequireAvailable()) {
|
|
72
|
+
// @ts-ignore - require.resolve is available in Node.js CJS context
|
|
73
|
+
const pkgExists = (() => { try {
|
|
74
|
+
require.resolve(String('pdfjs-dist'));
|
|
75
|
+
return true;
|
|
76
|
+
}
|
|
77
|
+
catch {
|
|
78
|
+
return false;
|
|
79
|
+
} })();
|
|
80
|
+
if (pkgExists) {
|
|
81
|
+
// node_modules present: try the legacy build path first for stability with
|
|
82
|
+
// ESM-only main, then fall back to the package root.
|
|
83
|
+
// Errors from these loaders are NOT swallowed into the bundled-fallback path.
|
|
84
|
+
try {
|
|
85
|
+
return await loadNodeEsmModule('pdfjs-dist/legacy/build/pdf.mjs');
|
|
86
|
+
}
|
|
87
|
+
catch {
|
|
88
|
+
return await loadNodeEsmModule('pdfjs-dist');
|
|
89
|
+
}
|
|
90
|
+
}
|
|
58
91
|
}
|
|
92
|
+
// Standalone/bundled fallback for Node.js (SEA, bundled CJS without node_modules, or native ESM context):
|
|
93
|
+
// Load both PDF.js and its worker, and register the worker on globalThis to enable the offline fake-worker.
|
|
94
|
+
// @ts-ignore - Ignore type check for local .mjs files in node_modules/packaging
|
|
95
|
+
const [pdfjs, pdfjsWorker] = await Promise.all([
|
|
96
|
+
// @ts-ignore - mjs imports may not have types
|
|
97
|
+
import('pdfjs-dist/legacy/build/pdf.mjs'),
|
|
98
|
+
// @ts-ignore - mjs imports may not have types
|
|
99
|
+
import('pdfjs-dist/legacy/build/pdf.worker.mjs')
|
|
100
|
+
]);
|
|
101
|
+
globalThis.pdfjsWorker = pdfjsWorker;
|
|
102
|
+
return pdfjs;
|
|
59
103
|
}
|
|
60
|
-
// Browser environment: esbuild handles standard
|
|
104
|
+
// Browser environment: bundler (esbuild/Vite) handles the standard dynamic import()
|
|
61
105
|
return import('pdfjs-dist');
|
|
62
106
|
}
|
package/dist/utils/zipUtils.d.ts
CHANGED
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
*
|
|
14
14
|
* @module zipUtils
|
|
15
15
|
*/
|
|
16
|
+
import { DecompressionLimits } from '../types.js';
|
|
16
17
|
/**
|
|
17
18
|
* Represents a file extracted from a ZIP archive.
|
|
18
19
|
* Contains the file's path within the archive and its content as a Buffer.
|
|
@@ -69,5 +70,5 @@ interface ZipFileContent {
|
|
|
69
70
|
*
|
|
70
71
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
71
72
|
*/
|
|
72
|
-
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
|
|
73
|
+
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
|
|
73
74
|
export {};
|
package/dist/utils/zipUtils.js
CHANGED
|
@@ -17,6 +17,8 @@
|
|
|
17
17
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
18
18
|
exports.extractFiles = void 0;
|
|
19
19
|
const fflate_1 = require("fflate");
|
|
20
|
+
const types_js_1 = require("../types.js");
|
|
21
|
+
const errorUtils_js_1 = require("./errorUtils.js");
|
|
20
22
|
/**
|
|
21
23
|
* Extracts files from a ZIP archive with optional filtering.
|
|
22
24
|
*
|
|
@@ -56,9 +58,39 @@ const fflate_1 = require("fflate");
|
|
|
56
58
|
*
|
|
57
59
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
58
60
|
*/
|
|
59
|
-
const extractFiles = (zipInput, filterFn) => {
|
|
61
|
+
const extractFiles = (zipInput, filterFn, limits) => {
|
|
62
|
+
const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
|
|
63
|
+
? limits.maxUncompressedBytes
|
|
64
|
+
: 512 * 1024 * 1024;
|
|
65
|
+
const maxZipEntries = limits?.maxZipEntries !== undefined && Number.isFinite(limits.maxZipEntries) && limits.maxZipEntries >= 0
|
|
66
|
+
? limits.maxZipEntries
|
|
67
|
+
: 10000;
|
|
60
68
|
return new Promise((resolve, reject) => {
|
|
61
|
-
|
|
69
|
+
let totalEntryCount = 0;
|
|
70
|
+
let entryCount = 0;
|
|
71
|
+
let declaredTotal = 0;
|
|
72
|
+
(0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), {
|
|
73
|
+
filter: (file) => {
|
|
74
|
+
totalEntryCount++;
|
|
75
|
+
if (totalEntryCount > maxZipEntries) {
|
|
76
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, undefined, maxZipEntries));
|
|
77
|
+
return false;
|
|
78
|
+
}
|
|
79
|
+
if (!filterFn(file.name))
|
|
80
|
+
return false;
|
|
81
|
+
if (typeof file.originalSize !== 'number' || !Number.isFinite(file.originalSize) || file.originalSize < 0) {
|
|
82
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE));
|
|
83
|
+
return false;
|
|
84
|
+
}
|
|
85
|
+
entryCount++;
|
|
86
|
+
declaredTotal += file.originalSize;
|
|
87
|
+
if (declaredTotal > maxUncompressedBytes) {
|
|
88
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, undefined, maxUncompressedBytes));
|
|
89
|
+
return false;
|
|
90
|
+
}
|
|
91
|
+
return true;
|
|
92
|
+
}
|
|
93
|
+
}, (err, decompressed) => {
|
|
62
94
|
if (err)
|
|
63
95
|
return reject(err);
|
|
64
96
|
resolve(Object.entries(decompressed).map(([path, data]) => ({
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "7.2.
|
|
3
|
+
"version": "7.2.2",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, and RAG-focused chunks.",
|
|
5
5
|
"funding": "https://github.com/sponsors/harshankur",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -31,16 +31,18 @@
|
|
|
31
31
|
"sync:versions": "node scripts/sync-pdfjs-versions.js",
|
|
32
32
|
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/ && mkdir -p docs/test/files && cp test/files/* docs/test/files/",
|
|
33
33
|
"lint": "eslint src",
|
|
34
|
-
"test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator",
|
|
34
|
+
"test": "npm run lint && npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser && npm run test:generator && npm run test:cli",
|
|
35
35
|
"test:baseline": "npm run test:parser:baseline && npm run test:generator:baseline",
|
|
36
36
|
"test:parser": "npx tsx test/parser/testOfficeParser.ts",
|
|
37
37
|
"test:parser:baseline": "npx tsx test/parser/testOfficeParser.ts baseline",
|
|
38
38
|
"test:generator": "npx tsx test/generator/testOfficeGenerator.ts",
|
|
39
39
|
"test:generator:baseline": "npx tsx test/generator/testOfficeGenerator.ts baseline",
|
|
40
40
|
"test:artifacts": "npx tsx test/testShippingArtifacts.ts",
|
|
41
|
+
"test:cli": "npx tsx test/cli/testCli.ts",
|
|
41
42
|
"test:visualizer": "node test/testVisualizer.js",
|
|
43
|
+
"test:integration": "node test/testIntegration.js",
|
|
42
44
|
"test:license": "npm run sbom && node scripts/validate-licenses.js",
|
|
43
|
-
"test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output",
|
|
45
|
+
"test:clean": "rm -rf test/results test/generator/results test/generator/output test/parser/results test/parser/output test/cli/results",
|
|
44
46
|
"clean": "rm -rf dist && npm run test:clean",
|
|
45
47
|
"sbom": "npx --yes @cyclonedx/cyclonedx-npm --output-format json --output-file dist/sbom.cdx.json --omit dev",
|
|
46
48
|
"prepublishOnly": "npm run build",
|
|
@@ -123,6 +125,7 @@
|
|
|
123
125
|
"esbuild-plugins-node-modules-polyfill": "^1.8.1",
|
|
124
126
|
"eslint": "^10.3.0",
|
|
125
127
|
"husky": "^9.1.7",
|
|
128
|
+
"postject": "^1.0.0-alpha.6",
|
|
126
129
|
"process": "^0.11.10",
|
|
127
130
|
"puppeteer": "^22.15.0",
|
|
128
131
|
"tsx": "^4.21.0",
|