officeparser 7.2.1 → 7.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/dist/defaults.js +4 -0
- package/dist/generators/ChunkingGenerator.js +1 -1
- package/dist/generators/HtmlGenerator.js +1 -1
- package/dist/officeparser.browser.d.ts +31 -1
- package/dist/officeparser.browser.iife.js +51 -51
- package/dist/officeparser.browser.mjs +51 -51
- package/dist/parsers/ExcelParser.js +1 -1
- package/dist/parsers/OpenOfficeParser.js +1 -1
- package/dist/parsers/PowerPointParser.js +1 -1
- package/dist/parsers/WordParser.js +1 -1
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +31 -1
- package/dist/types.js +8 -0
- package/dist/utils/configUtils.js +14 -2
- package/dist/utils/errorUtils.js +5 -1
- package/dist/utils/zipUtils.d.ts +2 -1
- package/dist/utils/zipUtils.js +34 -2
- package/package.json +1 -1
package/dist/utils/errorUtils.js
CHANGED
|
@@ -30,7 +30,11 @@ const ERROR_MESSAGES = {
|
|
|
30
30
|
[types_js_1.OfficeErrorType.INVALID_SELECTOR]: (selector) => `Invalid selector: ${selector}`,
|
|
31
31
|
[types_js_1.OfficeErrorType.INVALID_OUTPUT_MAPPING]: (output) => `Invalid output mapping: ${output}`,
|
|
32
32
|
[types_js_1.OfficeErrorType.MISSING_EMBEDDING_FUNCTION]: `Semantic chunking requires an "embeddingFunction" to be provided in chunksConfig. This function must accept a string and return a Promise resolving to a number array (vector).`,
|
|
33
|
-
[types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted
|
|
33
|
+
[types_js_1.OfficeErrorType.OPERATION_ABORTED]: `The operation was aborted.`,
|
|
34
|
+
[types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED]: (limit) => `ZIP entry count exceeds limit (${limit})`,
|
|
35
|
+
[types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE]: `ZIP entry missing a valid declared size`,
|
|
36
|
+
[types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED]: (limit) => `ZIP uncompressed size limit exceeded (${limit} bytes)`,
|
|
37
|
+
[types_js_1.OfficeErrorType.EMBEDDING_TIMEOUT]: (timeout) => `Embedding call timed out after ${timeout}ms`
|
|
34
38
|
};
|
|
35
39
|
/**
|
|
36
40
|
* Lookup table for warning messages.
|
package/dist/utils/zipUtils.d.ts
CHANGED
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
*
|
|
14
14
|
* @module zipUtils
|
|
15
15
|
*/
|
|
16
|
+
import { DecompressionLimits } from '../types.js';
|
|
16
17
|
/**
|
|
17
18
|
* Represents a file extracted from a ZIP archive.
|
|
18
19
|
* Contains the file's path within the archive and its content as a Buffer.
|
|
@@ -69,5 +70,5 @@ interface ZipFileContent {
|
|
|
69
70
|
*
|
|
70
71
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
71
72
|
*/
|
|
72
|
-
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean) => Promise<ZipFileContent[]>;
|
|
73
|
+
export declare const extractFiles: (zipInput: Buffer, filterFn: (fileName: string) => boolean, limits: DecompressionLimits) => Promise<ZipFileContent[]>;
|
|
73
74
|
export {};
|
package/dist/utils/zipUtils.js
CHANGED
|
@@ -17,6 +17,8 @@
|
|
|
17
17
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
18
18
|
exports.extractFiles = void 0;
|
|
19
19
|
const fflate_1 = require("fflate");
|
|
20
|
+
const types_js_1 = require("../types.js");
|
|
21
|
+
const errorUtils_js_1 = require("./errorUtils.js");
|
|
20
22
|
/**
|
|
21
23
|
* Extracts files from a ZIP archive with optional filtering.
|
|
22
24
|
*
|
|
@@ -56,9 +58,39 @@ const fflate_1 = require("fflate");
|
|
|
56
58
|
*
|
|
57
59
|
* @see https://pkware.cachefly.net/webdocs/casestudies/APPNOTE.TXT ZIP file format specification
|
|
58
60
|
*/
|
|
59
|
-
const extractFiles = (zipInput, filterFn) => {
|
|
61
|
+
const extractFiles = (zipInput, filterFn, limits) => {
|
|
62
|
+
const maxUncompressedBytes = limits?.maxUncompressedBytes !== undefined && Number.isFinite(limits.maxUncompressedBytes) && limits.maxUncompressedBytes >= 0
|
|
63
|
+
? limits.maxUncompressedBytes
|
|
64
|
+
: 512 * 1024 * 1024;
|
|
65
|
+
const maxZipEntries = limits?.maxZipEntries !== undefined && Number.isFinite(limits.maxZipEntries) && limits.maxZipEntries >= 0
|
|
66
|
+
? limits.maxZipEntries
|
|
67
|
+
: 10000;
|
|
60
68
|
return new Promise((resolve, reject) => {
|
|
61
|
-
|
|
69
|
+
let totalEntryCount = 0;
|
|
70
|
+
let entryCount = 0;
|
|
71
|
+
let declaredTotal = 0;
|
|
72
|
+
(0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), {
|
|
73
|
+
filter: (file) => {
|
|
74
|
+
totalEntryCount++;
|
|
75
|
+
if (totalEntryCount > maxZipEntries) {
|
|
76
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_COUNT_LIMIT_EXCEEDED, undefined, maxZipEntries));
|
|
77
|
+
return false;
|
|
78
|
+
}
|
|
79
|
+
if (!filterFn(file.name))
|
|
80
|
+
return false;
|
|
81
|
+
if (typeof file.originalSize !== 'number' || !Number.isFinite(file.originalSize) || file.originalSize < 0) {
|
|
82
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_ENTRY_INVALID_SIZE));
|
|
83
|
+
return false;
|
|
84
|
+
}
|
|
85
|
+
entryCount++;
|
|
86
|
+
declaredTotal += file.originalSize;
|
|
87
|
+
if (declaredTotal > maxUncompressedBytes) {
|
|
88
|
+
reject((0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.ZIP_SIZE_LIMIT_EXCEEDED, undefined, maxUncompressedBytes));
|
|
89
|
+
return false;
|
|
90
|
+
}
|
|
91
|
+
return true;
|
|
92
|
+
}
|
|
93
|
+
}, (err, decompressed) => {
|
|
62
94
|
if (err)
|
|
63
95
|
return reject(err);
|
|
64
96
|
resolve(Object.entries(decompressed).map(([path, data]) => ({
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "7.2.
|
|
3
|
+
"version": "7.2.2",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf, .csv, .md, .html) and generating high-fidelity outputs in Markdown, HTML, CSV, RTF, and RAG-focused chunks.",
|
|
5
5
|
"funding": "https://github.com/sponsors/harshankur",
|
|
6
6
|
"main": "dist/index.js",
|