officeparser 6.0.7 → 6.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +136 -52
- package/dist/OfficeParser.d.ts +10 -1
- package/dist/OfficeParser.js +44 -56
- package/dist/cli.d.ts +20 -0
- package/dist/cli.js +117 -0
- package/dist/index.d.ts +4 -4
- package/dist/index.js +7 -59
- package/dist/index.mjs +18 -0
- package/dist/officeparser.browser.d.ts +133 -3
- package/dist/officeparser.browser.iife.js +115 -0
- package/dist/officeparser.browser.mjs +114 -0
- package/dist/parsers/ExcelParser.d.ts +1 -1
- package/dist/parsers/ExcelParser.js +76 -68
- package/dist/parsers/OpenOfficeParser.d.ts +1 -1
- package/dist/parsers/OpenOfficeParser.js +224 -159
- package/dist/parsers/PdfParser.d.ts +1 -1
- package/dist/parsers/PdfParser.js +98 -94
- package/dist/parsers/PowerPointParser.d.ts +1 -1
- package/dist/parsers/PowerPointParser.js +188 -179
- package/dist/parsers/RtfParser.d.ts +21 -1
- package/dist/parsers/RtfParser.js +117 -48
- package/dist/parsers/WordParser.d.ts +2 -1
- package/dist/parsers/WordParser.js +214 -123
- package/dist/sbom.cdx.json +1807 -0
- package/dist/types.d.ts +123 -3
- package/dist/utils/chartUtils.js +2 -0
- package/dist/utils/dateUtils.d.ts +17 -0
- package/dist/utils/dateUtils.js +69 -0
- package/dist/utils/envUtils.d.ts +24 -0
- package/dist/utils/envUtils.js +69 -0
- package/dist/utils/moduleLoader.d.ts +2 -1
- package/dist/utils/moduleLoader.js +9 -39
- package/dist/utils/ocrUtils.d.ts +16 -12
- package/dist/utils/ocrUtils.js +186 -25
- package/dist/utils/xmlUtils.d.ts +80 -9
- package/dist/utils/xmlUtils.js +236 -18
- package/dist/utils/zipUtils.js +6 -47
- package/package.json +31 -16
- package/dist/officeParserBundle@6.0.7.js +0 -154
- package/dist/officeparser.browser.js +0 -154
package/dist/utils/zipUtils.js
CHANGED
|
@@ -14,13 +14,9 @@
|
|
|
14
14
|
*
|
|
15
15
|
* @module zipUtils
|
|
16
16
|
*/
|
|
17
|
-
var __importDefault = (this && this.__importDefault) || function (mod) {
|
|
18
|
-
return (mod && mod.__esModule) ? mod : { "default": mod };
|
|
19
|
-
};
|
|
20
17
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
21
18
|
exports.extractFiles = void 0;
|
|
22
|
-
const
|
|
23
|
-
const concat_stream_1 = __importDefault(require("concat-stream"));
|
|
19
|
+
const fflate_1 = require("fflate");
|
|
24
20
|
/**
|
|
25
21
|
* Extracts files from a ZIP archive with optional filtering.
|
|
26
22
|
*
|
|
@@ -62,50 +58,13 @@ const concat_stream_1 = __importDefault(require("concat-stream"));
|
|
|
62
58
|
*/
|
|
63
59
|
const extractFiles = (zipInput, filterFn) => {
|
|
64
60
|
return new Promise((resolve, reject) => {
|
|
65
|
-
|
|
66
|
-
// lazyEntries: true means we manually control when to read each entry (better memory usage)
|
|
67
|
-
yauzl_1.default.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
|
|
61
|
+
(0, fflate_1.unzip)(new Uint8Array(zipInput.buffer, zipInput.byteOffset, zipInput.byteLength), { filter: (file) => filterFn(file.name) }, (err, decompressed) => {
|
|
68
62
|
if (err)
|
|
69
63
|
return reject(err);
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
// Step 2: Start reading the first entry
|
|
75
|
-
// This triggers the 'entry' event
|
|
76
|
-
zipfile.readEntry();
|
|
77
|
-
// Step 3: Handle each entry (file or directory) in the ZIP
|
|
78
|
-
zipfile.on('entry', (entry) => {
|
|
79
|
-
// Step 3a: Check if this file should be extracted using the filter function
|
|
80
|
-
if (filterFn(entry.fileName)) {
|
|
81
|
-
// Step 3b: Open a read stream for this entry
|
|
82
|
-
zipfile.openReadStream(entry, (err, readStream) => {
|
|
83
|
-
if (err)
|
|
84
|
-
return reject(err);
|
|
85
|
-
if (!readStream)
|
|
86
|
-
return reject(new Error("Failed to open read stream"));
|
|
87
|
-
// Step 3c: Pipe the stream through concat to collect all data into a single Buffer
|
|
88
|
-
// This is necessary because streams deliver data in chunks
|
|
89
|
-
readStream.pipe((0, concat_stream_1.default)((data) => {
|
|
90
|
-
// Step 3d: Add the extracted file to our results
|
|
91
|
-
extractedFiles.push({
|
|
92
|
-
path: entry.fileName,
|
|
93
|
-
content: data
|
|
94
|
-
});
|
|
95
|
-
// Step 3e: Continue to the next entry
|
|
96
|
-
zipfile.readEntry();
|
|
97
|
-
}));
|
|
98
|
-
});
|
|
99
|
-
}
|
|
100
|
-
else {
|
|
101
|
-
// Step 3f: Skip this entry and move to the next one
|
|
102
|
-
zipfile.readEntry();
|
|
103
|
-
}
|
|
104
|
-
});
|
|
105
|
-
// Step 4: All entries have been processed
|
|
106
|
-
zipfile.on('end', () => resolve(extractedFiles));
|
|
107
|
-
// Step 5: Handle any errors during extraction
|
|
108
|
-
zipfile.on('error', reject);
|
|
64
|
+
resolve(Object.entries(decompressed).map(([path, data]) => ({
|
|
65
|
+
path,
|
|
66
|
+
content: Buffer.from(data)
|
|
67
|
+
})));
|
|
109
68
|
});
|
|
110
69
|
});
|
|
111
70
|
};
|
package/package.json
CHANGED
|
@@ -1,9 +1,20 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "6.
|
|
3
|
+
"version": "6.1.1",
|
|
4
4
|
"description": "A robust, strictly-typed Node.js and Browser library for parsing office files (.docx, .pptx, .xlsx, .odt, .odp, .ods, .pdf, .rtf) into structured AST with rich metadata, formatting, and attachment support.",
|
|
5
|
+
"funding": "https://github.com/sponsors/harshankur",
|
|
5
6
|
"main": "dist/index.js",
|
|
7
|
+
"module": "dist/index.mjs",
|
|
6
8
|
"types": "dist/index.d.ts",
|
|
9
|
+
"browser": "./dist/officeparser.browser.mjs",
|
|
10
|
+
"exports": {
|
|
11
|
+
".": {
|
|
12
|
+
"types": "./dist/index.d.ts",
|
|
13
|
+
"browser": "./dist/officeparser.browser.mjs",
|
|
14
|
+
"import": "./dist/index.mjs",
|
|
15
|
+
"require": "./dist/index.js"
|
|
16
|
+
}
|
|
17
|
+
},
|
|
7
18
|
"sideEffects": false,
|
|
8
19
|
"engines": {
|
|
9
20
|
"node": ">=18.0.0"
|
|
@@ -12,15 +23,21 @@
|
|
|
12
23
|
"dist"
|
|
13
24
|
],
|
|
14
25
|
"scripts": {
|
|
15
|
-
"build": "npm run sync:versions && npm run build:node && npm run build:browser:types && npm run build:browser",
|
|
26
|
+
"build": "npm run sync:versions && npm run build:node && npm run build:esm-wrapper && npm run build:browser:types && npm run build:browser",
|
|
16
27
|
"build:node": "tsc",
|
|
28
|
+
"build:esm-wrapper": "node scripts/generate-esm-wrapper.js",
|
|
17
29
|
"build:browser:types": "dts-bundle-generator --no-check -o dist/officeparser.browser.d.ts src/index.ts",
|
|
18
30
|
"build:browser": "node build_browser.js && npm run sync:docs",
|
|
19
31
|
"sync:versions": "node scripts/sync-pdfjs-versions.js",
|
|
20
|
-
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.js docs/dist/",
|
|
21
|
-
"test": "npm run test:clean && npm run build &&
|
|
32
|
+
"sync:docs": "mkdir -p docs/dist && cp dist/officeparser.browser.iife.js docs/dist/ && cp dist/officeparser.browser.mjs docs/dist/",
|
|
33
|
+
"test": "npm run test:clean && npm run build && npm run test:license && npm run test:artifacts && npm run test:parser",
|
|
34
|
+
"test:baseline": "npm run test baseline",
|
|
35
|
+
"test:parser": "npx tsx test/testOfficeParser.ts",
|
|
36
|
+
"test:artifacts": "npx tsx test/testShippingArtifacts.ts",
|
|
37
|
+
"test:license": "npm run sbom && node scripts/validate-licenses.js",
|
|
22
38
|
"test:clean": "rm -rf test/results",
|
|
23
39
|
"clean": "rm -rf dist && npm run test:clean",
|
|
40
|
+
"sbom": "npx --yes @cyclonedx/cyclonedx-npm --output-format json --output-file dist/sbom.cdx.json --omit dev",
|
|
24
41
|
"prepublishOnly": "npm run build",
|
|
25
42
|
"prepare": "husky"
|
|
26
43
|
},
|
|
@@ -29,7 +46,7 @@
|
|
|
29
46
|
"url": "git+https://github.com/harshankur/officeParser.git"
|
|
30
47
|
},
|
|
31
48
|
"bin": {
|
|
32
|
-
"officeparser": "dist/
|
|
49
|
+
"officeparser": "dist/cli.js"
|
|
33
50
|
},
|
|
34
51
|
"publishConfig": {
|
|
35
52
|
"access": "public",
|
|
@@ -74,22 +91,20 @@
|
|
|
74
91
|
},
|
|
75
92
|
"homepage": "https://officeparser.harshankur.com",
|
|
76
93
|
"dependencies": {
|
|
77
|
-
"@xmldom/xmldom": "^0.
|
|
78
|
-
"
|
|
79
|
-
"file-type": "^
|
|
80
|
-
"pdfjs-dist": "
|
|
81
|
-
"tesseract.js": "^7.0.0"
|
|
82
|
-
"yauzl": "^3.2.1"
|
|
94
|
+
"@xmldom/xmldom": "^0.9.10",
|
|
95
|
+
"fflate": "^0.8.2",
|
|
96
|
+
"file-type": "^22.0.1",
|
|
97
|
+
"pdfjs-dist": "5.6.205",
|
|
98
|
+
"tesseract.js": "^7.0.0"
|
|
83
99
|
},
|
|
84
100
|
"devDependencies": {
|
|
85
|
-
"@types/
|
|
86
|
-
"
|
|
87
|
-
"@types/xmldom": "^0.1.34",
|
|
88
|
-
"@types/yauzl": "^2.10.3",
|
|
101
|
+
"@types/node": "^25.5.0",
|
|
102
|
+
"buffer": "^6.0.3",
|
|
89
103
|
"dts-bundle-generator": "^9.5.1",
|
|
90
104
|
"esbuild": "^0.27.4",
|
|
91
|
-
"esbuild-
|
|
105
|
+
"esbuild-plugins-node-modules-polyfill": "^1.8.1",
|
|
92
106
|
"husky": "^9.1.7",
|
|
107
|
+
"process": "^0.11.10",
|
|
93
108
|
"tsx": "^4.21.0",
|
|
94
109
|
"typescript": "^6.0.2"
|
|
95
110
|
}
|