officeparser 5.0.0 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +42 -0
- package/officeParser.js +47 -28
- package/package.json +2 -1
- package/pdfjs-dist-build/pdf.js +0 -17644
- package/pdfjs-dist-build/pdf.js.map +0 -1
- package/pdfjs-dist-build/pdf.worker.js +0 -45163
- package/pdfjs-dist-build/pdf.worker.js.map +0 -1
package/README.md
CHANGED
|
@@ -13,6 +13,7 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2024/11/12 - Added ArrayBuffer as a type of file input. Generating bundle files now which exposes namespace officeParser to be able to access parseOffice and parseOfficeAsync directly on the browser. Extracting text out of pdf files does not work currently in browser bundles.
|
|
16
17
|
* 2024/10/21 - Replaced extracting zip files from decompress to yauzl. This means that we now extract files in memory and we no longer need to write them to disk. Removed config flags related to extracted files. Added flags for CLI execution.
|
|
17
18
|
* 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
|
|
18
19
|
* 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
|
|
@@ -185,10 +186,51 @@ function searchForTermInOfficeFile(searchterm: string, filepath: string): Promis
|
|
|
185
186
|
\
|
|
186
187
|
**Please take note: I have breached convention in placing err as second argument in my callback but please understand that I had to do it to not break other people's existing modules.**
|
|
187
188
|
|
|
189
|
+
## Browser Usage
|
|
190
|
+
Download the bundle file available as part of the release asset.
|
|
191
|
+
Include this bundle file in your browser html file and access `parseOffice` and `parseOfficeAsync` under the **`officeParser`** namespace.
|
|
192
|
+
|
|
193
|
+
**Example**
|
|
194
|
+
```html
|
|
195
|
+
<head>
|
|
196
|
+
...
|
|
197
|
+
<!-- Include bundle file in the script tag. -->
|
|
198
|
+
<script src="officeParserBundle@5.1.0.js"></script>
|
|
199
|
+
</head>
|
|
200
|
+
<body>
|
|
201
|
+
...
|
|
202
|
+
<input type="file" id="fileInput" />
|
|
203
|
+
...
|
|
204
|
+
<script>
|
|
205
|
+
document.getElementById('fileInput').addEventListener('change', async function(event) {
|
|
206
|
+
const outputDiv = document.getElementById('output');
|
|
207
|
+
const file = event.target.files[0];
|
|
208
|
+
try {
|
|
209
|
+
// Your configuration options for officeParser
|
|
210
|
+
const config = {
|
|
211
|
+
outputErrorToConsole: false,
|
|
212
|
+
newlineDelimiter: '\n',
|
|
213
|
+
ignoreNotes: false,
|
|
214
|
+
putNotesAtLast: false
|
|
215
|
+
};
|
|
216
|
+
|
|
217
|
+
const arrayBuffer = await file.arrayBuffer();
|
|
218
|
+
const result = await officeParser.parseOfficeAsync(arrayBuffer, config);
|
|
219
|
+
// result contains the extracted text.
|
|
220
|
+
}
|
|
221
|
+
catch (error) {
|
|
222
|
+
// Handle error
|
|
223
|
+
}
|
|
224
|
+
});
|
|
225
|
+
</script>
|
|
226
|
+
</body>
|
|
227
|
+
```
|
|
228
|
+
|
|
188
229
|
|
|
189
230
|
## Known Bugs
|
|
190
231
|
1. Inconsistency and incorrectness in the positioning of footnotes and endnotes in .docx files where the footnotes and endnotes would end up at the end of the parsed text whereas it would be positioned exactly after the referenced word in .odt files.
|
|
191
232
|
2. The charts and objects information of .odt files are not accurate and may end up showing a few NaN in some cases.
|
|
233
|
+
3. Extracting texts in browser bundles does not work for pdf files.
|
|
192
234
|
----------
|
|
193
235
|
|
|
194
236
|
**npm**
|
package/officeParser.js
CHANGED
|
@@ -6,9 +6,11 @@ const concat = require('concat-stream');
|
|
|
6
6
|
const { DOMParser } = require('@xmldom/xmldom');
|
|
7
7
|
const fileType = require('file-type');
|
|
8
8
|
const fs = require('fs');
|
|
9
|
-
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
10
9
|
const yauzl = require('yauzl');
|
|
11
10
|
|
|
11
|
+
/** Load pdfjs-dist once at module scope. This returns a Promise that resolves to the module. */
|
|
12
|
+
const pdfjsPromise = import('pdfjs-dist/legacy/build/pdf.mjs');
|
|
13
|
+
|
|
12
14
|
/** Header for error messages */
|
|
13
15
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
14
16
|
/** Error messages */
|
|
@@ -271,8 +273,10 @@ function parseExcel(file, callback, config) {
|
|
|
271
273
|
? sharedStrings[value]
|
|
272
274
|
: value;
|
|
273
275
|
}
|
|
274
|
-
//
|
|
275
|
-
// Not the case now but it could happen
|
|
276
|
+
// Should not reach here. If we do, it means we are not filtering out items that we are not ready to process.
|
|
277
|
+
// Not the case now but it could happen if we change the filtering logic without updating the processing logic.
|
|
278
|
+
// So, it is better to error out here.
|
|
279
|
+
handleError(`Invalid c node found in sheet xml content: ${cNode}`, callback, config.outputErrorToConsole);
|
|
276
280
|
return '';
|
|
277
281
|
})
|
|
278
282
|
// Join each cell text within a sheet with a space.
|
|
@@ -442,14 +446,17 @@ function parseOpenOffice(file, callback, config) {
|
|
|
442
446
|
* @param {string | Buffer} file File path or Buffers
|
|
443
447
|
* @param {function} callback Callback function that returns value or error
|
|
444
448
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
445
|
-
* @returns {void}
|
|
449
|
+
* @returns {Promise<void>}
|
|
446
450
|
*/
|
|
447
|
-
function parsePdf(file, callback, config) {
|
|
448
|
-
//
|
|
449
|
-
|
|
450
|
-
|
|
451
|
+
async function parsePdf(file, callback, config) {
|
|
452
|
+
// Wait for pdfjs module to be loaded once
|
|
453
|
+
const pdfjs = await pdfjsPromise;
|
|
454
|
+
|
|
455
|
+
// Get the pdfjs document for the filepath or Uint8Array buffers.
|
|
456
|
+
// pdfjs does not accept Buffers directly, so we convert them to Uint8Array.
|
|
457
|
+
pdfjs.getDocument(file instanceof Buffer ? new Uint8Array(file) : file).promise
|
|
451
458
|
// We go through each page and build our text content promise array.
|
|
452
|
-
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).
|
|
459
|
+
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => document.getPage(index + 1).then(page => page.getTextContent()))))
|
|
453
460
|
// Each textContent item has property 'items' which is an array of objects.
|
|
454
461
|
// Each object element in the array has text stored in their 'str' key.
|
|
455
462
|
// The concatenation of str is what makes our pdf content.
|
|
@@ -462,29 +469,35 @@ function parsePdf(file, callback, config) {
|
|
|
462
469
|
const responseText = textContentArray
|
|
463
470
|
.map(textContent => textContent.items) // Get all the items
|
|
464
471
|
.flat() // Flatten all the items object
|
|
465
|
-
.
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
472
|
+
.reduce((a, v) => (
|
|
473
|
+
// the items could be TextItem or a TextMarkedContent.
|
|
474
|
+
// We are only interested in the TextItem which has a str property.
|
|
475
|
+
'str' in v && v.str != ''
|
|
476
|
+
? {
|
|
477
|
+
text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
|
|
478
|
+
transform5: v.transform[5]
|
|
479
|
+
} : {
|
|
480
|
+
text: a.text,
|
|
481
|
+
transform5: a.transform5
|
|
482
|
+
}
|
|
483
|
+
),
|
|
484
|
+
{
|
|
485
|
+
text: '',
|
|
486
|
+
transform5: undefined
|
|
487
|
+
}).text;
|
|
488
|
+
|
|
476
489
|
callback(responseText, undefined);
|
|
477
490
|
})
|
|
478
491
|
.catch(e => callback(undefined, e));
|
|
479
492
|
}
|
|
480
493
|
|
|
481
494
|
/** Main async function with callback to execute parseOffice for supported files
|
|
482
|
-
* @param {string | Buffer}
|
|
483
|
-
* @param {function}
|
|
484
|
-
* @param {OfficeParserConfig}
|
|
495
|
+
* @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
|
|
496
|
+
* @param {function} callback Callback function that returns value or error
|
|
497
|
+
* @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
|
|
485
498
|
* @returns {void}
|
|
486
499
|
*/
|
|
487
|
-
function parseOffice(
|
|
500
|
+
function parseOffice(srcFile, callback, config = {}) {
|
|
488
501
|
// Make a clone of the config with default values such that none of the config flags are undefined.
|
|
489
502
|
/** @type {OfficeParserConfig} */
|
|
490
503
|
const internalConfig = {
|
|
@@ -494,6 +507,12 @@ function parseOffice(file, callback, config = {}) {
|
|
|
494
507
|
outputErrorToConsole: false,
|
|
495
508
|
...config
|
|
496
509
|
};
|
|
510
|
+
|
|
511
|
+
// Our internal code can process regular node Buffers or file path.
|
|
512
|
+
// So, if the src file was presented as ArrayBuffers, we create Buffers from them.
|
|
513
|
+
let file = srcFile instanceof ArrayBuffer ? Buffer.from(srcFile)
|
|
514
|
+
: srcFile;
|
|
515
|
+
|
|
497
516
|
/**
|
|
498
517
|
* Prepare file for processing
|
|
499
518
|
* @type {Promise<{ file:string | Buffer, ext: string}>}
|
|
@@ -559,13 +578,13 @@ function parseOffice(file, callback, config = {}) {
|
|
|
559
578
|
}
|
|
560
579
|
|
|
561
580
|
/** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
562
|
-
* @param {string | Buffer}
|
|
563
|
-
* @param {OfficeParserConfig}
|
|
581
|
+
* @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
|
|
582
|
+
* @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
|
|
564
583
|
* @returns {Promise<string>}
|
|
565
584
|
*/
|
|
566
|
-
function parseOfficeAsync(
|
|
585
|
+
function parseOfficeAsync(srcFile, config = {}) {
|
|
567
586
|
return new Promise((res, rej) => {
|
|
568
|
-
parseOffice(
|
|
587
|
+
parseOffice(srcFile, function (data, err) {
|
|
569
588
|
if (err)
|
|
570
589
|
return rej(err);
|
|
571
590
|
return res(data);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.2.0",
|
|
4
4
|
"description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
|
|
5
5
|
"main": "officeParser.js",
|
|
6
6
|
"files": [
|
|
@@ -47,6 +47,7 @@
|
|
|
47
47
|
"concat-stream": "^2.0.0",
|
|
48
48
|
"file-type": "^16.5.4",
|
|
49
49
|
"node-ensure": "^0.0.0",
|
|
50
|
+
"pdfjs-dist": "^5.3.31",
|
|
50
51
|
"yauzl": "^3.1.3"
|
|
51
52
|
},
|
|
52
53
|
"devDependencies": {
|