officeparser 4.0.8 → 4.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/officeParser.js +33 -17
- package/package.json +4 -3
- package/pdfjs-dist-build/pdf.js +17644 -0
- package/pdfjs-dist-build/pdf.js.map +1 -0
- package/pdfjs-dist-build/pdf.worker.js +45163 -0
- package/pdfjs-dist-build/pdf.worker.js.map +1 -0
package/README.md
CHANGED
|
@@ -13,6 +13,7 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
16
17
|
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
17
18
|
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
|
18
19
|
* 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
|
package/officeParser.js
CHANGED
|
@@ -4,7 +4,7 @@ const decompress = require('decompress');
|
|
|
4
4
|
const fs = require('fs');
|
|
5
5
|
const rimraf = require('rimraf');
|
|
6
6
|
const fileType = require('file-type');
|
|
7
|
-
const
|
|
7
|
+
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
8
8
|
const { DOMParser } = require('@xmldom/xmldom');
|
|
9
9
|
|
|
10
10
|
/** Header for error messages */
|
|
@@ -431,9 +431,6 @@ function parseOpenOffice(filepath, callback, config) {
|
|
|
431
431
|
.catch(e => callback(undefined, e));
|
|
432
432
|
}
|
|
433
433
|
|
|
434
|
-
/** Header for error messages */
|
|
435
|
-
const PDFPARSEERRORHEADER = "[pdf-parse]: ";
|
|
436
|
-
|
|
437
434
|
/** Main function for parsing text from pdf files
|
|
438
435
|
* @param {string} filepath File path
|
|
439
436
|
* @param {function} callback Callback function that returns value or error
|
|
@@ -441,17 +438,36 @@ const PDFPARSEERRORHEADER = "[pdf-parse]: ";
|
|
|
441
438
|
* @returns {void}
|
|
442
439
|
*/
|
|
443
440
|
function parsePdf(filepath, callback, config) {
|
|
444
|
-
// Get the
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
441
|
+
// Get the pdfjs document for the filepath.
|
|
442
|
+
pdfjs.getDocument(filepath).promise
|
|
443
|
+
// We go through each page and build our text content promise array.
|
|
444
|
+
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).map(pageNr => document.getPage(pageNr).then(page => page.getTextContent()))))
|
|
445
|
+
// Each textContent item has property 'items' which is an array of objects.
|
|
446
|
+
// Each object element in the array has text stored in their 'str' key.
|
|
447
|
+
// The concatenation of str is what makes our pdf content.
|
|
448
|
+
// str already contains any space that was in the text.
|
|
449
|
+
// So, we only care about when to add the new line.
|
|
450
|
+
// That we determine using transform[5] value which is the y-coordinate of the item object.
|
|
451
|
+
// So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
|
|
452
|
+
.then(textContentArray => {
|
|
453
|
+
/** Store all the text content to respond */
|
|
454
|
+
const responseText = textContentArray
|
|
455
|
+
.map(textContent => textContent.items) // Get all the items
|
|
456
|
+
.flat() // Flatten all the items object
|
|
457
|
+
.filter(item => item.str != '') // Ignore the empty string items.
|
|
458
|
+
.reduce((a, v) => (
|
|
459
|
+
{
|
|
460
|
+
text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
|
|
461
|
+
transform5: v.transform[5]
|
|
462
|
+
}),
|
|
463
|
+
{
|
|
464
|
+
text: '',
|
|
465
|
+
transform5: undefined
|
|
466
|
+
}).text;
|
|
467
|
+
|
|
468
|
+
callback(responseText, undefined);
|
|
469
|
+
})
|
|
470
|
+
.catch(e => callback(undefined, e));
|
|
455
471
|
}
|
|
456
472
|
|
|
457
473
|
/** Main async function with callback to execute parseOffice for supported files
|
|
@@ -540,7 +556,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
540
556
|
break;
|
|
541
557
|
|
|
542
558
|
default:
|
|
543
|
-
|
|
559
|
+
internalCallback(undefined, ERRORMSG.extensionUnsupported(extension)); // Call the internalCallback function which removes the temp files if required.
|
|
544
560
|
}
|
|
545
561
|
|
|
546
562
|
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
@@ -584,7 +600,7 @@ let globalFileNameIterator = 0;
|
|
|
584
600
|
* to allow the files to be sorted in chronological order
|
|
585
601
|
* @param {string} tempFilesLocation Directory whether this new file needs to be stored
|
|
586
602
|
* @param {string} ext File extension for this new generated file name
|
|
587
|
-
* @returns {string}
|
|
603
|
+
* @returns {string}
|
|
588
604
|
*/
|
|
589
605
|
function getNewFileName(tempFilesLocation, ext) {
|
|
590
606
|
// Get the iterator part of the file name
|
package/package.json
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.1.1",
|
|
4
4
|
"description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
|
|
5
5
|
"main": "officeParser.js",
|
|
6
6
|
"files": [
|
|
7
7
|
"officeParser.js",
|
|
8
|
-
"typings/officeParser.d.ts"
|
|
8
|
+
"typings/officeParser.d.ts",
|
|
9
|
+
"pdfjs-dist-build/*"
|
|
9
10
|
],
|
|
10
11
|
"types": "typings/officeParser.d.ts",
|
|
11
12
|
"scripts": {
|
|
@@ -45,7 +46,7 @@
|
|
|
45
46
|
"@xmldom/xmldom": "^0.8.10",
|
|
46
47
|
"decompress": "^4.2.0",
|
|
47
48
|
"file-type": "^16.5.4",
|
|
48
|
-
"
|
|
49
|
+
"node-ensure": "^0.0.0",
|
|
49
50
|
"rimraf": "^2.6.3"
|
|
50
51
|
},
|
|
51
52
|
"devDependencies": {
|