officeparser 4.0.7 → 4.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +40 -27
- package/package.json +4 -3
- package/pdfjs-dist-build/pdf.js +17644 -0
- package/pdfjs-dist-build/pdf.js.map +1 -0
- package/pdfjs-dist-build/pdf.worker.js +45163 -0
- package/pdfjs-dist-build/pdf.worker.js.map +1 -0
package/officeParser.js
CHANGED
|
@@ -4,7 +4,7 @@ const decompress = require('decompress');
|
|
|
4
4
|
const fs = require('fs');
|
|
5
5
|
const rimraf = require('rimraf');
|
|
6
6
|
const fileType = require('file-type');
|
|
7
|
-
const
|
|
7
|
+
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
8
8
|
const { DOMParser } = require('@xmldom/xmldom');
|
|
9
9
|
|
|
10
10
|
/** Header for error messages */
|
|
@@ -59,25 +59,23 @@ const parseString = (xml) => {
|
|
|
59
59
|
*/
|
|
60
60
|
function parseWord(filepath, callback, config) {
|
|
61
61
|
/** The target content xml file for the docx file. */
|
|
62
|
-
const
|
|
63
|
-
const
|
|
64
|
-
const
|
|
62
|
+
const mainContentFileRegex = /word\/document[\d+]?.xml/g;
|
|
63
|
+
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
|
|
64
|
+
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
|
|
65
65
|
/** The decompress location which contains the filename in it */
|
|
66
66
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
67
67
|
decompress(filepath,
|
|
68
68
|
decompressLocation,
|
|
69
|
-
{ filter: x => [
|
|
69
|
+
{ filter: x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.path.match(fileRegex)) }
|
|
70
70
|
)
|
|
71
71
|
.then(files => {
|
|
72
72
|
// Verify if atleast the document xml file exists in the extracted files list.
|
|
73
|
-
if (!files.
|
|
73
|
+
if (!files.some(file => file.path.match(mainContentFileRegex)))
|
|
74
74
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
75
75
|
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
]
|
|
80
|
-
.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
76
|
+
return files
|
|
77
|
+
.filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
|
|
78
|
+
.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
81
79
|
})
|
|
82
80
|
// ************************************* word xml files explanation *************************************
|
|
83
81
|
// Structure of xmlContent of a word file is simple.
|
|
@@ -204,7 +202,7 @@ function parseExcel(filepath, callback, config) {
|
|
|
204
202
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
205
203
|
decompress(filepath,
|
|
206
204
|
decompressLocation,
|
|
207
|
-
{ filter: x =>
|
|
205
|
+
{ filter: x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.path.match(fileRegex)) || x.path == stringsFilePath }
|
|
208
206
|
)
|
|
209
207
|
.then(files => {
|
|
210
208
|
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
@@ -433,9 +431,6 @@ function parseOpenOffice(filepath, callback, config) {
|
|
|
433
431
|
.catch(e => callback(undefined, e));
|
|
434
432
|
}
|
|
435
433
|
|
|
436
|
-
/** Header for error messages */
|
|
437
|
-
const PDFPARSEERRORHEADER = "[pdf-parse]: ";
|
|
438
|
-
|
|
439
434
|
/** Main function for parsing text from pdf files
|
|
440
435
|
* @param {string} filepath File path
|
|
441
436
|
* @param {function} callback Callback function that returns value or error
|
|
@@ -443,17 +438,35 @@ const PDFPARSEERRORHEADER = "[pdf-parse]: ";
|
|
|
443
438
|
* @returns {void}
|
|
444
439
|
*/
|
|
445
440
|
function parsePdf(filepath, callback, config) {
|
|
446
|
-
// Get the
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
441
|
+
// Get the pdfjs document for the filepath.
|
|
442
|
+
pdfjs.getDocument(filepath).promise
|
|
443
|
+
// We go through each page and build our text content promise array.
|
|
444
|
+
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).map(pageNr => document.getPage(pageNr).then(page => page.getTextContent()))))
|
|
445
|
+
// Each textContent item has property 'items' which is an array of objects.
|
|
446
|
+
// Each object element in the array has text stored in their 'str' key.
|
|
447
|
+
// The concatenation of str is what makes our pdf content.
|
|
448
|
+
// str already contains any space that was in the text.
|
|
449
|
+
// So, we only care about when to add the new line.
|
|
450
|
+
// That we determine using transform[5] value which is the y-coordinate of the item object.
|
|
451
|
+
// So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
|
|
452
|
+
.then(textContentArray => {
|
|
453
|
+
/** Store all the text content to respond */
|
|
454
|
+
const responseText = textContentArray
|
|
455
|
+
.map(textContent => textContent.items) // Get all the items
|
|
456
|
+
.flat() // Flatten all the items object
|
|
457
|
+
.reduce((a, v) => (
|
|
458
|
+
{
|
|
459
|
+
text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
|
|
460
|
+
transform5: v.transform[5]
|
|
461
|
+
}),
|
|
462
|
+
{
|
|
463
|
+
text: '',
|
|
464
|
+
transform5: undefined
|
|
465
|
+
}).text;
|
|
466
|
+
|
|
467
|
+
callback(responseText, undefined);
|
|
468
|
+
})
|
|
469
|
+
.catch(e => callback(undefined, e));
|
|
457
470
|
}
|
|
458
471
|
|
|
459
472
|
/** Main async function with callback to execute parseOffice for supported files
|
|
@@ -586,7 +599,7 @@ let globalFileNameIterator = 0;
|
|
|
586
599
|
* to allow the files to be sorted in chronological order
|
|
587
600
|
* @param {string} tempFilesLocation Directory whether this new file needs to be stored
|
|
588
601
|
* @param {string} ext File extension for this new generated file name
|
|
589
|
-
* @returns {string}
|
|
602
|
+
* @returns {string}
|
|
590
603
|
*/
|
|
591
604
|
function getNewFileName(tempFilesLocation, ext) {
|
|
592
605
|
// Get the iterator part of the file name
|
package/package.json
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "4.0
|
|
3
|
+
"version": "4.1.0",
|
|
4
4
|
"description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
|
|
5
5
|
"main": "officeParser.js",
|
|
6
6
|
"files": [
|
|
7
7
|
"officeParser.js",
|
|
8
|
-
"typings/officeParser.d.ts"
|
|
8
|
+
"typings/officeParser.d.ts",
|
|
9
|
+
"pdfjs-dist-build/*"
|
|
9
10
|
],
|
|
10
11
|
"types": "typings/officeParser.d.ts",
|
|
11
12
|
"scripts": {
|
|
@@ -45,7 +46,7 @@
|
|
|
45
46
|
"@xmldom/xmldom": "^0.8.10",
|
|
46
47
|
"decompress": "^4.2.0",
|
|
47
48
|
"file-type": "^16.5.4",
|
|
48
|
-
"
|
|
49
|
+
"node-ensure": "^0.0.0",
|
|
49
50
|
"rimraf": "^2.6.3"
|
|
50
51
|
},
|
|
51
52
|
"devDependencies": {
|