officeparser 4.1.2 → 4.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/officeParser.js +58 -23
- package/package.json +3 -3
package/README.md
CHANGED
|
@@ -13,6 +13,8 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2024/10/15 - Fixed erroring out while deleting temp files when multiple worker threads make parallel executions resulting in same file name for multiple files. Fixed erroring out when multiple executions are made without waiting for the previous execution to finish which resulted in deleting the file from other execution. Upgraded dependencies.
|
|
17
|
+
* 2024/10/13 - Fixed parsing text from xlsx files which contain no shared strings file and files which have inlineStr based strings.
|
|
16
18
|
* 2024/05/06 - Replaced pdf parsing support from pdf-parse library to natively building it using pdf.js library from Mozilla by analyzing its output. Added pdfjs-dist build as a local library.
|
|
17
19
|
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
18
20
|
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
package/officeParser.js
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
|
-
const decompress
|
|
4
|
-
const fs
|
|
5
|
-
const
|
|
6
|
-
const fileType
|
|
7
|
-
const pdfjs
|
|
8
|
-
const { DOMParser }
|
|
3
|
+
const decompress = require('decompress');
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const { rimrafSync } = require('rimraf');
|
|
6
|
+
const fileType = require('file-type');
|
|
7
|
+
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
8
|
+
const { DOMParser } = require('@xmldom/xmldom');
|
|
9
9
|
|
|
10
10
|
/** Header for error messages */
|
|
11
11
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
@@ -543,7 +543,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
543
543
|
}
|
|
544
544
|
|
|
545
545
|
// create temp file subdirectory if it does not exist
|
|
546
|
-
fs.mkdirSync(
|
|
546
|
+
fs.mkdirSync(getTempFilesDirectory(internalConfig.tempFilesLocation), { recursive: true });
|
|
547
547
|
|
|
548
548
|
// Check if buffer
|
|
549
549
|
if (Buffer.isBuffer(file)) {
|
|
@@ -552,11 +552,11 @@ function parseOffice(file, callback, config = {}) {
|
|
|
552
552
|
.then(data =>
|
|
553
553
|
{
|
|
554
554
|
// temp file name
|
|
555
|
-
const
|
|
555
|
+
const newFileName = getNewFileName(data.ext.toLowerCase());
|
|
556
556
|
// write new file
|
|
557
|
-
fs.writeFileSync(
|
|
557
|
+
fs.writeFileSync(getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName), file);
|
|
558
558
|
// resolve promise
|
|
559
|
-
res(
|
|
559
|
+
res(newFileName);
|
|
560
560
|
})
|
|
561
561
|
.catch(() => rej(ERRORMSG.improperBuffers));
|
|
562
562
|
return;
|
|
@@ -569,16 +569,18 @@ function parseOffice(file, callback, config = {}) {
|
|
|
569
569
|
throw ERRORMSG.fileDoesNotExist(file);
|
|
570
570
|
|
|
571
571
|
// temp file name
|
|
572
|
-
const
|
|
572
|
+
const newFileName = getNewFileName(file.split(".").pop().toLowerCase());
|
|
573
573
|
// Copy the file into a temp location with the temp name
|
|
574
|
-
fs.copyFileSync(file,
|
|
574
|
+
fs.copyFileSync(file, getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), newFileName))
|
|
575
575
|
// resolve promise
|
|
576
|
-
res(
|
|
576
|
+
res(newFileName);
|
|
577
577
|
});
|
|
578
578
|
|
|
579
579
|
// Process filePreparedPromise resolution.
|
|
580
580
|
filePreparedPromise
|
|
581
|
-
.then(
|
|
581
|
+
.then(filename => {
|
|
582
|
+
// The file path
|
|
583
|
+
const filepath = getFilePath(getTempFilesDirectory(internalConfig.tempFilesLocation), filename);
|
|
582
584
|
// File extension. Already in lowercase when we prepared the temp file above.
|
|
583
585
|
const extension = filepath.split(".").pop();
|
|
584
586
|
|
|
@@ -609,9 +611,33 @@ function parseOffice(file, callback, config = {}) {
|
|
|
609
611
|
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
610
612
|
function internalCallback(data, err) {
|
|
611
613
|
// Check if we need to preserve unzipped content files or delete them.
|
|
612
|
-
if (!internalConfig.preserveTempFiles)
|
|
613
|
-
|
|
614
|
-
|
|
614
|
+
if (!internalConfig.preserveTempFiles) {
|
|
615
|
+
/** Safely delete location */
|
|
616
|
+
function safelyDeleteLocation(location, deleteDirIfEmpty) {
|
|
617
|
+
if (!fs.existsSync(location))
|
|
618
|
+
return;
|
|
619
|
+
|
|
620
|
+
if (!fs.lstatSync(location).isDirectory() // If not directory
|
|
621
|
+
|| !deleteDirIfEmpty // or if the deleteDirIfEmpty is false or undefined
|
|
622
|
+
|| fs.readdirSync(location).length == 0) { // or if it is true, we check if the contents are empty.
|
|
623
|
+
try {
|
|
624
|
+
rimrafSync(location);
|
|
625
|
+
}
|
|
626
|
+
catch(rimrafErr) {
|
|
627
|
+
consoleError(rimrafErr, internalConfig.outputErrorToConsole);
|
|
628
|
+
}
|
|
629
|
+
}
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
// We delete our file as well as the extracted files inside the office files.
|
|
633
|
+
// There is no extraction for pdf files because we don't support any unpacking of it.
|
|
634
|
+
// After removing files, we check if the folders containing them are empty.
|
|
635
|
+
// If yes, we remove those folders too.
|
|
636
|
+
safelyDeleteLocation(filepath);
|
|
637
|
+
safelyDeleteLocation(getFilePath(internalConfig.tempFilesLocation, filename));
|
|
638
|
+
safelyDeleteLocation(getTempFilesDirectory(internalConfig.tempFilesLocation), true);
|
|
639
|
+
safelyDeleteLocation(internalConfig.tempFilesLocation, true);
|
|
640
|
+
}
|
|
615
641
|
|
|
616
642
|
// Check if there is an error. Throw if there is an error.
|
|
617
643
|
if (err)
|
|
@@ -644,21 +670,30 @@ function parseOfficeAsync(file, config = {}) {
|
|
|
644
670
|
let globalFileNameIterator = 0;
|
|
645
671
|
/**
|
|
646
672
|
* File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
|
|
647
|
-
* to allow the files to be sorted in chronological order
|
|
648
|
-
*
|
|
649
|
-
* @param {string} ext
|
|
673
|
+
* to allow the files to be sorted in chronological order. We also prefix them with ppid and pid of the process to support
|
|
674
|
+
* common destination from worker threads as well as multiple processes running them together.
|
|
675
|
+
* @param {string} ext File extension for this new generated file name
|
|
650
676
|
* @returns {string}
|
|
651
677
|
*/
|
|
652
|
-
function getNewFileName(
|
|
678
|
+
function getNewFileName(ext) {
|
|
653
679
|
// Get the iterator part of the file name
|
|
654
680
|
let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
|
|
655
681
|
// We want the iterator part of the file name to be of 5 digits.
|
|
656
682
|
// Therefore, when the iterator crosses into 6 digits, we reset it to 0.
|
|
657
683
|
if (globalFileNameIterator > 99999)
|
|
658
684
|
globalFileNameIterator = 0;
|
|
685
|
+
// Return the file name with ppid and pid to allow unique names even with worker threads.
|
|
686
|
+
return `${process.ppid}_${process.pid}_${new Date().getTime().toString() + iteratorPart}.${ext}`;
|
|
687
|
+
}
|
|
688
|
+
|
|
689
|
+
/** Gets directory for storing files. */
|
|
690
|
+
function getTempFilesDirectory(root) {
|
|
691
|
+
return `${root}/tempfiles`;
|
|
692
|
+
}
|
|
659
693
|
|
|
660
|
-
|
|
661
|
-
|
|
694
|
+
/** Gets file path for the supplied directory and the file name. */
|
|
695
|
+
function getFilePath(directory, fileName) {
|
|
696
|
+
return `${directory}/${fileName}`;
|
|
662
697
|
}
|
|
663
698
|
|
|
664
699
|
/**
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "officeparser",
|
|
3
|
-
"version": "4.
|
|
3
|
+
"version": "4.2.0",
|
|
4
4
|
"description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
|
|
5
5
|
"main": "officeParser.js",
|
|
6
6
|
"files": [
|
|
@@ -44,10 +44,10 @@
|
|
|
44
44
|
"homepage": "https://github.com/harshankur/officeParser#readme",
|
|
45
45
|
"dependencies": {
|
|
46
46
|
"@xmldom/xmldom": "^0.8.10",
|
|
47
|
-
"decompress": "^4.2.
|
|
47
|
+
"decompress": "^4.2.1",
|
|
48
48
|
"file-type": "^16.5.4",
|
|
49
49
|
"node-ensure": "^0.0.0",
|
|
50
|
-
"rimraf": "^
|
|
50
|
+
"rimraf": "^5.0.10"
|
|
51
51
|
},
|
|
52
52
|
"devDependencies": {
|
|
53
53
|
"@types/decompress": "^4.2.6",
|