officeparser 4.0.3 → 4.0.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -13
- package/officeParser.js +40 -9
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -13,6 +13,7 @@ A Node.js library to parse text out of any office file.
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
#### Update
|
|
16
|
+
* 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
|
|
16
17
|
* 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
|
|
17
18
|
* 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
|
|
18
19
|
* 2023/04/07 - Added typings to methods to help with Typescript projects.
|
|
@@ -95,14 +96,15 @@ officeParser.parseOfficeAsync(fileBuffers);
|
|
|
95
96
|
|
|
96
97
|
### Configuration Object: OfficeParserConfig
|
|
97
98
|
*Optionally add a config object as 3rd variable to parseOffice for the following configurations*
|
|
98
|
-
|
|
|
99
|
-
|
|
100
|
-
| tempFilesLocation | string | The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. **Please ensure that this directory actually exists.** Default is
|
|
101
|
-
| preserveTempFiles | boolean | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
|
|
102
|
-
| outputErrorToConsole | boolean | Flag to show all the logs to console in case of an error.
|
|
103
|
-
| newlineDelimiter | string | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
|
|
104
|
-
| ignoreNotes | boolean | Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default. |
|
|
105
|
-
| putNotesAtLast | boolean | Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored. |
|
|
99
|
+
| Flag | DataType | Default | Explanation |
|
|
100
|
+
|----------------------|----------|------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
|
|
101
|
+
| tempFilesLocation | string | officeParserTemp | The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. **Please ensure that this directory actually exists.** Default is officeParserTemp. |
|
|
102
|
+
| preserveTempFiles | boolean | false | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
|
|
103
|
+
| outputErrorToConsole | boolean | false | Flag to show all the logs to console in case of an error. Default is false. |
|
|
104
|
+
| newlineDelimiter | string | \n | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
|
|
105
|
+
| ignoreNotes | boolean | false | Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default. |
|
|
106
|
+
| putNotesAtLast | boolean | false | Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored. |
|
|
107
|
+
<br>
|
|
106
108
|
|
|
107
109
|
```js
|
|
108
110
|
const config = {
|
|
@@ -121,8 +123,8 @@ officeParser.parseOffice("/path/to/officeFile", function(data, err){
|
|
|
121
123
|
|
|
122
124
|
// promise
|
|
123
125
|
officeParser.parseOfficeAsync("/path/to/officeFile", config);
|
|
124
|
-
.then(
|
|
125
|
-
.catch(
|
|
126
|
+
.then(data => console.log(data))
|
|
127
|
+
.catch(err => console.error(err))
|
|
126
128
|
```
|
|
127
129
|
|
|
128
130
|
**Example - JavaScript**
|
|
@@ -140,7 +142,7 @@ officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config
|
|
|
140
142
|
const newText = data + " look, I can parse a powerpoint file";
|
|
141
143
|
callSomeOtherFunction(newText);
|
|
142
144
|
})
|
|
143
|
-
.catch(
|
|
145
|
+
.catch(err => console.error(err));
|
|
144
146
|
|
|
145
147
|
// Search for a term in the parsed text.
|
|
146
148
|
function searchForTermInOfficeFile(searchterm, filepath) {
|
|
@@ -165,10 +167,10 @@ officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config
|
|
|
165
167
|
const newText = data + " look, I can parse a powerpoint file";
|
|
166
168
|
callSomeOtherFunction(newText);
|
|
167
169
|
})
|
|
168
|
-
.catch(
|
|
170
|
+
.catch(err => console.error(err));
|
|
169
171
|
|
|
170
172
|
// Search for a term in the parsed text.
|
|
171
|
-
function searchForTermInOfficeFile(searchterm, filepath): Promise<boolean> {
|
|
173
|
+
function searchForTermInOfficeFile(searchterm: string, filepath: string): Promise<boolean> {
|
|
172
174
|
return officeParser.parseOfficeAsync(filepath)
|
|
173
175
|
.then(data => data.indexOf(searchterm) != -1)
|
|
174
176
|
}
|
package/officeParser.js
CHANGED
|
@@ -72,7 +72,7 @@ function parseWord(filepath, callback, config) {
|
|
|
72
72
|
if (files.length == 0)
|
|
73
73
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
74
74
|
|
|
75
|
-
return [...files.filter(file => file.path == mainContentFile),
|
|
75
|
+
return [...files.filter(file => file.path == mainContentFile),
|
|
76
76
|
...files.filter(file => file.path == footnotesFile),
|
|
77
77
|
...files.filter(file => file.path == endnotesFile)
|
|
78
78
|
]
|
|
@@ -475,7 +475,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
475
475
|
.then(data =>
|
|
476
476
|
{
|
|
477
477
|
// temp file name
|
|
478
|
-
const newfilepath =
|
|
478
|
+
const newfilepath = getNewFileName(internalConfig.tempFilesLocation, data.ext.toLowerCase());
|
|
479
479
|
// write new file
|
|
480
480
|
fs.writeFileSync(newfilepath, file);
|
|
481
481
|
// resolve promise
|
|
@@ -492,7 +492,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
492
492
|
throw ERRORMSG.fileDoesNotExist(file);
|
|
493
493
|
|
|
494
494
|
// temp file name
|
|
495
|
-
const newfilepath =
|
|
495
|
+
const newfilepath = getNewFileName(internalConfig.tempFilesLocation, file.split(".").pop().toLowerCase());
|
|
496
496
|
// Copy the file into a temp location with the temp name
|
|
497
497
|
fs.copyFileSync(file, newfilepath)
|
|
498
498
|
// resolve promise
|
|
@@ -538,16 +538,13 @@ function parseOffice(file, callback, config = {}) {
|
|
|
538
538
|
|
|
539
539
|
// Check if there is an error. Throw if there is an error.
|
|
540
540
|
if (err)
|
|
541
|
-
|
|
541
|
+
return handleError(err, callback, internalConfig.outputErrorToConsole);
|
|
542
542
|
|
|
543
543
|
// Call the original callback
|
|
544
544
|
callback(data, undefined);
|
|
545
545
|
}
|
|
546
546
|
})
|
|
547
|
-
.catch(error =>
|
|
548
|
-
consoleError(error, internalConfig.outputErrorToConsole);
|
|
549
|
-
callback(undefined, ERRORHEADER + error);
|
|
550
|
-
});
|
|
547
|
+
.catch(error => handleError(error, callback, internalConfig.outputErrorToConsole));
|
|
551
548
|
}
|
|
552
549
|
|
|
553
550
|
/**
|
|
@@ -556,7 +553,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
556
553
|
* @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
|
|
557
554
|
* @returns {Promise<string>}
|
|
558
555
|
*/
|
|
559
|
-
function parseOfficeAsync
|
|
556
|
+
function parseOfficeAsync(file, config = {}) {
|
|
560
557
|
return new Promise((res, rej) => {
|
|
561
558
|
parseOffice(file, function (data, err) {
|
|
562
559
|
if (err)
|
|
@@ -566,6 +563,40 @@ function parseOfficeAsync (file, config = {}) {
|
|
|
566
563
|
});
|
|
567
564
|
}
|
|
568
565
|
|
|
566
|
+
/** Global file name iterator. */
|
|
567
|
+
let globalFileNameIterator = 0;
|
|
568
|
+
/**
|
|
569
|
+
* File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
|
|
570
|
+
* to allow the files to be sorted in chronological order
|
|
571
|
+
* @param {string} tempFilesLocation Directory whether this new file needs to be stored
|
|
572
|
+
* @param {string} ext File extension for this new generated file name
|
|
573
|
+
* @returns {string}
|
|
574
|
+
*/
|
|
575
|
+
function getNewFileName(tempFilesLocation, ext) {
|
|
576
|
+
// Get the iterator part of the file name
|
|
577
|
+
let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
|
|
578
|
+
// We want the iterator part of the file name to be of 5 digits.
|
|
579
|
+
// Therefore, when the iterator crosses into 6 digits, we reset it to 0.
|
|
580
|
+
if (globalFileNameIterator > 99999)
|
|
581
|
+
globalFileNameIterator = 0;
|
|
582
|
+
|
|
583
|
+
// Return the file name
|
|
584
|
+
return `${tempFilesLocation}/tempfiles/${new Date().getTime().toString() + iteratorPart}.${ext}`;
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
/**
|
|
588
|
+
* Handle error by logging it to console if permitted by the config.
|
|
589
|
+
* And after that, trigger the callback function with the error value.
|
|
590
|
+
* @param {string} error Error text
|
|
591
|
+
* @param {function} callback Callback function provided by the caller
|
|
592
|
+
* @param {boolean} outputErrorToConsole Flag to log error to console.
|
|
593
|
+
* @returns {void}
|
|
594
|
+
*/
|
|
595
|
+
function handleError(error, callback, outputErrorToConsole) {
|
|
596
|
+
consoleError(error, outputErrorToConsole);
|
|
597
|
+
callback(undefined, ERRORHEADER + error);
|
|
598
|
+
}
|
|
599
|
+
|
|
569
600
|
|
|
570
601
|
// Export functions
|
|
571
602
|
module.exports.parseOffice = parseOffice;
|
package/package.json
CHANGED