officeparser 4.0.3 → 4.0.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -13,6 +13,7 @@ A Node.js library to parse text out of any office file.
13
13
 
14
14
 
15
15
  #### Update
16
+ * 2023/11/25 - Fixed error catching when an error occurs within the parsing of a file, especially after decompressing it. Also fixed the problem with parallel parsing of files as we were using only timestamp in file names.
16
17
  * 2023/10/24 - Revamped content parsing code. Fixed order of content in files, especially in word files where table information would always land up at the end of the text. Added config object as argument for parseOffice which can be used to set new line delimiter and multiple other configurations. Added support for parsing pdf files using the popular npm library pdf-parse. Removed support for individual file parsing functions.
17
18
  * 2023/04/26 - Added support for file buffers as argument for filepath for parseOffice and parseOfficeAsync
18
19
  * 2023/04/07 - Added typings to methods to help with Typescript projects.
@@ -95,14 +96,15 @@ officeParser.parseOfficeAsync(fileBuffers);
95
96
 
96
97
  ### Configuration Object: OfficeParserConfig
97
98
  *Optionally add a config object as 3rd variable to parseOffice for the following configurations*
98
- | flag | datatype | explanation |
99
- |----------------------|----------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
100
- | tempFilesLocation | string | The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. **Please ensure that this directory actually exists.** Default is officeParsertemp. |
101
- | preserveTempFiles | boolean | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
102
- | outputErrorToConsole | boolean | Flag to show all the logs to console in case of an error. |
103
- | newlineDelimiter | string | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
104
- | ignoreNotes | boolean | Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default. |
105
- | putNotesAtLast | boolean | Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored. |
99
+ | Flag | DataType | Default | Explanation |
100
+ |----------------------|----------|------------------|-------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|
101
+ | tempFilesLocation | string | officeParserTemp | The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. **Please ensure that this directory actually exists.** Default is officeParserTemp. |
102
+ | preserveTempFiles | boolean | false | Flag to not delete the internal content files and the possible duplicate temp files that it uses after unzipping office files. Default is false. It always deletes all of those files. |
103
+ | outputErrorToConsole | boolean | false | Flag to show all the logs to console in case of an error. Default is false. |
104
+ | newlineDelimiter | string | \n | The delimiter used for every new line in places that allow multiline text like word. Default is \n. |
105
+ | ignoreNotes | boolean | false | Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default. |
106
+ | putNotesAtLast | boolean | false | Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored. |
107
+ <br>
106
108
 
107
109
  ```js
108
110
  const config = {
@@ -121,8 +123,8 @@ officeParser.parseOffice("/path/to/officeFile", function(data, err){
121
123
 
122
124
  // promise
123
125
  officeParser.parseOfficeAsync("/path/to/officeFile", config);
124
- .then((data) => console.log(data))
125
- .catch((err) => console.error(err))
126
+ .then(data => console.log(data))
127
+ .catch(err => console.error(err))
126
128
  ```
127
129
 
128
130
  **Example - JavaScript**
@@ -140,7 +142,7 @@ officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config
140
142
  const newText = data + " look, I can parse a powerpoint file";
141
143
  callSomeOtherFunction(newText);
142
144
  })
143
- .catch((err) => console.error(err));
145
+ .catch(err => console.error(err));
144
146
 
145
147
  // Search for a term in the parsed text.
146
148
  function searchForTermInOfficeFile(searchterm, filepath) {
@@ -165,10 +167,10 @@ officeParser.parseOfficeAsync("/Users/harsh/Desktop/files/mySlides.pptx", config
165
167
  const newText = data + " look, I can parse a powerpoint file";
166
168
  callSomeOtherFunction(newText);
167
169
  })
168
- .catch((err) => console.error(err));
170
+ .catch(err => console.error(err));
169
171
 
170
172
  // Search for a term in the parsed text.
171
- function searchForTermInOfficeFile(searchterm, filepath): Promise<boolean> {
173
+ function searchForTermInOfficeFile(searchterm: string, filepath: string): Promise<boolean> {
172
174
  return officeParser.parseOfficeAsync(filepath)
173
175
  .then(data => data.indexOf(searchterm) != -1)
174
176
  }
package/officeParser.js CHANGED
@@ -72,7 +72,7 @@ function parseWord(filepath, callback, config) {
72
72
  if (files.length == 0)
73
73
  throw ERRORMSG.fileCorrupted(filepath);
74
74
 
75
- return [...files.filter(file => file.path == mainContentFile),
75
+ return [...files.filter(file => file.path == mainContentFile),
76
76
  ...files.filter(file => file.path == footnotesFile),
77
77
  ...files.filter(file => file.path == endnotesFile)
78
78
  ]
@@ -475,7 +475,7 @@ function parseOffice(file, callback, config = {}) {
475
475
  .then(data =>
476
476
  {
477
477
  // temp file name
478
- const newfilepath = `${internalConfig.tempFilesLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
478
+ const newfilepath = getNewFileName(internalConfig.tempFilesLocation, data.ext.toLowerCase());
479
479
  // write new file
480
480
  fs.writeFileSync(newfilepath, file);
481
481
  // resolve promise
@@ -492,7 +492,7 @@ function parseOffice(file, callback, config = {}) {
492
492
  throw ERRORMSG.fileDoesNotExist(file);
493
493
 
494
494
  // temp file name
495
- const newfilepath = `${internalConfig.tempFilesLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
495
+ const newfilepath = getNewFileName(internalConfig.tempFilesLocation, file.split(".").pop().toLowerCase());
496
496
  // Copy the file into a temp location with the temp name
497
497
  fs.copyFileSync(file, newfilepath)
498
498
  // resolve promise
@@ -538,16 +538,13 @@ function parseOffice(file, callback, config = {}) {
538
538
 
539
539
  // Check if there is an error. Throw if there is an error.
540
540
  if (err)
541
- throw err;
541
+ return handleError(err, callback, internalConfig.outputErrorToConsole);
542
542
 
543
543
  // Call the original callback
544
544
  callback(data, undefined);
545
545
  }
546
546
  })
547
- .catch(error => {
548
- consoleError(error, internalConfig.outputErrorToConsole);
549
- callback(undefined, ERRORHEADER + error);
550
- });
547
+ .catch(error => handleError(error, callback, internalConfig.outputErrorToConsole));
551
548
  }
552
549
 
553
550
  /**
@@ -556,7 +553,7 @@ function parseOffice(file, callback, config = {}) {
556
553
  * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
557
554
  * @returns {Promise<string>}
558
555
  */
559
- function parseOfficeAsync (file, config = {}) {
556
+ function parseOfficeAsync(file, config = {}) {
560
557
  return new Promise((res, rej) => {
561
558
  parseOffice(file, function (data, err) {
562
559
  if (err)
@@ -566,6 +563,40 @@ function parseOfficeAsync (file, config = {}) {
566
563
  });
567
564
  }
568
565
 
566
+ /** Global file name iterator. */
567
+ let globalFileNameIterator = 0;
568
+ /**
569
+ * File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
570
+ * to allow the files to be sorted in chronological order
571
+ * @param {string} tempFilesLocation Directory whether this new file needs to be stored
572
+ * @param {string} ext File extension for this new generated file name
573
+ * @returns {string}
574
+ */
575
+ function getNewFileName(tempFilesLocation, ext) {
576
+ // Get the iterator part of the file name
577
+ let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
578
+ // We want the iterator part of the file name to be of 5 digits.
579
+ // Therefore, when the iterator crosses into 6 digits, we reset it to 0.
580
+ if (globalFileNameIterator > 99999)
581
+ globalFileNameIterator = 0;
582
+
583
+ // Return the file name
584
+ return `${tempFilesLocation}/tempfiles/${new Date().getTime().toString() + iteratorPart}.${ext}`;
585
+ }
586
+
587
+ /**
588
+ * Handle error by logging it to console if permitted by the config.
589
+ * And after that, trigger the callback function with the error value.
590
+ * @param {string} error Error text
591
+ * @param {function} callback Callback function provided by the caller
592
+ * @param {boolean} outputErrorToConsole Flag to log error to console.
593
+ * @returns {void}
594
+ */
595
+ function handleError(error, callback, outputErrorToConsole) {
596
+ consoleError(error, outputErrorToConsole);
597
+ callback(undefined, ERRORHEADER + error);
598
+ }
599
+
569
600
 
570
601
  // Export functions
571
602
  module.exports.parseOffice = parseOffice;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.0.3",
3
+ "version": "4.0.5",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [