officeparser 3.3.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js CHANGED
@@ -1,517 +1,552 @@
1
1
  #!/usr/bin/env node
2
2
 
3
- const decompress = require('decompress');
4
- const xml2js = require('xml2js')
5
- const fs = require('fs')
6
- const rimraf = require('rimraf');
7
- const fileType = require('file-type');
3
+ const decompress = require('decompress');
4
+ const fs = require('fs');
5
+ const rimraf = require('rimraf');
6
+ const fileType = require('file-type');
7
+ const pdfParse = require('pdf-parse');
8
+ const { DOMParser } = require('xmldom');
8
9
 
9
10
  /** Header for error messages */
10
11
  const ERRORHEADER = "[OfficeParser]: ";
11
12
  /** Error messages */
12
13
  const ERRORMSG = {
13
- extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
14
- fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
15
- fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
16
- locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
17
- improperArguments: `${ERRORHEADER}Improper arguments`,
18
- improperBuffers: `${ERRORHEADER}Error occured while reading the file buffers`
14
+ extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
15
+ fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
16
+ fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
17
+ locationNotFound: (location) => `Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
18
+ improperArguments: `Improper arguments`,
19
+ improperBuffers: `Error occured while reading the file buffers`
19
20
  }
20
21
  /** Default sublocation for decompressing files under the current directory. */
21
22
  const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
22
23
  /** Location for decompressing files. Default is "officeDist" */
23
24
  let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
24
- /** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
25
- let outputErrorToConsole = false;
26
25
 
27
26
  /** Console error if allowed
28
- * @param {string} errorMessage Error message to show on the console
27
+ * @param {string} errorMessage Error message to show on the console
28
+ * @param {string} outputErrorToConsole Flag to show log on console. Ignore if not true.
29
29
  * @returns {void}
30
30
  */
31
- function consoleError(errorMessage) {
32
- if (outputErrorToConsole)
33
- console.error(errorMessage);
31
+ function consoleError(errorMessage, outputErrorToConsole) {
32
+ if (!errorMessage || !outputErrorToConsole)
33
+ return;
34
+ console.error(ERRORHEADER + errorMessage);
34
35
  }
35
36
 
36
- /** Custom parseString promise as the native has bugs
37
+ /** Returns parsed xml document for a given xml text.
37
38
  * @param {string} xml The xml string from the doc file
38
- * @param {boolean} [ignoreAttrs=true] Optional: Ignore attributes part of xml and focus only on the content.
39
- * @returns {Promise<string>}
39
+ * @returns {XMLDocument}
40
+ */
41
+ const parseString = (xml) => {
42
+ let parser = new DOMParser();
43
+ return parser.parseFromString(xml, "text/xml");
44
+ };
45
+
46
+ /** @typedef {Object} OfficeParserConfig
47
+ * @property {boolean} preserveTempFiles Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
48
+ * @property {boolean} outputErrorToConsole Flag to show all the logs to console in case of an error irrespective of your own handling.
49
+ * @property {string} newlineDelimiter The delimiter used for every new line in places that allow multiline text like word. Default is \n.
50
+ * @property {boolean} ignoreNotes Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
51
+ * @property {boolean} putNotesAtLast Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
40
52
  */
41
- const parseStringPromise = (xml, ignoreAttrs = true) => new Promise((resolve, reject) => {
42
- xml2js.parseString(xml, { "ignoreAttrs": ignoreAttrs }, (err, result) => {
43
- if (err)
44
- reject(err);
45
- resolve(result);
46
- });
47
- });
48
53
 
49
54
 
50
55
  /** Main function for parsing text from word files
51
- * @param {string} filename File path
52
- * @param {function} callback Callback function that returns value or error
53
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
56
+ * @param {string} filepath File path
57
+ * @param {function} callback Callback function that returns value or error
58
+ * @param {OfficeParserConfig} config Config Object for officeParser
54
59
  * @returns {void}
55
60
  */
56
- function parseWord(filename, callback, deleteOfficeDist = true) {
57
- if (!fs.existsSync(filename)) {
58
- consoleError(ERRORMSG.fileDoesNotExist(filename));
59
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
60
- }
61
- const ext = filename.split(".").pop().toLowerCase();
62
- if (ext != 'docx') {
63
- consoleError(ERRORMSG.extensionUnsupported(extension));
64
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
65
- }
66
-
67
- /** Store all the text content to respond */
68
- let responseText = [];
69
-
70
- /** Extracting text from Word files xml objects converted to js */
71
- function extractTextFromWordXmlObjects(xmlObjects) {
72
- // specifically for Arrays
73
- if (Array.isArray(xmlObjects)) {
74
- xmlObjects.forEach(item =>
75
- (typeof item == "string") && (item != "")
76
- ? responseText.push(item)
77
- : extractTextFromWordXmlObjects(item))
78
- }
79
- // for other JS Object
80
- else if (typeof xmlObjects == "object") {
81
- for (const [key, value] of Object.entries(xmlObjects)) {
82
- (typeof value == "string") || (typeof value[0] == "string")
83
- ? (key == "w:t" || key == "_") && value != ""
84
- ? responseText.push(value)
85
- : undefined
86
- : extractTextFromWordXmlObjects(value);
87
- }
88
- }
89
- }
90
-
91
- const contentFile = 'word/document.xml';
92
- decompress(filename,
93
- decompressSubLocation,
94
- { filter: x => x.path == contentFile }
61
+ function parseWord(filepath, callback, config) {
62
+ /** The target content xml file for the docx file. */
63
+ const mainContentFile = 'word/document.xml';
64
+ const footnotesFile = 'word/footnotes.xml';
65
+ const endnotesFile = 'word/endnotes.xml';
66
+ /** The decompress location which contains the filename in it */
67
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
68
+ decompress(filepath,
69
+ decompressLocation,
70
+ { filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
95
71
  )
96
72
  .then(files => {
97
- if (files.length != 1) {
98
- consoleError(ERRORMSG.fileCorrupted(filename));
99
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
100
- }
101
-
102
- return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
73
+ if (files.length == 0)
74
+ throw ERRORMSG.fileCorrupted(filepath);
75
+
76
+ return [...files.filter(file => file.path == mainContentFile),
77
+ ...files.filter(file => file.path == footnotesFile),
78
+ ...files.filter(file => file.path == endnotesFile)
79
+ ]
80
+ .map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
103
81
  })
104
- .then(xmlContent => parseStringPromise(xmlContent))
105
- .then(xmlObjects => {
106
- extractTextFromWordXmlObjects(xmlObjects);
107
- const returnCallbackPromise = new Promise((res, rej) => {
108
- if (deleteOfficeDist)
109
- rimraf(decompressSubLocation, err => {
110
- if (err)
111
- consoleError(err);
112
- res();
113
- });
114
- else
115
- res();
82
+ // ************************************* word xml files explanation *************************************
83
+ // Structure of xmlContent of a word file is simple.
84
+ // All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
85
+ // So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
86
+ // ******************************************************************************************************
87
+ .then(xmlContentArray => {
88
+ /** Store all the text content to respond */
89
+ let responseText = [];
90
+
91
+ xmlContentArray.forEach(xmlContent => {
92
+ /** Find text nodes with w:p tags */
93
+ const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
94
+ /** Store all the text content to respond */
95
+ responseText.push(
96
+ Array.from(xmlParagraphNodesList)
97
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
98
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
99
+ .map(paragraphNode => {
100
+ // Find text nodes with w:t tags
101
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
102
+ // Join the texts within this paragraph node without any spaces or delimiters.
103
+ return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
104
+ })
105
+ // Join each paragraph text with a new line delimiter.
106
+ .join(config.newlineDelimiter ?? "\n")
107
+ );
116
108
  });
117
109
 
118
- returnCallbackPromise
119
- .then(() => callback(responseText.join(" "), undefined));
120
-
110
+ // Join all responseText array
111
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
112
+ // Respond by calling the Callback function.
113
+ callback(responseText, undefined);
121
114
  })
122
- .catch(error => {
123
- consoleError(error)
124
- return callback(undefined, error);
125
- });
115
+ .catch(e => callback(undefined, e));
126
116
  }
127
117
 
128
118
  /** Main function for parsing text from PowerPoint files
129
- * @param {string} filename File path
130
- * @param {function} callback Callback function that returns value or error
131
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
119
+ * @param {string} filepath File path
120
+ * @param {function} callback Callback function that returns value or error
121
+ * @param {OfficeParserConfig} config Config Object for officeParser
132
122
  * @returns {void}
133
123
  */
134
- function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
135
- if (!fs.existsSync(filename)) {
136
- consoleError(ERRORMSG.fileDoesNotExist(filename));
137
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
138
- }
139
- const ext = filename.split(".").pop().toLowerCase();
140
- if (ext != 'pptx') {
141
- consoleError(ERRORMSG.extensionUnsupported(extension));
142
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
143
- }
144
-
145
- /** Store all the text content to respond */
146
- let responseText = [];
147
-
148
- /** Extracting text from powerpoint files xml objects converted to js */
149
- function extractTextFromPowerPointXmlObjects(xmlObjects) {
150
- // specifically for Arrays
151
- if (Array.isArray(xmlObjects)) {
152
- xmlObjects.forEach(item =>
153
- (typeof item == "string") && (item != "")
154
- ? responseText.push(item)
155
- : extractTextFromPowerPointXmlObjects(item))
156
- }
157
- // for other JS Object
158
- else if (typeof xmlObjects == "object") {
159
- for (const [key, value] of Object.entries(xmlObjects)) {
160
- (typeof value == "string") || (typeof value[0] == "string")
161
- ? (key == "a:t" || key == "_") && value != ""
162
- ? responseText.push(value)
163
- : undefined
164
- : extractTextFromPowerPointXmlObjects(value);
165
- }
166
- }
167
- }
168
-
124
+ function parsePowerPoint(filepath, callback, config) {
169
125
  // Files regex that hold our content of interest
170
- const contentFiles = [
171
- /ppt\/slides\/slide\d+.xml/g,
172
- /ppt\/notesSlides\/notesSlide\d+.xml/g
173
- ]
174
-
175
- decompress(filename,
176
- decompressSubLocation,
177
- { filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
126
+ const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
127
+ const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
128
+
129
+ /** The decompress location which contains the filename in it */
130
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
131
+ decompress(filepath,
132
+ decompressLocation,
133
+ { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
178
134
  )
179
135
  .then(files => {
180
- // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
181
- files.sort((a,b) => contentFiles.findIndex(fileRegex => a.path.match(fileRegex)) - contentFiles.findIndex(fileRegex => b.path.match(fileRegex)))
136
+ // Check if files is corrupted
137
+ if (files.length == 0)
138
+ throw ERRORMSG.fileCorrupted(filepath);
182
139
 
183
- if (files.length == 0) {
184
- consoleError(ERRORMSG.fileCorrupted(filename));
185
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
186
- }
140
+ // Check if any sorting is required.
141
+ if (!config.ignoreNotes && config.putNotesAtLast)
142
+ // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
143
+ // For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
144
+ files.sort((a,b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
187
145
 
188
146
  // Returning an array of all the xml contents read using fs.readFileSync
189
- return files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8'))
147
+ return files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
190
148
  })
191
- .then(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent)))) // Returning an array of all parseStringPromise responses
192
- .then(xmlObjectsArray => {
193
- xmlObjectsArray.forEach(xmlObjects => extractTextFromPowerPointXmlObjects(xmlObjects)); // Extracting text from all xml js objects with our conditions
194
-
195
- const returnCallbackPromise = new Promise((res, rej) => {
196
- if (deleteOfficeDist)
197
- rimraf(decompressSubLocation, err => {
198
- if (err)
199
- consoleError(err);
200
- res();
201
- });
202
- else
203
- res();
149
+ // ******************************** powerpoint xml files explanation ************************************
150
+ // Structure of xmlContent of a powerpoint file is simple.
151
+ // There are multiple xml files for each slide and correspondingly their notesSlide files.
152
+ // All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
153
+ // So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
154
+ // ******************************************************************************************************
155
+ .then(xmlContentArray => {
156
+ /** Store all the text content to respond */
157
+ let responseText = [];
158
+
159
+ xmlContentArray.forEach(xmlContent => {
160
+ /** Find text nodes with a:p tags */
161
+ const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
162
+ /** Store all the text content to respond */
163
+ responseText.push(
164
+ Array.from(xmlParagraphNodesList)
165
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
166
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
167
+ .map(paragraphNode => {
168
+ /** Find text nodes with a:t tags */
169
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
170
+ return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
171
+ })
172
+ .join(config.newlineDelimiter ?? "\n")
173
+ );
204
174
  });
205
175
 
206
- returnCallbackPromise
207
- .then(() => callback(responseText.join(" "), undefined));
208
-
176
+ // Join all responseText array
177
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
178
+ // Respond by calling the Callback function.
179
+ callback(responseText, undefined);
209
180
  })
210
- .catch(error => {
211
- consoleError(error)
212
- return callback(undefined, error);
213
- });
181
+ .catch(e => callback(undefined, e));
214
182
  }
215
183
 
216
184
  /** Main function for parsing text from Excel files
217
- * @param {string} filename File path
218
- * @param {function} callback Callback function that returns value or error
219
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
185
+ * @param {string} filepath File path
186
+ * @param {function} callback Callback function that returns value or error
187
+ * @param {OfficeParserConfig} config Config Object for officeParser
220
188
  * @returns {void}
221
189
  */
222
- function parseExcel(filename, callback, deleteOfficeDist = true) {
223
- if (!fs.existsSync(filename)) {
224
- consoleError(ERRORMSG.fileDoesNotExist(filename));
225
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
226
- }
227
- const ext = filename.split(".").pop().toLowerCase();
228
- if (ext != 'xlsx') {
229
- consoleError(ERRORMSG.extensionUnsupported(extension));
230
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
231
- }
232
-
233
- /** Store all the text content to respond */
234
- let responseText = [];
235
-
236
- function extractTextFromExcelXmlObjects2dArray(xmlObjects2dArray) {
237
- xmlObjects2dArray[0].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 0));
238
- xmlObjects2dArray[1].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 1));
239
- xmlObjects2dArray[2].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 2));
240
- }
241
-
242
- /** Extracting text from Excel files xml objects converted to js */
243
- function extractTextFromExcelXmlObjects(xmlObjects, contentFilesIndex) {
244
- switch(contentFilesIndex) {
245
- case 0: { // worksheet
246
- // specifically for Arrays
247
- if (Array.isArray(xmlObjects)) {
248
- xmlObjects.forEach(item =>
249
- item["v"]
250
- ? ((item["$"]["t"] != "s"))
251
- ? responseText.push(item["v"][0])
252
- : undefined
253
- : extractTextFromExcelXmlObjects(item, contentFilesIndex))
254
- }
255
- // for other JS Object
256
- else if (typeof xmlObjects == "object") {
257
- for (const [key, value] of Object.entries(xmlObjects)) {
258
- value["v"]
259
- ? ((value["$"]["t"] == "s"))
260
- ? responseText.push(value["v"][0])
261
- : undefined
262
- : extractTextFromExcelXmlObjects(value, contentFilesIndex);
263
- }
264
- }
265
- break;
266
- }
267
- case 1: { // sharedStrings
268
- // specifically for Arrays
269
- if (Array.isArray(xmlObjects)) {
270
- xmlObjects.forEach(item =>
271
- (typeof item == "string") && (item != "")
272
- ? responseText.push(item)
273
- : extractTextFromExcelXmlObjects(item, contentFilesIndex))
274
- }
275
- // for other JS Object
276
- else if (typeof xmlObjects == "object") {
277
- for (const [key, value] of Object.entries(xmlObjects)) {
278
- (typeof value == "string") || (typeof value[0] == "string")
279
- ? (key == "t" || key == "_") && (value != "")
280
- ? responseText.push(value)
281
- : undefined
282
- : extractTextFromExcelXmlObjects(value, contentFilesIndex);
283
- }
284
- }
285
- break;
286
- }
287
- case 2: { // drawings
288
- // specifically for Arrays
289
- if (Array.isArray(xmlObjects)) {
290
- xmlObjects.forEach(item =>
291
- (typeof item == "string") && (item != "")
292
- ? responseText.push(item)
293
- : extractTextFromExcelXmlObjects(item, contentFilesIndex))
294
- }
295
- // for other JS Object
296
- else if (typeof xmlObjects == "object") {
297
- for (const [key, value] of Object.entries(xmlObjects)) {
298
- (typeof value == "string") || (typeof value[0] == "string")
299
- ? (key == "a:t" || key == "_") && (value != "")
300
- ? responseText.push(value)
301
- : undefined
302
- : extractTextFromExcelXmlObjects(value, contentFilesIndex);
303
- }
304
- }
305
- break;
306
- }
307
- }
308
- }
309
-
190
+ function parseExcel(filepath, callback, config) {
310
191
  // Files regex that hold our content of interest
311
- const contentFiles = [
312
- /xl\/worksheets\/sheet\d+.xml/g,
313
- /xl\/sharedStrings.xml/g,
314
- /xl\/drawings\/drawing\d+.xml/g,
315
- ]
316
-
317
- decompress(filename,
318
- decompressSubLocation,
319
- { filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
192
+ const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
193
+ const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
194
+ const chartsRegex = /xl\/charts\/chart\d+.xml/g;
195
+ const stringsFilePath = 'xl/sharedStrings.xml';
196
+
197
+ /** The decompress location which contains the filename in it */
198
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
199
+ decompress(filepath,
200
+ decompressLocation,
201
+ { filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
320
202
  )
321
203
  .then(files => {
322
- // arrange files into 2d array of files organized in contentFiles order, separated by array elements
323
- const files2dArray = [];
324
- contentFiles.forEach(fileRegex => files2dArray.push(files.filter(file => file.path.match(fileRegex))))
325
-
326
- if (files.length == 0) {
327
- consoleError(ERRORMSG.fileCorrupted(filename));
328
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
329
- }
330
-
331
- // Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
332
- return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
204
+ if (files.length == 0)
205
+ throw ERRORMSG.fileCorrupted(filepath);
206
+
207
+ return {
208
+ sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
209
+ drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
210
+ chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
211
+ sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
212
+ };
333
213
  })
334
- .then(xmlContent2dArray => Promise.all(xmlContent2dArray.map(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent, false)))))) // Returning a 2dArray of all parseStringPromise responses
335
- .then(xmlObjects2dArray => {
336
- extractTextFromExcelXmlObjects2dArray(xmlObjects2dArray); // Extracting text from all xml js objects with our conditions
337
-
338
- const returnCallbackPromise = new Promise((res, rej) => {
339
- if (deleteOfficeDist)
340
- rimraf(decompressSubLocation, err => {
341
- if (err)
342
- consoleError(err);
343
- res();
344
- });
345
- else
346
- res();
214
+ // ********************************** excel xml files explanation ***************************************
215
+ // Structure of xmlContent of an excel file is a bit complex.
216
+ // We have a sharedStrings.xml file which has strings inside t tags
217
+ // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
218
+ // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
219
+ // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
220
+ // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
221
+ // ******************************************************************************************************
222
+ .then(xmlContentFilesObject => {
223
+ /** Store all the text content to respond */
224
+ let responseText = [];
225
+
226
+ /** Find text nodes with t tags in sharedStrings xml file */
227
+ const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
228
+ /** Create shared string array. This will be used as a map to get strings from within sheet files. */
229
+ const sharedStrings = Array.from(sharedStringsXmlTNodesList).map(tNode => tNode.childNodes[0].nodeValue);
230
+
231
+ // Parse Sheet files
232
+ xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
233
+ /** Find text nodes with c tags in sharedStrings xml file */
234
+ const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
235
+ // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
236
+ responseText.push(
237
+ Array.from(sheetsXmlCNodesList)
238
+ // Filter c nodes than do not have any v nodes
239
+ .filter(cNode => cNode.getElementsByTagName("v").length != 0)
240
+ .map(cNode => {
241
+ /** Flag whether this node's value represents a string index */
242
+ const isString = cNode.getAttribute("t") == "s";
243
+ /** Find value nodes represented by v tags */
244
+ const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
245
+ // Validate text
246
+ if (isString && value >= sharedStrings.length)
247
+ throw ERRORMSG.fileCorrupted(filepath);
248
+
249
+ return isString
250
+ ? sharedStrings[value]
251
+ : value;
252
+ })
253
+ // Join each cell text within a sheet with a space.
254
+ .join(config.newlineDelimiter ?? "\n")
255
+ );
256
+ });
257
+
258
+ // Parse Drawing files
259
+ xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
260
+ /** Find text nodes with a:p tags */
261
+ const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
262
+ /** Store all the text content to respond */
263
+ responseText.push(
264
+ Array.from(drawingsXmlParagraphNodesList)
265
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
266
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
267
+ .map(paragraphNode => {
268
+ /** Find text nodes with a:t tags */
269
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
270
+ return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
271
+ })
272
+ .join(config.newlineDelimiter ?? "\n")
273
+ );
347
274
  });
348
275
 
349
- returnCallbackPromise
350
- .then(() => callback(responseText.join(" "), undefined));
276
+ // Parse Chart files
277
+ xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
278
+ /** Find text nodes with c:v tags */
279
+ const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
280
+ /** Store all the text content to respond */
281
+ responseText.push(
282
+ Array.from(chartsXmlCVNodesList)
283
+ .map(cVNode => cVNode.childNodes[0].nodeValue)
284
+ .join(config.newlineDelimiter ?? "\n")
285
+ );
286
+ });
351
287
 
288
+ // Join all responseText array
289
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
290
+ // Respond by calling the Callback function.
291
+ callback(responseText, undefined);
352
292
  })
353
- .catch(error => {
354
- consoleError(error)
355
- return callback(undefined, error);
356
- });
293
+ .catch(e => callback(undefined, e));
357
294
  }
358
295
 
359
296
 
360
297
  /** Main function for parsing text from open office files
361
- * @param {string} filename File path
362
- * @param {function} callback Callback function that returns value or error
363
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
298
+ * @param {string} filepath File path
299
+ * @param {function} callback Callback function that returns value or error
300
+ * @param {OfficeParserConfig} config Config Object for officeParser
364
301
  * @returns {void}
365
302
  */
366
- function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
367
- if (!fs.existsSync(filename)) {
368
- consoleError(ERRORMSG.fileDoesNotExist(filename));
369
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
370
- }
371
- const ext = filename.split(".").pop().toLowerCase();
372
- if (!["odt", "odp", "ods"].includes(ext)) {
373
- consoleError(ERRORMSG.extensionUnsupported(extension));
374
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
375
- }
303
+ function parseOpenOffice(filepath, callback, config) {
304
+ /** The target content xml file for the openoffice file. */
305
+ const mainContentFilePath = 'content.xml';
306
+ const objectContentFilesRegex = /Object \d+\/content.xml/g;
307
+
308
+ /** The decompress location which contains the filename in it */
309
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
310
+ decompress(filepath,
311
+ decompressLocation,
312
+ { filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
313
+ )
314
+ .then(files => {
315
+ if (files.length == 0)
316
+ throw ERRORMSG.fileCorrupted(filepath);
376
317
 
377
- /** Store all the text content to respond */
378
- let responseText = [];
379
- /** Extracting text from Open Office files xml objects converted to js */
380
- function extractTextFromOpenOfficeXmlObjects(xmlObjects) {
381
- // specifically for Arrays
382
- if (Array.isArray(xmlObjects)) {
383
- xmlObjects.forEach(item =>
384
- (typeof item == "string") && (item != "")
385
- ? responseText.push(item)
386
- : extractTextFromOpenOfficeXmlObjects(item))
318
+ return {
319
+ mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
320
+ objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
321
+ }
322
+ })
323
+ // ********************************** openoffice xml files explanation **********************************
324
+ // Structure of xmlContent of openoffice files is simple.
325
+ // All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
326
+ // All text in these tags are separated by new line delimiters.
327
+ // Objects like charts in ods files are in Object d+/content.xml with the same way as above.
328
+ // ******************************************************************************************************
329
+ .then(xmlContentFilesObject => {
330
+ /** Store all the notes text content to respond */
331
+ let notesText = [];
332
+ /** Store all the text content to respond */
333
+ let responseText = [];
334
+
335
+ /** List of allowed text tags */
336
+ const allowedTextTags = ["text:p", "text:h"];
337
+ /** List of notes tags */
338
+ const notesTag = "presentation:notes";
339
+
340
+ /** Main dfs traversal function that goes from one node to its children and returns the value out. */
341
+ function extractAllTextsFromNode(root) {
342
+ let xmlTextArray = []
343
+ for (let i = 0; i < root.childNodes.length; i++)
344
+ traversal(root.childNodes[i], xmlTextArray, true);
345
+ return xmlTextArray.join("");
387
346
  }
388
- // for other JS Object
389
- else if (typeof xmlObjects == "object") {
390
- for (const [key, value] of Object.entries(xmlObjects)) {
391
- typeof value == "string"
392
- ? value != ""
393
- ? responseText.push(value)
394
- : undefined
395
- : extractTextFromOpenOfficeXmlObjects(value);
347
+ /** Traversal function that gets recursive calling. */
348
+ function traversal(node, xmlTextArray, isFirstRecursion) {
349
+ if(!node.childNodes || node.childNodes.length == 0)
350
+ {
351
+ if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
352
+ if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
353
+ notesText.push(node.nodeValue);
354
+ if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
355
+ notesText.push(config.newlineDelimiter ?? "\n");
356
+ }
357
+ else {
358
+ xmlTextArray.push(node.nodeValue);
359
+ if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
360
+ xmlTextArray.push(config.newlineDelimiter ?? "\n");
361
+ }
362
+ }
363
+ return;
396
364
  }
365
+
366
+ for (let i = 0; i < node.childNodes.length; i++)
367
+ traversal(node.childNodes[i], xmlTextArray, false);
397
368
  }
398
- }
399
369
 
400
- const contentFile = 'content.xml';
401
- decompress(filename,
402
- decompressSubLocation,
403
- { filter: x => x.path == contentFile }
404
- )
405
- .then(files => {
406
- if (files.length != 1) {
407
- consoleError(ERRORMSG.fileCorrupted(filename));
408
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
370
+ /** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
371
+ function isNotesNode(node) {
372
+ if (node.tagName == notesTag)
373
+ return true;
374
+ if (node.parentNode)
375
+ return isNotesNode(node.parentNode);
376
+ return false;
409
377
  }
410
378
 
411
- return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
412
- })
413
- .then(xmlContent => parseStringPromise(xmlContent))
414
- .then(xmlObjects => {
415
- extractTextFromOpenOfficeXmlObjects(xmlObjects);
416
- const returnCallbackPromise = new Promise((res, rej) => {
417
- if (deleteOfficeDist)
418
- rimraf(decompressSubLocation, err => {
419
- if (err)
420
- consoleError(err);
421
- res();
422
- });
423
- else
424
- res();
379
+ /** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
380
+ function isInvalidTextNode(node) {
381
+ if (allowedTextTags.includes(node.tagName))
382
+ return true;
383
+ if (node.parentNode)
384
+ return isInvalidTextNode(node.parentNode);
385
+ return false;
386
+ }
387
+
388
+ /** The xml string parsed as xml array */
389
+ const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
390
+ // Iterate over each xmlContent and extract text from them.
391
+ xmlContentArray.forEach(xmlContent => {
392
+ /** Find text nodes with text:h and text:p tags in xmlContent */
393
+ const xmlTextNodesList = [...Array.from(xmlContent
394
+ .getElementsByTagName("*"))
395
+ .filter(node => allowedTextTags.includes(node.tagName)
396
+ && !isInvalidTextNode(node.parentNode))
397
+ ];
398
+ /** Store all the text content to respond */
399
+ responseText.push(
400
+ xmlTextNodesList
401
+ // Add every text information from within this textNode and combine them together.
402
+ .map(textNode => extractAllTextsFromNode(textNode))
403
+ .filter(text => text != "")
404
+ .join(config.newlineDelimiter ?? "\n")
405
+ );
425
406
  });
426
407
 
427
- returnCallbackPromise
428
- .then(() => callback(responseText.join(" "), undefined));
408
+ // Add notes text at the end if the user config says so.
409
+ // Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
410
+ if (!config.ignoreNotes && config.putNotesAtLast)
411
+ responseText = [...responseText, ...notesText];
429
412
 
413
+ // Join all responseText array
414
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
415
+ // Respond by calling the Callback function.
416
+ callback(responseText, undefined);
430
417
  })
431
- .catch(error => {
432
- consoleError(error)
433
- return callback(undefined, error);
434
- });
418
+ .catch(e => callback(undefined, e));
435
419
  }
436
420
 
421
+ /** Header for error messages */
422
+ const PDFPARSEERRORHEADER = "[pdf-parse]: ";
423
+
424
+ /** Main function for parsing text from pdf files
425
+ * @param {string} filepath File path
426
+ * @param {function} callback Callback function that returns value or error
427
+ * @param {OfficeParserConfig} config Config Object for officeParser
428
+ * @returns {void}
429
+ */
430
+ function parsePdf(filepath, callback, config) {
431
+ // Get the data buffer for the given file path.
432
+ const dataBuffer = fs.readFileSync(filepath);
433
+
434
+ pdfParse(dataBuffer)
435
+ .then(data => {
436
+ let text = data.text;
437
+ if (!!config.newlineDelimiter && config.newlineDelimiter != "\n")
438
+ text = text.replaceAll("\n", config.newlineDelimiter)
439
+ callback(text, undefined);
440
+ })
441
+ .catch(e => callback(undefined, PDFPARSEERRORHEADER + e));
442
+ }
437
443
 
438
444
  /** Main async function with callback to execute parseOffice for supported files
439
- * @param {string | Buffer} file File path or file buffers
440
- * @param {function} callback Callback function that returns value or error
441
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
445
+ * @param {string | Buffer} file File path or file buffers
446
+ * @param {function} callback Callback function that returns value or error
447
+ * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
442
448
  * @returns {void}
443
449
  */
444
- function parseOffice(file, callback, deleteOfficeDist = true) {
445
- // filename that is to be filled below depending on file input from argument
446
- let filename = "";
450
+ function parseOffice(file, callback, config = {}) {
447
451
  // Prepare file for processing
448
- const filePreparedPromise = new Promise((res, rej) =>
449
- {
452
+ const filePreparedPromise = new Promise((res, rej) => {
453
+ // create temp file subdirectory if it does not exist
454
+ fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
455
+
450
456
  // Check if buffer
451
- if (Buffer.isBuffer(file))
452
- {
457
+ if (Buffer.isBuffer(file)) {
453
458
  // Guess file type from buffer
454
459
  fileType.fromBuffer(file)
455
460
  .then(data =>
456
461
  {
457
462
  // temp file name
458
- filename = `${decompressSubLocation}/tempfiles/${Math.floor(Math.random()*100000000)}.${data.ext}`;
459
- // create directory if it does not exist
460
- fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
463
+ const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
461
464
  // write new file
462
- fs.writeFileSync(filename, file);
465
+ fs.writeFileSync(newfilepath, file);
463
466
  // resolve promise
464
- res();
467
+ res(newfilepath);
465
468
  })
466
- .catch(() => rej());
469
+ .catch(() => rej(ERRORMSG.improperBuffers));
467
470
  return;
468
471
  }
469
472
 
470
- // Treat as filepath
471
- filename = file;
473
+ // Not buffers but real file path.
474
+
475
+ // Check if file exists
476
+ if (!fs.existsSync(file))
477
+ throw ERRORMSG.fileDoesNotExist(file);
478
+
479
+ // temp file name
480
+ const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
481
+ // Copy the file into a temp location with the temp name
482
+ fs.copyFileSync(file, newfilepath)
472
483
  // resolve promise
473
- res();
474
- })
484
+ res(newfilepath);
485
+ });
475
486
 
476
487
  // Process filePreparedPromise resolution.
477
488
  filePreparedPromise
478
- .then(() =>
479
- {
480
- // Check if file exists
481
- if (!fs.existsSync(filename)) {
482
- consoleError(ERRORMSG.fileDoesNotExist(filename));
483
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
484
- }
485
- var extension = filename.split(".").pop().toLowerCase();
489
+ .then(filepath => {
490
+ // File extension. Already in lowercase when we prepared the temp file above.
491
+ const extension = filepath.split(".").pop();
486
492
 
487
493
  // Switch between parsing functions depending on extension.
488
- switch(extension)
489
- {
494
+ switch(extension) {
490
495
  case "docx":
491
- parseWord(filename, (data, err) => callback(data, err), deleteOfficeDist);
492
- return;
496
+ parseWord(filepath, internalCallback, config);
497
+ break;
493
498
  case "pptx":
494
- parsePowerPoint(filename, (data, err) => callback(data, err), deleteOfficeDist);
495
- return;
499
+ parsePowerPoint(filepath, internalCallback, config);
500
+ break;
496
501
  case "xlsx":
497
- parseExcel(filename, (data, err) => callback(data, err), deleteOfficeDist);
498
- return;
502
+ parseExcel(filepath, internalCallback, config);
503
+ break;
499
504
  case "odt":
500
505
  case "odp":
501
506
  case "ods":
502
- parseOpenOffice(filename, (data, err) => callback(data, err), deleteOfficeDist);
503
- return;
504
-
507
+ parseOpenOffice(filepath, internalCallback, config);
508
+ break;
509
+ case "pdf":
510
+ parsePdf(filepath, internalCallback, config);
511
+ break;
512
+
505
513
  default:
506
- consoleError(ERRORMSG.extensionUnsupported(extension));
507
- callback(undefined, ERRORMSG.extensionUnsupported(extension));
514
+ throw ERRORMSG.extensionUnsupported(extension);
515
+ }
516
+
517
+ /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
518
+ function internalCallback(data, err) {
519
+ if (err)
520
+ consoleError(err, config.outputErrorToConsole)
521
+ // Call the original callback
522
+ callback(data, err);
523
+ // Check if we need to preserve unzipped content files or delete them.
524
+ if (config.preserveTempFiles)
525
+ return;
526
+ // Delete decompress sublocation.
527
+ rimraf(decompressSubLocation, rimrafErr => consoleError(rimrafErr, config.outputErrorToConsole));
508
528
  }
509
529
  })
510
- .catch(() =>
511
- {
512
- consoleError(ERRORMSG.improperBuffers);
513
- callback(undefined, ERRORMSG.improperBuffers);
514
- })
530
+ .catch(error => {
531
+ consoleError(error, config.outputErrorToConsole);
532
+ callback(undefined, error);
533
+ });
534
+ }
535
+
536
+ /**
537
+ * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
538
+ * @param {string | Buffer} file File path or file buffers
539
+ * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
540
+ * @returns {Promise<string>}
541
+ */
542
+ function parseOfficeAsync (file, config) {
543
+ return new Promise((res, rej) => {
544
+ parseOffice(file, function (data, err) {
545
+ if (err)
546
+ return rej(err);
547
+ return res(data);
548
+ }, config);
549
+ });
515
550
  }
516
551
 
517
552
  /**
@@ -522,147 +557,18 @@ function parseOffice(file, callback, deleteOfficeDist = true) {
522
557
  function setDecompressionLocation(newLocation) {
523
558
  if (newLocation != undefined) {
524
559
  newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
525
-
526
560
  if (fs.existsSync(newLocation))
527
561
  decompressSubLocation = newLocation;
528
562
  return;
529
563
  }
530
- consoleError(ERRORMSG.locationNotFound(newLocation));
564
+ consoleError(ERRORMSG.locationNotFound(newLocation), config.outputErrorToConsole);
531
565
  decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
532
566
  }
533
567
 
534
- /** Enable console output
535
- * @returns {void}
536
- */
537
- function enableConsoleOutput() {
538
- outputErrorToConsole = true;
539
- }
540
-
541
- /** Disabled console output
542
- * @returns {void}
543
- */
544
- function disableConsoleOutput() {
545
- outputErrorToConsole = false;
546
- }
547
-
548
-
549
- // #region Promise versions of above functions
550
-
551
- /** Async function that can be used with await to execute parseWord. Or it can be used with promises.
552
- * @param {string} filename File path
553
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
554
- * @returns {Promise<string>}
555
- */
556
- var parseWordAsync = function (filename, deleteOfficeDist = true) {
557
- return new Promise((resolve, reject) => {
558
- try {
559
- parseWord(filename, function (data, error) {
560
- if (error)
561
- return reject(error);
562
- return resolve(data);
563
- }, deleteOfficeDist);
564
- }
565
- catch (error) {
566
- return reject(error);
567
- }
568
- })
569
- }
570
-
571
- /** Async function that can be used with await to execute parsePowerPoint. Or it can be used with promises.
572
- * @param {string} filename File path
573
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
574
- * @returns {Promise<string>}
575
- */
576
- var parsePowerPointAsync = function (filename, deleteOfficeDist = true) {
577
- return new Promise((resolve, reject) => {
578
- try {
579
- parsePowerPoint(filename, function (data, err) {
580
- if (err)
581
- return reject(err);
582
- return resolve(data);
583
- }, deleteOfficeDist);
584
- }
585
- catch (error) {
586
- return reject(error);
587
- }
588
- })
589
- }
590
-
591
- /** Async function that can be used with await to execute parseExcel. Or it can be used with promises.
592
- * @param {string} filename File path
593
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
594
- * @returns {Promise<string>}
595
- */
596
- var parseExcelAsync = function (filename, deleteOfficeDist = true) {
597
- return new Promise((resolve, reject) => {
598
- try {
599
- parseExcel(filename, function (data, err) {
600
- if (err)
601
- return reject(err);
602
- return resolve(data);
603
- }, deleteOfficeDist);
604
- }
605
- catch (error) {
606
- return reject(error);
607
- }
608
- })
609
- }
610
-
611
- /** Async function that can be used with await to execute parseOpenOffice. Or it can be used with promises.
612
- * @param {string} filename File path
613
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
614
- * @returns {Promise<string>}
615
- */
616
- var parseOpenOfficeAsync = function (filename, deleteOfficeDist = true) {
617
- return new Promise((resolve, reject) => {
618
- try {
619
- parseOpenOffice(filename, function (data, err) {
620
- if (err)
621
- return reject(err);
622
- return resolve(data);
623
- }, deleteOfficeDist);
624
- }
625
- catch (error) {
626
- return reject(error);
627
- }
628
- })
629
- }
630
-
631
- /**
632
- * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
633
- * @param {string | Buffer} file File path or file buffers
634
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
635
- * @returns {Promise<string>}
636
- */
637
- var parseOfficeAsync = function (file, deleteOfficeDist = true) {
638
- return new Promise((resolve, reject) => {
639
- try {
640
- parseOffice(file, function (data, err) {
641
- if (err)
642
- return reject(err);
643
- return resolve(data);
644
- }, deleteOfficeDist);
645
- }
646
- catch (error) {
647
- return reject(error);
648
- }
649
- })
650
- }
651
- // #endregion Async Versions
652
-
653
- module.exports.parseWord = parseWord;
654
- module.exports.parsePowerPoint = parsePowerPoint;
655
- module.exports.parseExcel = parseExcel;
656
- module.exports.parseOpenOffice = parseOpenOffice;
657
- module.exports.parseOffice = parseOffice;
658
- module.exports.parseWordAsync = parseWordAsync;
659
- module.exports.parsePowerPointAsync = parsePowerPointAsync;
660
- module.exports.parseExcelAsync = parseExcelAsync;
661
- module.exports.parseOpenOfficeAsync = parseOpenOfficeAsync;
662
- module.exports.parseOfficeAsync = parseOfficeAsync;
568
+ // Export functions
569
+ module.exports.parseOffice = parseOffice;
570
+ module.exports.parseOfficeAsync = parseOfficeAsync;
663
571
  module.exports.setDecompressionLocation = setDecompressionLocation;
664
- module.exports.enableConsoleOutput = enableConsoleOutput;
665
- module.exports.disableConsoleOutput = disableConsoleOutput;
666
572
 
667
573
 
668
574
  // Run this library on CLI
@@ -672,8 +578,8 @@ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').po
672
578
  }
673
579
  else if (process.argv.length == 3)
674
580
  parseOfficeAsync(process.argv[2])
675
- .then(text => console.log(text))
676
- .catch(error => console.error(error))
581
+ .then(text => console.log(text))
582
+ .catch(error => console.error(ERRORHEADER + error))
677
583
  else
678
584
  console.error(ERRORMSG.improperArguments)
679
585
  }