officeparser 3.2.2 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js CHANGED
@@ -1,623 +1,574 @@
1
1
  #!/usr/bin/env node
2
2
 
3
- const decompress = require('decompress');
4
- const xml2js = require('xml2js')
5
- const fs = require('fs')
6
- const rimraf = require('rimraf');
3
+ const decompress = require('decompress');
4
+ const fs = require('fs');
5
+ const rimraf = require('rimraf');
6
+ const fileType = require('file-type');
7
+ const pdfParse = require('pdf-parse');
8
+ const { DOMParser } = require('xmldom');
7
9
 
8
10
  /** Header for error messages */
9
11
  const ERRORHEADER = "[OfficeParser]: ";
10
12
  /** Error messages */
11
13
  const ERRORMSG = {
12
- extensionUnsupported: (ext) => `${ERRORHEADER}Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
13
- fileCorrupted: (filename) => `${ERRORHEADER}Your file ${filename} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
14
- fileDoesNotExist: (filename) => `${ERRORHEADER}File ${filename} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
15
- locationNotFound: (location) => `${ERRORHEADER}Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
16
- improperArguments: `${ERRORHEADER}Improper arguments`
14
+ extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
15
+ fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
16
+ fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
17
+ locationNotFound: (location) => `Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
18
+ improperArguments: `Improper arguments`,
19
+ improperBuffers: `Error occured while reading the file buffers`
17
20
  }
18
21
  /** Default sublocation for decompressing files under the current directory. */
19
22
  const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
20
23
  /** Location for decompressing files. Default is "officeDist" */
21
24
  let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
22
- /** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
23
- let outputErrorToConsole = false;
24
25
 
25
26
  /** Console error if allowed
26
- * @param {string} errorMessage Error message to show on the console
27
+ * @param {string} errorMessage Error message to show on the console
28
+ * @param {string} outputErrorToConsole Flag to show log on console. Ignore if not true.
27
29
  * @returns {void}
28
30
  */
29
- function consoleError(errorMessage) {
30
- if (outputErrorToConsole)
31
- console.error(errorMessage);
31
+ function consoleError(errorMessage, outputErrorToConsole) {
32
+ if (!errorMessage || !outputErrorToConsole)
33
+ return;
34
+ console.error(ERRORHEADER + errorMessage);
32
35
  }
33
36
 
34
- /** Custom parseString promise as the native has bugs
37
+ /** Returns parsed xml document for a given xml text.
35
38
  * @param {string} xml The xml string from the doc file
36
- * @param {boolean} [ignoreAttrs=true] Optional: Ignore attributes part of xml and focus only on the content.
37
- * @returns {Promise<string>}
39
+ * @returns {XMLDocument}
40
+ */
41
+ const parseString = (xml) => {
42
+ let parser = new DOMParser();
43
+ return parser.parseFromString(xml, "text/xml");
44
+ };
45
+
46
+ /** @typedef {Object} OfficeParserConfig
47
+ * @property {boolean} preserveTempFiles Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
48
+ * @property {boolean} outputErrorToConsole Flag to show all the logs to console in case of an error irrespective of your own handling.
49
+ * @property {string} newlineDelimiter The delimiter used for every new line in places that allow multiline text like word. Default is \n.
50
+ * @property {boolean} ignoreNotes Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
51
+ * @property {boolean} putNotesAtLast Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
38
52
  */
39
- const parseStringPromise = (xml, ignoreAttrs = true) => new Promise((resolve, reject) => {
40
- xml2js.parseString(xml, { "ignoreAttrs": ignoreAttrs }, (err, result) => {
41
- if (err)
42
- reject(err);
43
- resolve(result);
44
- });
45
- });
46
53
 
47
54
 
48
55
  /** Main function for parsing text from word files
49
- * @param {string} filename File path
50
- * @param {function} callback Callback function that returns value or error
51
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
56
+ * @param {string} filepath File path
57
+ * @param {function} callback Callback function that returns value or error
58
+ * @param {OfficeParserConfig} config Config Object for officeParser
52
59
  * @returns {void}
53
60
  */
54
- function parseWord(filename, callback, deleteOfficeDist = true) {
55
- if (!fs.existsSync(filename)) {
56
- consoleError(ERRORMSG.fileDoesNotExist(filename));
57
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
58
- }
59
- const ext = filename.split(".").pop().toLowerCase();
60
- if (ext != 'docx') {
61
- consoleError(ERRORMSG.extensionUnsupported(extension));
62
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
63
- }
64
-
65
- /** Store all the text content to respond */
66
- let responseText = [];
67
-
68
- /** Extracting text from Word files xml objects converted to js */
69
- function extractTextFromWordXmlObjects(xmlObjects) {
70
- // specifically for Arrays
71
- if (Array.isArray(xmlObjects)) {
72
- xmlObjects.forEach(item =>
73
- (typeof item == "string") && (item != "")
74
- ? responseText.push(item)
75
- : extractTextFromWordXmlObjects(item))
76
- }
77
- // for other JS Object
78
- else if (typeof xmlObjects == "object") {
79
- for (const [key, value] of Object.entries(xmlObjects)) {
80
- (typeof value == "string") || (typeof value[0] == "string")
81
- ? (key == "w:t" || key == "_") && value != ""
82
- ? responseText.push(value)
83
- : undefined
84
- : extractTextFromWordXmlObjects(value);
85
- }
86
- }
87
- }
88
-
89
- const contentFile = 'word/document.xml';
90
- decompress(filename,
91
- decompressSubLocation,
92
- { filter: x => x.path == contentFile }
61
+ function parseWord(filepath, callback, config) {
62
+ /** The target content xml file for the docx file. */
63
+ const mainContentFile = 'word/document.xml';
64
+ const footnotesFile = 'word/footnotes.xml';
65
+ const endnotesFile = 'word/endnotes.xml';
66
+ /** The decompress location which contains the filename in it */
67
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
68
+ decompress(filepath,
69
+ decompressLocation,
70
+ { filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
93
71
  )
94
72
  .then(files => {
95
- if (files.length != 1) {
96
- consoleError(ERRORMSG.fileCorrupted(filename));
97
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
98
- }
99
-
100
- return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
73
+ if (files.length == 0)
74
+ throw ERRORMSG.fileCorrupted(filepath);
75
+
76
+ return [...files.filter(file => file.path == mainContentFile),
77
+ ...files.filter(file => file.path == footnotesFile),
78
+ ...files.filter(file => file.path == endnotesFile)
79
+ ]
80
+ .map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
101
81
  })
102
- .then(xmlContent => parseStringPromise(xmlContent))
103
- .then(xmlObjects => {
104
- extractTextFromWordXmlObjects(xmlObjects);
105
- const returnCallbackPromise = new Promise((res, rej) => {
106
- if (deleteOfficeDist)
107
- rimraf(decompressSubLocation, err => {
108
- if (err)
109
- consoleError(err);
110
- res();
111
- });
112
- else
113
- res();
82
+ // ************************************* word xml files explanation *************************************
83
+ // Structure of xmlContent of a word file is simple.
84
+ // All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
85
+ // So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
86
+ // ******************************************************************************************************
87
+ .then(xmlContentArray => {
88
+ /** Store all the text content to respond */
89
+ let responseText = [];
90
+
91
+ xmlContentArray.forEach(xmlContent => {
92
+ /** Find text nodes with w:p tags */
93
+ const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
94
+ /** Store all the text content to respond */
95
+ responseText.push(
96
+ Array.from(xmlParagraphNodesList)
97
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
98
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
99
+ .map(paragraphNode => {
100
+ // Find text nodes with w:t tags
101
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
102
+ // Join the texts within this paragraph node without any spaces or delimiters.
103
+ return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
104
+ })
105
+ // Join each paragraph text with a new line delimiter.
106
+ .join(config.newlineDelimiter ?? "\n")
107
+ );
114
108
  });
115
109
 
116
- returnCallbackPromise
117
- .then(() => callback(responseText.join(" "), undefined));
118
-
110
+ // Join all responseText array
111
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
112
+ // Respond by calling the Callback function.
113
+ callback(responseText, undefined);
119
114
  })
120
- .catch(error => {
121
- consoleError(error)
122
- return callback(undefined, error);
123
- });
115
+ .catch(e => callback(undefined, e));
124
116
  }
125
117
 
126
118
  /** Main function for parsing text from PowerPoint files
127
- * @param {string} filename File path
128
- * @param {function} callback Callback function that returns value or error
129
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
119
+ * @param {string} filepath File path
120
+ * @param {function} callback Callback function that returns value or error
121
+ * @param {OfficeParserConfig} config Config Object for officeParser
130
122
  * @returns {void}
131
123
  */
132
- function parsePowerPoint(filename, callback, deleteOfficeDist = true) {
133
- if (!fs.existsSync(filename)) {
134
- consoleError(ERRORMSG.fileDoesNotExist(filename));
135
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
136
- }
137
- const ext = filename.split(".").pop().toLowerCase();
138
- if (ext != 'pptx') {
139
- consoleError(ERRORMSG.extensionUnsupported(extension));
140
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
141
- }
142
-
143
- /** Store all the text content to respond */
144
- let responseText = [];
145
-
146
- /** Extracting text from powerpoint files xml objects converted to js */
147
- function extractTextFromPowerPointXmlObjects(xmlObjects) {
148
- // specifically for Arrays
149
- if (Array.isArray(xmlObjects)) {
150
- xmlObjects.forEach(item =>
151
- (typeof item == "string") && (item != "")
152
- ? responseText.push(item)
153
- : extractTextFromPowerPointXmlObjects(item))
154
- }
155
- // for other JS Object
156
- else if (typeof xmlObjects == "object") {
157
- for (const [key, value] of Object.entries(xmlObjects)) {
158
- (typeof value == "string") || (typeof value[0] == "string")
159
- ? (key == "a:t" || key == "_") && value != ""
160
- ? responseText.push(value)
161
- : undefined
162
- : extractTextFromPowerPointXmlObjects(value);
163
- }
164
- }
165
- }
166
-
124
+ function parsePowerPoint(filepath, callback, config) {
167
125
  // Files regex that hold our content of interest
168
- const contentFiles = [
169
- /ppt\/slides\/slide\d+.xml/g,
170
- /ppt\/notesSlides\/notesSlide\d+.xml/g
171
- ]
172
-
173
- decompress(filename,
174
- decompressSubLocation,
175
- { filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
126
+ const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
127
+ const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
128
+
129
+ /** The decompress location which contains the filename in it */
130
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
131
+ decompress(filepath,
132
+ decompressLocation,
133
+ { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
176
134
  )
177
135
  .then(files => {
178
- // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
179
- files.sort((a,b) => contentFiles.findIndex(fileRegex => a.path.match(fileRegex)) - contentFiles.findIndex(fileRegex => b.path.match(fileRegex)))
136
+ // Check if files is corrupted
137
+ if (files.length == 0)
138
+ throw ERRORMSG.fileCorrupted(filepath);
180
139
 
181
- if (files.length == 0) {
182
- consoleError(ERRORMSG.fileCorrupted(filename));
183
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
184
- }
140
+ // Check if any sorting is required.
141
+ if (!config.ignoreNotes && config.putNotesAtLast)
142
+ // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
143
+ // For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
144
+ files.sort((a,b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
185
145
 
186
146
  // Returning an array of all the xml contents read using fs.readFileSync
187
- return files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8'))
147
+ return files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
188
148
  })
189
- .then(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent)))) // Returning an array of all parseStringPromise responses
190
- .then(xmlObjectsArray => {
191
- xmlObjectsArray.forEach(xmlObjects => extractTextFromPowerPointXmlObjects(xmlObjects)); // Extracting text from all xml js objects with our conditions
192
-
193
- const returnCallbackPromise = new Promise((res, rej) => {
194
- if (deleteOfficeDist)
195
- rimraf(decompressSubLocation, err => {
196
- if (err)
197
- consoleError(err);
198
- res();
199
- });
200
- else
201
- res();
149
+ // ******************************** powerpoint xml files explanation ************************************
150
+ // Structure of xmlContent of a powerpoint file is simple.
151
+ // There are multiple xml files for each slide and correspondingly their notesSlide files.
152
+ // All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
153
+ // So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
154
+ // ******************************************************************************************************
155
+ .then(xmlContentArray => {
156
+ /** Store all the text content to respond */
157
+ let responseText = [];
158
+
159
+ xmlContentArray.forEach(xmlContent => {
160
+ /** Find text nodes with a:p tags */
161
+ const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
162
+ /** Store all the text content to respond */
163
+ responseText.push(
164
+ Array.from(xmlParagraphNodesList)
165
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
166
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
167
+ .map(paragraphNode => {
168
+ /** Find text nodes with a:t tags */
169
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
170
+ return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
171
+ })
172
+ .join(config.newlineDelimiter ?? "\n")
173
+ );
202
174
  });
203
175
 
204
- returnCallbackPromise
205
- .then(() => callback(responseText.join(" "), undefined));
206
-
176
+ // Join all responseText array
177
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
178
+ // Respond by calling the Callback function.
179
+ callback(responseText, undefined);
207
180
  })
208
- .catch(error => {
209
- consoleError(error)
210
- return callback(undefined, error);
211
- });
181
+ .catch(e => callback(undefined, e));
212
182
  }
213
183
 
214
184
  /** Main function for parsing text from Excel files
215
- * @param {string} filename File path
216
- * @param {function} callback Callback function that returns value or error
217
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
185
+ * @param {string} filepath File path
186
+ * @param {function} callback Callback function that returns value or error
187
+ * @param {OfficeParserConfig} config Config Object for officeParser
218
188
  * @returns {void}
219
189
  */
220
- function parseExcel(filename, callback, deleteOfficeDist = true) {
221
- if (!fs.existsSync(filename)) {
222
- consoleError(ERRORMSG.fileDoesNotExist(filename));
223
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
224
- }
225
- const ext = filename.split(".").pop().toLowerCase();
226
- if (ext != 'xlsx') {
227
- consoleError(ERRORMSG.extensionUnsupported(extension));
228
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
229
- }
230
-
231
- /** Store all the text content to respond */
232
- let responseText = [];
233
-
234
- function extractTextFromExcelXmlObjects2dArray(xmlObjects2dArray) {
235
- xmlObjects2dArray[0].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 0));
236
- xmlObjects2dArray[1].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 1));
237
- xmlObjects2dArray[2].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 2));
238
- }
239
-
240
- /** Extracting text from Excel files xml objects converted to js */
241
- function extractTextFromExcelXmlObjects(xmlObjects, contentFilesIndex) {
242
- switch(contentFilesIndex) {
243
- case 0: { // worksheet
244
- // specifically for Arrays
245
- if (Array.isArray(xmlObjects)) {
246
- xmlObjects.forEach(item =>
247
- item["v"]
248
- ? ((item["$"]["t"] != "s"))
249
- ? responseText.push(item["v"][0])
250
- : undefined
251
- : extractTextFromExcelXmlObjects(item, contentFilesIndex))
252
- }
253
- // for other JS Object
254
- else if (typeof xmlObjects == "object") {
255
- for (const [key, value] of Object.entries(xmlObjects)) {
256
- value["v"]
257
- ? ((value["$"]["t"] == "s"))
258
- ? responseText.push(value["v"][0])
259
- : undefined
260
- : extractTextFromExcelXmlObjects(value, contentFilesIndex);
261
- }
262
- }
263
- break;
264
- }
265
- case 1: { // sharedStrings
266
- // specifically for Arrays
267
- if (Array.isArray(xmlObjects)) {
268
- xmlObjects.forEach(item =>
269
- (typeof item == "string") && (item != "")
270
- ? responseText.push(item)
271
- : extractTextFromExcelXmlObjects(item, contentFilesIndex))
272
- }
273
- // for other JS Object
274
- else if (typeof xmlObjects == "object") {
275
- for (const [key, value] of Object.entries(xmlObjects)) {
276
- (typeof value == "string") || (typeof value[0] == "string")
277
- ? (key == "t" || key == "_") && (value != "")
278
- ? responseText.push(value)
279
- : undefined
280
- : extractTextFromExcelXmlObjects(value, contentFilesIndex);
281
- }
282
- }
283
- break;
284
- }
285
- case 2: { // drawings
286
- // specifically for Arrays
287
- if (Array.isArray(xmlObjects)) {
288
- xmlObjects.forEach(item =>
289
- (typeof item == "string") && (item != "")
290
- ? responseText.push(item)
291
- : extractTextFromExcelXmlObjects(item, contentFilesIndex))
292
- }
293
- // for other JS Object
294
- else if (typeof xmlObjects == "object") {
295
- for (const [key, value] of Object.entries(xmlObjects)) {
296
- (typeof value == "string") || (typeof value[0] == "string")
297
- ? (key == "a:t" || key == "_") && (value != "")
298
- ? responseText.push(value)
299
- : undefined
300
- : extractTextFromExcelXmlObjects(value, contentFilesIndex);
301
- }
302
- }
303
- break;
304
- }
305
- }
306
- }
307
-
190
+ function parseExcel(filepath, callback, config) {
308
191
  // Files regex that hold our content of interest
309
- const contentFiles = [
310
- /xl\/worksheets\/sheet\d+.xml/g,
311
- /xl\/sharedStrings.xml/g,
312
- /xl\/drawings\/drawing\d+.xml/g,
313
- ]
314
-
315
- decompress(filename,
316
- decompressSubLocation,
317
- { filter: x => contentFiles.findIndex(fileRegex => x.path.match(fileRegex)) > -1 }
192
+ const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
193
+ const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
194
+ const chartsRegex = /xl\/charts\/chart\d+.xml/g;
195
+ const stringsFilePath = 'xl/sharedStrings.xml';
196
+
197
+ /** The decompress location which contains the filename in it */
198
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
199
+ decompress(filepath,
200
+ decompressLocation,
201
+ { filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
318
202
  )
319
203
  .then(files => {
320
- // arrange files into 2d array of files organized in contentFiles order, separated by array elements
321
- const files2dArray = [];
322
- contentFiles.forEach(fileRegex => files2dArray.push(files.filter(file => file.path.match(fileRegex))))
323
-
324
- if (files.length == 0) {
325
- consoleError(ERRORMSG.fileCorrupted(filename));
326
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
327
- }
328
-
329
- // Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
330
- return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
204
+ if (files.length == 0)
205
+ throw ERRORMSG.fileCorrupted(filepath);
206
+
207
+ return {
208
+ sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
209
+ drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
210
+ chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
211
+ sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
212
+ };
331
213
  })
332
- .then(xmlContent2dArray => Promise.all(xmlContent2dArray.map(xmlContentArray => Promise.all(xmlContentArray.map(xmlContent => parseStringPromise(xmlContent, false)))))) // Returning a 2dArray of all parseStringPromise responses
333
- .then(xmlObjects2dArray => {
334
- extractTextFromExcelXmlObjects2dArray(xmlObjects2dArray); // Extracting text from all xml js objects with our conditions
335
-
336
- const returnCallbackPromise = new Promise((res, rej) => {
337
- if (deleteOfficeDist)
338
- rimraf(decompressSubLocation, err => {
339
- if (err)
340
- consoleError(err);
341
- res();
342
- });
343
- else
344
- res();
214
+ // ********************************** excel xml files explanation ***************************************
215
+ // Structure of xmlContent of an excel file is a bit complex.
216
+ // We have a sharedStrings.xml file which has strings inside t tags
217
+ // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
218
+ // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
219
+ // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
220
+ // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
221
+ // ******************************************************************************************************
222
+ .then(xmlContentFilesObject => {
223
+ /** Store all the text content to respond */
224
+ let responseText = [];
225
+
226
+ /** Find text nodes with t tags in sharedStrings xml file */
227
+ const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
228
+ /** Create shared string array. This will be used as a map to get strings from within sheet files. */
229
+ const sharedStrings = Array.from(sharedStringsXmlTNodesList).map(tNode => tNode.childNodes[0].nodeValue);
230
+
231
+ // Parse Sheet files
232
+ xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
233
+ /** Find text nodes with c tags in sharedStrings xml file */
234
+ const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
235
+ // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
236
+ responseText.push(
237
+ Array.from(sheetsXmlCNodesList)
238
+ // Filter c nodes than do not have any v nodes
239
+ .filter(cNode => cNode.getElementsByTagName("v").length != 0)
240
+ .map(cNode => {
241
+ /** Flag whether this node's value represents a string index */
242
+ const isString = cNode.getAttribute("t") == "s";
243
+ /** Find value nodes represented by v tags */
244
+ const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
245
+ // Validate text
246
+ if (isString && value >= sharedStrings.length)
247
+ throw ERRORMSG.fileCorrupted(filepath);
248
+
249
+ return isString
250
+ ? sharedStrings[value]
251
+ : value;
252
+ })
253
+ // Join each cell text within a sheet with a space.
254
+ .join(config.newlineDelimiter ?? "\n")
255
+ );
345
256
  });
346
257
 
347
- returnCallbackPromise
348
- .then(() => callback(responseText.join(" "), undefined));
258
+ // Parse Drawing files
259
+ xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
260
+ /** Find text nodes with a:p tags */
261
+ const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
262
+ /** Store all the text content to respond */
263
+ responseText.push(
264
+ Array.from(drawingsXmlParagraphNodesList)
265
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
266
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
267
+ .map(paragraphNode => {
268
+ /** Find text nodes with a:t tags */
269
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
270
+ return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
271
+ })
272
+ .join(config.newlineDelimiter ?? "\n")
273
+ );
274
+ });
275
+
276
+ // Parse Chart files
277
+ xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
278
+ /** Find text nodes with c:v tags */
279
+ const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
280
+ /** Store all the text content to respond */
281
+ responseText.push(
282
+ Array.from(chartsXmlCVNodesList)
283
+ .map(cVNode => cVNode.childNodes[0].nodeValue)
284
+ .join(config.newlineDelimiter ?? "\n")
285
+ );
286
+ });
349
287
 
288
+ // Join all responseText array
289
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
290
+ // Respond by calling the Callback function.
291
+ callback(responseText, undefined);
350
292
  })
351
- .catch(error => {
352
- consoleError(error)
353
- return callback(undefined, error);
354
- });
293
+ .catch(e => callback(undefined, e));
355
294
  }
356
295
 
357
296
 
358
297
  /** Main function for parsing text from open office files
359
- * @param {string} filename File path
360
- * @param {function} callback Callback function that returns value or error
361
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
298
+ * @param {string} filepath File path
299
+ * @param {function} callback Callback function that returns value or error
300
+ * @param {OfficeParserConfig} config Config Object for officeParser
362
301
  * @returns {void}
363
302
  */
364
- function parseOpenOffice(filename, callback, deleteOfficeDist = true) {
365
- if (!fs.existsSync(filename)) {
366
- consoleError(ERRORMSG.fileDoesNotExist(filename));
367
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
368
- }
369
- const ext = filename.split(".").pop().toLowerCase();
370
- if (!["odt", "odp", "ods"].includes(ext)) {
371
- consoleError(ERRORMSG.extensionUnsupported(extension));
372
- return callback(undefined, ERRORMSG.extensionUnsupported(ext));
373
- }
303
+ function parseOpenOffice(filepath, callback, config) {
304
+ /** The target content xml file for the openoffice file. */
305
+ const mainContentFilePath = 'content.xml';
306
+ const objectContentFilesRegex = /Object \d+\/content.xml/g;
307
+
308
+ /** The decompress location which contains the filename in it */
309
+ const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
310
+ decompress(filepath,
311
+ decompressLocation,
312
+ { filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
313
+ )
314
+ .then(files => {
315
+ if (files.length == 0)
316
+ throw ERRORMSG.fileCorrupted(filepath);
374
317
 
375
- /** Store all the text content to respond */
376
- let responseText = [];
377
- /** Extracting text from Open Office files xml objects converted to js */
378
- function extractTextFromOpenOfficeXmlObjects(xmlObjects) {
379
- // specifically for Arrays
380
- if (Array.isArray(xmlObjects)) {
381
- xmlObjects.forEach(item =>
382
- (typeof item == "string") && (item != "")
383
- ? responseText.push(item)
384
- : extractTextFromOpenOfficeXmlObjects(item))
318
+ return {
319
+ mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
320
+ objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
385
321
  }
386
- // for other JS Object
387
- else if (typeof xmlObjects == "object") {
388
- for (const [key, value] of Object.entries(xmlObjects)) {
389
- typeof value == "string"
390
- ? value != ""
391
- ? responseText.push(value)
392
- : undefined
393
- : extractTextFromOpenOfficeXmlObjects(value);
322
+ })
323
+ // ********************************** openoffice xml files explanation **********************************
324
+ // Structure of xmlContent of openoffice files is simple.
325
+ // All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
326
+ // All text in these tags are separated by new line delimiters.
327
+ // Objects like charts in ods files are in Object d+/content.xml with the same way as above.
328
+ // ******************************************************************************************************
329
+ .then(xmlContentFilesObject => {
330
+ /** Store all the notes text content to respond */
331
+ let notesText = [];
332
+ /** Store all the text content to respond */
333
+ let responseText = [];
334
+
335
+ /** List of allowed text tags */
336
+ const allowedTextTags = ["text:p", "text:h"];
337
+ /** List of notes tags */
338
+ const notesTag = "presentation:notes";
339
+
340
+ /** Main dfs traversal function that goes from one node to its children and returns the value out. */
341
+ function extractAllTextsFromNode(root) {
342
+ let xmlTextArray = []
343
+ for (let i = 0; i < root.childNodes.length; i++)
344
+ traversal(root.childNodes[i], xmlTextArray, true);
345
+ return xmlTextArray.join("");
346
+ }
347
+ /** Traversal function that gets recursive calling. */
348
+ function traversal(node, xmlTextArray, isFirstRecursion) {
349
+ if(!node.childNodes || node.childNodes.length == 0)
350
+ {
351
+ if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
352
+ if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
353
+ notesText.push(node.nodeValue);
354
+ if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
355
+ notesText.push(config.newlineDelimiter ?? "\n");
356
+ }
357
+ else {
358
+ xmlTextArray.push(node.nodeValue);
359
+ if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
360
+ xmlTextArray.push(config.newlineDelimiter ?? "\n");
361
+ }
362
+ }
363
+ return;
394
364
  }
365
+
366
+ for (let i = 0; i < node.childNodes.length; i++)
367
+ traversal(node.childNodes[i], xmlTextArray, false);
395
368
  }
396
- }
397
369
 
398
- const contentFile = 'content.xml';
399
- decompress(filename,
400
- decompressSubLocation,
401
- { filter: x => x.path == contentFile }
402
- )
403
- .then(files => {
404
- if (files.length != 1) {
405
- consoleError(ERRORMSG.fileCorrupted(filename));
406
- return callback(undefined, ERRORMSG.fileCorrupted(filename));
370
+ /** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
371
+ function isNotesNode(node) {
372
+ if (node.tagName == notesTag)
373
+ return true;
374
+ if (node.parentNode)
375
+ return isNotesNode(node.parentNode);
376
+ return false;
407
377
  }
408
378
 
409
- return fs.readFileSync(`${decompressSubLocation}/${contentFile}`, 'utf8');
410
- })
411
- .then(xmlContent => parseStringPromise(xmlContent))
412
- .then(xmlObjects => {
413
- extractTextFromOpenOfficeXmlObjects(xmlObjects);
414
- const returnCallbackPromise = new Promise((res, rej) => {
415
- if (deleteOfficeDist)
416
- rimraf(decompressSubLocation, err => {
417
- if (err)
418
- consoleError(err);
419
- res();
420
- });
421
- else
422
- res();
379
+ /** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
380
+ function isInvalidTextNode(node) {
381
+ if (allowedTextTags.includes(node.tagName))
382
+ return true;
383
+ if (node.parentNode)
384
+ return isInvalidTextNode(node.parentNode);
385
+ return false;
386
+ }
387
+
388
+ /** The xml string parsed as xml array */
389
+ const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
390
+ // Iterate over each xmlContent and extract text from them.
391
+ xmlContentArray.forEach(xmlContent => {
392
+ /** Find text nodes with text:h and text:p tags in xmlContent */
393
+ const xmlTextNodesList = [...Array.from(xmlContent
394
+ .getElementsByTagName("*"))
395
+ .filter(node => allowedTextTags.includes(node.tagName)
396
+ && !isInvalidTextNode(node.parentNode))
397
+ ];
398
+ /** Store all the text content to respond */
399
+ responseText.push(
400
+ xmlTextNodesList
401
+ // Add every text information from within this textNode and combine them together.
402
+ .map(textNode => extractAllTextsFromNode(textNode))
403
+ .filter(text => text != "")
404
+ .join(config.newlineDelimiter ?? "\n")
405
+ );
423
406
  });
424
407
 
425
- returnCallbackPromise
426
- .then(() => callback(responseText.join(" "), undefined));
408
+ // Add notes text at the end if the user config says so.
409
+ // Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
410
+ if (!config.ignoreNotes && config.putNotesAtLast)
411
+ responseText = [...responseText, ...notesText];
427
412
 
413
+ // Join all responseText array
414
+ responseText = responseText.join(config.newlineDelimiter ?? "\n");
415
+ // Respond by calling the Callback function.
416
+ callback(responseText, undefined);
428
417
  })
429
- .catch(error => {
430
- consoleError(error)
431
- return callback(undefined, error);
432
- });
433
- }
434
-
435
-
436
- /** Main async function with callback to execute parseOffice for supported files
437
- * @param {string} filename File path
438
- * @param {function} callback Callback function that returns value or error
439
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
440
- * @returns {void}
441
- */
442
- function parseOffice(filename, callback, deleteOfficeDist = true) {
443
- if (!fs.existsSync(filename)) {
444
- consoleError(ERRORMSG.fileDoesNotExist(filename));
445
- return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
446
- }
447
- var extension = filename.split(".").pop().toLowerCase();
448
-
449
- switch(extension)
450
- {
451
- case "docx":
452
- parseWord(filename, (data, err) => callback(data, err), deleteOfficeDist);
453
- return;
454
- case "pptx":
455
- parsePowerPoint(filename, (data, err) => callback(data, err), deleteOfficeDist);
456
- return;
457
- case "xlsx":
458
- parseExcel(filename, (data, err) => callback(data, err), deleteOfficeDist);
459
- return;
460
- case "odt":
461
- case "odp":
462
- case "ods":
463
- parseOpenOffice(filename, (data, err) => callback(data, err), deleteOfficeDist);
464
- return;
465
-
466
- default:
467
- consoleError(ERRORMSG.extensionUnsupported(extension));
468
- callback(undefined, ERRORMSG.extensionUnsupported(extension));
469
- }
418
+ .catch(e => callback(undefined, e));
470
419
  }
471
420
 
472
- /**
473
- * Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
474
- * @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
475
- * @returns {void}
476
- */
477
- function setDecompressionLocation(newLocation) {
478
- if (newLocation != undefined) {
479
- newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
480
-
481
- if (fs.existsSync(newLocation))
482
- decompressSubLocation = newLocation;
483
- return;
484
- }
485
- consoleError(ERRORMSG.locationNotFound(newLocation));
486
- decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
487
- }
421
+ /** Header for error messages */
422
+ const PDFPARSEERRORHEADER = "[pdf-parse]: ";
488
423
 
489
- /** Enable console output
424
+ /** Main function for parsing text from pdf files
425
+ * @param {string} filepath File path
426
+ * @param {function} callback Callback function that returns value or error
427
+ * @param {OfficeParserConfig} config Config Object for officeParser
490
428
  * @returns {void}
491
429
  */
492
- function enableConsoleOutput() {
493
- outputErrorToConsole = true;
430
+ function parsePdf(filepath, callback, config) {
431
+ // Get the data buffer for the given file path.
432
+ const dataBuffer = fs.readFileSync(filepath);
433
+
434
+ pdfParse(dataBuffer)
435
+ .then(data => {
436
+ let text = data.text;
437
+ if (!!config.newlineDelimiter && config.newlineDelimiter != "\n")
438
+ text = text.replaceAll("\n", config.newlineDelimiter)
439
+ callback(text, undefined);
440
+ })
441
+ .catch(e => callback(undefined, PDFPARSEERRORHEADER + e));
494
442
  }
495
443
 
496
- /** Disabled console output
444
+ /** Main async function with callback to execute parseOffice for supported files
445
+ * @param {string | Buffer} file File path or file buffers
446
+ * @param {function} callback Callback function that returns value or error
447
+ * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
497
448
  * @returns {void}
498
449
  */
499
- function disableConsoleOutput() {
500
- outputErrorToConsole = false;
501
- }
450
+ function parseOffice(file, callback, config = {}) {
451
+ // Prepare file for processing
452
+ const filePreparedPromise = new Promise((res, rej) => {
453
+ // create temp file subdirectory if it does not exist
454
+ fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
455
+
456
+ // Check if buffer
457
+ if (Buffer.isBuffer(file)) {
458
+ // Guess file type from buffer
459
+ fileType.fromBuffer(file)
460
+ .then(data =>
461
+ {
462
+ // temp file name
463
+ const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
464
+ // write new file
465
+ fs.writeFileSync(newfilepath, file);
466
+ // resolve promise
467
+ res(newfilepath);
468
+ })
469
+ .catch(() => rej(ERRORMSG.improperBuffers));
470
+ return;
471
+ }
502
472
 
473
+ // Not buffers but real file path.
503
474
 
504
- // #region Promise versions of above functions
475
+ // Check if file exists
476
+ if (!fs.existsSync(file))
477
+ throw ERRORMSG.fileDoesNotExist(file);
505
478
 
506
- /** Async function that can be used with await to execute parseWord. Or it can be used with promises.
507
- * @param {string} filename File path
508
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
509
- * @returns {Promise<string>}
510
- */
511
- var parseWordAsync = function (filename, deleteOfficeDist = true) {
512
- return new Promise((resolve, reject) => {
513
- try {
514
- parseWord(filename, function (data, error) {
515
- if (error)
516
- return reject(error);
517
- return resolve(data);
518
- }, deleteOfficeDist);
519
- }
520
- catch (error) {
521
- return reject(error);
522
- }
523
- })
524
- }
479
+ // temp file name
480
+ const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
481
+ // Copy the file into a temp location with the temp name
482
+ fs.copyFileSync(file, newfilepath)
483
+ // resolve promise
484
+ res(newfilepath);
485
+ });
525
486
 
526
- /** Async function that can be used with await to execute parsePowerPoint. Or it can be used with promises.
527
- * @param {string} filename File path
528
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
529
- * @returns {Promise<string>}
530
- */
531
- var parsePowerPointAsync = function (filename, deleteOfficeDist = true) {
532
- return new Promise((resolve, reject) => {
533
- try {
534
- parsePowerPoint(filename, function (data, err) {
535
- if (err)
536
- return reject(err);
537
- return resolve(data);
538
- }, deleteOfficeDist);
539
- }
540
- catch (error) {
541
- return reject(error);
542
- }
543
- })
544
- }
487
+ // Process filePreparedPromise resolution.
488
+ filePreparedPromise
489
+ .then(filepath => {
490
+ // File extension. Already in lowercase when we prepared the temp file above.
491
+ const extension = filepath.split(".").pop();
492
+
493
+ // Switch between parsing functions depending on extension.
494
+ switch(extension) {
495
+ case "docx":
496
+ parseWord(filepath, internalCallback, config);
497
+ break;
498
+ case "pptx":
499
+ parsePowerPoint(filepath, internalCallback, config);
500
+ break;
501
+ case "xlsx":
502
+ parseExcel(filepath, internalCallback, config);
503
+ break;
504
+ case "odt":
505
+ case "odp":
506
+ case "ods":
507
+ parseOpenOffice(filepath, internalCallback, config);
508
+ break;
509
+ case "pdf":
510
+ parsePdf(filepath, internalCallback, config);
511
+ break;
512
+
513
+ default:
514
+ throw ERRORMSG.extensionUnsupported(extension);
515
+ }
545
516
 
546
- /** Async function that can be used with await to execute parseExcel. Or it can be used with promises.
547
- * @param {string} filename File path
548
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
549
- * @returns {Promise<string>}
550
- */
551
- var parseExcelAsync = function (filename, deleteOfficeDist = true) {
552
- return new Promise((resolve, reject) => {
553
- try {
554
- parseExcel(filename, function (data, err) {
517
+ /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
518
+ function internalCallback(data, err) {
555
519
  if (err)
556
- return reject(err);
557
- return resolve(data);
558
- }, deleteOfficeDist);
559
- }
560
- catch (error) {
561
- return reject(error);
562
- }
563
- })
520
+ consoleError(err, config.outputErrorToConsole)
521
+ // Call the original callback
522
+ callback(data, err);
523
+ // Check if we need to preserve unzipped content files or delete them.
524
+ if (config.preserveTempFiles)
525
+ return;
526
+ // Delete decompress sublocation.
527
+ rimraf(decompressSubLocation, rimrafErr => consoleError(rimrafErr, config.outputErrorToConsole));
528
+ }
529
+ })
530
+ .catch(error => {
531
+ consoleError(error, config.outputErrorToConsole);
532
+ callback(undefined, error);
533
+ });
564
534
  }
565
535
 
566
- /** Async function that can be used with await to execute parseOpenOffice. Or it can be used with promises.
567
- * @param {string} filename File path
568
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
536
+ /**
537
+ * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
538
+ * @param {string | Buffer} file File path or file buffers
539
+ * @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
569
540
  * @returns {Promise<string>}
570
541
  */
571
- var parseOpenOfficeAsync = function (filename, deleteOfficeDist = true) {
572
- return new Promise((resolve, reject) => {
573
- try {
574
- parseOpenOffice(filename, function (data, err) {
575
- if (err)
576
- return reject(err);
577
- return resolve(data);
578
- }, deleteOfficeDist);
579
- }
580
- catch (error) {
581
- return reject(error);
582
- }
583
- })
542
+ function parseOfficeAsync (file, config) {
543
+ return new Promise((res, rej) => {
544
+ parseOffice(file, function (data, err) {
545
+ if (err)
546
+ return rej(err);
547
+ return res(data);
548
+ }, config);
549
+ });
584
550
  }
585
551
 
586
552
  /**
587
- * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
588
- * @param {string} filename File path
589
- * @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
590
- * @returns {Promise<string>}
553
+ * Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
554
+ * @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
555
+ * @returns {void}
591
556
  */
592
- var parseOfficeAsync = function (filename, deleteOfficeDist = true) {
593
- return new Promise((resolve, reject) => {
594
- try {
595
- parseOffice(filename, function (data, err) {
596
- if (err)
597
- return reject(err);
598
- return resolve(data);
599
- }, deleteOfficeDist);
600
- }
601
- catch (error) {
602
- return reject(error);
603
- }
604
- })
557
+ function setDecompressionLocation(newLocation) {
558
+ if (newLocation != undefined) {
559
+ newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
560
+ if (fs.existsSync(newLocation))
561
+ decompressSubLocation = newLocation;
562
+ return;
563
+ }
564
+ consoleError(ERRORMSG.locationNotFound(newLocation), config.outputErrorToConsole);
565
+ decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
605
566
  }
606
- // #endregion Async Versions
607
-
608
- module.exports.parseWord = parseWord;
609
- module.exports.parsePowerPoint = parsePowerPoint;
610
- module.exports.parseExcel = parseExcel;
611
- module.exports.parseOpenOffice = parseOpenOffice;
612
- module.exports.parseOffice = parseOffice;
613
- module.exports.parseWordAsync = parseWordAsync;
614
- module.exports.parsePowerPointAsync = parsePowerPointAsync;
615
- module.exports.parseExcelAsync = parseExcelAsync;
616
- module.exports.parseOpenOfficeAsync = parseOpenOfficeAsync;
617
- module.exports.parseOfficeAsync = parseOfficeAsync;
567
+
568
+ // Export functions
569
+ module.exports.parseOffice = parseOffice;
570
+ module.exports.parseOfficeAsync = parseOfficeAsync;
618
571
  module.exports.setDecompressionLocation = setDecompressionLocation;
619
- module.exports.enableConsoleOutput = enableConsoleOutput;
620
- module.exports.disableConsoleOutput = disableConsoleOutput;
621
572
 
622
573
 
623
574
  // Run this library on CLI
@@ -627,8 +578,8 @@ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').po
627
578
  }
628
579
  else if (process.argv.length == 3)
629
580
  parseOfficeAsync(process.argv[2])
630
- .then(text => console.log(text))
631
- .catch(error => console.error(error))
581
+ .then(text => console.log(text))
582
+ .catch(error => console.error(ERRORHEADER + error))
632
583
  else
633
584
  console.error(ERRORMSG.improperArguments)
634
585
  }