officeparser 4.1.2 → 5.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js CHANGED
@@ -1,11 +1,13 @@
1
1
  #!/usr/bin/env node
2
2
 
3
- const decompress = require('decompress');
4
- const fs = require('fs');
5
- const rimraf = require('rimraf');
3
+ // @ts-check
4
+
5
+ const concat = require('concat-stream');
6
+ const { DOMParser } = require('@xmldom/xmldom');
6
7
  const fileType = require('file-type');
8
+ const fs = require('fs');
7
9
  const pdfjs = require('./pdfjs-dist-build/pdf.js');
8
- const { DOMParser } = require('@xmldom/xmldom');
10
+ const yauzl = require('yauzl');
9
11
 
10
12
  /** Header for error messages */
11
13
  const ERRORHEADER = "[OfficeParser]: ";
@@ -16,20 +18,8 @@ const ERRORMSG = {
16
18
  fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
17
19
  locationNotFound: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
18
20
  improperArguments: `Improper arguments`,
19
- improperBuffers: `Error occured while reading the file buffers`
20
- }
21
- /** Default sublocation for decompressing files under the current directory. */
22
- const DEFAULTDECOMPRESSSUBLOCATION = "officeParserTemp";
23
-
24
- /** Console error if allowed
25
- * @param {string} errorMessage Error message to show on the console
26
- * @param {string} outputErrorToConsole Flag to show log on console. Ignore if not true.
27
- * @returns {void}
28
- */
29
- function consoleError(errorMessage, outputErrorToConsole) {
30
- if (!errorMessage || !outputErrorToConsole)
31
- return;
32
- console.error(ERRORHEADER + errorMessage);
21
+ improperBuffers: `Error occured while reading the file buffers`,
22
+ invalidInput: `Invalid input type: Expected a Buffer or a valid file path`
33
23
  }
34
24
 
35
25
  /** Returns parsed xml document for a given xml text.
@@ -42,9 +32,7 @@ const parseString = (xml) => {
42
32
  };
43
33
 
44
34
  /** @typedef {Object} OfficeParserConfig
45
- * @property {string} [tempFilesLocation] The directory where officeparser stores the temp files . The final decompressed data will be put inside officeParserTemp folder within your directory. Please ensure that this directory actually exists. Default is officeParsertemp.
46
- * @property {boolean} [preserveTempFiles] Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
47
- * @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling.
35
+ * @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling. Default is false.
48
36
  * @property {string} [newlineDelimiter] The delimiter used for every new line in places that allow multiline text like word. Default is \n.
49
37
  * @property {boolean} [ignoreNotes] Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
50
38
  * @property {boolean} [putNotesAtLast] Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
@@ -52,469 +40,442 @@ const parseString = (xml) => {
52
40
 
53
41
 
54
42
  /** Main function for parsing text from word files
55
- * @param {string} filepath File path
43
+ * @param {string | Buffer} file File path or Buffers
56
44
  * @param {function} callback Callback function that returns value or error
57
45
  * @param {OfficeParserConfig} config Config Object for officeParser
58
46
  * @returns {void}
59
47
  */
60
- function parseWord(filepath, callback, config) {
48
+ function parseWord(file, callback, config) {
61
49
  /** The target content xml file for the docx file. */
62
50
  const mainContentFileRegex = /word\/document[\d+]?.xml/g;
63
51
  const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
64
52
  const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
65
- /** The decompress location which contains the filename in it */
66
- const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
67
- decompress(filepath,
68
- decompressLocation,
69
- { filter: x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.path.match(fileRegex)) }
70
- )
71
- .then(files => {
72
- // Verify if atleast the document xml file exists in the extracted files list.
73
- if (!files.some(file => file.path.match(mainContentFileRegex)))
74
- throw ERRORMSG.fileCorrupted(filepath);
53
+
54
+ extractFiles(file, x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.match(fileRegex)))
55
+ .then(files => {
56
+ // Verify if atleast the document xml file exists in the extracted files list.
57
+ if (!files.some(file => file.path.match(mainContentFileRegex)))
58
+ throw ERRORMSG.fileCorrupted(file);
75
59
 
76
60
  return files
77
61
  .filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
78
- .map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
79
- })
80
- // ************************************* word xml files explanation *************************************
81
- // Structure of xmlContent of a word file is simple.
82
- // All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
83
- // So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
84
- // ******************************************************************************************************
85
- .then(xmlContentArray => {
86
- /** Store all the text content to respond */
87
- let responseText = [];
88
-
89
- xmlContentArray.forEach(xmlContent => {
90
- /** Find text nodes with w:p tags */
91
- const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
92
- /** Store all the text content to respond */
93
- responseText.push(
94
- Array.from(xmlParagraphNodesList)
95
- // Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
96
- .filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
97
- .map(paragraphNode => {
98
- // Find text nodes with w:t tags
99
- const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
100
- // Join the texts within this paragraph node without any spaces or delimiters.
101
- return Array.from(xmlTextNodeList)
102
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
103
- .map(textNode => textNode.childNodes[0].nodeValue)
104
- .join("");
105
- })
106
- // Join each paragraph text with a new line delimiter.
107
- .join(config.newlineDelimiter ?? "\n")
108
- );
109
- });
110
-
111
- // Join all responseText array
112
- responseText = responseText.join(config.newlineDelimiter ?? "\n");
113
- // Respond by calling the Callback function.
114
- callback(responseText, undefined);
115
- })
116
- .catch(e => callback(undefined, e));
62
+ .map(file => file.content);
63
+ })
64
+ // ************************************* word xml files explanation *************************************
65
+ // Structure of xmlContent of a word file is simple.
66
+ // All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
67
+ // So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
68
+ // ******************************************************************************************************
69
+ .then(xmlContentArray => {
70
+ /** Store all the text content to respond. */
71
+ let responseText = [];
72
+
73
+ xmlContentArray.forEach(xmlContent => {
74
+ /** Find text nodes with w:p tags */
75
+ const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
76
+ /** Store all the text content to respond */
77
+ responseText.push(
78
+ Array.from(xmlParagraphNodesList)
79
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
80
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
81
+ .map(paragraphNode => {
82
+ // Find text nodes with w:t tags
83
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
84
+ // Join the texts within this paragraph node without any spaces or delimiters.
85
+ return Array.from(xmlTextNodeList)
86
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
87
+ .map(textNode => textNode.childNodes[0].nodeValue)
88
+ .join("");
89
+ })
90
+ // Join each paragraph text with a new line delimiter.
91
+ .join(config.newlineDelimiter ?? "\n")
92
+ );
93
+ });
94
+
95
+ // Respond by calling the Callback function.
96
+ callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
97
+ })
98
+ .catch(e => callback(undefined, e));
117
99
  }
118
100
 
119
101
  /** Main function for parsing text from PowerPoint files
120
- * @param {string} filepath File path
102
+ * @param {string | Buffer} file File path or Buffers
121
103
  * @param {function} callback Callback function that returns value or error
122
104
  * @param {OfficeParserConfig} config Config Object for officeParser
123
105
  * @returns {void}
124
106
  */
125
- function parsePowerPoint(filepath, callback, config) {
107
+ function parsePowerPoint(file, callback, config) {
126
108
  // Files regex that hold our content of interest
127
109
  const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
128
110
  const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
129
111
  const slideNumberRegex = /lide(\d+)\.xml/;
130
112
 
131
- /** The decompress location which contains the filename in it */
132
- const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
133
- decompress(filepath,
134
- decompressLocation,
135
- { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
136
- )
137
- .then(files => {
138
- // Sort files by slide number and their notes (if any).
139
- files.sort((a, b) => {
140
- const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
141
- const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
142
-
143
- const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
144
- const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
145
-
146
- return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
147
- });
148
-
149
- // Verify if atleast the slides xml files exist in the extracted files list.
150
- if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
151
- throw ERRORMSG.fileCorrupted(filepath);
152
-
153
- // Check if any sorting is required.
154
- if (!config.ignoreNotes && config.putNotesAtLast)
155
- // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
156
- // For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
157
- files.sort((a,b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
158
-
159
- // Returning an array of all the xml contents read using fs.readFileSync
160
- return files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
161
- })
162
- // ******************************** powerpoint xml files explanation ************************************
163
- // Structure of xmlContent of a powerpoint file is simple.
164
- // There are multiple xml files for each slide and correspondingly their notesSlide files.
165
- // All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
166
- // So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
167
- // ******************************************************************************************************
168
- .then(xmlContentArray => {
169
- /** Store all the text content to respond */
170
- let responseText = [];
171
-
172
- xmlContentArray.forEach(xmlContent => {
173
- /** Find text nodes with a:p tags */
174
- const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
113
+ extractFiles(file, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex))
114
+ .then(files => {
115
+ // Sort files by slide number and their notes (if any).
116
+ files.sort((a, b) => {
117
+ const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
118
+ const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
119
+
120
+ const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
121
+ const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
122
+
123
+ return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
124
+ });
125
+
126
+ // Verify if atleast the slides xml files exist in the extracted files list.
127
+ if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
128
+ throw ERRORMSG.fileCorrupted(file);
129
+
130
+ // Check if any sorting is required.
131
+ if (!config.ignoreNotes && config.putNotesAtLast)
132
+ // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
133
+ // For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
134
+ files.sort((a, b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
135
+
136
+ // Returning an array of all the xml contents read using fs.readFileSync
137
+ return files.map(file => file.content);
138
+ })
139
+ // ******************************** powerpoint xml files explanation ************************************
140
+ // Structure of xmlContent of a powerpoint file is simple.
141
+ // There are multiple xml files for each slide and correspondingly their notesSlide files.
142
+ // All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
143
+ // So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
144
+ // ******************************************************************************************************
145
+ .then(xmlContentArray => {
175
146
  /** Store all the text content to respond */
176
- responseText.push(
177
- Array.from(xmlParagraphNodesList)
178
- // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
179
- .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
180
- .map(paragraphNode => {
181
- /** Find text nodes with a:t tags */
182
- const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
183
- return Array.from(xmlTextNodeList)
184
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
185
- .map(textNode => textNode.childNodes[0].nodeValue)
186
- .join("");
187
- })
188
- .join(config.newlineDelimiter ?? "\n")
189
- );
190
- });
191
-
192
- // Join all responseText array
193
- responseText = responseText.join(config.newlineDelimiter ?? "\n");
194
- // Respond by calling the Callback function.
195
- callback(responseText, undefined);
196
- })
197
- .catch(e => callback(undefined, e));
147
+ let responseText = [];
148
+
149
+ xmlContentArray.forEach(xmlContent => {
150
+ /** Find text nodes with a:p tags */
151
+ const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
152
+ /** Store all the text content to respond */
153
+ responseText.push(
154
+ Array.from(xmlParagraphNodesList)
155
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
156
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
157
+ .map(paragraphNode => {
158
+ /** Find text nodes with a:t tags */
159
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
160
+ return Array.from(xmlTextNodeList)
161
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
162
+ .map(textNode => textNode.childNodes[0].nodeValue)
163
+ .join("");
164
+ })
165
+ .join(config.newlineDelimiter ?? "\n")
166
+ );
167
+ });
168
+
169
+ // Respond by calling the Callback function.
170
+ callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
171
+ })
172
+ .catch(e => callback(undefined, e));
198
173
  }
199
174
 
200
175
  /** Main function for parsing text from Excel files
201
- * @param {string} filepath File path
176
+ * @param {string | Buffer} file File path or Buffers
202
177
  * @param {function} callback Callback function that returns value or error
203
178
  * @param {OfficeParserConfig} config Config Object for officeParser
204
179
  * @returns {void}
205
180
  */
206
- function parseExcel(filepath, callback, config) {
181
+ function parseExcel(file, callback, config) {
207
182
  // Files regex that hold our content of interest
208
183
  const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
209
184
  const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
210
185
  const chartsRegex = /xl\/charts\/chart\d+.xml/g;
211
186
  const stringsFilePath = 'xl/sharedStrings.xml';
212
187
 
213
- /** The decompress location which contains the filename in it */
214
- const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
215
- decompress(filepath,
216
- decompressLocation,
217
- { filter: x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.path.match(fileRegex)) || x.path == stringsFilePath }
218
- )
219
- .then(files => {
220
- // Verify if atleast the slides xml files exist in the extracted files list.
221
- if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
222
- throw ERRORMSG.fileCorrupted(filepath);
223
-
224
- return {
225
- sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
226
- drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
227
- chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
228
- sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
229
- };
230
- })
231
- // ********************************** excel xml files explanation ***************************************
232
- // Structure of xmlContent of an excel file is a bit complex.
233
- // We usually have a sharedStrings.xml file which has strings inside t tags
234
- // However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
235
- // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
236
- // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
237
- // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
238
- // However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
239
- // We extract either the inline strings or use the value to get numbers of text from shared strings.
240
- // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
241
- // ******************************************************************************************************
242
- .then(xmlContentFilesObject => {
243
- /** Store all the text content to respond */
244
- let responseText = [];
245
-
246
- /** Function to check if the given c node is a valid inline string node. */
247
- function isValidInlineStringCNode(cNode) {
248
- // Initial check to see if the passed node is a cNode
249
- if (cNode.tagName.toLowerCase() != 'c')
250
- return false;
251
- if (cNode.getAttribute("t") != 'inlineStr')
252
- return false;
253
- const childNodesNamedIs = cNode.getElementsByTagName('is');
254
- if (childNodesNamedIs.length != 1)
255
- return false;
256
- const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
257
- if (childNodesNamedT.length != 1)
258
- return false;
259
- return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
260
- }
188
+ extractFiles(file, x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.match(fileRegex)) || x == stringsFilePath)
189
+ .then(files => {
190
+ // Verify if atleast the slides xml files exist in the extracted files list.
191
+ if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
192
+ throw ERRORMSG.fileCorrupted(file);
193
+
194
+ return {
195
+ sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => file.content),
196
+ drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => file.content),
197
+ chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => file.content),
198
+ sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => file.content)[0],
199
+ };
200
+ })
201
+ // ********************************** excel xml files explanation ***************************************
202
+ // Structure of xmlContent of an excel file is a bit complex.
203
+ // We usually have a sharedStrings.xml file which has strings inside t tags
204
+ // However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
205
+ // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
206
+ // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
207
+ // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
208
+ // However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
209
+ // We extract either the inline strings or use the value to get numbers of text from shared strings.
210
+ // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
211
+ // ******************************************************************************************************
212
+ .then(xmlContentFilesObject => {
213
+ /** Store all the text content to respond */
214
+ let responseText = [];
215
+
216
+ /** Function to check if the given c node is a valid inline string node. */
217
+ function isValidInlineStringCNode(cNode) {
218
+ // Initial check to see if the passed node is a cNode
219
+ if (cNode.tagName.toLowerCase() != 'c')
220
+ return false;
221
+ if (cNode.getAttribute("t") != 'inlineStr')
222
+ return false;
223
+ const childNodesNamedIs = cNode.getElementsByTagName('is');
224
+ if (childNodesNamedIs.length != 1)
225
+ return false;
226
+ const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
227
+ if (childNodesNamedT.length != 1)
228
+ return false;
229
+ return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
230
+ }
261
231
 
262
- /** Function to check if the given c node has a valid v node */
263
- function hasValidVNodeInCNode(cNode) {
264
- return cNode.getElementsByTagName("v")[0]
265
- && cNode.getElementsByTagName("v")[0].childNodes[0]
266
- && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
267
- }
232
+ /** Function to check if the given c node has a valid v node */
233
+ function hasValidVNodeInCNode(cNode) {
234
+ return cNode.getElementsByTagName("v")[0]
235
+ && cNode.getElementsByTagName("v")[0].childNodes[0]
236
+ && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
237
+ }
268
238
 
269
- /** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
270
- const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
271
- : [];
272
- /** Create shared string array. This will be used as a map to get strings from within sheet files. */
273
- const sharedStrings = Array.from(sharedStringsXmlTNodesList)
274
- .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
275
-
276
- // Parse Sheet files
277
- xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
278
- /** Find text nodes with c tags in sharedStrings xml file */
279
- const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
280
- // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
281
- responseText.push(
282
- Array.from(sheetsXmlCNodesList)
283
- // Filter out invalid c nodes
284
- .filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
285
- .map(cNode => {
286
- // Processing if this is a valid inline string c node.
287
- if (isValidInlineStringCNode(cNode))
288
- return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
289
-
290
- // Processing if this c node has a valid v node.
291
- if (hasValidVNodeInCNode(cNode)) {
292
- /** Flag whether this node's value represents an index in the shared string array */
293
- const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
294
- /** Find value nodes represented by v tags */
295
- const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
296
- // Validate text
297
- if (isIndexInSharedStrings && value >= sharedStrings.length)
298
- throw ERRORMSG.fileCorrupted(filepath);
299
-
300
- return isIndexInSharedStrings
301
- ? sharedStrings[value]
302
- : value;
303
- }
304
- // TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
305
- // Not the case now but it could happen and it is better to be safe.
306
- return '';
307
- })
308
- // Join each cell text within a sheet with a space.
309
- .join(config.newlineDelimiter ?? "\n")
310
- );
311
- });
312
-
313
- // Parse Drawing files
314
- xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
315
- /** Find text nodes with a:p tags */
316
- const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
317
- /** Store all the text content to respond */
318
- responseText.push(
319
- Array.from(drawingsXmlParagraphNodesList)
320
- // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
321
- .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
322
- .map(paragraphNode => {
323
- /** Find text nodes with a:t tags */
324
- const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
325
- return Array.from(xmlTextNodeList)
326
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
327
- .map(textNode => textNode.childNodes[0].nodeValue)
328
- .join("");
329
- })
330
- .join(config.newlineDelimiter ?? "\n")
331
- );
332
- });
333
-
334
- // Parse Chart files
335
- xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
336
- /** Find text nodes with c:v tags */
337
- const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
338
- /** Store all the text content to respond */
339
- responseText.push(
340
- Array.from(chartsXmlCVNodesList)
341
- .filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
342
- .map(cVNode => cVNode.childNodes[0].nodeValue)
343
- .join(config.newlineDelimiter ?? "\n")
344
- );
345
- });
346
-
347
- // Join all responseText array
348
- responseText = responseText.join(config.newlineDelimiter ?? "\n");
349
- // Respond by calling the Callback function.
350
- callback(responseText, undefined);
351
- })
352
- .catch(e => callback(undefined, e));
239
+ /** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
240
+ const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
241
+ : [];
242
+ /** Create shared string array. This will be used as a map to get strings from within sheet files. */
243
+ const sharedStrings = Array.from(sharedStringsXmlTNodesList)
244
+ .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
245
+
246
+ // Parse Sheet files
247
+ xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
248
+ /** Find text nodes with c tags in sharedStrings xml file */
249
+ const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
250
+ // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
251
+ responseText.push(
252
+ Array.from(sheetsXmlCNodesList)
253
+ // Filter out invalid c nodes
254
+ .filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
255
+ .map(cNode => {
256
+ // Processing if this is a valid inline string c node.
257
+ if (isValidInlineStringCNode(cNode))
258
+ return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
259
+
260
+ // Processing if this c node has a valid v node.
261
+ if (hasValidVNodeInCNode(cNode)) {
262
+ /** Flag whether this node's value represents an index in the shared string array */
263
+ const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
264
+ /** Find value nodes represented by v tags */
265
+ const value = parseInt(cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue, 10);
266
+ // Validate text
267
+ if (isIndexInSharedStrings && value >= sharedStrings.length)
268
+ throw ERRORMSG.fileCorrupted(file);
269
+
270
+ return isIndexInSharedStrings
271
+ ? sharedStrings[value]
272
+ : value;
273
+ }
274
+ // TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
275
+ // Not the case now but it could happen and it is better to be safe.
276
+ return '';
277
+ })
278
+ // Join each cell text within a sheet with a space.
279
+ .join(config.newlineDelimiter ?? "\n")
280
+ );
281
+ });
282
+
283
+ // Parse Drawing files
284
+ xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
285
+ /** Find text nodes with a:p tags */
286
+ const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
287
+ /** Store all the text content to respond */
288
+ responseText.push(
289
+ Array.from(drawingsXmlParagraphNodesList)
290
+ // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
291
+ .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
292
+ .map(paragraphNode => {
293
+ /** Find text nodes with a:t tags */
294
+ const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
295
+ return Array.from(xmlTextNodeList)
296
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
297
+ .map(textNode => textNode.childNodes[0].nodeValue)
298
+ .join("");
299
+ })
300
+ .join(config.newlineDelimiter ?? "\n")
301
+ );
302
+ });
303
+
304
+ // Parse Chart files
305
+ xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
306
+ /** Find text nodes with c:v tags */
307
+ const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
308
+ /** Store all the text content to respond */
309
+ responseText.push(
310
+ Array.from(chartsXmlCVNodesList)
311
+ .filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
312
+ .map(cVNode => cVNode.childNodes[0].nodeValue)
313
+ .join(config.newlineDelimiter ?? "\n")
314
+ );
315
+ });
316
+
317
+ // Respond by calling the Callback function.
318
+ callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
319
+ })
320
+ .catch(e => callback(undefined, e));
353
321
  }
354
322
 
355
323
 
356
324
  /** Main function for parsing text from open office files
357
- * @param {string} filepath File path
325
+ * @param {string | Buffer} file File path or Buffers
358
326
  * @param {function} callback Callback function that returns value or error
359
327
  * @param {OfficeParserConfig} config Config Object for officeParser
360
328
  * @returns {void}
361
329
  */
362
- function parseOpenOffice(filepath, callback, config) {
330
+ function parseOpenOffice(file, callback, config) {
363
331
  /** The target content xml file for the openoffice file. */
364
332
  const mainContentFilePath = 'content.xml';
365
333
  const objectContentFilesRegex = /Object \d+\/content.xml/g;
366
334
 
367
- /** The decompress location which contains the filename in it */
368
- const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
369
- decompress(filepath,
370
- decompressLocation,
371
- { filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
372
- )
373
- .then(files => {
374
- // Verify if atleast the content xml file exists in the extracted files list.
375
- if (!files.map(file => file.path).includes(mainContentFilePath))
376
- throw ERRORMSG.fileCorrupted(filepath);
377
-
378
- return {
379
- mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
380
- objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
381
- }
382
- })
383
- // ********************************** openoffice xml files explanation **********************************
384
- // Structure of xmlContent of openoffice files is simple.
385
- // All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
386
- // All text in these tags are separated by new line delimiters.
387
- // Objects like charts in ods files are in Object d+/content.xml with the same way as above.
388
- // ******************************************************************************************************
389
- .then(xmlContentFilesObject => {
390
- /** Store all the notes text content to respond */
391
- let notesText = [];
392
- /** Store all the text content to respond */
393
- let responseText = [];
394
-
395
- /** List of allowed text tags */
396
- const allowedTextTags = ["text:p", "text:h"];
397
- /** List of notes tags */
398
- const notesTag = "presentation:notes";
399
-
400
- /** Main dfs traversal function that goes from one node to its children and returns the value out. */
401
- function extractAllTextsFromNode(root) {
402
- let xmlTextArray = []
403
- for (let i = 0; i < root.childNodes.length; i++)
404
- traversal(root.childNodes[i], xmlTextArray, true);
405
- return xmlTextArray.join("");
406
- }
407
- /** Traversal function that gets recursive calling. */
408
- function traversal(node, xmlTextArray, isFirstRecursion) {
409
- if(!node.childNodes || node.childNodes.length == 0)
410
- {
411
- if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
412
- if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
413
- notesText.push(node.nodeValue);
414
- if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
415
- notesText.push(config.newlineDelimiter ?? "\n");
416
- }
417
- else {
418
- xmlTextArray.push(node.nodeValue);
419
- if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
420
- xmlTextArray.push(config.newlineDelimiter ?? "\n");
335
+ extractFiles(file, x => x == mainContentFilePath || !!x.match(objectContentFilesRegex))
336
+ .then(files => {
337
+ // Verify if atleast the content xml file exists in the extracted files list.
338
+ if (!files.map(file => file.path).includes(mainContentFilePath))
339
+ throw ERRORMSG.fileCorrupted(file);
340
+
341
+ return {
342
+ mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => file.content)[0],
343
+ objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => file.content),
344
+ }
345
+ })
346
+ // ********************************** openoffice xml files explanation **********************************
347
+ // Structure of xmlContent of openoffice files is simple.
348
+ // All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
349
+ // All text in these tags are separated by new line delimiters.
350
+ // Objects like charts in ods files are in Object d+/content.xml with the same way as above.
351
+ // ******************************************************************************************************
352
+ .then(xmlContentFilesObject => {
353
+ /** Store all the notes text content to respond */
354
+ let notesText = [];
355
+ /** Store all the text content to respond */
356
+ let responseText = [];
357
+
358
+ /** List of allowed text tags */
359
+ const allowedTextTags = ["text:p", "text:h"];
360
+ /** List of notes tags */
361
+ const notesTag = "presentation:notes";
362
+
363
+ /** Main dfs traversal function that goes from one node to its children and returns the value out. */
364
+ function extractAllTextsFromNode(root) {
365
+ let xmlTextArray = []
366
+ for (let i = 0; i < root.childNodes.length; i++)
367
+ traversal(root.childNodes[i], xmlTextArray, true);
368
+ return xmlTextArray.join("");
369
+ }
370
+ /** Traversal function that gets recursive calling. */
371
+ function traversal(node, xmlTextArray, isFirstRecursion) {
372
+ if (!node.childNodes || node.childNodes.length == 0) {
373
+ if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
374
+ if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
375
+ notesText.push(node.nodeValue);
376
+ if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
377
+ notesText.push(config.newlineDelimiter ?? "\n");
378
+ }
379
+ else {
380
+ xmlTextArray.push(node.nodeValue);
381
+ if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
382
+ xmlTextArray.push(config.newlineDelimiter ?? "\n");
383
+ }
421
384
  }
385
+ return;
422
386
  }
423
- return;
424
- }
425
387
 
426
- for (let i = 0; i < node.childNodes.length; i++)
427
- traversal(node.childNodes[i], xmlTextArray, false);
428
- }
388
+ for (let i = 0; i < node.childNodes.length; i++)
389
+ traversal(node.childNodes[i], xmlTextArray, false);
390
+ }
429
391
 
430
- /** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
431
- function isNotesNode(node) {
432
- if (node.tagName == notesTag)
433
- return true;
434
- if (node.parentNode)
435
- return isNotesNode(node.parentNode);
436
- return false;
437
- }
392
+ /** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
393
+ function isNotesNode(node) {
394
+ if (node.tagName == notesTag)
395
+ return true;
396
+ if (node.parentNode)
397
+ return isNotesNode(node.parentNode);
398
+ return false;
399
+ }
438
400
 
439
- /** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
440
- function isInvalidTextNode(node) {
441
- if (allowedTextTags.includes(node.tagName))
442
- return true;
443
- if (node.parentNode)
444
- return isInvalidTextNode(node.parentNode);
445
- return false;
446
- }
401
+ /** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
402
+ function isInvalidTextNode(node) {
403
+ if (allowedTextTags.includes(node.tagName))
404
+ return true;
405
+ if (node.parentNode)
406
+ return isInvalidTextNode(node.parentNode);
407
+ return false;
408
+ }
447
409
 
448
- /** The xml string parsed as xml array */
449
- const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
450
- // Iterate over each xmlContent and extract text from them.
451
- xmlContentArray.forEach(xmlContent => {
452
- /** Find text nodes with text:h and text:p tags in xmlContent */
453
- const xmlTextNodesList = [...Array.from(xmlContent
454
- .getElementsByTagName("*"))
455
- .filter(node => allowedTextTags.includes(node.tagName)
456
- && !isInvalidTextNode(node.parentNode))
457
- ];
458
- /** Store all the text content to respond */
459
- responseText.push(
460
- xmlTextNodesList
461
- // Add every text information from within this textNode and combine them together.
462
- .map(textNode => extractAllTextsFromNode(textNode))
463
- .filter(text => text != "")
464
- .join(config.newlineDelimiter ?? "\n")
465
- );
466
- });
467
-
468
- // Add notes text at the end if the user config says so.
469
- // Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
470
- if (!config.ignoreNotes && config.putNotesAtLast)
471
- responseText = [...responseText, ...notesText];
472
-
473
- // Join all responseText array
474
- responseText = responseText.join(config.newlineDelimiter ?? "\n");
475
- // Respond by calling the Callback function.
476
- callback(responseText, undefined);
477
- })
478
- .catch(e => callback(undefined, e));
410
+ /** The xml string parsed as xml array */
411
+ const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
412
+ // Iterate over each xmlContent and extract text from them.
413
+ xmlContentArray.forEach(xmlContent => {
414
+ /** Find text nodes with text:h and text:p tags in xmlContent */
415
+ const xmlTextNodesList = [...Array.from(xmlContent
416
+ .getElementsByTagName("*"))
417
+ .filter(node => allowedTextTags.includes(node.tagName)
418
+ && !isInvalidTextNode(node.parentNode))
419
+ ];
420
+ /** Store all the text content to respond */
421
+ responseText.push(
422
+ xmlTextNodesList
423
+ // Add every text information from within this textNode and combine them together.
424
+ .map(textNode => extractAllTextsFromNode(textNode))
425
+ .filter(text => text != "")
426
+ .join(config.newlineDelimiter ?? "\n")
427
+ );
428
+ });
429
+
430
+ // Add notes text at the end if the user config says so.
431
+ // Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
432
+ if (!config.ignoreNotes && config.putNotesAtLast)
433
+ responseText = [...responseText, ...notesText];
434
+
435
+ // Respond by calling the Callback function.
436
+ callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
437
+ })
438
+ .catch(e => callback(undefined, e));
479
439
  }
480
440
 
481
441
  /** Main function for parsing text from pdf files
482
- * @param {string} filepath File path
442
+ * @param {string | Buffer} file File path or Buffers
483
443
  * @param {function} callback Callback function that returns value or error
484
444
  * @param {OfficeParserConfig} config Config Object for officeParser
485
445
  * @returns {void}
486
446
  */
487
- function parsePdf(filepath, callback, config) {
488
- // Get the pdfjs document for the filepath.
489
- pdfjs.getDocument(filepath).promise
490
- // We go through each page and build our text content promise array.
491
- .then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).map(pageNr => document.getPage(pageNr).then(page => page.getTextContent()))))
492
- // Each textContent item has property 'items' which is an array of objects.
493
- // Each object element in the array has text stored in their 'str' key.
494
- // The concatenation of str is what makes our pdf content.
495
- // str already contains any space that was in the text.
496
- // So, we only care about when to add the new line.
497
- // That we determine using transform[5] value which is the y-coordinate of the item object.
498
- // So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
499
- .then(textContentArray => {
500
- /** Store all the text content to respond */
501
- const responseText = textContentArray
502
- .map(textContent => textContent.items) // Get all the items
503
- .flat() // Flatten all the items object
504
- .filter(item => item.str != '') // Ignore the empty string items.
505
- .reduce((a, v) => (
506
- {
507
- text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
508
- transform5: v.transform[5]
509
- }),
510
- {
511
- text: '',
512
- transform5: undefined
513
- }).text;
514
-
515
- callback(responseText, undefined);
516
- })
517
- .catch(e => callback(undefined, e));
447
+ function parsePdf(file, callback, config) {
448
+ // Get the pdfjs document for the filepath or buffers.
449
+ // @ts-ignore
450
+ pdfjs.getDocument(file).promise
451
+ // We go through each page and build our text content promise array.
452
+ .then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).map(pageNr => document.getPage(pageNr).then(page => page.getTextContent()))))
453
+ // Each textContent item has property 'items' which is an array of objects.
454
+ // Each object element in the array has text stored in their 'str' key.
455
+ // The concatenation of str is what makes our pdf content.
456
+ // str already contains any space that was in the text.
457
+ // So, we only care about when to add the new line.
458
+ // That we determine using transform[5] value which is the y-coordinate of the item object.
459
+ // So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
460
+ .then(textContentArray => {
461
+ /** Store all the text content to respond */
462
+ const responseText = textContentArray
463
+ .map(textContent => textContent.items) // Get all the items
464
+ .flat() // Flatten all the items object
465
+ .filter(item => item.str != '') // Ignore the empty string items.
466
+ .reduce((a, v) => (
467
+ {
468
+ text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
469
+ transform5: v.transform[5]
470
+ }),
471
+ {
472
+ text: '',
473
+ transform5: undefined
474
+ }).text;
475
+
476
+ callback(responseText, undefined);
477
+ })
478
+ .catch(e => callback(undefined, e));
518
479
  }
519
480
 
520
481
  /** Main async function with callback to execute parseOffice for supported files
@@ -524,95 +485,68 @@ function parsePdf(filepath, callback, config) {
524
485
  * @returns {void}
525
486
  */
526
487
  function parseOffice(file, callback, config = {}) {
527
- // Make a clone of the config.
528
- const internalConfig = { ...config };
529
- // Prepare file for processing
488
+ // Make a clone of the config with default values such that none of the config flags are undefined.
489
+ /** @type {OfficeParserConfig} */
490
+ const internalConfig = {
491
+ ignoreNotes: false,
492
+ newlineDelimiter: '\n',
493
+ putNotesAtLast: false,
494
+ outputErrorToConsole: false,
495
+ ...config
496
+ };
497
+ /**
498
+ * Prepare file for processing
499
+ * @type {Promise<{ file:string | Buffer, ext: string}>}
500
+ */
530
501
  const filePreparedPromise = new Promise((res, rej) => {
531
- // Check if decompress location in the config is present.
532
- // If it is valid, we set the final decompression location in the config.
533
- // If it is not valid, we reject the promise with appropriate error message.
534
- if (!internalConfig.tempFilesLocation)
535
- internalConfig.tempFilesLocation = DEFAULTDECOMPRESSSUBLOCATION;
536
- else {
537
- if (!fs.existsSync(internalConfig.tempFilesLocation))
538
- {
539
- rej(ERRORMSG.locationNotFound(internalConfig.tempFilesLocation));
540
- return;
541
- }
542
- internalConfig.tempFilesLocation = `${internalConfig.tempFilesLocation}${internalConfig.tempFilesLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`;
543
- }
544
-
545
- // create temp file subdirectory if it does not exist
546
- fs.mkdirSync(`${internalConfig.tempFilesLocation}/tempfiles`, { recursive: true });
547
-
548
502
  // Check if buffer
549
- if (Buffer.isBuffer(file)) {
503
+ if (Buffer.isBuffer(file))
550
504
  // Guess file type from buffer
551
- fileType.fromBuffer(file)
552
- .then(data =>
553
- {
554
- // temp file name
555
- const newfilepath = getNewFileName(internalConfig.tempFilesLocation, data.ext.toLowerCase());
556
- // write new file
557
- fs.writeFileSync(newfilepath, file);
558
- // resolve promise
559
- res(newfilepath);
560
- })
505
+ return fileType.fromBuffer(file)
506
+ .then(data => res({ file: file, ext: data.ext.toLowerCase() }))
561
507
  .catch(() => rej(ERRORMSG.improperBuffers));
562
- return;
508
+ else if (typeof file === 'string') {
509
+ // Not buffers but real file path.
510
+ // Check if file exists
511
+ if (!fs.existsSync(file))
512
+ throw ERRORMSG.fileDoesNotExist(file);
513
+
514
+ // resolve promise
515
+ res({ file: file, ext: file.split(".").pop() });
563
516
  }
564
-
565
- // Not buffers but real file path.
566
-
567
- // Check if file exists
568
- if (!fs.existsSync(file))
569
- throw ERRORMSG.fileDoesNotExist(file);
570
-
571
- // temp file name
572
- const newfilepath = getNewFileName(internalConfig.tempFilesLocation, file.split(".").pop().toLowerCase());
573
- // Copy the file into a temp location with the temp name
574
- fs.copyFileSync(file, newfilepath)
575
- // resolve promise
576
- res(newfilepath);
517
+ else
518
+ rej(ERRORMSG.invalidInput);
577
519
  });
578
520
 
579
521
  // Process filePreparedPromise resolution.
580
522
  filePreparedPromise
581
- .then(filepath => {
582
- // File extension. Already in lowercase when we prepared the temp file above.
583
- const extension = filepath.split(".").pop();
584
-
523
+ .then(({ file, ext }) => {
585
524
  // Switch between parsing functions depending on extension.
586
- switch(extension) {
525
+ switch (ext) {
587
526
  case "docx":
588
- parseWord(filepath, internalCallback, internalConfig);
527
+ parseWord(file, internalCallback, internalConfig);
589
528
  break;
590
529
  case "pptx":
591
- parsePowerPoint(filepath, internalCallback, internalConfig);
530
+ parsePowerPoint(file, internalCallback, internalConfig);
592
531
  break;
593
532
  case "xlsx":
594
- parseExcel(filepath, internalCallback, internalConfig);
533
+ parseExcel(file, internalCallback, internalConfig);
595
534
  break;
596
535
  case "odt":
597
536
  case "odp":
598
537
  case "ods":
599
- parseOpenOffice(filepath, internalCallback, internalConfig);
538
+ parseOpenOffice(file, internalCallback, internalConfig);
600
539
  break;
601
540
  case "pdf":
602
- parsePdf(filepath, internalCallback, internalConfig);
541
+ parsePdf(file, internalCallback, internalConfig);
603
542
  break;
604
543
 
605
544
  default:
606
- internalCallback(undefined, ERRORMSG.extensionUnsupported(extension)); // Call the internalCallback function which removes the temp files if required.
545
+ internalCallback(undefined, ERRORMSG.extensionUnsupported(ext)); // Call the internalCallback function which removes the temp files if required.
607
546
  }
608
547
 
609
548
  /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
610
549
  function internalCallback(data, err) {
611
- // Check if we need to preserve unzipped content files or delete them.
612
- if (!internalConfig.preserveTempFiles)
613
- // Delete decompress sublocation.
614
- rimraf(internalConfig.tempFilesLocation, rimrafErr => consoleError(rimrafErr, internalConfig.outputErrorToConsole));
615
-
616
550
  // Check if there is an error. Throw if there is an error.
617
551
  if (err)
618
552
  return handleError(err, callback, internalConfig.outputErrorToConsole);
@@ -624,8 +558,7 @@ function parseOffice(file, callback, config = {}) {
624
558
  .catch(error => handleError(error, callback, internalConfig.outputErrorToConsole));
625
559
  }
626
560
 
627
- /**
628
- * Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
561
+ /** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
629
562
  * @param {string | Buffer} file File path or file buffers
630
563
  * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
631
564
  * @returns {Promise<string>}
@@ -640,29 +573,69 @@ function parseOfficeAsync(file, config = {}) {
640
573
  });
641
574
  }
642
575
 
643
- /** Global file name iterator. */
644
- let globalFileNameIterator = 0;
645
- /**
646
- * File Name generator that takes the extension as an input and returns a file name that comprises a timestamp and an incrementing number
647
- * to allow the files to be sorted in chronological order
648
- * @param {string} tempFilesLocation Directory whether this new file needs to be stored
649
- * @param {string} ext File extension for this new generated file name
650
- * @returns {string}
576
+ /** Extract specific files from either a ZIP file buffer or file path based on a filter function.
577
+ * @param {Buffer|string} zipInput ZIP file input, either a Buffer or a file path (string).
578
+ * @param {(x: string) => boolean} filterFn A function that receives the entry object and returns true if the file should be extracted.
579
+ * @returns {Promise<{ path: string, content: string }[]>} Resolves to an array of object
651
580
  */
652
- function getNewFileName(tempFilesLocation, ext) {
653
- // Get the iterator part of the file name
654
- let iteratorPart = (globalFileNameIterator++).toString().padStart(5, '0');
655
- // We want the iterator part of the file name to be of 5 digits.
656
- // Therefore, when the iterator crosses into 6 digits, we reset it to 0.
657
- if (globalFileNameIterator > 99999)
658
- globalFileNameIterator = 0;
659
-
660
- // Return the file name
661
- return `${tempFilesLocation}/tempfiles/${new Date().getTime().toString() + iteratorPart}.${ext}`;
581
+ function extractFiles(zipInput, filterFn) {
582
+ return new Promise((res, rej) => {
583
+ /** Processes zip file and resolves with the path of file and their content.
584
+ * @param {yauzl.ZipFile} zipfile
585
+ */
586
+ const processZipfile = (zipfile) => {
587
+ /** @type {{ path: string, content: string }[]} */
588
+ const extractedFiles = [];
589
+ zipfile.readEntry();
590
+
591
+ /** @param {yauzl.Entry} entry */
592
+ function processEntry(entry) {
593
+ // Use the filter function to determine if the file should be extracted
594
+ if (filterFn(entry.fileName)) {
595
+ zipfile.openReadStream(entry, (err, readStream) => {
596
+ if (err)
597
+ return rej(err);
598
+
599
+ // Use concat-stream to collect the data into a single Buffer
600
+ readStream.pipe(concat(data => {
601
+ extractedFiles.push({
602
+ path: entry.fileName,
603
+ content: data.toString()
604
+ });
605
+ zipfile.readEntry(); // Continue reading entries
606
+ }));
607
+ });
608
+ }
609
+ else
610
+ zipfile.readEntry(); // Skip entries that don't match the filter
611
+ }
612
+
613
+ zipfile.on('entry', processEntry);
614
+ zipfile.on('end', () => res(extractedFiles));
615
+ zipfile.on('error', rej);
616
+ };
617
+
618
+ // Determine whether the input is a buffer or file path
619
+ if (Buffer.isBuffer(zipInput)) {
620
+ // Process ZIP from Buffer
621
+ yauzl.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
622
+ if (err) return rej(err);
623
+ processZipfile(zipfile);
624
+ });
625
+ }
626
+ else if (typeof zipInput === 'string') {
627
+ // Process ZIP from File Path
628
+ yauzl.open(zipInput, { lazyEntries: true }, (err, zipfile) => {
629
+ if (err) return rej(err);
630
+ processZipfile(zipfile);
631
+ });
632
+ }
633
+ else
634
+ rej(ERRORMSG.invalidInput);
635
+ });
662
636
  }
663
637
 
664
- /**
665
- * Handle error by logging it to console if permitted by the config.
638
+ /** Handle error by logging it to console if permitted by the config.
666
639
  * And after that, trigger the callback function with the error value.
667
640
  * @param {string} error Error text
668
641
  * @param {function} callback Callback function provided by the caller
@@ -670,8 +643,10 @@ function getNewFileName(tempFilesLocation, ext) {
670
643
  * @returns {void}
671
644
  */
672
645
  function handleError(error, callback, outputErrorToConsole) {
673
- consoleError(error, outputErrorToConsole);
674
- callback(undefined, ERRORHEADER + error);
646
+ if (error && outputErrorToConsole)
647
+ console.error(ERRORHEADER + error);
648
+
649
+ callback(undefined, new Error(ERRORHEADER + error));
675
650
  }
676
651
 
677
652
 
@@ -681,14 +656,104 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
681
656
 
682
657
 
683
658
  // Run this library on CLI
684
- if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx") && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser")) {
685
- if (process.argv.length == 2) {
686
- // continue
659
+ if ((typeof process.argv[0] == 'string' && (process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx")) &&
660
+ (typeof process.argv[1] == 'string' && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser"))) {
661
+
662
+ // Extract arguments after the script is called
663
+ /** Stores the list of arguments for this CLI call
664
+ * @type {string[]}
665
+ */
666
+ const args = process.argv.slice(2);
667
+ /** Stores the file argument for this CLI call
668
+ * @type {string | Buffer | undefined}
669
+ */
670
+ let fileArg = undefined;
671
+ /** Stores the config arguments for this CLI call
672
+ * @type {string[]}
673
+ */
674
+ const configArgs = [];
675
+
676
+ /** Function to identify if an argument is a config option (i.e., --key=value)
677
+ * @param {string} arg Argument passed in the CLI call.
678
+ */
679
+ function isConfigOption(arg) {
680
+ return arg.startsWith('--') && arg.includes('=');
687
681
  }
688
- else if (process.argv.length == 3)
689
- parseOfficeAsync(process.argv[2])
682
+
683
+ // Loop through arguments to separate file path and config options
684
+ args.forEach(arg => {
685
+ if (isConfigOption(arg))
686
+ // It's a config option
687
+ configArgs.push(arg);
688
+ else if (!fileArg)
689
+ // First non-config argument is assumed to be the file path
690
+ fileArg = arg;
691
+ });
692
+
693
+ // Check if we have a valid file argument
694
+ // If not, we return error and we write the instructions on how to use the library on the terminal.
695
+ if (fileArg != undefined) {
696
+ /** Helper function to parse config arguments from CLI
697
+ * @param {string[]} args List of string arguments that we need to parse to understand the config flag they represent.
698
+ */
699
+ function parseCLIConfigArgs(args) {
700
+ /** @type {OfficeParserConfig} */
701
+ const config = {};
702
+ args.forEach(arg => {
703
+ // Split the argument by '=' to differentiate between the key and value
704
+ const [key, value] = arg.split('=');
705
+
706
+ // We only care about the keys that are important to us. We ignore any other key.
707
+ switch (key) {
708
+ case '--ignoreNotes':
709
+ config.ignoreNotes = value.toLowerCase() === 'true';
710
+ break;
711
+ case '--newlineDelimiter':
712
+ config.newlineDelimiter = value;
713
+ break;
714
+ case '--putNotesAtLast':
715
+ config.putNotesAtLast = value.toLowerCase() === 'true';
716
+ break;
717
+ case '--outputErrorToConsole':
718
+ config.outputErrorToConsole = value.toLowerCase() === 'true';
719
+ break;
720
+ }
721
+ });
722
+
723
+ return config;
724
+ }
725
+
726
+ // Parse CLI config arguments
727
+ const config = parseCLIConfigArgs(configArgs);
728
+
729
+ // Execute parseOfficeAsync with file and config
730
+ parseOfficeAsync(fileArg, config)
690
731
  .then(text => console.log(text))
691
- .catch(error => console.error(ERRORHEADER + error))
692
- else
693
- console.error(ERRORMSG.improperArguments)
732
+ .catch(error => console.error(ERRORHEADER + error));
733
+ }
734
+ else {
735
+ console.error(ERRORMSG.improperArguments);
736
+
737
+ const CLI_INSTRUCTIONS =
738
+ `
739
+ === How to Use officeParser CLI ===
740
+
741
+ Usage:
742
+ node officeparser [--configOption=value] [FILE_PATH]
743
+
744
+ Example:
745
+ node officeparser --ignoreNotes=true --putNotesAtLast=true ./example.docx
746
+
747
+ Config Options:
748
+ --ignoreNotes=[true|false] Flag to ignore notes from files like PowerPoint. Default is false.
749
+ --newlineDelimiter=[delimiter] The delimiter to use for new lines. Default is '\\n'.
750
+ --putNotesAtLast=[true|false] Flag to collect notes at the end of files like PowerPoint. Default is false.
751
+ --outputErrorToConsole=[true|false] Flag to output errors to the console. Default is false.
752
+
753
+ Note:
754
+ The order of file path and config options doesn't matter.
755
+ `;
756
+ // Usage instructions for the user
757
+ console.log(CLI_INSTRUCTIONS);
758
+ }
694
759
  }