officeparser 5.2.2 → 6.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js DELETED
@@ -1,790 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- // @ts-check
4
-
5
- const concat = require('concat-stream');
6
- const { DOMParser } = require('@xmldom/xmldom');
7
- const fileType = require('file-type');
8
- const fs = require('fs');
9
- const yauzl = require('yauzl');
10
-
11
- /** Header for error messages */
12
- const ERRORHEADER = "[OfficeParser]: ";
13
- /** Error messages */
14
- const ERRORMSG = {
15
- extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
16
- fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
17
- fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
18
- locationNotFound: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
19
- improperArguments: `Improper arguments`,
20
- improperBuffers: `Error occured while reading the file buffers`,
21
- invalidInput: `Invalid input type: Expected a Buffer or a valid file path`
22
- }
23
-
24
- /** Returns parsed xml document for a given xml text.
25
- * @param {string} xml The xml string from the doc file
26
- * @returns {XMLDocument}
27
- */
28
- const parseString = (xml) => {
29
- let parser = new DOMParser();
30
- return parser.parseFromString(xml, "text/xml");
31
- };
32
-
33
- /** @typedef {Object} OfficeParserConfig
34
- * @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling. Default is false.
35
- * @property {string} [newlineDelimiter] The delimiter used for every new line in places that allow multiline text like word. Default is \n.
36
- * @property {boolean} [ignoreNotes] Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
37
- * @property {boolean} [putNotesAtLast] Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
38
- */
39
-
40
-
41
- /** Main function for parsing text from word files
42
- * @param {string | Buffer} file File path or Buffers
43
- * @param {function} callback Callback function that returns value or error
44
- * @param {OfficeParserConfig} config Config Object for officeParser
45
- * @returns {void}
46
- */
47
- function parseWord(file, callback, config) {
48
- /** The target content xml file for the docx file. */
49
- const mainContentFileRegex = /word\/document[\d+]?.xml/g;
50
- const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
51
- const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
52
-
53
- extractFiles(file, x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.match(fileRegex)))
54
- .then(files => {
55
- // Verify if atleast the document xml file exists in the extracted files list.
56
- if (!files.some(file => file.path.match(mainContentFileRegex)))
57
- throw ERRORMSG.fileCorrupted(file);
58
-
59
- return files
60
- .filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
61
- .map(file => file.content);
62
- })
63
- // ************************************* word xml files explanation *************************************
64
- // Structure of xmlContent of a word file is simple.
65
- // All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
66
- // So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
67
- // ******************************************************************************************************
68
- .then(xmlContentArray => {
69
- /** Store all the text content to respond. */
70
- let responseText = [];
71
-
72
- xmlContentArray.forEach(xmlContent => {
73
- /** Find text nodes with w:p tags */
74
- const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
75
- /** Store all the text content to respond */
76
- responseText.push(
77
- Array.from(xmlParagraphNodesList)
78
- // Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
79
- .filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
80
- .map(paragraphNode => {
81
- // Find text nodes with w:t tags
82
- const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
83
- // Join the texts within this paragraph node without any spaces or delimiters.
84
- return Array.from(xmlTextNodeList)
85
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
86
- .map(textNode => textNode.childNodes[0].nodeValue)
87
- .join("");
88
- })
89
- // Join each paragraph text with a new line delimiter.
90
- .join(config.newlineDelimiter ?? "\n")
91
- );
92
- });
93
-
94
- // Respond by calling the Callback function.
95
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
96
- })
97
- .catch(e => callback(undefined, e));
98
- }
99
-
100
- /** Main function for parsing text from PowerPoint files
101
- * @param {string | Buffer} file File path or Buffers
102
- * @param {function} callback Callback function that returns value or error
103
- * @param {OfficeParserConfig} config Config Object for officeParser
104
- * @returns {void}
105
- */
106
- function parsePowerPoint(file, callback, config) {
107
- // Files regex that hold our content of interest
108
- const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
109
- const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
110
- const slideNumberRegex = /lide(\d+)\.xml/;
111
-
112
- extractFiles(file, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex))
113
- .then(files => {
114
- // Sort files by slide number and their notes (if any).
115
- files.sort((a, b) => {
116
- const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
117
- const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
118
-
119
- const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
120
- const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
121
-
122
- return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
123
- });
124
-
125
- // Verify if atleast the slides xml files exist in the extracted files list.
126
- if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
127
- throw ERRORMSG.fileCorrupted(file);
128
-
129
- // Check if any sorting is required.
130
- if (!config.ignoreNotes && config.putNotesAtLast)
131
- // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
132
- // For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
133
- files.sort((a, b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
134
-
135
- // Returning an array of all the xml contents read using fs.readFileSync
136
- return files.map(file => file.content);
137
- })
138
- // ******************************** powerpoint xml files explanation ************************************
139
- // Structure of xmlContent of a powerpoint file is simple.
140
- // There are multiple xml files for each slide and correspondingly their notesSlide files.
141
- // All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
142
- // So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
143
- // ******************************************************************************************************
144
- .then(xmlContentArray => {
145
- /** Store all the text content to respond */
146
- let responseText = [];
147
-
148
- xmlContentArray.forEach(xmlContent => {
149
- /** Find text nodes with a:p tags */
150
- const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
151
- /** Store all the text content to respond */
152
- responseText.push(
153
- Array.from(xmlParagraphNodesList)
154
- // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
155
- .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
156
- .map(paragraphNode => {
157
- /** Find text nodes with a:t tags */
158
- const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
159
- return Array.from(xmlTextNodeList)
160
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
161
- .map(textNode => textNode.childNodes[0].nodeValue)
162
- .join("");
163
- })
164
- .join(config.newlineDelimiter ?? "\n")
165
- );
166
- });
167
-
168
- // Respond by calling the Callback function.
169
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
170
- })
171
- .catch(e => callback(undefined, e));
172
- }
173
-
174
- /** Main function for parsing text from Excel files
175
- * @param {string | Buffer} file File path or Buffers
176
- * @param {function} callback Callback function that returns value or error
177
- * @param {OfficeParserConfig} config Config Object for officeParser
178
- * @returns {void}
179
- */
180
- function parseExcel(file, callback, config) {
181
- // Files regex that hold our content of interest
182
- const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
183
- const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
184
- const chartsRegex = /xl\/charts\/chart\d+.xml/g;
185
- const stringsFilePath = 'xl/sharedStrings.xml';
186
-
187
- extractFiles(file, x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.match(fileRegex)) || x == stringsFilePath)
188
- .then(files => {
189
- // Verify if atleast the slides xml files exist in the extracted files list.
190
- if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
191
- throw ERRORMSG.fileCorrupted(file);
192
-
193
- return {
194
- sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => file.content),
195
- drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => file.content),
196
- chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => file.content),
197
- sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => file.content)[0],
198
- };
199
- })
200
- // ********************************** excel xml files explanation ***************************************
201
- // Structure of xmlContent of an excel file is a bit complex.
202
- // We usually have a sharedStrings.xml file which has strings inside t tags
203
- // However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
204
- // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
205
- // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
206
- // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
207
- // However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
208
- // We extract either the inline strings or use the value to get numbers of text from shared strings.
209
- // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
210
- // ******************************************************************************************************
211
- .then(xmlContentFilesObject => {
212
- /** Store all the text content to respond */
213
- let responseText = [];
214
-
215
- /** Function to check if the given c node is a valid inline string node. */
216
- function isValidInlineStringCNode(cNode) {
217
- // Initial check to see if the passed node is a cNode
218
- if (cNode.tagName.toLowerCase() != 'c')
219
- return false;
220
- if (cNode.getAttribute("t") != 'inlineStr')
221
- return false;
222
- const childNodesNamedIs = cNode.getElementsByTagName('is');
223
- if (childNodesNamedIs.length != 1)
224
- return false;
225
- const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
226
- if (childNodesNamedT.length != 1)
227
- return false;
228
- return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
229
- }
230
-
231
- /** Function to check if the given c node has a valid v node */
232
- function hasValidVNodeInCNode(cNode) {
233
- return cNode.getElementsByTagName("v")[0]
234
- && cNode.getElementsByTagName("v")[0].childNodes[0]
235
- && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
236
- }
237
-
238
- /** Find text nodes with t tags in sharedStrings.xml file. If the sharedStringsFile is not present, we return an empty array. */
239
- const sharedStringsXmlSiNodesList = xmlContentFilesObject.sharedStringsFile != undefined
240
- ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("si")
241
- : [];
242
-
243
- /** Create shared string array. This will be used as a map to get strings from within sheet files. */
244
- const sharedStrings = Array.from(sharedStringsXmlSiNodesList)
245
- .map(siNode => {
246
- // Concatenate all <t> nodes within the <si> node
247
- return Array.from(siNode.getElementsByTagName("t"))
248
- .map(tNode => tNode.childNodes[0]?.nodeValue ?? '') // Extract text content from each <t> node
249
- .join(''); // Combine all <t> node text into a single string
250
- });
251
-
252
- // Parse Sheet files
253
- xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
254
- /** Find text nodes with c tags in sharedStrings xml file */
255
- const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
256
- // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
257
- responseText.push(
258
- Array.from(sheetsXmlCNodesList)
259
- // Filter out invalid c nodes
260
- .filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
261
- .map(cNode => {
262
- // Processing if this is a valid inline string c node.
263
- if (isValidInlineStringCNode(cNode))
264
- return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
265
-
266
- // Processing if this c node has a valid v node.
267
- if (hasValidVNodeInCNode(cNode)) {
268
- /** Flag whether this node's value represents an index in the shared string array */
269
- const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
270
- /** Find value nodes represented by v tags */
271
- const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
272
- const valueAsIndex = Number(value);
273
- // Validate text
274
- if (isIndexInSharedStrings && (valueAsIndex != parseInt(value, 10) || valueAsIndex >= sharedStrings.length))
275
- throw ERRORMSG.fileCorrupted(file);
276
-
277
- return isIndexInSharedStrings
278
- ? sharedStrings[valueAsIndex]
279
- : value;
280
- }
281
- // Should not reach here. If we do, it means we are not filtering out items that we are not ready to process.
282
- // Not the case now but it could happen if we change the filtering logic without updating the processing logic.
283
- // So, it is better to error out here.
284
- handleError(`Invalid c node found in sheet xml content: ${cNode}`, callback, config.outputErrorToConsole);
285
- return '';
286
- })
287
- // Join each cell text within a sheet with a space.
288
- .join(config.newlineDelimiter ?? "\n")
289
- );
290
- });
291
-
292
- // Parse Drawing files
293
- xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
294
- /** Find text nodes with a:p tags */
295
- const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
296
- /** Store all the text content to respond */
297
- responseText.push(
298
- Array.from(drawingsXmlParagraphNodesList)
299
- // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
300
- .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
301
- .map(paragraphNode => {
302
- /** Find text nodes with a:t tags */
303
- const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
304
- return Array.from(xmlTextNodeList)
305
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
306
- .map(textNode => textNode.childNodes[0].nodeValue)
307
- .join("");
308
- })
309
- .join(config.newlineDelimiter ?? "\n")
310
- );
311
- });
312
-
313
- // Parse Chart files
314
- xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
315
- /** Find text nodes with c:v tags */
316
- const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
317
- /** Store all the text content to respond */
318
- responseText.push(
319
- Array.from(chartsXmlCVNodesList)
320
- .filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
321
- .map(cVNode => cVNode.childNodes[0].nodeValue)
322
- .join(config.newlineDelimiter ?? "\n")
323
- );
324
- });
325
-
326
- // Respond by calling the Callback function.
327
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
328
- })
329
- .catch(e => callback(undefined, e));
330
- }
331
-
332
-
333
- /** Main function for parsing text from open office files
334
- * @param {string | Buffer} file File path or Buffers
335
- * @param {function} callback Callback function that returns value or error
336
- * @param {OfficeParserConfig} config Config Object for officeParser
337
- * @returns {void}
338
- */
339
- function parseOpenOffice(file, callback, config) {
340
- /** The target content xml file for the openoffice file. */
341
- const mainContentFilePath = 'content.xml';
342
- const objectContentFilesRegex = /Object \d+\/content.xml/g;
343
-
344
- extractFiles(file, x => x == mainContentFilePath || !!x.match(objectContentFilesRegex))
345
- .then(files => {
346
- // Verify if atleast the content xml file exists in the extracted files list.
347
- if (!files.map(file => file.path).includes(mainContentFilePath))
348
- throw ERRORMSG.fileCorrupted(file);
349
-
350
- return {
351
- mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => file.content)[0],
352
- objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => file.content),
353
- }
354
- })
355
- // ********************************** openoffice xml files explanation **********************************
356
- // Structure of xmlContent of openoffice files is simple.
357
- // All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
358
- // All text in these tags are separated by new line delimiters.
359
- // Objects like charts in ods files are in Object d+/content.xml with the same way as above.
360
- // ******************************************************************************************************
361
- .then(xmlContentFilesObject => {
362
- /** Store all the notes text content to respond */
363
- let notesText = [];
364
- /** Store all the text content to respond */
365
- let responseText = [];
366
-
367
- /** List of allowed text tags */
368
- const allowedTextTags = ["text:p", "text:h"];
369
- /** List of notes tags */
370
- const notesTag = "presentation:notes";
371
-
372
- /** Main dfs traversal function that goes from one node to its children and returns the value out. */
373
- function extractAllTextsFromNode(root) {
374
- let xmlTextArray = []
375
- for (let i = 0; i < root.childNodes.length; i++)
376
- traversal(root.childNodes[i], xmlTextArray, true);
377
- return xmlTextArray.join("");
378
- }
379
- /** Traversal function that gets recursive calling. */
380
- function traversal(node, xmlTextArray, isFirstRecursion) {
381
- if (!node.childNodes || node.childNodes.length == 0) {
382
- if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
383
- // If the corresponding value is of type float, we take the value from office:value attribute.
384
- // However, it is not on the parentNode but rather grandparentNode.
385
- const value = node.parentNode.parentNode?.getAttribute('office:value-type') == 'float'
386
- ? Number(node.parentNode.parentNode.getAttribute('office:value'))
387
- : node.nodeValue;
388
-
389
- if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
390
- notesText.push(value);
391
- if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
392
- notesText.push(config.newlineDelimiter ?? "\n");
393
- }
394
- else {
395
- xmlTextArray.push(value);
396
- if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
397
- xmlTextArray.push(config.newlineDelimiter ?? "\n");
398
- }
399
- }
400
- return;
401
- }
402
-
403
- for (let i = 0; i < node.childNodes.length; i++)
404
- traversal(node.childNodes[i], xmlTextArray, false);
405
- }
406
-
407
- /** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
408
- function isNotesNode(node) {
409
- if (node.tagName == notesTag)
410
- return true;
411
- if (node.parentNode)
412
- return isNotesNode(node.parentNode);
413
- return false;
414
- }
415
-
416
- /** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
417
- function isInvalidTextNode(node) {
418
- if (allowedTextTags.includes(node.tagName))
419
- return true;
420
- if (node.parentNode)
421
- return isInvalidTextNode(node.parentNode);
422
- return false;
423
- }
424
-
425
- /** The xml string parsed as xml array */
426
- const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
427
- // Iterate over each xmlContent and extract text from them.
428
- xmlContentArray.forEach(xmlContent => {
429
- /** Find text nodes with text:h and text:p tags in xmlContent */
430
- const xmlTextNodesList = [...Array.from(xmlContent
431
- .getElementsByTagName("*"))
432
- .filter(node => allowedTextTags.includes(node.tagName)
433
- && !isInvalidTextNode(node.parentNode))
434
- ];
435
- /** Store all the text content to respond */
436
- responseText.push(
437
- xmlTextNodesList
438
- // Add every text information from within this textNode and combine them together.
439
- .map(textNode => extractAllTextsFromNode(textNode))
440
- .filter(text => text != "")
441
- .join(config.newlineDelimiter ?? "\n")
442
- );
443
- });
444
-
445
- // Add notes text at the end if the user config says so.
446
- // Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
447
- if (!config.ignoreNotes && config.putNotesAtLast)
448
- responseText = [...responseText, ...notesText];
449
-
450
- // Respond by calling the Callback function.
451
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
452
- })
453
- .catch(e => callback(undefined, e));
454
- }
455
-
456
- /** Main function for parsing text from pdf files
457
- * @param {string | Buffer} file File path or Buffers
458
- * @param {function} callback Callback function that returns value or error
459
- * @param {OfficeParserConfig} config Config Object for officeParser
460
- * @returns {Promise<void>}
461
- */
462
- async function parsePdf(file, callback, config) {
463
- // Wait for pdfjs module to be loaded once
464
- // Lazy import pdfjs to avoid Node startup issues for environments that don't use PDF parsing
465
- const pdfjs = await import('pdfjs-dist/legacy/build/pdf.mjs');
466
-
467
- // Get the pdfjs document for the filepath or Uint8Array buffers.
468
- // pdfjs does not accept Buffers directly, so we convert them to Uint8Array.
469
- pdfjs.getDocument(file instanceof Buffer ? new Uint8Array(file) : file).promise
470
- // We go through each page and build our text content promise array.
471
- .then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => document.getPage(index + 1).then(page => page.getTextContent()))))
472
- // Each textContent item has property 'items' which is an array of objects.
473
- // Each object element in the array has text stored in their 'str' key.
474
- // The concatenation of str is what makes our pdf content.
475
- // str already contains any space that was in the text.
476
- // So, we only care about when to add the new line.
477
- // That we determine using transform[5] value which is the y-coordinate of the item object.
478
- // So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
479
- .then(textContentArray => {
480
- /** Store all the text content to respond */
481
- const responseText = textContentArray
482
- .map(textContent => textContent.items) // Get all the items
483
- .flat() // Flatten all the items object
484
- .reduce((a, v) => (
485
- // the items could be TextItem or a TextMarkedContent.
486
- // We are only interested in the TextItem which has a str property.
487
- 'str' in v && v.str != ''
488
- ? {
489
- text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
490
- transform5: v.transform[5]
491
- } : {
492
- text: a.text,
493
- transform5: a.transform5
494
- }
495
- ),
496
- {
497
- text: '',
498
- transform5: undefined
499
- }).text;
500
-
501
- callback(responseText, undefined);
502
- })
503
- .catch(e => callback(undefined, e));
504
- }
505
-
506
- /** Main async function with callback to execute parseOffice for supported files
507
- * @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
508
- * @param {function} callback Callback function that returns value or error
509
- * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
510
- * @returns {void}
511
- */
512
- function parseOffice(srcFile, callback, config = {}) {
513
- // Make a clone of the config with default values such that none of the config flags are undefined.
514
- /** @type {OfficeParserConfig} */
515
- const internalConfig = {
516
- ignoreNotes: false,
517
- newlineDelimiter: '\n',
518
- putNotesAtLast: false,
519
- outputErrorToConsole: false,
520
- ...config
521
- };
522
-
523
- // Our internal code can process regular node Buffers or file path.
524
- // So, if the src file was presented as ArrayBuffers, we create Buffers from them.
525
- let file = srcFile instanceof ArrayBuffer ? Buffer.from(srcFile)
526
- : srcFile;
527
-
528
- /**
529
- * Prepare file for processing
530
- * @type {Promise<{ file:string | Buffer, ext: string}>}
531
- */
532
- const filePreparedPromise = new Promise((res, rej) => {
533
- // Check if buffer
534
- if (Buffer.isBuffer(file))
535
- // Guess file type from buffer
536
- return fileType.fromBuffer(file)
537
- .then(data => res({ file: file, ext: data.ext.toLowerCase() }))
538
- .catch(() => rej(ERRORMSG.improperBuffers));
539
- else if (typeof file === 'string') {
540
- // Not buffers but real file path.
541
- // Check if file exists
542
- if (!fs.existsSync(file))
543
- throw ERRORMSG.fileDoesNotExist(file);
544
-
545
- // resolve promise
546
- res({ file: file, ext: file.split(".").pop() });
547
- }
548
- else
549
- rej(ERRORMSG.invalidInput);
550
- });
551
-
552
- // Process filePreparedPromise resolution.
553
- filePreparedPromise
554
- .then(({ file, ext }) => {
555
- // Switch between parsing functions depending on extension.
556
- switch (ext) {
557
- case "docx":
558
- parseWord(file, internalCallback, internalConfig);
559
- break;
560
- case "pptx":
561
- parsePowerPoint(file, internalCallback, internalConfig);
562
- break;
563
- case "xlsx":
564
- parseExcel(file, internalCallback, internalConfig);
565
- break;
566
- case "odt":
567
- case "odp":
568
- case "ods":
569
- parseOpenOffice(file, internalCallback, internalConfig);
570
- break;
571
- case "pdf":
572
- parsePdf(file, internalCallback, internalConfig);
573
- break;
574
-
575
- default:
576
- internalCallback(undefined, ERRORMSG.extensionUnsupported(ext)); // Call the internalCallback function which removes the temp files if required.
577
- }
578
-
579
- /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
580
- function internalCallback(data, err) {
581
- // Check if there is an error. Throw if there is an error.
582
- if (err)
583
- return handleError(err, callback, internalConfig.outputErrorToConsole);
584
-
585
- // Call the original callback
586
- callback(data, undefined);
587
- }
588
- })
589
- .catch(error => handleError(error, callback, internalConfig.outputErrorToConsole));
590
- }
591
-
592
- /** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
593
- * @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
594
- * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
595
- * @returns {Promise<string>}
596
- */
597
- function parseOfficeAsync(srcFile, config = {}) {
598
- return new Promise((res, rej) => {
599
- parseOffice(srcFile, function (data, err) {
600
- if (err)
601
- return rej(err);
602
- return res(data);
603
- }, config);
604
- });
605
- }
606
-
607
- /** Extract specific files from either a ZIP file buffer or file path based on a filter function.
608
- * @param {Buffer|string} zipInput ZIP file input, either a Buffer or a file path (string).
609
- * @param {(x: string) => boolean} filterFn A function that receives the entry object and returns true if the file should be extracted.
610
- * @returns {Promise<{ path: string, content: string }[]>} Resolves to an array of object
611
- */
612
- function extractFiles(zipInput, filterFn) {
613
- return new Promise((res, rej) => {
614
- /** Processes zip file and resolves with the path of file and their content.
615
- * @param {yauzl.ZipFile} zipfile
616
- */
617
- const processZipfile = (zipfile) => {
618
- /** @type {{ path: string, content: string }[]} */
619
- const extractedFiles = [];
620
- zipfile.readEntry();
621
-
622
- /** @param {yauzl.Entry} entry */
623
- function processEntry(entry) {
624
- // Use the filter function to determine if the file should be extracted
625
- if (filterFn(entry.fileName)) {
626
- zipfile.openReadStream(entry, (err, readStream) => {
627
- if (err)
628
- return rej(err);
629
-
630
- // Use concat-stream to collect the data into a single Buffer
631
- readStream.pipe(concat(data => {
632
- extractedFiles.push({
633
- path: entry.fileName,
634
- content: data.toString()
635
- });
636
- zipfile.readEntry(); // Continue reading entries
637
- }));
638
- });
639
- }
640
- else
641
- zipfile.readEntry(); // Skip entries that don't match the filter
642
- }
643
-
644
- zipfile.on('entry', processEntry);
645
- zipfile.on('end', () => res(extractedFiles));
646
- zipfile.on('error', rej);
647
- };
648
-
649
- // Determine whether the input is a buffer or file path
650
- if (Buffer.isBuffer(zipInput)) {
651
- // Process ZIP from Buffer
652
- yauzl.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
653
- if (err) return rej(err);
654
- processZipfile(zipfile);
655
- });
656
- }
657
- else if (typeof zipInput === 'string') {
658
- // Process ZIP from File Path
659
- yauzl.open(zipInput, { lazyEntries: true }, (err, zipfile) => {
660
- if (err) return rej(err);
661
- processZipfile(zipfile);
662
- });
663
- }
664
- else
665
- rej(ERRORMSG.invalidInput);
666
- });
667
- }
668
-
669
- /** Handle error by logging it to console if permitted by the config.
670
- * And after that, trigger the callback function with the error value.
671
- * @param {string} error Error text
672
- * @param {function} callback Callback function provided by the caller
673
- * @param {boolean} outputErrorToConsole Flag to log error to console.
674
- * @returns {void}
675
- */
676
- function handleError(error, callback, outputErrorToConsole) {
677
- if (error && outputErrorToConsole)
678
- console.error(ERRORHEADER + error);
679
-
680
- callback(undefined, new Error(ERRORHEADER + error));
681
- }
682
-
683
-
684
- // Export functions
685
- module.exports.parseOffice = parseOffice;
686
- module.exports.parseOfficeAsync = parseOfficeAsync;
687
-
688
-
689
- // Run this library on CLI
690
- if ((typeof process.argv[0] == 'string' && (process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx")) &&
691
- (typeof process.argv[1] == 'string' && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser"))) {
692
-
693
- // Extract arguments after the script is called
694
- /** Stores the list of arguments for this CLI call
695
- * @type {string[]}
696
- */
697
- const args = process.argv.slice(2);
698
- /** Stores the file argument for this CLI call
699
- * @type {string | Buffer | undefined}
700
- */
701
- let fileArg = undefined;
702
- /** Stores the config arguments for this CLI call
703
- * @type {string[]}
704
- */
705
- const configArgs = [];
706
-
707
- /** Function to identify if an argument is a config option (i.e., --key=value)
708
- * @param {string} arg Argument passed in the CLI call.
709
- */
710
- function isConfigOption(arg) {
711
- return arg.startsWith('--') && arg.includes('=');
712
- }
713
-
714
- // Loop through arguments to separate file path and config options
715
- args.forEach(arg => {
716
- if (isConfigOption(arg))
717
- // It's a config option
718
- configArgs.push(arg);
719
- else if (!fileArg)
720
- // First non-config argument is assumed to be the file path
721
- fileArg = arg;
722
- });
723
-
724
- // Check if we have a valid file argument
725
- // If not, we return error and we write the instructions on how to use the library on the terminal.
726
- if (fileArg != undefined) {
727
- /** Helper function to parse config arguments from CLI
728
- * @param {string[]} args List of string arguments that we need to parse to understand the config flag they represent.
729
- */
730
- function parseCLIConfigArgs(args) {
731
- /** @type {OfficeParserConfig} */
732
- const config = {};
733
- args.forEach(arg => {
734
- // Split the argument by '=' to differentiate between the key and value
735
- const [key, value] = arg.split('=');
736
-
737
- // We only care about the keys that are important to us. We ignore any other key.
738
- switch (key) {
739
- case '--ignoreNotes':
740
- config.ignoreNotes = value.toLowerCase() === 'true';
741
- break;
742
- case '--newlineDelimiter':
743
- config.newlineDelimiter = value;
744
- break;
745
- case '--putNotesAtLast':
746
- config.putNotesAtLast = value.toLowerCase() === 'true';
747
- break;
748
- case '--outputErrorToConsole':
749
- config.outputErrorToConsole = value.toLowerCase() === 'true';
750
- break;
751
- }
752
- });
753
-
754
- return config;
755
- }
756
-
757
- // Parse CLI config arguments
758
- const config = parseCLIConfigArgs(configArgs);
759
-
760
- // Execute parseOfficeAsync with file and config
761
- parseOfficeAsync(fileArg, config)
762
- .then(text => console.log(text))
763
- .catch(error => console.error(ERRORHEADER + error));
764
- }
765
- else {
766
- console.error(ERRORMSG.improperArguments);
767
-
768
- const CLI_INSTRUCTIONS =
769
- `
770
- === How to Use officeParser CLI ===
771
-
772
- Usage:
773
- node officeparser [--configOption=value] [FILE_PATH]
774
-
775
- Example:
776
- node officeparser --ignoreNotes=true --putNotesAtLast=true ./example.docx
777
-
778
- Config Options:
779
- --ignoreNotes=[true|false] Flag to ignore notes from files like PowerPoint. Default is false.
780
- --newlineDelimiter=[delimiter] The delimiter to use for new lines. Default is '\\n'.
781
- --putNotesAtLast=[true|false] Flag to collect notes at the end of files like PowerPoint. Default is false.
782
- --outputErrorToConsole=[true|false] Flag to output errors to the console. Default is false.
783
-
784
- Note:
785
- The order of file path and config options doesn't matter.
786
- `;
787
- // Usage instructions for the user
788
- console.log(CLI_INSTRUCTIONS);
789
- }
790
- }