officeparser 5.2.1 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/officeParser.js DELETED
@@ -1,776 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- // @ts-check
4
-
5
- const concat = require('concat-stream');
6
- const { DOMParser } = require('@xmldom/xmldom');
7
- const fileType = require('file-type');
8
- const fs = require('fs');
9
- const yauzl = require('yauzl');
10
-
11
- /** Header for error messages */
12
- const ERRORHEADER = "[OfficeParser]: ";
13
- /** Error messages */
14
- const ERRORMSG = {
15
- extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
16
- fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
17
- fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
18
- locationNotFound: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
19
- improperArguments: `Improper arguments`,
20
- improperBuffers: `Error occured while reading the file buffers`,
21
- invalidInput: `Invalid input type: Expected a Buffer or a valid file path`
22
- }
23
-
24
- /** Returns parsed xml document for a given xml text.
25
- * @param {string} xml The xml string from the doc file
26
- * @returns {XMLDocument}
27
- */
28
- const parseString = (xml) => {
29
- let parser = new DOMParser();
30
- return parser.parseFromString(xml, "text/xml");
31
- };
32
-
33
- /** @typedef {Object} OfficeParserConfig
34
- * @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling. Default is false.
35
- * @property {string} [newlineDelimiter] The delimiter used for every new line in places that allow multiline text like word. Default is \n.
36
- * @property {boolean} [ignoreNotes] Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
37
- * @property {boolean} [putNotesAtLast] Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
38
- */
39
-
40
-
41
- /** Main function for parsing text from word files
42
- * @param {string | Buffer} file File path or Buffers
43
- * @param {function} callback Callback function that returns value or error
44
- * @param {OfficeParserConfig} config Config Object for officeParser
45
- * @returns {void}
46
- */
47
- function parseWord(file, callback, config) {
48
- /** The target content xml file for the docx file. */
49
- const mainContentFileRegex = /word\/document[\d+]?.xml/g;
50
- const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
51
- const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
52
-
53
- extractFiles(file, x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.match(fileRegex)))
54
- .then(files => {
55
- // Verify if atleast the document xml file exists in the extracted files list.
56
- if (!files.some(file => file.path.match(mainContentFileRegex)))
57
- throw ERRORMSG.fileCorrupted(file);
58
-
59
- return files
60
- .filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
61
- .map(file => file.content);
62
- })
63
- // ************************************* word xml files explanation *************************************
64
- // Structure of xmlContent of a word file is simple.
65
- // All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
66
- // So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
67
- // ******************************************************************************************************
68
- .then(xmlContentArray => {
69
- /** Store all the text content to respond. */
70
- let responseText = [];
71
-
72
- xmlContentArray.forEach(xmlContent => {
73
- /** Find text nodes with w:p tags */
74
- const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
75
- /** Store all the text content to respond */
76
- responseText.push(
77
- Array.from(xmlParagraphNodesList)
78
- // Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
79
- .filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
80
- .map(paragraphNode => {
81
- // Find text nodes with w:t tags
82
- const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
83
- // Join the texts within this paragraph node without any spaces or delimiters.
84
- return Array.from(xmlTextNodeList)
85
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
86
- .map(textNode => textNode.childNodes[0].nodeValue)
87
- .join("");
88
- })
89
- // Join each paragraph text with a new line delimiter.
90
- .join(config.newlineDelimiter ?? "\n")
91
- );
92
- });
93
-
94
- // Respond by calling the Callback function.
95
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
96
- })
97
- .catch(e => callback(undefined, e));
98
- }
99
-
100
- /** Main function for parsing text from PowerPoint files
101
- * @param {string | Buffer} file File path or Buffers
102
- * @param {function} callback Callback function that returns value or error
103
- * @param {OfficeParserConfig} config Config Object for officeParser
104
- * @returns {void}
105
- */
106
- function parsePowerPoint(file, callback, config) {
107
- // Files regex that hold our content of interest
108
- const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
109
- const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
110
- const slideNumberRegex = /lide(\d+)\.xml/;
111
-
112
- extractFiles(file, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex))
113
- .then(files => {
114
- // Sort files by slide number and their notes (if any).
115
- files.sort((a, b) => {
116
- const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
117
- const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
118
-
119
- const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
120
- const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
121
-
122
- return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
123
- });
124
-
125
- // Verify if atleast the slides xml files exist in the extracted files list.
126
- if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
127
- throw ERRORMSG.fileCorrupted(file);
128
-
129
- // Check if any sorting is required.
130
- if (!config.ignoreNotes && config.putNotesAtLast)
131
- // Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
132
- // For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
133
- files.sort((a, b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
134
-
135
- // Returning an array of all the xml contents read using fs.readFileSync
136
- return files.map(file => file.content);
137
- })
138
- // ******************************** powerpoint xml files explanation ************************************
139
- // Structure of xmlContent of a powerpoint file is simple.
140
- // There are multiple xml files for each slide and correspondingly their notesSlide files.
141
- // All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
142
- // So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
143
- // ******************************************************************************************************
144
- .then(xmlContentArray => {
145
- /** Store all the text content to respond */
146
- let responseText = [];
147
-
148
- xmlContentArray.forEach(xmlContent => {
149
- /** Find text nodes with a:p tags */
150
- const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
151
- /** Store all the text content to respond */
152
- responseText.push(
153
- Array.from(xmlParagraphNodesList)
154
- // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
155
- .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
156
- .map(paragraphNode => {
157
- /** Find text nodes with a:t tags */
158
- const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
159
- return Array.from(xmlTextNodeList)
160
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
161
- .map(textNode => textNode.childNodes[0].nodeValue)
162
- .join("");
163
- })
164
- .join(config.newlineDelimiter ?? "\n")
165
- );
166
- });
167
-
168
- // Respond by calling the Callback function.
169
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
170
- })
171
- .catch(e => callback(undefined, e));
172
- }
173
-
174
- /** Main function for parsing text from Excel files
175
- * @param {string | Buffer} file File path or Buffers
176
- * @param {function} callback Callback function that returns value or error
177
- * @param {OfficeParserConfig} config Config Object for officeParser
178
- * @returns {void}
179
- */
180
- function parseExcel(file, callback, config) {
181
- // Files regex that hold our content of interest
182
- const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
183
- const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
184
- const chartsRegex = /xl\/charts\/chart\d+.xml/g;
185
- const stringsFilePath = 'xl/sharedStrings.xml';
186
-
187
- extractFiles(file, x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.match(fileRegex)) || x == stringsFilePath)
188
- .then(files => {
189
- // Verify if atleast the slides xml files exist in the extracted files list.
190
- if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
191
- throw ERRORMSG.fileCorrupted(file);
192
-
193
- return {
194
- sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => file.content),
195
- drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => file.content),
196
- chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => file.content),
197
- sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => file.content)[0],
198
- };
199
- })
200
- // ********************************** excel xml files explanation ***************************************
201
- // Structure of xmlContent of an excel file is a bit complex.
202
- // We usually have a sharedStrings.xml file which has strings inside t tags
203
- // However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
204
- // Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
205
- // Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
206
- // If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
207
- // However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
208
- // We extract either the inline strings or use the value to get numbers of text from shared strings.
209
- // Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
210
- // ******************************************************************************************************
211
- .then(xmlContentFilesObject => {
212
- /** Store all the text content to respond */
213
- let responseText = [];
214
-
215
- /** Function to check if the given c node is a valid inline string node. */
216
- function isValidInlineStringCNode(cNode) {
217
- // Initial check to see if the passed node is a cNode
218
- if (cNode.tagName.toLowerCase() != 'c')
219
- return false;
220
- if (cNode.getAttribute("t") != 'inlineStr')
221
- return false;
222
- const childNodesNamedIs = cNode.getElementsByTagName('is');
223
- if (childNodesNamedIs.length != 1)
224
- return false;
225
- const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
226
- if (childNodesNamedT.length != 1)
227
- return false;
228
- return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
229
- }
230
-
231
- /** Function to check if the given c node has a valid v node */
232
- function hasValidVNodeInCNode(cNode) {
233
- return cNode.getElementsByTagName("v")[0]
234
- && cNode.getElementsByTagName("v")[0].childNodes[0]
235
- && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
236
- }
237
-
238
- /** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
239
- const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
240
- : [];
241
- /** Create shared string array. This will be used as a map to get strings from within sheet files. */
242
- const sharedStrings = Array.from(sharedStringsXmlTNodesList)
243
- .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
244
-
245
- // Parse Sheet files
246
- xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
247
- /** Find text nodes with c tags in sharedStrings xml file */
248
- const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
249
- // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
250
- responseText.push(
251
- Array.from(sheetsXmlCNodesList)
252
- // Filter out invalid c nodes
253
- .filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
254
- .map(cNode => {
255
- // Processing if this is a valid inline string c node.
256
- if (isValidInlineStringCNode(cNode))
257
- return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
258
-
259
- // Processing if this c node has a valid v node.
260
- if (hasValidVNodeInCNode(cNode)) {
261
- /** Flag whether this node's value represents an index in the shared string array */
262
- const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
263
- /** Find value nodes represented by v tags */
264
- const value = parseInt(cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue, 10);
265
- // Validate text
266
- if (isIndexInSharedStrings && value >= sharedStrings.length)
267
- throw ERRORMSG.fileCorrupted(file);
268
-
269
- return isIndexInSharedStrings
270
- ? sharedStrings[value]
271
- : value;
272
- }
273
- // Should not reach here. If we do, it means we are not filtering out items that we are not ready to process.
274
- // Not the case now but it could happen if we change the filtering logic without updating the processing logic.
275
- // So, it is better to error out here.
276
- handleError(`Invalid c node found in sheet xml content: ${cNode}`, callback, config.outputErrorToConsole);
277
- return '';
278
- })
279
- // Join each cell text within a sheet with a space.
280
- .join(config.newlineDelimiter ?? "\n")
281
- );
282
- });
283
-
284
- // Parse Drawing files
285
- xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
286
- /** Find text nodes with a:p tags */
287
- const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
288
- /** Store all the text content to respond */
289
- responseText.push(
290
- Array.from(drawingsXmlParagraphNodesList)
291
- // Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
292
- .filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
293
- .map(paragraphNode => {
294
- /** Find text nodes with a:t tags */
295
- const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
296
- return Array.from(xmlTextNodeList)
297
- .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
298
- .map(textNode => textNode.childNodes[0].nodeValue)
299
- .join("");
300
- })
301
- .join(config.newlineDelimiter ?? "\n")
302
- );
303
- });
304
-
305
- // Parse Chart files
306
- xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
307
- /** Find text nodes with c:v tags */
308
- const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
309
- /** Store all the text content to respond */
310
- responseText.push(
311
- Array.from(chartsXmlCVNodesList)
312
- .filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
313
- .map(cVNode => cVNode.childNodes[0].nodeValue)
314
- .join(config.newlineDelimiter ?? "\n")
315
- );
316
- });
317
-
318
- // Respond by calling the Callback function.
319
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
320
- })
321
- .catch(e => callback(undefined, e));
322
- }
323
-
324
-
325
- /** Main function for parsing text from open office files
326
- * @param {string | Buffer} file File path or Buffers
327
- * @param {function} callback Callback function that returns value or error
328
- * @param {OfficeParserConfig} config Config Object for officeParser
329
- * @returns {void}
330
- */
331
- function parseOpenOffice(file, callback, config) {
332
- /** The target content xml file for the openoffice file. */
333
- const mainContentFilePath = 'content.xml';
334
- const objectContentFilesRegex = /Object \d+\/content.xml/g;
335
-
336
- extractFiles(file, x => x == mainContentFilePath || !!x.match(objectContentFilesRegex))
337
- .then(files => {
338
- // Verify if atleast the content xml file exists in the extracted files list.
339
- if (!files.map(file => file.path).includes(mainContentFilePath))
340
- throw ERRORMSG.fileCorrupted(file);
341
-
342
- return {
343
- mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => file.content)[0],
344
- objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => file.content),
345
- }
346
- })
347
- // ********************************** openoffice xml files explanation **********************************
348
- // Structure of xmlContent of openoffice files is simple.
349
- // All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
350
- // All text in these tags are separated by new line delimiters.
351
- // Objects like charts in ods files are in Object d+/content.xml with the same way as above.
352
- // ******************************************************************************************************
353
- .then(xmlContentFilesObject => {
354
- /** Store all the notes text content to respond */
355
- let notesText = [];
356
- /** Store all the text content to respond */
357
- let responseText = [];
358
-
359
- /** List of allowed text tags */
360
- const allowedTextTags = ["text:p", "text:h"];
361
- /** List of notes tags */
362
- const notesTag = "presentation:notes";
363
-
364
- /** Main dfs traversal function that goes from one node to its children and returns the value out. */
365
- function extractAllTextsFromNode(root) {
366
- let xmlTextArray = []
367
- for (let i = 0; i < root.childNodes.length; i++)
368
- traversal(root.childNodes[i], xmlTextArray, true);
369
- return xmlTextArray.join("");
370
- }
371
- /** Traversal function that gets recursive calling. */
372
- function traversal(node, xmlTextArray, isFirstRecursion) {
373
- if (!node.childNodes || node.childNodes.length == 0) {
374
- if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
375
- if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
376
- notesText.push(node.nodeValue);
377
- if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
378
- notesText.push(config.newlineDelimiter ?? "\n");
379
- }
380
- else {
381
- xmlTextArray.push(node.nodeValue);
382
- if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
383
- xmlTextArray.push(config.newlineDelimiter ?? "\n");
384
- }
385
- }
386
- return;
387
- }
388
-
389
- for (let i = 0; i < node.childNodes.length; i++)
390
- traversal(node.childNodes[i], xmlTextArray, false);
391
- }
392
-
393
- /** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
394
- function isNotesNode(node) {
395
- if (node.tagName == notesTag)
396
- return true;
397
- if (node.parentNode)
398
- return isNotesNode(node.parentNode);
399
- return false;
400
- }
401
-
402
- /** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
403
- function isInvalidTextNode(node) {
404
- if (allowedTextTags.includes(node.tagName))
405
- return true;
406
- if (node.parentNode)
407
- return isInvalidTextNode(node.parentNode);
408
- return false;
409
- }
410
-
411
- /** The xml string parsed as xml array */
412
- const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
413
- // Iterate over each xmlContent and extract text from them.
414
- xmlContentArray.forEach(xmlContent => {
415
- /** Find text nodes with text:h and text:p tags in xmlContent */
416
- const xmlTextNodesList = [...Array.from(xmlContent
417
- .getElementsByTagName("*"))
418
- .filter(node => allowedTextTags.includes(node.tagName)
419
- && !isInvalidTextNode(node.parentNode))
420
- ];
421
- /** Store all the text content to respond */
422
- responseText.push(
423
- xmlTextNodesList
424
- // Add every text information from within this textNode and combine them together.
425
- .map(textNode => extractAllTextsFromNode(textNode))
426
- .filter(text => text != "")
427
- .join(config.newlineDelimiter ?? "\n")
428
- );
429
- });
430
-
431
- // Add notes text at the end if the user config says so.
432
- // Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
433
- if (!config.ignoreNotes && config.putNotesAtLast)
434
- responseText = [...responseText, ...notesText];
435
-
436
- // Respond by calling the Callback function.
437
- callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
438
- })
439
- .catch(e => callback(undefined, e));
440
- }
441
-
442
- /** Main function for parsing text from pdf files
443
- * @param {string | Buffer} file File path or Buffers
444
- * @param {function} callback Callback function that returns value or error
445
- * @param {OfficeParserConfig} config Config Object for officeParser
446
- * @returns {Promise<void>}
447
- */
448
- async function parsePdf(file, callback, config) {
449
- // Wait for pdfjs module to be loaded once
450
- // Lazy import pdfjs to avoid Node startup issues for environments that don't use PDF parsing
451
- const pdfjs = await import('pdfjs-dist/legacy/build/pdf.mjs');
452
-
453
- // Get the pdfjs document for the filepath or Uint8Array buffers.
454
- // pdfjs does not accept Buffers directly, so we convert them to Uint8Array.
455
- pdfjs.getDocument(file instanceof Buffer ? new Uint8Array(file) : file).promise
456
- // We go through each page and build our text content promise array.
457
- .then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => document.getPage(index + 1).then(page => page.getTextContent()))))
458
- // Each textContent item has property 'items' which is an array of objects.
459
- // Each object element in the array has text stored in their 'str' key.
460
- // The concatenation of str is what makes our pdf content.
461
- // str already contains any space that was in the text.
462
- // So, we only care about when to add the new line.
463
- // That we determine using transform[5] value which is the y-coordinate of the item object.
464
- // So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
465
- .then(textContentArray => {
466
- /** Store all the text content to respond */
467
- const responseText = textContentArray
468
- .map(textContent => textContent.items) // Get all the items
469
- .flat() // Flatten all the items object
470
- .reduce((a, v) => (
471
- // the items could be TextItem or a TextMarkedContent.
472
- // We are only interested in the TextItem which has a str property.
473
- 'str' in v && v.str != ''
474
- ? {
475
- text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
476
- transform5: v.transform[5]
477
- } : {
478
- text: a.text,
479
- transform5: a.transform5
480
- }
481
- ),
482
- {
483
- text: '',
484
- transform5: undefined
485
- }).text;
486
-
487
- callback(responseText, undefined);
488
- })
489
- .catch(e => callback(undefined, e));
490
- }
491
-
492
- /** Main async function with callback to execute parseOffice for supported files
493
- * @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
494
- * @param {function} callback Callback function that returns value or error
495
- * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
496
- * @returns {void}
497
- */
498
- function parseOffice(srcFile, callback, config = {}) {
499
- // Make a clone of the config with default values such that none of the config flags are undefined.
500
- /** @type {OfficeParserConfig} */
501
- const internalConfig = {
502
- ignoreNotes: false,
503
- newlineDelimiter: '\n',
504
- putNotesAtLast: false,
505
- outputErrorToConsole: false,
506
- ...config
507
- };
508
-
509
- // Our internal code can process regular node Buffers or file path.
510
- // So, if the src file was presented as ArrayBuffers, we create Buffers from them.
511
- let file = srcFile instanceof ArrayBuffer ? Buffer.from(srcFile)
512
- : srcFile;
513
-
514
- /**
515
- * Prepare file for processing
516
- * @type {Promise<{ file:string | Buffer, ext: string}>}
517
- */
518
- const filePreparedPromise = new Promise((res, rej) => {
519
- // Check if buffer
520
- if (Buffer.isBuffer(file))
521
- // Guess file type from buffer
522
- return fileType.fromBuffer(file)
523
- .then(data => res({ file: file, ext: data.ext.toLowerCase() }))
524
- .catch(() => rej(ERRORMSG.improperBuffers));
525
- else if (typeof file === 'string') {
526
- // Not buffers but real file path.
527
- // Check if file exists
528
- if (!fs.existsSync(file))
529
- throw ERRORMSG.fileDoesNotExist(file);
530
-
531
- // resolve promise
532
- res({ file: file, ext: file.split(".").pop() });
533
- }
534
- else
535
- rej(ERRORMSG.invalidInput);
536
- });
537
-
538
- // Process filePreparedPromise resolution.
539
- filePreparedPromise
540
- .then(({ file, ext }) => {
541
- // Switch between parsing functions depending on extension.
542
- switch (ext) {
543
- case "docx":
544
- parseWord(file, internalCallback, internalConfig);
545
- break;
546
- case "pptx":
547
- parsePowerPoint(file, internalCallback, internalConfig);
548
- break;
549
- case "xlsx":
550
- parseExcel(file, internalCallback, internalConfig);
551
- break;
552
- case "odt":
553
- case "odp":
554
- case "ods":
555
- parseOpenOffice(file, internalCallback, internalConfig);
556
- break;
557
- case "pdf":
558
- parsePdf(file, internalCallback, internalConfig);
559
- break;
560
-
561
- default:
562
- internalCallback(undefined, ERRORMSG.extensionUnsupported(ext)); // Call the internalCallback function which removes the temp files if required.
563
- }
564
-
565
- /** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
566
- function internalCallback(data, err) {
567
- // Check if there is an error. Throw if there is an error.
568
- if (err)
569
- return handleError(err, callback, internalConfig.outputErrorToConsole);
570
-
571
- // Call the original callback
572
- callback(data, undefined);
573
- }
574
- })
575
- .catch(error => handleError(error, callback, internalConfig.outputErrorToConsole));
576
- }
577
-
578
- /** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
579
- * @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
580
- * @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
581
- * @returns {Promise<string>}
582
- */
583
- function parseOfficeAsync(srcFile, config = {}) {
584
- return new Promise((res, rej) => {
585
- parseOffice(srcFile, function (data, err) {
586
- if (err)
587
- return rej(err);
588
- return res(data);
589
- }, config);
590
- });
591
- }
592
-
593
- /** Extract specific files from either a ZIP file buffer or file path based on a filter function.
594
- * @param {Buffer|string} zipInput ZIP file input, either a Buffer or a file path (string).
595
- * @param {(x: string) => boolean} filterFn A function that receives the entry object and returns true if the file should be extracted.
596
- * @returns {Promise<{ path: string, content: string }[]>} Resolves to an array of object
597
- */
598
- function extractFiles(zipInput, filterFn) {
599
- return new Promise((res, rej) => {
600
- /** Processes zip file and resolves with the path of file and their content.
601
- * @param {yauzl.ZipFile} zipfile
602
- */
603
- const processZipfile = (zipfile) => {
604
- /** @type {{ path: string, content: string }[]} */
605
- const extractedFiles = [];
606
- zipfile.readEntry();
607
-
608
- /** @param {yauzl.Entry} entry */
609
- function processEntry(entry) {
610
- // Use the filter function to determine if the file should be extracted
611
- if (filterFn(entry.fileName)) {
612
- zipfile.openReadStream(entry, (err, readStream) => {
613
- if (err)
614
- return rej(err);
615
-
616
- // Use concat-stream to collect the data into a single Buffer
617
- readStream.pipe(concat(data => {
618
- extractedFiles.push({
619
- path: entry.fileName,
620
- content: data.toString()
621
- });
622
- zipfile.readEntry(); // Continue reading entries
623
- }));
624
- });
625
- }
626
- else
627
- zipfile.readEntry(); // Skip entries that don't match the filter
628
- }
629
-
630
- zipfile.on('entry', processEntry);
631
- zipfile.on('end', () => res(extractedFiles));
632
- zipfile.on('error', rej);
633
- };
634
-
635
- // Determine whether the input is a buffer or file path
636
- if (Buffer.isBuffer(zipInput)) {
637
- // Process ZIP from Buffer
638
- yauzl.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
639
- if (err) return rej(err);
640
- processZipfile(zipfile);
641
- });
642
- }
643
- else if (typeof zipInput === 'string') {
644
- // Process ZIP from File Path
645
- yauzl.open(zipInput, { lazyEntries: true }, (err, zipfile) => {
646
- if (err) return rej(err);
647
- processZipfile(zipfile);
648
- });
649
- }
650
- else
651
- rej(ERRORMSG.invalidInput);
652
- });
653
- }
654
-
655
- /** Handle error by logging it to console if permitted by the config.
656
- * And after that, trigger the callback function with the error value.
657
- * @param {string} error Error text
658
- * @param {function} callback Callback function provided by the caller
659
- * @param {boolean} outputErrorToConsole Flag to log error to console.
660
- * @returns {void}
661
- */
662
- function handleError(error, callback, outputErrorToConsole) {
663
- if (error && outputErrorToConsole)
664
- console.error(ERRORHEADER + error);
665
-
666
- callback(undefined, new Error(ERRORHEADER + error));
667
- }
668
-
669
-
670
- // Export functions
671
- module.exports.parseOffice = parseOffice;
672
- module.exports.parseOfficeAsync = parseOfficeAsync;
673
-
674
-
675
- // Run this library on CLI
676
- if ((typeof process.argv[0] == 'string' && (process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx")) &&
677
- (typeof process.argv[1] == 'string' && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser"))) {
678
-
679
- // Extract arguments after the script is called
680
- /** Stores the list of arguments for this CLI call
681
- * @type {string[]}
682
- */
683
- const args = process.argv.slice(2);
684
- /** Stores the file argument for this CLI call
685
- * @type {string | Buffer | undefined}
686
- */
687
- let fileArg = undefined;
688
- /** Stores the config arguments for this CLI call
689
- * @type {string[]}
690
- */
691
- const configArgs = [];
692
-
693
- /** Function to identify if an argument is a config option (i.e., --key=value)
694
- * @param {string} arg Argument passed in the CLI call.
695
- */
696
- function isConfigOption(arg) {
697
- return arg.startsWith('--') && arg.includes('=');
698
- }
699
-
700
- // Loop through arguments to separate file path and config options
701
- args.forEach(arg => {
702
- if (isConfigOption(arg))
703
- // It's a config option
704
- configArgs.push(arg);
705
- else if (!fileArg)
706
- // First non-config argument is assumed to be the file path
707
- fileArg = arg;
708
- });
709
-
710
- // Check if we have a valid file argument
711
- // If not, we return error and we write the instructions on how to use the library on the terminal.
712
- if (fileArg != undefined) {
713
- /** Helper function to parse config arguments from CLI
714
- * @param {string[]} args List of string arguments that we need to parse to understand the config flag they represent.
715
- */
716
- function parseCLIConfigArgs(args) {
717
- /** @type {OfficeParserConfig} */
718
- const config = {};
719
- args.forEach(arg => {
720
- // Split the argument by '=' to differentiate between the key and value
721
- const [key, value] = arg.split('=');
722
-
723
- // We only care about the keys that are important to us. We ignore any other key.
724
- switch (key) {
725
- case '--ignoreNotes':
726
- config.ignoreNotes = value.toLowerCase() === 'true';
727
- break;
728
- case '--newlineDelimiter':
729
- config.newlineDelimiter = value;
730
- break;
731
- case '--putNotesAtLast':
732
- config.putNotesAtLast = value.toLowerCase() === 'true';
733
- break;
734
- case '--outputErrorToConsole':
735
- config.outputErrorToConsole = value.toLowerCase() === 'true';
736
- break;
737
- }
738
- });
739
-
740
- return config;
741
- }
742
-
743
- // Parse CLI config arguments
744
- const config = parseCLIConfigArgs(configArgs);
745
-
746
- // Execute parseOfficeAsync with file and config
747
- parseOfficeAsync(fileArg, config)
748
- .then(text => console.log(text))
749
- .catch(error => console.error(ERRORHEADER + error));
750
- }
751
- else {
752
- console.error(ERRORMSG.improperArguments);
753
-
754
- const CLI_INSTRUCTIONS =
755
- `
756
- === How to Use officeParser CLI ===
757
-
758
- Usage:
759
- node officeparser [--configOption=value] [FILE_PATH]
760
-
761
- Example:
762
- node officeparser --ignoreNotes=true --putNotesAtLast=true ./example.docx
763
-
764
- Config Options:
765
- --ignoreNotes=[true|false] Flag to ignore notes from files like PowerPoint. Default is false.
766
- --newlineDelimiter=[delimiter] The delimiter to use for new lines. Default is '\\n'.
767
- --putNotesAtLast=[true|false] Flag to collect notes at the end of files like PowerPoint. Default is false.
768
- --outputErrorToConsole=[true|false] Flag to output errors to the console. Default is false.
769
-
770
- Note:
771
- The order of file path and config options doesn't matter.
772
- `;
773
- // Usage instructions for the user
774
- console.log(CLI_INSTRUCTIONS);
775
- }
776
- }