officeparser 4.1.2 → 5.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +15 -9
- package/officeParser.js +578 -513
- package/package.json +5 -4
- package/typings/officeParser.d.ts +2 -11
package/officeParser.js
CHANGED
|
@@ -1,11 +1,13 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
const
|
|
3
|
+
// @ts-check
|
|
4
|
+
|
|
5
|
+
const concat = require('concat-stream');
|
|
6
|
+
const { DOMParser } = require('@xmldom/xmldom');
|
|
6
7
|
const fileType = require('file-type');
|
|
8
|
+
const fs = require('fs');
|
|
7
9
|
const pdfjs = require('./pdfjs-dist-build/pdf.js');
|
|
8
|
-
const
|
|
10
|
+
const yauzl = require('yauzl');
|
|
9
11
|
|
|
10
12
|
/** Header for error messages */
|
|
11
13
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
@@ -16,20 +18,8 @@ const ERRORMSG = {
|
|
|
16
18
|
fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
17
19
|
locationNotFound: (location) => `Entered location ${location} is not reachable! Please make sure that the entered directory location exists. Check relative paths and reenter.`,
|
|
18
20
|
improperArguments: `Improper arguments`,
|
|
19
|
-
improperBuffers: `Error occured while reading the file buffers
|
|
20
|
-
|
|
21
|
-
/** Default sublocation for decompressing files under the current directory. */
|
|
22
|
-
const DEFAULTDECOMPRESSSUBLOCATION = "officeParserTemp";
|
|
23
|
-
|
|
24
|
-
/** Console error if allowed
|
|
25
|
-
* @param {string} errorMessage Error message to show on the console
|
|
26
|
-
* @param {string} outputErrorToConsole Flag to show log on console. Ignore if not true.
|
|
27
|
-
* @returns {void}
|
|
28
|
-
*/
|
|
29
|
-
function consoleError(errorMessage, outputErrorToConsole) {
|
|
30
|
-
if (!errorMessage || !outputErrorToConsole)
|
|
31
|
-
return;
|
|
32
|
-
console.error(ERRORHEADER + errorMessage);
|
|
21
|
+
improperBuffers: `Error occured while reading the file buffers`,
|
|
22
|
+
invalidInput: `Invalid input type: Expected a Buffer or a valid file path`
|
|
33
23
|
}
|
|
34
24
|
|
|
35
25
|
/** Returns parsed xml document for a given xml text.
|
|
@@ -42,9 +32,7 @@ const parseString = (xml) => {
|
|
|
42
32
|
};
|
|
43
33
|
|
|
44
34
|
/** @typedef {Object} OfficeParserConfig
|
|
45
|
-
* @property {
|
|
46
|
-
* @property {boolean} [preserveTempFiles] Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
|
|
47
|
-
* @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling.
|
|
35
|
+
* @property {boolean} [outputErrorToConsole] Flag to show all the logs to console in case of an error irrespective of your own handling. Default is false.
|
|
48
36
|
* @property {string} [newlineDelimiter] The delimiter used for every new line in places that allow multiline text like word. Default is \n.
|
|
49
37
|
* @property {boolean} [ignoreNotes] Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
|
|
50
38
|
* @property {boolean} [putNotesAtLast] Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
|
|
@@ -52,469 +40,442 @@ const parseString = (xml) => {
|
|
|
52
40
|
|
|
53
41
|
|
|
54
42
|
/** Main function for parsing text from word files
|
|
55
|
-
* @param {string}
|
|
43
|
+
* @param {string | Buffer} file File path or Buffers
|
|
56
44
|
* @param {function} callback Callback function that returns value or error
|
|
57
45
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
58
46
|
* @returns {void}
|
|
59
47
|
*/
|
|
60
|
-
function parseWord(
|
|
48
|
+
function parseWord(file, callback, config) {
|
|
61
49
|
/** The target content xml file for the docx file. */
|
|
62
50
|
const mainContentFileRegex = /word\/document[\d+]?.xml/g;
|
|
63
51
|
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
|
|
64
52
|
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
.then(files => {
|
|
72
|
-
// Verify if atleast the document xml file exists in the extracted files list.
|
|
73
|
-
if (!files.some(file => file.path.match(mainContentFileRegex)))
|
|
74
|
-
throw ERRORMSG.fileCorrupted(filepath);
|
|
53
|
+
|
|
54
|
+
extractFiles(file, x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.match(fileRegex)))
|
|
55
|
+
.then(files => {
|
|
56
|
+
// Verify if atleast the document xml file exists in the extracted files list.
|
|
57
|
+
if (!files.some(file => file.path.match(mainContentFileRegex)))
|
|
58
|
+
throw ERRORMSG.fileCorrupted(file);
|
|
75
59
|
|
|
76
60
|
return files
|
|
77
61
|
.filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
|
|
78
|
-
.map(file =>
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
callback(
|
|
115
|
-
})
|
|
116
|
-
.catch(e => callback(undefined, e));
|
|
62
|
+
.map(file => file.content);
|
|
63
|
+
})
|
|
64
|
+
// ************************************* word xml files explanation *************************************
|
|
65
|
+
// Structure of xmlContent of a word file is simple.
|
|
66
|
+
// All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
|
|
67
|
+
// So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
|
|
68
|
+
// ******************************************************************************************************
|
|
69
|
+
.then(xmlContentArray => {
|
|
70
|
+
/** Store all the text content to respond. */
|
|
71
|
+
let responseText = [];
|
|
72
|
+
|
|
73
|
+
xmlContentArray.forEach(xmlContent => {
|
|
74
|
+
/** Find text nodes with w:p tags */
|
|
75
|
+
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
|
|
76
|
+
/** Store all the text content to respond */
|
|
77
|
+
responseText.push(
|
|
78
|
+
Array.from(xmlParagraphNodesList)
|
|
79
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
|
|
80
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
|
|
81
|
+
.map(paragraphNode => {
|
|
82
|
+
// Find text nodes with w:t tags
|
|
83
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
|
|
84
|
+
// Join the texts within this paragraph node without any spaces or delimiters.
|
|
85
|
+
return Array.from(xmlTextNodeList)
|
|
86
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
87
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
88
|
+
.join("");
|
|
89
|
+
})
|
|
90
|
+
// Join each paragraph text with a new line delimiter.
|
|
91
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
92
|
+
);
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
// Respond by calling the Callback function.
|
|
96
|
+
callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
|
|
97
|
+
})
|
|
98
|
+
.catch(e => callback(undefined, e));
|
|
117
99
|
}
|
|
118
100
|
|
|
119
101
|
/** Main function for parsing text from PowerPoint files
|
|
120
|
-
* @param {string}
|
|
102
|
+
* @param {string | Buffer} file File path or Buffers
|
|
121
103
|
* @param {function} callback Callback function that returns value or error
|
|
122
104
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
123
105
|
* @returns {void}
|
|
124
106
|
*/
|
|
125
|
-
function parsePowerPoint(
|
|
107
|
+
function parsePowerPoint(file, callback, config) {
|
|
126
108
|
// Files regex that hold our content of interest
|
|
127
109
|
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
128
110
|
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
129
111
|
const slideNumberRegex = /lide(\d+)\.xml/;
|
|
130
112
|
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
//
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
// There are multiple xml files for each slide and correspondingly their notesSlide files.
|
|
165
|
-
// All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
|
|
166
|
-
// So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
|
|
167
|
-
// ******************************************************************************************************
|
|
168
|
-
.then(xmlContentArray => {
|
|
169
|
-
/** Store all the text content to respond */
|
|
170
|
-
let responseText = [];
|
|
171
|
-
|
|
172
|
-
xmlContentArray.forEach(xmlContent => {
|
|
173
|
-
/** Find text nodes with a:p tags */
|
|
174
|
-
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
|
|
113
|
+
extractFiles(file, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex))
|
|
114
|
+
.then(files => {
|
|
115
|
+
// Sort files by slide number and their notes (if any).
|
|
116
|
+
files.sort((a, b) => {
|
|
117
|
+
const matchedANumber = parseInt(a.path.match(slideNumberRegex)?.at(1), 10);
|
|
118
|
+
const matchedBNumber = parseInt(b.path.match(slideNumberRegex)?.at(1), 10);
|
|
119
|
+
|
|
120
|
+
const aNumber = isNaN(matchedANumber) ? Infinity : matchedANumber;
|
|
121
|
+
const bNumber = isNaN(matchedBNumber) ? Infinity : matchedBNumber;
|
|
122
|
+
|
|
123
|
+
return aNumber - bNumber || Number(a.path.includes('notes')) - Number(b.path.includes('notes'));
|
|
124
|
+
});
|
|
125
|
+
|
|
126
|
+
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
127
|
+
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
|
|
128
|
+
throw ERRORMSG.fileCorrupted(file);
|
|
129
|
+
|
|
130
|
+
// Check if any sorting is required.
|
|
131
|
+
if (!config.ignoreNotes && config.putNotesAtLast)
|
|
132
|
+
// Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
|
|
133
|
+
// For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
|
|
134
|
+
files.sort((a, b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
|
|
135
|
+
|
|
136
|
+
// Returning an array of all the xml contents read using fs.readFileSync
|
|
137
|
+
return files.map(file => file.content);
|
|
138
|
+
})
|
|
139
|
+
// ******************************** powerpoint xml files explanation ************************************
|
|
140
|
+
// Structure of xmlContent of a powerpoint file is simple.
|
|
141
|
+
// There are multiple xml files for each slide and correspondingly their notesSlide files.
|
|
142
|
+
// All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
|
|
143
|
+
// So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
|
|
144
|
+
// ******************************************************************************************************
|
|
145
|
+
.then(xmlContentArray => {
|
|
175
146
|
/** Store all the text content to respond */
|
|
176
|
-
responseText
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
147
|
+
let responseText = [];
|
|
148
|
+
|
|
149
|
+
xmlContentArray.forEach(xmlContent => {
|
|
150
|
+
/** Find text nodes with a:p tags */
|
|
151
|
+
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
|
|
152
|
+
/** Store all the text content to respond */
|
|
153
|
+
responseText.push(
|
|
154
|
+
Array.from(xmlParagraphNodesList)
|
|
155
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
|
|
156
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
|
|
157
|
+
.map(paragraphNode => {
|
|
158
|
+
/** Find text nodes with a:t tags */
|
|
159
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
160
|
+
return Array.from(xmlTextNodeList)
|
|
161
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
162
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
163
|
+
.join("");
|
|
164
|
+
})
|
|
165
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
166
|
+
);
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
// Respond by calling the Callback function.
|
|
170
|
+
callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
|
|
171
|
+
})
|
|
172
|
+
.catch(e => callback(undefined, e));
|
|
198
173
|
}
|
|
199
174
|
|
|
200
175
|
/** Main function for parsing text from Excel files
|
|
201
|
-
* @param {string}
|
|
176
|
+
* @param {string | Buffer} file File path or Buffers
|
|
202
177
|
* @param {function} callback Callback function that returns value or error
|
|
203
178
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
204
179
|
* @returns {void}
|
|
205
180
|
*/
|
|
206
|
-
function parseExcel(
|
|
181
|
+
function parseExcel(file, callback, config) {
|
|
207
182
|
// Files regex that hold our content of interest
|
|
208
183
|
const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
|
|
209
184
|
const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
|
|
210
185
|
const chartsRegex = /xl\/charts\/chart\d+.xml/g;
|
|
211
186
|
const stringsFilePath = 'xl/sharedStrings.xml';
|
|
212
187
|
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
|
|
257
|
-
if (childNodesNamedT.length != 1)
|
|
258
|
-
return false;
|
|
259
|
-
return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
|
|
260
|
-
}
|
|
188
|
+
extractFiles(file, x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.match(fileRegex)) || x == stringsFilePath)
|
|
189
|
+
.then(files => {
|
|
190
|
+
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
191
|
+
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
|
|
192
|
+
throw ERRORMSG.fileCorrupted(file);
|
|
193
|
+
|
|
194
|
+
return {
|
|
195
|
+
sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => file.content),
|
|
196
|
+
drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => file.content),
|
|
197
|
+
chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => file.content),
|
|
198
|
+
sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => file.content)[0],
|
|
199
|
+
};
|
|
200
|
+
})
|
|
201
|
+
// ********************************** excel xml files explanation ***************************************
|
|
202
|
+
// Structure of xmlContent of an excel file is a bit complex.
|
|
203
|
+
// We usually have a sharedStrings.xml file which has strings inside t tags
|
|
204
|
+
// However, this file is not necessary to be present. It is sometimes absent if the file has no shared strings indices represented in v nodes.
|
|
205
|
+
// Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
|
|
206
|
+
// Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
|
|
207
|
+
// If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
|
|
208
|
+
// However, if the "t" attribute of c tag is "inlineStr", strings can be inline inside "is"(probably inside String) > "t".
|
|
209
|
+
// We extract either the inline strings or use the value to get numbers of text from shared strings.
|
|
210
|
+
// Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
|
|
211
|
+
// ******************************************************************************************************
|
|
212
|
+
.then(xmlContentFilesObject => {
|
|
213
|
+
/** Store all the text content to respond */
|
|
214
|
+
let responseText = [];
|
|
215
|
+
|
|
216
|
+
/** Function to check if the given c node is a valid inline string node. */
|
|
217
|
+
function isValidInlineStringCNode(cNode) {
|
|
218
|
+
// Initial check to see if the passed node is a cNode
|
|
219
|
+
if (cNode.tagName.toLowerCase() != 'c')
|
|
220
|
+
return false;
|
|
221
|
+
if (cNode.getAttribute("t") != 'inlineStr')
|
|
222
|
+
return false;
|
|
223
|
+
const childNodesNamedIs = cNode.getElementsByTagName('is');
|
|
224
|
+
if (childNodesNamedIs.length != 1)
|
|
225
|
+
return false;
|
|
226
|
+
const childNodesNamedT = childNodesNamedIs[0].getElementsByTagName('t');
|
|
227
|
+
if (childNodesNamedT.length != 1)
|
|
228
|
+
return false;
|
|
229
|
+
return childNodesNamedT[0].childNodes[0] && childNodesNamedT[0].childNodes[0].nodeValue != '';
|
|
230
|
+
}
|
|
261
231
|
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
232
|
+
/** Function to check if the given c node has a valid v node */
|
|
233
|
+
function hasValidVNodeInCNode(cNode) {
|
|
234
|
+
return cNode.getElementsByTagName("v")[0]
|
|
235
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
236
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
|
|
237
|
+
}
|
|
268
238
|
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
callback(
|
|
351
|
-
})
|
|
352
|
-
.catch(e => callback(undefined, e));
|
|
239
|
+
/** Find text nodes with t tags in sharedStrings xml file. If the sharedStringsFile is not present, we return an empty array. */
|
|
240
|
+
const sharedStringsXmlTNodesList = xmlContentFilesObject.sharedStringsFile != undefined ? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t")
|
|
241
|
+
: [];
|
|
242
|
+
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
243
|
+
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
244
|
+
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
|
|
245
|
+
|
|
246
|
+
// Parse Sheet files
|
|
247
|
+
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
248
|
+
/** Find text nodes with c tags in sharedStrings xml file */
|
|
249
|
+
const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
|
|
250
|
+
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings or an inline string.
|
|
251
|
+
responseText.push(
|
|
252
|
+
Array.from(sheetsXmlCNodesList)
|
|
253
|
+
// Filter out invalid c nodes
|
|
254
|
+
.filter(cNode => isValidInlineStringCNode(cNode) || hasValidVNodeInCNode(cNode))
|
|
255
|
+
.map(cNode => {
|
|
256
|
+
// Processing if this is a valid inline string c node.
|
|
257
|
+
if (isValidInlineStringCNode(cNode))
|
|
258
|
+
return cNode.getElementsByTagName('is')[0].getElementsByTagName('t')[0].childNodes[0].nodeValue;
|
|
259
|
+
|
|
260
|
+
// Processing if this c node has a valid v node.
|
|
261
|
+
if (hasValidVNodeInCNode(cNode)) {
|
|
262
|
+
/** Flag whether this node's value represents an index in the shared string array */
|
|
263
|
+
const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
|
|
264
|
+
/** Find value nodes represented by v tags */
|
|
265
|
+
const value = parseInt(cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue, 10);
|
|
266
|
+
// Validate text
|
|
267
|
+
if (isIndexInSharedStrings && value >= sharedStrings.length)
|
|
268
|
+
throw ERRORMSG.fileCorrupted(file);
|
|
269
|
+
|
|
270
|
+
return isIndexInSharedStrings
|
|
271
|
+
? sharedStrings[value]
|
|
272
|
+
: value;
|
|
273
|
+
}
|
|
274
|
+
// TODO: Add debug asserts for if we reach here which would mean we are filtering more items than we are processing.
|
|
275
|
+
// Not the case now but it could happen and it is better to be safe.
|
|
276
|
+
return '';
|
|
277
|
+
})
|
|
278
|
+
// Join each cell text within a sheet with a space.
|
|
279
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
280
|
+
);
|
|
281
|
+
});
|
|
282
|
+
|
|
283
|
+
// Parse Drawing files
|
|
284
|
+
xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
|
|
285
|
+
/** Find text nodes with a:p tags */
|
|
286
|
+
const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
|
|
287
|
+
/** Store all the text content to respond */
|
|
288
|
+
responseText.push(
|
|
289
|
+
Array.from(drawingsXmlParagraphNodesList)
|
|
290
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
|
|
291
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
|
|
292
|
+
.map(paragraphNode => {
|
|
293
|
+
/** Find text nodes with a:t tags */
|
|
294
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
295
|
+
return Array.from(xmlTextNodeList)
|
|
296
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
297
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
298
|
+
.join("");
|
|
299
|
+
})
|
|
300
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
301
|
+
);
|
|
302
|
+
});
|
|
303
|
+
|
|
304
|
+
// Parse Chart files
|
|
305
|
+
xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
|
|
306
|
+
/** Find text nodes with c:v tags */
|
|
307
|
+
const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
|
|
308
|
+
/** Store all the text content to respond */
|
|
309
|
+
responseText.push(
|
|
310
|
+
Array.from(chartsXmlCVNodesList)
|
|
311
|
+
.filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
|
|
312
|
+
.map(cVNode => cVNode.childNodes[0].nodeValue)
|
|
313
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
314
|
+
);
|
|
315
|
+
});
|
|
316
|
+
|
|
317
|
+
// Respond by calling the Callback function.
|
|
318
|
+
callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
|
|
319
|
+
})
|
|
320
|
+
.catch(e => callback(undefined, e));
|
|
353
321
|
}
|
|
354
322
|
|
|
355
323
|
|
|
356
324
|
/** Main function for parsing text from open office files
|
|
357
|
-
* @param {string}
|
|
325
|
+
* @param {string | Buffer} file File path or Buffers
|
|
358
326
|
* @param {function} callback Callback function that returns value or error
|
|
359
327
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
360
328
|
* @returns {void}
|
|
361
329
|
*/
|
|
362
|
-
function parseOpenOffice(
|
|
330
|
+
function parseOpenOffice(file, callback, config) {
|
|
363
331
|
/** The target content xml file for the openoffice file. */
|
|
364
332
|
const mainContentFilePath = 'content.xml';
|
|
365
333
|
const objectContentFilesRegex = /Object \d+\/content.xml/g;
|
|
366
334
|
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
}
|
|
417
|
-
else {
|
|
418
|
-
xmlTextArray.push(node.nodeValue);
|
|
419
|
-
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
420
|
-
xmlTextArray.push(config.newlineDelimiter ?? "\n");
|
|
335
|
+
extractFiles(file, x => x == mainContentFilePath || !!x.match(objectContentFilesRegex))
|
|
336
|
+
.then(files => {
|
|
337
|
+
// Verify if atleast the content xml file exists in the extracted files list.
|
|
338
|
+
if (!files.map(file => file.path).includes(mainContentFilePath))
|
|
339
|
+
throw ERRORMSG.fileCorrupted(file);
|
|
340
|
+
|
|
341
|
+
return {
|
|
342
|
+
mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => file.content)[0],
|
|
343
|
+
objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => file.content),
|
|
344
|
+
}
|
|
345
|
+
})
|
|
346
|
+
// ********************************** openoffice xml files explanation **********************************
|
|
347
|
+
// Structure of xmlContent of openoffice files is simple.
|
|
348
|
+
// All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
|
|
349
|
+
// All text in these tags are separated by new line delimiters.
|
|
350
|
+
// Objects like charts in ods files are in Object d+/content.xml with the same way as above.
|
|
351
|
+
// ******************************************************************************************************
|
|
352
|
+
.then(xmlContentFilesObject => {
|
|
353
|
+
/** Store all the notes text content to respond */
|
|
354
|
+
let notesText = [];
|
|
355
|
+
/** Store all the text content to respond */
|
|
356
|
+
let responseText = [];
|
|
357
|
+
|
|
358
|
+
/** List of allowed text tags */
|
|
359
|
+
const allowedTextTags = ["text:p", "text:h"];
|
|
360
|
+
/** List of notes tags */
|
|
361
|
+
const notesTag = "presentation:notes";
|
|
362
|
+
|
|
363
|
+
/** Main dfs traversal function that goes from one node to its children and returns the value out. */
|
|
364
|
+
function extractAllTextsFromNode(root) {
|
|
365
|
+
let xmlTextArray = []
|
|
366
|
+
for (let i = 0; i < root.childNodes.length; i++)
|
|
367
|
+
traversal(root.childNodes[i], xmlTextArray, true);
|
|
368
|
+
return xmlTextArray.join("");
|
|
369
|
+
}
|
|
370
|
+
/** Traversal function that gets recursive calling. */
|
|
371
|
+
function traversal(node, xmlTextArray, isFirstRecursion) {
|
|
372
|
+
if (!node.childNodes || node.childNodes.length == 0) {
|
|
373
|
+
if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
|
|
374
|
+
if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
|
|
375
|
+
notesText.push(node.nodeValue);
|
|
376
|
+
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
377
|
+
notesText.push(config.newlineDelimiter ?? "\n");
|
|
378
|
+
}
|
|
379
|
+
else {
|
|
380
|
+
xmlTextArray.push(node.nodeValue);
|
|
381
|
+
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
382
|
+
xmlTextArray.push(config.newlineDelimiter ?? "\n");
|
|
383
|
+
}
|
|
421
384
|
}
|
|
385
|
+
return;
|
|
422
386
|
}
|
|
423
|
-
return;
|
|
424
|
-
}
|
|
425
387
|
|
|
426
|
-
|
|
427
|
-
|
|
428
|
-
|
|
388
|
+
for (let i = 0; i < node.childNodes.length; i++)
|
|
389
|
+
traversal(node.childNodes[i], xmlTextArray, false);
|
|
390
|
+
}
|
|
429
391
|
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
|
|
437
|
-
|
|
392
|
+
/** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
|
|
393
|
+
function isNotesNode(node) {
|
|
394
|
+
if (node.tagName == notesTag)
|
|
395
|
+
return true;
|
|
396
|
+
if (node.parentNode)
|
|
397
|
+
return isNotesNode(node.parentNode);
|
|
398
|
+
return false;
|
|
399
|
+
}
|
|
438
400
|
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
401
|
+
/** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
|
|
402
|
+
function isInvalidTextNode(node) {
|
|
403
|
+
if (allowedTextTags.includes(node.tagName))
|
|
404
|
+
return true;
|
|
405
|
+
if (node.parentNode)
|
|
406
|
+
return isInvalidTextNode(node.parentNode);
|
|
407
|
+
return false;
|
|
408
|
+
}
|
|
447
409
|
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
callback(
|
|
477
|
-
})
|
|
478
|
-
.catch(e => callback(undefined, e));
|
|
410
|
+
/** The xml string parsed as xml array */
|
|
411
|
+
const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
|
|
412
|
+
// Iterate over each xmlContent and extract text from them.
|
|
413
|
+
xmlContentArray.forEach(xmlContent => {
|
|
414
|
+
/** Find text nodes with text:h and text:p tags in xmlContent */
|
|
415
|
+
const xmlTextNodesList = [...Array.from(xmlContent
|
|
416
|
+
.getElementsByTagName("*"))
|
|
417
|
+
.filter(node => allowedTextTags.includes(node.tagName)
|
|
418
|
+
&& !isInvalidTextNode(node.parentNode))
|
|
419
|
+
];
|
|
420
|
+
/** Store all the text content to respond */
|
|
421
|
+
responseText.push(
|
|
422
|
+
xmlTextNodesList
|
|
423
|
+
// Add every text information from within this textNode and combine them together.
|
|
424
|
+
.map(textNode => extractAllTextsFromNode(textNode))
|
|
425
|
+
.filter(text => text != "")
|
|
426
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
427
|
+
);
|
|
428
|
+
});
|
|
429
|
+
|
|
430
|
+
// Add notes text at the end if the user config says so.
|
|
431
|
+
// Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
|
|
432
|
+
if (!config.ignoreNotes && config.putNotesAtLast)
|
|
433
|
+
responseText = [...responseText, ...notesText];
|
|
434
|
+
|
|
435
|
+
// Respond by calling the Callback function.
|
|
436
|
+
callback(responseText.join(config.newlineDelimiter ?? "\n"), undefined);
|
|
437
|
+
})
|
|
438
|
+
.catch(e => callback(undefined, e));
|
|
479
439
|
}
|
|
480
440
|
|
|
481
441
|
/** Main function for parsing text from pdf files
|
|
482
|
-
* @param {string}
|
|
442
|
+
* @param {string | Buffer} file File path or Buffers
|
|
483
443
|
* @param {function} callback Callback function that returns value or error
|
|
484
444
|
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
485
445
|
* @returns {void}
|
|
486
446
|
*/
|
|
487
|
-
function parsePdf(
|
|
488
|
-
// Get the pdfjs document for the filepath.
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
447
|
+
function parsePdf(file, callback, config) {
|
|
448
|
+
// Get the pdfjs document for the filepath or buffers.
|
|
449
|
+
// @ts-ignore
|
|
450
|
+
pdfjs.getDocument(file).promise
|
|
451
|
+
// We go through each page and build our text content promise array.
|
|
452
|
+
.then(document => Promise.all(Array.from({ length: document.numPages }, (_, index) => index + 1).map(pageNr => document.getPage(pageNr).then(page => page.getTextContent()))))
|
|
453
|
+
// Each textContent item has property 'items' which is an array of objects.
|
|
454
|
+
// Each object element in the array has text stored in their 'str' key.
|
|
455
|
+
// The concatenation of str is what makes our pdf content.
|
|
456
|
+
// str already contains any space that was in the text.
|
|
457
|
+
// So, we only care about when to add the new line.
|
|
458
|
+
// That we determine using transform[5] value which is the y-coordinate of the item object.
|
|
459
|
+
// So, if there is a mismatch in the transform[5] value between the current item and the previous item, we put a line break.
|
|
460
|
+
.then(textContentArray => {
|
|
461
|
+
/** Store all the text content to respond */
|
|
462
|
+
const responseText = textContentArray
|
|
463
|
+
.map(textContent => textContent.items) // Get all the items
|
|
464
|
+
.flat() // Flatten all the items object
|
|
465
|
+
.filter(item => item.str != '') // Ignore the empty string items.
|
|
466
|
+
.reduce((a, v) => (
|
|
467
|
+
{
|
|
468
|
+
text: a.text + (v.transform[5] != a.transform5 ? (config.newlineDelimiter ?? "\n") : '') + v.str,
|
|
469
|
+
transform5: v.transform[5]
|
|
470
|
+
}),
|
|
471
|
+
{
|
|
472
|
+
text: '',
|
|
473
|
+
transform5: undefined
|
|
474
|
+
}).text;
|
|
475
|
+
|
|
476
|
+
callback(responseText, undefined);
|
|
477
|
+
})
|
|
478
|
+
.catch(e => callback(undefined, e));
|
|
518
479
|
}
|
|
519
480
|
|
|
520
481
|
/** Main async function with callback to execute parseOffice for supported files
|
|
@@ -524,95 +485,68 @@ function parsePdf(filepath, callback, config) {
|
|
|
524
485
|
* @returns {void}
|
|
525
486
|
*/
|
|
526
487
|
function parseOffice(file, callback, config = {}) {
|
|
527
|
-
// Make a clone of the config.
|
|
528
|
-
|
|
529
|
-
|
|
488
|
+
// Make a clone of the config with default values such that none of the config flags are undefined.
|
|
489
|
+
/** @type {OfficeParserConfig} */
|
|
490
|
+
const internalConfig = {
|
|
491
|
+
ignoreNotes: false,
|
|
492
|
+
newlineDelimiter: '\n',
|
|
493
|
+
putNotesAtLast: false,
|
|
494
|
+
outputErrorToConsole: false,
|
|
495
|
+
...config
|
|
496
|
+
};
|
|
497
|
+
/**
|
|
498
|
+
* Prepare file for processing
|
|
499
|
+
* @type {Promise<{ file:string | Buffer, ext: string}>}
|
|
500
|
+
*/
|
|
530
501
|
const filePreparedPromise = new Promise((res, rej) => {
|
|
531
|
-
// Check if decompress location in the config is present.
|
|
532
|
-
// If it is valid, we set the final decompression location in the config.
|
|
533
|
-
// If it is not valid, we reject the promise with appropriate error message.
|
|
534
|
-
if (!internalConfig.tempFilesLocation)
|
|
535
|
-
internalConfig.tempFilesLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
536
|
-
else {
|
|
537
|
-
if (!fs.existsSync(internalConfig.tempFilesLocation))
|
|
538
|
-
{
|
|
539
|
-
rej(ERRORMSG.locationNotFound(internalConfig.tempFilesLocation));
|
|
540
|
-
return;
|
|
541
|
-
}
|
|
542
|
-
internalConfig.tempFilesLocation = `${internalConfig.tempFilesLocation}${internalConfig.tempFilesLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`;
|
|
543
|
-
}
|
|
544
|
-
|
|
545
|
-
// create temp file subdirectory if it does not exist
|
|
546
|
-
fs.mkdirSync(`${internalConfig.tempFilesLocation}/tempfiles`, { recursive: true });
|
|
547
|
-
|
|
548
502
|
// Check if buffer
|
|
549
|
-
if (Buffer.isBuffer(file))
|
|
503
|
+
if (Buffer.isBuffer(file))
|
|
550
504
|
// Guess file type from buffer
|
|
551
|
-
fileType.fromBuffer(file)
|
|
552
|
-
.then(data =>
|
|
553
|
-
{
|
|
554
|
-
// temp file name
|
|
555
|
-
const newfilepath = getNewFileName(internalConfig.tempFilesLocation, data.ext.toLowerCase());
|
|
556
|
-
// write new file
|
|
557
|
-
fs.writeFileSync(newfilepath, file);
|
|
558
|
-
// resolve promise
|
|
559
|
-
res(newfilepath);
|
|
560
|
-
})
|
|
505
|
+
return fileType.fromBuffer(file)
|
|
506
|
+
.then(data => res({ file: file, ext: data.ext.toLowerCase() }))
|
|
561
507
|
.catch(() => rej(ERRORMSG.improperBuffers));
|
|
562
|
-
|
|
508
|
+
else if (typeof file === 'string') {
|
|
509
|
+
// Not buffers but real file path.
|
|
510
|
+
// Check if file exists
|
|
511
|
+
if (!fs.existsSync(file))
|
|
512
|
+
throw ERRORMSG.fileDoesNotExist(file);
|
|
513
|
+
|
|
514
|
+
// resolve promise
|
|
515
|
+
res({ file: file, ext: file.split(".").pop() });
|
|
563
516
|
}
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
// Check if file exists
|
|
568
|
-
if (!fs.existsSync(file))
|
|
569
|
-
throw ERRORMSG.fileDoesNotExist(file);
|
|
570
|
-
|
|
571
|
-
// temp file name
|
|
572
|
-
const newfilepath = getNewFileName(internalConfig.tempFilesLocation, file.split(".").pop().toLowerCase());
|
|
573
|
-
// Copy the file into a temp location with the temp name
|
|
574
|
-
fs.copyFileSync(file, newfilepath)
|
|
575
|
-
// resolve promise
|
|
576
|
-
res(newfilepath);
|
|
517
|
+
else
|
|
518
|
+
rej(ERRORMSG.invalidInput);
|
|
577
519
|
});
|
|
578
520
|
|
|
579
521
|
// Process filePreparedPromise resolution.
|
|
580
522
|
filePreparedPromise
|
|
581
|
-
.then(
|
|
582
|
-
// File extension. Already in lowercase when we prepared the temp file above.
|
|
583
|
-
const extension = filepath.split(".").pop();
|
|
584
|
-
|
|
523
|
+
.then(({ file, ext }) => {
|
|
585
524
|
// Switch between parsing functions depending on extension.
|
|
586
|
-
switch(
|
|
525
|
+
switch (ext) {
|
|
587
526
|
case "docx":
|
|
588
|
-
parseWord(
|
|
527
|
+
parseWord(file, internalCallback, internalConfig);
|
|
589
528
|
break;
|
|
590
529
|
case "pptx":
|
|
591
|
-
parsePowerPoint(
|
|
530
|
+
parsePowerPoint(file, internalCallback, internalConfig);
|
|
592
531
|
break;
|
|
593
532
|
case "xlsx":
|
|
594
|
-
parseExcel(
|
|
533
|
+
parseExcel(file, internalCallback, internalConfig);
|
|
595
534
|
break;
|
|
596
535
|
case "odt":
|
|
597
536
|
case "odp":
|
|
598
537
|
case "ods":
|
|
599
|
-
parseOpenOffice(
|
|
538
|
+
parseOpenOffice(file, internalCallback, internalConfig);
|
|
600
539
|
break;
|
|
601
540
|
case "pdf":
|
|
602
|
-
parsePdf(
|
|
541
|
+
parsePdf(file, internalCallback, internalConfig);
|
|
603
542
|
break;
|
|
604
543
|
|
|
605
544
|
default:
|
|
606
|
-
internalCallback(undefined, ERRORMSG.extensionUnsupported(
|
|
545
|
+
internalCallback(undefined, ERRORMSG.extensionUnsupported(ext)); // Call the internalCallback function which removes the temp files if required.
|
|
607
546
|
}
|
|
608
547
|
|
|
609
548
|
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
610
549
|
function internalCallback(data, err) {
|
|
611
|
-
// Check if we need to preserve unzipped content files or delete them.
|
|
612
|
-
if (!internalConfig.preserveTempFiles)
|
|
613
|
-
// Delete decompress sublocation.
|
|
614
|
-
rimraf(internalConfig.tempFilesLocation, rimrafErr => consoleError(rimrafErr, internalConfig.outputErrorToConsole));
|
|
615
|
-
|
|
616
550
|
// Check if there is an error. Throw if there is an error.
|
|
617
551
|
if (err)
|
|
618
552
|
return handleError(err, callback, internalConfig.outputErrorToConsole);
|
|
@@ -624,8 +558,7 @@ function parseOffice(file, callback, config = {}) {
|
|
|
624
558
|
.catch(error => handleError(error, callback, internalConfig.outputErrorToConsole));
|
|
625
559
|
}
|
|
626
560
|
|
|
627
|
-
/**
|
|
628
|
-
* Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
561
|
+
/** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
629
562
|
* @param {string | Buffer} file File path or file buffers
|
|
630
563
|
* @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
|
|
631
564
|
* @returns {Promise<string>}
|
|
@@ -640,29 +573,69 @@ function parseOfficeAsync(file, config = {}) {
|
|
|
640
573
|
});
|
|
641
574
|
}
|
|
642
575
|
|
|
643
|
-
/**
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
*
|
|
647
|
-
* to allow the files to be sorted in chronological order
|
|
648
|
-
* @param {string} tempFilesLocation Directory whether this new file needs to be stored
|
|
649
|
-
* @param {string} ext File extension for this new generated file name
|
|
650
|
-
* @returns {string}
|
|
576
|
+
/** Extract specific files from either a ZIP file buffer or file path based on a filter function.
|
|
577
|
+
* @param {Buffer|string} zipInput ZIP file input, either a Buffer or a file path (string).
|
|
578
|
+
* @param {(x: string) => boolean} filterFn A function that receives the entry object and returns true if the file should be extracted.
|
|
579
|
+
* @returns {Promise<{ path: string, content: string }[]>} Resolves to an array of object
|
|
651
580
|
*/
|
|
652
|
-
function
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
581
|
+
function extractFiles(zipInput, filterFn) {
|
|
582
|
+
return new Promise((res, rej) => {
|
|
583
|
+
/** Processes zip file and resolves with the path of file and their content.
|
|
584
|
+
* @param {yauzl.ZipFile} zipfile
|
|
585
|
+
*/
|
|
586
|
+
const processZipfile = (zipfile) => {
|
|
587
|
+
/** @type {{ path: string, content: string }[]} */
|
|
588
|
+
const extractedFiles = [];
|
|
589
|
+
zipfile.readEntry();
|
|
590
|
+
|
|
591
|
+
/** @param {yauzl.Entry} entry */
|
|
592
|
+
function processEntry(entry) {
|
|
593
|
+
// Use the filter function to determine if the file should be extracted
|
|
594
|
+
if (filterFn(entry.fileName)) {
|
|
595
|
+
zipfile.openReadStream(entry, (err, readStream) => {
|
|
596
|
+
if (err)
|
|
597
|
+
return rej(err);
|
|
598
|
+
|
|
599
|
+
// Use concat-stream to collect the data into a single Buffer
|
|
600
|
+
readStream.pipe(concat(data => {
|
|
601
|
+
extractedFiles.push({
|
|
602
|
+
path: entry.fileName,
|
|
603
|
+
content: data.toString()
|
|
604
|
+
});
|
|
605
|
+
zipfile.readEntry(); // Continue reading entries
|
|
606
|
+
}));
|
|
607
|
+
});
|
|
608
|
+
}
|
|
609
|
+
else
|
|
610
|
+
zipfile.readEntry(); // Skip entries that don't match the filter
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
zipfile.on('entry', processEntry);
|
|
614
|
+
zipfile.on('end', () => res(extractedFiles));
|
|
615
|
+
zipfile.on('error', rej);
|
|
616
|
+
};
|
|
617
|
+
|
|
618
|
+
// Determine whether the input is a buffer or file path
|
|
619
|
+
if (Buffer.isBuffer(zipInput)) {
|
|
620
|
+
// Process ZIP from Buffer
|
|
621
|
+
yauzl.fromBuffer(zipInput, { lazyEntries: true }, (err, zipfile) => {
|
|
622
|
+
if (err) return rej(err);
|
|
623
|
+
processZipfile(zipfile);
|
|
624
|
+
});
|
|
625
|
+
}
|
|
626
|
+
else if (typeof zipInput === 'string') {
|
|
627
|
+
// Process ZIP from File Path
|
|
628
|
+
yauzl.open(zipInput, { lazyEntries: true }, (err, zipfile) => {
|
|
629
|
+
if (err) return rej(err);
|
|
630
|
+
processZipfile(zipfile);
|
|
631
|
+
});
|
|
632
|
+
}
|
|
633
|
+
else
|
|
634
|
+
rej(ERRORMSG.invalidInput);
|
|
635
|
+
});
|
|
662
636
|
}
|
|
663
637
|
|
|
664
|
-
/**
|
|
665
|
-
* Handle error by logging it to console if permitted by the config.
|
|
638
|
+
/** Handle error by logging it to console if permitted by the config.
|
|
666
639
|
* And after that, trigger the callback function with the error value.
|
|
667
640
|
* @param {string} error Error text
|
|
668
641
|
* @param {function} callback Callback function provided by the caller
|
|
@@ -670,8 +643,10 @@ function getNewFileName(tempFilesLocation, ext) {
|
|
|
670
643
|
* @returns {void}
|
|
671
644
|
*/
|
|
672
645
|
function handleError(error, callback, outputErrorToConsole) {
|
|
673
|
-
|
|
674
|
-
|
|
646
|
+
if (error && outputErrorToConsole)
|
|
647
|
+
console.error(ERRORHEADER + error);
|
|
648
|
+
|
|
649
|
+
callback(undefined, new Error(ERRORHEADER + error));
|
|
675
650
|
}
|
|
676
651
|
|
|
677
652
|
|
|
@@ -681,14 +656,104 @@ module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
|
681
656
|
|
|
682
657
|
|
|
683
658
|
// Run this library on CLI
|
|
684
|
-
if ((process.argv[0]
|
|
685
|
-
|
|
686
|
-
|
|
659
|
+
if ((typeof process.argv[0] == 'string' && (process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').pop() == "npx")) &&
|
|
660
|
+
(typeof process.argv[1] == 'string' && (process.argv[1].split('/').pop() == "officeParser.js" || process.argv[1].split('/').pop().toLowerCase() == "officeparser"))) {
|
|
661
|
+
|
|
662
|
+
// Extract arguments after the script is called
|
|
663
|
+
/** Stores the list of arguments for this CLI call
|
|
664
|
+
* @type {string[]}
|
|
665
|
+
*/
|
|
666
|
+
const args = process.argv.slice(2);
|
|
667
|
+
/** Stores the file argument for this CLI call
|
|
668
|
+
* @type {string | Buffer | undefined}
|
|
669
|
+
*/
|
|
670
|
+
let fileArg = undefined;
|
|
671
|
+
/** Stores the config arguments for this CLI call
|
|
672
|
+
* @type {string[]}
|
|
673
|
+
*/
|
|
674
|
+
const configArgs = [];
|
|
675
|
+
|
|
676
|
+
/** Function to identify if an argument is a config option (i.e., --key=value)
|
|
677
|
+
* @param {string} arg Argument passed in the CLI call.
|
|
678
|
+
*/
|
|
679
|
+
function isConfigOption(arg) {
|
|
680
|
+
return arg.startsWith('--') && arg.includes('=');
|
|
687
681
|
}
|
|
688
|
-
|
|
689
|
-
|
|
682
|
+
|
|
683
|
+
// Loop through arguments to separate file path and config options
|
|
684
|
+
args.forEach(arg => {
|
|
685
|
+
if (isConfigOption(arg))
|
|
686
|
+
// It's a config option
|
|
687
|
+
configArgs.push(arg);
|
|
688
|
+
else if (!fileArg)
|
|
689
|
+
// First non-config argument is assumed to be the file path
|
|
690
|
+
fileArg = arg;
|
|
691
|
+
});
|
|
692
|
+
|
|
693
|
+
// Check if we have a valid file argument
|
|
694
|
+
// If not, we return error and we write the instructions on how to use the library on the terminal.
|
|
695
|
+
if (fileArg != undefined) {
|
|
696
|
+
/** Helper function to parse config arguments from CLI
|
|
697
|
+
* @param {string[]} args List of string arguments that we need to parse to understand the config flag they represent.
|
|
698
|
+
*/
|
|
699
|
+
function parseCLIConfigArgs(args) {
|
|
700
|
+
/** @type {OfficeParserConfig} */
|
|
701
|
+
const config = {};
|
|
702
|
+
args.forEach(arg => {
|
|
703
|
+
// Split the argument by '=' to differentiate between the key and value
|
|
704
|
+
const [key, value] = arg.split('=');
|
|
705
|
+
|
|
706
|
+
// We only care about the keys that are important to us. We ignore any other key.
|
|
707
|
+
switch (key) {
|
|
708
|
+
case '--ignoreNotes':
|
|
709
|
+
config.ignoreNotes = value.toLowerCase() === 'true';
|
|
710
|
+
break;
|
|
711
|
+
case '--newlineDelimiter':
|
|
712
|
+
config.newlineDelimiter = value;
|
|
713
|
+
break;
|
|
714
|
+
case '--putNotesAtLast':
|
|
715
|
+
config.putNotesAtLast = value.toLowerCase() === 'true';
|
|
716
|
+
break;
|
|
717
|
+
case '--outputErrorToConsole':
|
|
718
|
+
config.outputErrorToConsole = value.toLowerCase() === 'true';
|
|
719
|
+
break;
|
|
720
|
+
}
|
|
721
|
+
});
|
|
722
|
+
|
|
723
|
+
return config;
|
|
724
|
+
}
|
|
725
|
+
|
|
726
|
+
// Parse CLI config arguments
|
|
727
|
+
const config = parseCLIConfigArgs(configArgs);
|
|
728
|
+
|
|
729
|
+
// Execute parseOfficeAsync with file and config
|
|
730
|
+
parseOfficeAsync(fileArg, config)
|
|
690
731
|
.then(text => console.log(text))
|
|
691
|
-
.catch(error => console.error(ERRORHEADER + error))
|
|
692
|
-
|
|
693
|
-
|
|
732
|
+
.catch(error => console.error(ERRORHEADER + error));
|
|
733
|
+
}
|
|
734
|
+
else {
|
|
735
|
+
console.error(ERRORMSG.improperArguments);
|
|
736
|
+
|
|
737
|
+
const CLI_INSTRUCTIONS =
|
|
738
|
+
`
|
|
739
|
+
=== How to Use officeParser CLI ===
|
|
740
|
+
|
|
741
|
+
Usage:
|
|
742
|
+
node officeparser [--configOption=value] [FILE_PATH]
|
|
743
|
+
|
|
744
|
+
Example:
|
|
745
|
+
node officeparser --ignoreNotes=true --putNotesAtLast=true ./example.docx
|
|
746
|
+
|
|
747
|
+
Config Options:
|
|
748
|
+
--ignoreNotes=[true|false] Flag to ignore notes from files like PowerPoint. Default is false.
|
|
749
|
+
--newlineDelimiter=[delimiter] The delimiter to use for new lines. Default is '\\n'.
|
|
750
|
+
--putNotesAtLast=[true|false] Flag to collect notes at the end of files like PowerPoint. Default is false.
|
|
751
|
+
--outputErrorToConsole=[true|false] Flag to output errors to the console. Default is false.
|
|
752
|
+
|
|
753
|
+
Note:
|
|
754
|
+
The order of file path and config options doesn't matter.
|
|
755
|
+
`;
|
|
756
|
+
// Usage instructions for the user
|
|
757
|
+
console.log(CLI_INSTRUCTIONS);
|
|
758
|
+
}
|
|
694
759
|
}
|