officeparser 3.3.0 → 4.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -214
- package/officeParser.js +445 -539
- package/package.json +8 -4
- package/typings/officeParser.d.ts +29 -67
package/officeParser.js
CHANGED
|
@@ -1,517 +1,552 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
|
-
const decompress
|
|
4
|
-
const
|
|
5
|
-
const
|
|
6
|
-
const
|
|
7
|
-
const
|
|
3
|
+
const decompress = require('decompress');
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const rimraf = require('rimraf');
|
|
6
|
+
const fileType = require('file-type');
|
|
7
|
+
const pdfParse = require('pdf-parse');
|
|
8
|
+
const { DOMParser } = require('@xmldom/xmldom');
|
|
8
9
|
|
|
9
10
|
/** Header for error messages */
|
|
10
11
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
11
12
|
/** Error messages */
|
|
12
13
|
const ERRORMSG = {
|
|
13
|
-
extensionUnsupported: (ext) =>
|
|
14
|
-
fileCorrupted: (
|
|
15
|
-
fileDoesNotExist: (
|
|
16
|
-
locationNotFound: (location) =>
|
|
17
|
-
improperArguments:
|
|
18
|
-
improperBuffers:
|
|
14
|
+
extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
15
|
+
fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
16
|
+
fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
17
|
+
locationNotFound: (location) => `Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
|
|
18
|
+
improperArguments: `Improper arguments`,
|
|
19
|
+
improperBuffers: `Error occured while reading the file buffers`
|
|
19
20
|
}
|
|
20
21
|
/** Default sublocation for decompressing files under the current directory. */
|
|
21
22
|
const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
|
|
22
23
|
/** Location for decompressing files. Default is "officeDist" */
|
|
23
24
|
let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
24
|
-
/** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
|
|
25
|
-
let outputErrorToConsole = false;
|
|
26
25
|
|
|
27
26
|
/** Console error if allowed
|
|
28
|
-
* @param {string} errorMessage
|
|
27
|
+
* @param {string} errorMessage Error message to show on the console
|
|
28
|
+
* @param {string} outputErrorToConsole Flag to show log on console. Ignore if not true.
|
|
29
29
|
* @returns {void}
|
|
30
30
|
*/
|
|
31
|
-
function consoleError(errorMessage) {
|
|
32
|
-
if (outputErrorToConsole)
|
|
33
|
-
|
|
31
|
+
function consoleError(errorMessage, outputErrorToConsole) {
|
|
32
|
+
if (!errorMessage || !outputErrorToConsole)
|
|
33
|
+
return;
|
|
34
|
+
console.error(ERRORHEADER + errorMessage);
|
|
34
35
|
}
|
|
35
36
|
|
|
36
|
-
/**
|
|
37
|
+
/** Returns parsed xml document for a given xml text.
|
|
37
38
|
* @param {string} xml The xml string from the doc file
|
|
38
|
-
* @
|
|
39
|
-
|
|
39
|
+
* @returns {XMLDocument}
|
|
40
|
+
*/
|
|
41
|
+
const parseString = (xml) => {
|
|
42
|
+
let parser = new DOMParser();
|
|
43
|
+
return parser.parseFromString(xml, "text/xml");
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
/** @typedef {Object} OfficeParserConfig
|
|
47
|
+
* @property {boolean} preserveTempFiles Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
|
|
48
|
+
* @property {boolean} outputErrorToConsole Flag to show all the logs to console in case of an error irrespective of your own handling.
|
|
49
|
+
* @property {string} newlineDelimiter The delimiter used for every new line in places that allow multiline text like word. Default is \n.
|
|
50
|
+
* @property {boolean} ignoreNotes Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
|
|
51
|
+
* @property {boolean} putNotesAtLast Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
|
|
40
52
|
*/
|
|
41
|
-
const parseStringPromise = (xml, ignoreAttrs = true) => new Promise((resolve, reject) => {
|
|
42
|
-
xml2js.parseString(xml, { "ignoreAttrs": ignoreAttrs }, (err, result) => {
|
|
43
|
-
if (err)
|
|
44
|
-
reject(err);
|
|
45
|
-
resolve(result);
|
|
46
|
-
});
|
|
47
|
-
});
|
|
48
53
|
|
|
49
54
|
|
|
50
55
|
/** Main function for parsing text from word files
|
|
51
|
-
* @param {string}
|
|
52
|
-
* @param {function}
|
|
53
|
-
* @param {
|
|
56
|
+
* @param {string} filepath File path
|
|
57
|
+
* @param {function} callback Callback function that returns value or error
|
|
58
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
54
59
|
* @returns {void}
|
|
55
60
|
*/
|
|
56
|
-
function parseWord(
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
/** Store all the text content to respond */
|
|
68
|
-
let responseText = [];
|
|
69
|
-
|
|
70
|
-
/** Extracting text from Word files xml objects converted to js */
|
|
71
|
-
function extractTextFromWordXmlObjects(xmlObjects) {
|
|
72
|
-
// specifically for Arrays
|
|
73
|
-
if (Array.isArray(xmlObjects)) {
|
|
74
|
-
xmlObjects.forEach(item =>
|
|
75
|
-
(typeof item == "string") && (item != "")
|
|
76
|
-
? responseText.push(item)
|
|
77
|
-
: extractTextFromWordXmlObjects(item))
|
|
78
|
-
}
|
|
79
|
-
// for other JS Object
|
|
80
|
-
else if (typeof xmlObjects == "object") {
|
|
81
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
82
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
83
|
-
? (key == "w:t" || key == "_") && value != ""
|
|
84
|
-
? responseText.push(value)
|
|
85
|
-
: undefined
|
|
86
|
-
: extractTextFromWordXmlObjects(value);
|
|
87
|
-
}
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
const contentFile = 'word/document.xml';
|
|
92
|
-
decompress(filename,
|
|
93
|
-
decompressSubLocation,
|
|
94
|
-
{ filter: x => x.path == contentFile }
|
|
61
|
+
function parseWord(filepath, callback, config) {
|
|
62
|
+
/** The target content xml file for the docx file. */
|
|
63
|
+
const mainContentFile = 'word/document.xml';
|
|
64
|
+
const footnotesFile = 'word/footnotes.xml';
|
|
65
|
+
const endnotesFile = 'word/endnotes.xml';
|
|
66
|
+
/** The decompress location which contains the filename in it */
|
|
67
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
68
|
+
decompress(filepath,
|
|
69
|
+
decompressLocation,
|
|
70
|
+
{ filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
|
|
95
71
|
)
|
|
96
72
|
.then(files => {
|
|
97
|
-
if (files.length
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
73
|
+
if (files.length == 0)
|
|
74
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
75
|
+
|
|
76
|
+
return [...files.filter(file => file.path == mainContentFile),
|
|
77
|
+
...files.filter(file => file.path == footnotesFile),
|
|
78
|
+
...files.filter(file => file.path == endnotesFile)
|
|
79
|
+
]
|
|
80
|
+
.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
103
81
|
})
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
82
|
+
// ************************************* word xml files explanation *************************************
|
|
83
|
+
// Structure of xmlContent of a word file is simple.
|
|
84
|
+
// All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
|
|
85
|
+
// So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
|
|
86
|
+
// ******************************************************************************************************
|
|
87
|
+
.then(xmlContentArray => {
|
|
88
|
+
/** Store all the text content to respond */
|
|
89
|
+
let responseText = [];
|
|
90
|
+
|
|
91
|
+
xmlContentArray.forEach(xmlContent => {
|
|
92
|
+
/** Find text nodes with w:p tags */
|
|
93
|
+
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
|
|
94
|
+
/** Store all the text content to respond */
|
|
95
|
+
responseText.push(
|
|
96
|
+
Array.from(xmlParagraphNodesList)
|
|
97
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
|
|
98
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
|
|
99
|
+
.map(paragraphNode => {
|
|
100
|
+
// Find text nodes with w:t tags
|
|
101
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
|
|
102
|
+
// Join the texts within this paragraph node without any spaces or delimiters.
|
|
103
|
+
return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
|
|
104
|
+
})
|
|
105
|
+
// Join each paragraph text with a new line delimiter.
|
|
106
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
107
|
+
);
|
|
116
108
|
});
|
|
117
109
|
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
110
|
+
// Join all responseText array
|
|
111
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
112
|
+
// Respond by calling the Callback function.
|
|
113
|
+
callback(responseText, undefined);
|
|
121
114
|
})
|
|
122
|
-
.catch(
|
|
123
|
-
consoleError(error)
|
|
124
|
-
return callback(undefined, error);
|
|
125
|
-
});
|
|
115
|
+
.catch(e => callback(undefined, e));
|
|
126
116
|
}
|
|
127
117
|
|
|
128
118
|
/** Main function for parsing text from PowerPoint files
|
|
129
|
-
* @param {string}
|
|
130
|
-
* @param {function}
|
|
131
|
-
* @param {
|
|
119
|
+
* @param {string} filepath File path
|
|
120
|
+
* @param {function} callback Callback function that returns value or error
|
|
121
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
132
122
|
* @returns {void}
|
|
133
123
|
*/
|
|
134
|
-
function parsePowerPoint(
|
|
135
|
-
if (!fs.existsSync(filename)) {
|
|
136
|
-
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
137
|
-
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
138
|
-
}
|
|
139
|
-
const ext = filename.split(".").pop().toLowerCase();
|
|
140
|
-
if (ext != 'pptx') {
|
|
141
|
-
consoleError(ERRORMSG.extensionUnsupported(extension));
|
|
142
|
-
return callback(undefined, ERRORMSG.extensionUnsupported(ext));
|
|
143
|
-
}
|
|
144
|
-
|
|
145
|
-
/** Store all the text content to respond */
|
|
146
|
-
let responseText = [];
|
|
147
|
-
|
|
148
|
-
/** Extracting text from powerpoint files xml objects converted to js */
|
|
149
|
-
function extractTextFromPowerPointXmlObjects(xmlObjects) {
|
|
150
|
-
// specifically for Arrays
|
|
151
|
-
if (Array.isArray(xmlObjects)) {
|
|
152
|
-
xmlObjects.forEach(item =>
|
|
153
|
-
(typeof item == "string") && (item != "")
|
|
154
|
-
? responseText.push(item)
|
|
155
|
-
: extractTextFromPowerPointXmlObjects(item))
|
|
156
|
-
}
|
|
157
|
-
// for other JS Object
|
|
158
|
-
else if (typeof xmlObjects == "object") {
|
|
159
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
160
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
161
|
-
? (key == "a:t" || key == "_") && value != ""
|
|
162
|
-
? responseText.push(value)
|
|
163
|
-
: undefined
|
|
164
|
-
: extractTextFromPowerPointXmlObjects(value);
|
|
165
|
-
}
|
|
166
|
-
}
|
|
167
|
-
}
|
|
168
|
-
|
|
124
|
+
function parsePowerPoint(filepath, callback, config) {
|
|
169
125
|
// Files regex that hold our content of interest
|
|
170
|
-
const
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
decompress(
|
|
176
|
-
|
|
177
|
-
{ filter: x =>
|
|
126
|
+
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
127
|
+
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
128
|
+
|
|
129
|
+
/** The decompress location which contains the filename in it */
|
|
130
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
131
|
+
decompress(filepath,
|
|
132
|
+
decompressLocation,
|
|
133
|
+
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
178
134
|
)
|
|
179
135
|
.then(files => {
|
|
180
|
-
//
|
|
181
|
-
|
|
136
|
+
// Check if files is corrupted
|
|
137
|
+
if (files.length == 0)
|
|
138
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
182
139
|
|
|
183
|
-
if
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
140
|
+
// Check if any sorting is required.
|
|
141
|
+
if (!config.ignoreNotes && config.putNotesAtLast)
|
|
142
|
+
// Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
|
|
143
|
+
// For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
|
|
144
|
+
files.sort((a,b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
|
|
187
145
|
|
|
188
146
|
// Returning an array of all the xml contents read using fs.readFileSync
|
|
189
|
-
return files.map(file => fs.readFileSync(`${
|
|
147
|
+
return files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
190
148
|
})
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
149
|
+
// ******************************** powerpoint xml files explanation ************************************
|
|
150
|
+
// Structure of xmlContent of a powerpoint file is simple.
|
|
151
|
+
// There are multiple xml files for each slide and correspondingly their notesSlide files.
|
|
152
|
+
// All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
|
|
153
|
+
// So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
|
|
154
|
+
// ******************************************************************************************************
|
|
155
|
+
.then(xmlContentArray => {
|
|
156
|
+
/** Store all the text content to respond */
|
|
157
|
+
let responseText = [];
|
|
158
|
+
|
|
159
|
+
xmlContentArray.forEach(xmlContent => {
|
|
160
|
+
/** Find text nodes with a:p tags */
|
|
161
|
+
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
|
|
162
|
+
/** Store all the text content to respond */
|
|
163
|
+
responseText.push(
|
|
164
|
+
Array.from(xmlParagraphNodesList)
|
|
165
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
|
|
166
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
|
|
167
|
+
.map(paragraphNode => {
|
|
168
|
+
/** Find text nodes with a:t tags */
|
|
169
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
170
|
+
return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
|
|
171
|
+
})
|
|
172
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
173
|
+
);
|
|
204
174
|
});
|
|
205
175
|
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
176
|
+
// Join all responseText array
|
|
177
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
178
|
+
// Respond by calling the Callback function.
|
|
179
|
+
callback(responseText, undefined);
|
|
209
180
|
})
|
|
210
|
-
.catch(
|
|
211
|
-
consoleError(error)
|
|
212
|
-
return callback(undefined, error);
|
|
213
|
-
});
|
|
181
|
+
.catch(e => callback(undefined, e));
|
|
214
182
|
}
|
|
215
183
|
|
|
216
184
|
/** Main function for parsing text from Excel files
|
|
217
|
-
* @param {string}
|
|
218
|
-
* @param {function}
|
|
219
|
-
* @param {
|
|
185
|
+
* @param {string} filepath File path
|
|
186
|
+
* @param {function} callback Callback function that returns value or error
|
|
187
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
220
188
|
* @returns {void}
|
|
221
189
|
*/
|
|
222
|
-
function parseExcel(
|
|
223
|
-
if (!fs.existsSync(filename)) {
|
|
224
|
-
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
225
|
-
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
226
|
-
}
|
|
227
|
-
const ext = filename.split(".").pop().toLowerCase();
|
|
228
|
-
if (ext != 'xlsx') {
|
|
229
|
-
consoleError(ERRORMSG.extensionUnsupported(extension));
|
|
230
|
-
return callback(undefined, ERRORMSG.extensionUnsupported(ext));
|
|
231
|
-
}
|
|
232
|
-
|
|
233
|
-
/** Store all the text content to respond */
|
|
234
|
-
let responseText = [];
|
|
235
|
-
|
|
236
|
-
function extractTextFromExcelXmlObjects2dArray(xmlObjects2dArray) {
|
|
237
|
-
xmlObjects2dArray[0].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 0));
|
|
238
|
-
xmlObjects2dArray[1].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 1));
|
|
239
|
-
xmlObjects2dArray[2].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 2));
|
|
240
|
-
}
|
|
241
|
-
|
|
242
|
-
/** Extracting text from Excel files xml objects converted to js */
|
|
243
|
-
function extractTextFromExcelXmlObjects(xmlObjects, contentFilesIndex) {
|
|
244
|
-
switch(contentFilesIndex) {
|
|
245
|
-
case 0: { // worksheet
|
|
246
|
-
// specifically for Arrays
|
|
247
|
-
if (Array.isArray(xmlObjects)) {
|
|
248
|
-
xmlObjects.forEach(item =>
|
|
249
|
-
item["v"]
|
|
250
|
-
? ((item["$"]["t"] != "s"))
|
|
251
|
-
? responseText.push(item["v"][0])
|
|
252
|
-
: undefined
|
|
253
|
-
: extractTextFromExcelXmlObjects(item, contentFilesIndex))
|
|
254
|
-
}
|
|
255
|
-
// for other JS Object
|
|
256
|
-
else if (typeof xmlObjects == "object") {
|
|
257
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
258
|
-
value["v"]
|
|
259
|
-
? ((value["$"]["t"] == "s"))
|
|
260
|
-
? responseText.push(value["v"][0])
|
|
261
|
-
: undefined
|
|
262
|
-
: extractTextFromExcelXmlObjects(value, contentFilesIndex);
|
|
263
|
-
}
|
|
264
|
-
}
|
|
265
|
-
break;
|
|
266
|
-
}
|
|
267
|
-
case 1: { // sharedStrings
|
|
268
|
-
// specifically for Arrays
|
|
269
|
-
if (Array.isArray(xmlObjects)) {
|
|
270
|
-
xmlObjects.forEach(item =>
|
|
271
|
-
(typeof item == "string") && (item != "")
|
|
272
|
-
? responseText.push(item)
|
|
273
|
-
: extractTextFromExcelXmlObjects(item, contentFilesIndex))
|
|
274
|
-
}
|
|
275
|
-
// for other JS Object
|
|
276
|
-
else if (typeof xmlObjects == "object") {
|
|
277
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
278
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
279
|
-
? (key == "t" || key == "_") && (value != "")
|
|
280
|
-
? responseText.push(value)
|
|
281
|
-
: undefined
|
|
282
|
-
: extractTextFromExcelXmlObjects(value, contentFilesIndex);
|
|
283
|
-
}
|
|
284
|
-
}
|
|
285
|
-
break;
|
|
286
|
-
}
|
|
287
|
-
case 2: { // drawings
|
|
288
|
-
// specifically for Arrays
|
|
289
|
-
if (Array.isArray(xmlObjects)) {
|
|
290
|
-
xmlObjects.forEach(item =>
|
|
291
|
-
(typeof item == "string") && (item != "")
|
|
292
|
-
? responseText.push(item)
|
|
293
|
-
: extractTextFromExcelXmlObjects(item, contentFilesIndex))
|
|
294
|
-
}
|
|
295
|
-
// for other JS Object
|
|
296
|
-
else if (typeof xmlObjects == "object") {
|
|
297
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
298
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
299
|
-
? (key == "a:t" || key == "_") && (value != "")
|
|
300
|
-
? responseText.push(value)
|
|
301
|
-
: undefined
|
|
302
|
-
: extractTextFromExcelXmlObjects(value, contentFilesIndex);
|
|
303
|
-
}
|
|
304
|
-
}
|
|
305
|
-
break;
|
|
306
|
-
}
|
|
307
|
-
}
|
|
308
|
-
}
|
|
309
|
-
|
|
190
|
+
function parseExcel(filepath, callback, config) {
|
|
310
191
|
// Files regex that hold our content of interest
|
|
311
|
-
const
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
192
|
+
const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
|
|
193
|
+
const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
|
|
194
|
+
const chartsRegex = /xl\/charts\/chart\d+.xml/g;
|
|
195
|
+
const stringsFilePath = 'xl/sharedStrings.xml';
|
|
196
|
+
|
|
197
|
+
/** The decompress location which contains the filename in it */
|
|
198
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
199
|
+
decompress(filepath,
|
|
200
|
+
decompressLocation,
|
|
201
|
+
{ filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
|
|
320
202
|
)
|
|
321
203
|
.then(files => {
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
// Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
|
|
332
|
-
return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
|
|
204
|
+
if (files.length == 0)
|
|
205
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
206
|
+
|
|
207
|
+
return {
|
|
208
|
+
sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
209
|
+
drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
210
|
+
chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
211
|
+
sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
|
|
212
|
+
};
|
|
333
213
|
})
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
214
|
+
// ********************************** excel xml files explanation ***************************************
|
|
215
|
+
// Structure of xmlContent of an excel file is a bit complex.
|
|
216
|
+
// We have a sharedStrings.xml file which has strings inside t tags
|
|
217
|
+
// Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
|
|
218
|
+
// Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
|
|
219
|
+
// If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
|
|
220
|
+
// Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
|
|
221
|
+
// ******************************************************************************************************
|
|
222
|
+
.then(xmlContentFilesObject => {
|
|
223
|
+
/** Store all the text content to respond */
|
|
224
|
+
let responseText = [];
|
|
225
|
+
|
|
226
|
+
/** Find text nodes with t tags in sharedStrings xml file */
|
|
227
|
+
const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
|
|
228
|
+
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
229
|
+
const sharedStrings = Array.from(sharedStringsXmlTNodesList).map(tNode => tNode.childNodes[0].nodeValue);
|
|
230
|
+
|
|
231
|
+
// Parse Sheet files
|
|
232
|
+
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
233
|
+
/** Find text nodes with c tags in sharedStrings xml file */
|
|
234
|
+
const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
|
|
235
|
+
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
236
|
+
responseText.push(
|
|
237
|
+
Array.from(sheetsXmlCNodesList)
|
|
238
|
+
// Filter c nodes than do not have any v nodes
|
|
239
|
+
.filter(cNode => cNode.getElementsByTagName("v").length != 0)
|
|
240
|
+
.map(cNode => {
|
|
241
|
+
/** Flag whether this node's value represents a string index */
|
|
242
|
+
const isString = cNode.getAttribute("t") == "s";
|
|
243
|
+
/** Find value nodes represented by v tags */
|
|
244
|
+
const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
|
|
245
|
+
// Validate text
|
|
246
|
+
if (isString && value >= sharedStrings.length)
|
|
247
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
248
|
+
|
|
249
|
+
return isString
|
|
250
|
+
? sharedStrings[value]
|
|
251
|
+
: value;
|
|
252
|
+
})
|
|
253
|
+
// Join each cell text within a sheet with a space.
|
|
254
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
255
|
+
);
|
|
256
|
+
});
|
|
257
|
+
|
|
258
|
+
// Parse Drawing files
|
|
259
|
+
xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
|
|
260
|
+
/** Find text nodes with a:p tags */
|
|
261
|
+
const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
|
|
262
|
+
/** Store all the text content to respond */
|
|
263
|
+
responseText.push(
|
|
264
|
+
Array.from(drawingsXmlParagraphNodesList)
|
|
265
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
|
|
266
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
|
|
267
|
+
.map(paragraphNode => {
|
|
268
|
+
/** Find text nodes with a:t tags */
|
|
269
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
270
|
+
return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
|
|
271
|
+
})
|
|
272
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
273
|
+
);
|
|
347
274
|
});
|
|
348
275
|
|
|
349
|
-
|
|
350
|
-
.
|
|
276
|
+
// Parse Chart files
|
|
277
|
+
xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
|
|
278
|
+
/** Find text nodes with c:v tags */
|
|
279
|
+
const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
|
|
280
|
+
/** Store all the text content to respond */
|
|
281
|
+
responseText.push(
|
|
282
|
+
Array.from(chartsXmlCVNodesList)
|
|
283
|
+
.map(cVNode => cVNode.childNodes[0].nodeValue)
|
|
284
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
285
|
+
);
|
|
286
|
+
});
|
|
351
287
|
|
|
288
|
+
// Join all responseText array
|
|
289
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
290
|
+
// Respond by calling the Callback function.
|
|
291
|
+
callback(responseText, undefined);
|
|
352
292
|
})
|
|
353
|
-
.catch(
|
|
354
|
-
consoleError(error)
|
|
355
|
-
return callback(undefined, error);
|
|
356
|
-
});
|
|
293
|
+
.catch(e => callback(undefined, e));
|
|
357
294
|
}
|
|
358
295
|
|
|
359
296
|
|
|
360
297
|
/** Main function for parsing text from open office files
|
|
361
|
-
* @param {string}
|
|
362
|
-
* @param {function}
|
|
363
|
-
* @param {
|
|
298
|
+
* @param {string} filepath File path
|
|
299
|
+
* @param {function} callback Callback function that returns value or error
|
|
300
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
364
301
|
* @returns {void}
|
|
365
302
|
*/
|
|
366
|
-
function parseOpenOffice(
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
303
|
+
function parseOpenOffice(filepath, callback, config) {
|
|
304
|
+
/** The target content xml file for the openoffice file. */
|
|
305
|
+
const mainContentFilePath = 'content.xml';
|
|
306
|
+
const objectContentFilesRegex = /Object \d+\/content.xml/g;
|
|
307
|
+
|
|
308
|
+
/** The decompress location which contains the filename in it */
|
|
309
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
310
|
+
decompress(filepath,
|
|
311
|
+
decompressLocation,
|
|
312
|
+
{ filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
|
|
313
|
+
)
|
|
314
|
+
.then(files => {
|
|
315
|
+
if (files.length == 0)
|
|
316
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
376
317
|
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
318
|
+
return {
|
|
319
|
+
mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
|
|
320
|
+
objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
321
|
+
}
|
|
322
|
+
})
|
|
323
|
+
// ********************************** openoffice xml files explanation **********************************
|
|
324
|
+
// Structure of xmlContent of openoffice files is simple.
|
|
325
|
+
// All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
|
|
326
|
+
// All text in these tags are separated by new line delimiters.
|
|
327
|
+
// Objects like charts in ods files are in Object d+/content.xml with the same way as above.
|
|
328
|
+
// ******************************************************************************************************
|
|
329
|
+
.then(xmlContentFilesObject => {
|
|
330
|
+
/** Store all the notes text content to respond */
|
|
331
|
+
let notesText = [];
|
|
332
|
+
/** Store all the text content to respond */
|
|
333
|
+
let responseText = [];
|
|
334
|
+
|
|
335
|
+
/** List of allowed text tags */
|
|
336
|
+
const allowedTextTags = ["text:p", "text:h"];
|
|
337
|
+
/** List of notes tags */
|
|
338
|
+
const notesTag = "presentation:notes";
|
|
339
|
+
|
|
340
|
+
/** Main dfs traversal function that goes from one node to its children and returns the value out. */
|
|
341
|
+
function extractAllTextsFromNode(root) {
|
|
342
|
+
let xmlTextArray = []
|
|
343
|
+
for (let i = 0; i < root.childNodes.length; i++)
|
|
344
|
+
traversal(root.childNodes[i], xmlTextArray, true);
|
|
345
|
+
return xmlTextArray.join("");
|
|
387
346
|
}
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
347
|
+
/** Traversal function that gets recursive calling. */
|
|
348
|
+
function traversal(node, xmlTextArray, isFirstRecursion) {
|
|
349
|
+
if(!node.childNodes || node.childNodes.length == 0)
|
|
350
|
+
{
|
|
351
|
+
if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
|
|
352
|
+
if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
|
|
353
|
+
notesText.push(node.nodeValue);
|
|
354
|
+
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
355
|
+
notesText.push(config.newlineDelimiter ?? "\n");
|
|
356
|
+
}
|
|
357
|
+
else {
|
|
358
|
+
xmlTextArray.push(node.nodeValue);
|
|
359
|
+
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
360
|
+
xmlTextArray.push(config.newlineDelimiter ?? "\n");
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
return;
|
|
396
364
|
}
|
|
365
|
+
|
|
366
|
+
for (let i = 0; i < node.childNodes.length; i++)
|
|
367
|
+
traversal(node.childNodes[i], xmlTextArray, false);
|
|
397
368
|
}
|
|
398
|
-
}
|
|
399
369
|
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
consoleError(ERRORMSG.fileCorrupted(filename));
|
|
408
|
-
return callback(undefined, ERRORMSG.fileCorrupted(filename));
|
|
370
|
+
/** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
|
|
371
|
+
function isNotesNode(node) {
|
|
372
|
+
if (node.tagName == notesTag)
|
|
373
|
+
return true;
|
|
374
|
+
if (node.parentNode)
|
|
375
|
+
return isNotesNode(node.parentNode);
|
|
376
|
+
return false;
|
|
409
377
|
}
|
|
410
378
|
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
379
|
+
/** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
|
|
380
|
+
function isInvalidTextNode(node) {
|
|
381
|
+
if (allowedTextTags.includes(node.tagName))
|
|
382
|
+
return true;
|
|
383
|
+
if (node.parentNode)
|
|
384
|
+
return isInvalidTextNode(node.parentNode);
|
|
385
|
+
return false;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** The xml string parsed as xml array */
|
|
389
|
+
const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
|
|
390
|
+
// Iterate over each xmlContent and extract text from them.
|
|
391
|
+
xmlContentArray.forEach(xmlContent => {
|
|
392
|
+
/** Find text nodes with text:h and text:p tags in xmlContent */
|
|
393
|
+
const xmlTextNodesList = [...Array.from(xmlContent
|
|
394
|
+
.getElementsByTagName("*"))
|
|
395
|
+
.filter(node => allowedTextTags.includes(node.tagName)
|
|
396
|
+
&& !isInvalidTextNode(node.parentNode))
|
|
397
|
+
];
|
|
398
|
+
/** Store all the text content to respond */
|
|
399
|
+
responseText.push(
|
|
400
|
+
xmlTextNodesList
|
|
401
|
+
// Add every text information from within this textNode and combine them together.
|
|
402
|
+
.map(textNode => extractAllTextsFromNode(textNode))
|
|
403
|
+
.filter(text => text != "")
|
|
404
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
405
|
+
);
|
|
425
406
|
});
|
|
426
407
|
|
|
427
|
-
|
|
428
|
-
|
|
408
|
+
// Add notes text at the end if the user config says so.
|
|
409
|
+
// Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
|
|
410
|
+
if (!config.ignoreNotes && config.putNotesAtLast)
|
|
411
|
+
responseText = [...responseText, ...notesText];
|
|
429
412
|
|
|
413
|
+
// Join all responseText array
|
|
414
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
415
|
+
// Respond by calling the Callback function.
|
|
416
|
+
callback(responseText, undefined);
|
|
430
417
|
})
|
|
431
|
-
.catch(
|
|
432
|
-
consoleError(error)
|
|
433
|
-
return callback(undefined, error);
|
|
434
|
-
});
|
|
418
|
+
.catch(e => callback(undefined, e));
|
|
435
419
|
}
|
|
436
420
|
|
|
421
|
+
/** Header for error messages */
|
|
422
|
+
const PDFPARSEERRORHEADER = "[pdf-parse]: ";
|
|
423
|
+
|
|
424
|
+
/** Main function for parsing text from pdf files
|
|
425
|
+
* @param {string} filepath File path
|
|
426
|
+
* @param {function} callback Callback function that returns value or error
|
|
427
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
428
|
+
* @returns {void}
|
|
429
|
+
*/
|
|
430
|
+
function parsePdf(filepath, callback, config) {
|
|
431
|
+
// Get the data buffer for the given file path.
|
|
432
|
+
const dataBuffer = fs.readFileSync(filepath);
|
|
433
|
+
|
|
434
|
+
pdfParse(dataBuffer)
|
|
435
|
+
.then(data => {
|
|
436
|
+
let text = data.text;
|
|
437
|
+
if (!!config.newlineDelimiter && config.newlineDelimiter != "\n")
|
|
438
|
+
text = text.replaceAll("\n", config.newlineDelimiter)
|
|
439
|
+
callback(text, undefined);
|
|
440
|
+
})
|
|
441
|
+
.catch(e => callback(undefined, PDFPARSEERRORHEADER + e));
|
|
442
|
+
}
|
|
437
443
|
|
|
438
444
|
/** Main async function with callback to execute parseOffice for supported files
|
|
439
|
-
* @param {string | Buffer}
|
|
440
|
-
* @param {function}
|
|
441
|
-
* @param {
|
|
445
|
+
* @param {string | Buffer} file File path or file buffers
|
|
446
|
+
* @param {function} callback Callback function that returns value or error
|
|
447
|
+
* @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
|
|
442
448
|
* @returns {void}
|
|
443
449
|
*/
|
|
444
|
-
function parseOffice(file, callback,
|
|
445
|
-
// filename that is to be filled below depending on file input from argument
|
|
446
|
-
let filename = "";
|
|
450
|
+
function parseOffice(file, callback, config = {}) {
|
|
447
451
|
// Prepare file for processing
|
|
448
|
-
const filePreparedPromise = new Promise((res, rej) =>
|
|
449
|
-
|
|
452
|
+
const filePreparedPromise = new Promise((res, rej) => {
|
|
453
|
+
// create temp file subdirectory if it does not exist
|
|
454
|
+
fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
|
|
455
|
+
|
|
450
456
|
// Check if buffer
|
|
451
|
-
if (Buffer.isBuffer(file))
|
|
452
|
-
{
|
|
457
|
+
if (Buffer.isBuffer(file)) {
|
|
453
458
|
// Guess file type from buffer
|
|
454
459
|
fileType.fromBuffer(file)
|
|
455
460
|
.then(data =>
|
|
456
461
|
{
|
|
457
462
|
// temp file name
|
|
458
|
-
|
|
459
|
-
// create directory if it does not exist
|
|
460
|
-
fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
|
|
463
|
+
const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
|
|
461
464
|
// write new file
|
|
462
|
-
fs.writeFileSync(
|
|
465
|
+
fs.writeFileSync(newfilepath, file);
|
|
463
466
|
// resolve promise
|
|
464
|
-
res();
|
|
467
|
+
res(newfilepath);
|
|
465
468
|
})
|
|
466
|
-
.catch(() => rej());
|
|
469
|
+
.catch(() => rej(ERRORMSG.improperBuffers));
|
|
467
470
|
return;
|
|
468
471
|
}
|
|
469
472
|
|
|
470
|
-
//
|
|
471
|
-
|
|
473
|
+
// Not buffers but real file path.
|
|
474
|
+
|
|
475
|
+
// Check if file exists
|
|
476
|
+
if (!fs.existsSync(file))
|
|
477
|
+
throw ERRORMSG.fileDoesNotExist(file);
|
|
478
|
+
|
|
479
|
+
// temp file name
|
|
480
|
+
const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
|
|
481
|
+
// Copy the file into a temp location with the temp name
|
|
482
|
+
fs.copyFileSync(file, newfilepath)
|
|
472
483
|
// resolve promise
|
|
473
|
-
res();
|
|
474
|
-
})
|
|
484
|
+
res(newfilepath);
|
|
485
|
+
});
|
|
475
486
|
|
|
476
487
|
// Process filePreparedPromise resolution.
|
|
477
488
|
filePreparedPromise
|
|
478
|
-
.then(
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
if (!fs.existsSync(filename)) {
|
|
482
|
-
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
483
|
-
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
484
|
-
}
|
|
485
|
-
var extension = filename.split(".").pop().toLowerCase();
|
|
489
|
+
.then(filepath => {
|
|
490
|
+
// File extension. Already in lowercase when we prepared the temp file above.
|
|
491
|
+
const extension = filepath.split(".").pop();
|
|
486
492
|
|
|
487
493
|
// Switch between parsing functions depending on extension.
|
|
488
|
-
switch(extension)
|
|
489
|
-
{
|
|
494
|
+
switch(extension) {
|
|
490
495
|
case "docx":
|
|
491
|
-
parseWord(
|
|
492
|
-
|
|
496
|
+
parseWord(filepath, internalCallback, config);
|
|
497
|
+
break;
|
|
493
498
|
case "pptx":
|
|
494
|
-
parsePowerPoint(
|
|
495
|
-
|
|
499
|
+
parsePowerPoint(filepath, internalCallback, config);
|
|
500
|
+
break;
|
|
496
501
|
case "xlsx":
|
|
497
|
-
parseExcel(
|
|
498
|
-
|
|
502
|
+
parseExcel(filepath, internalCallback, config);
|
|
503
|
+
break;
|
|
499
504
|
case "odt":
|
|
500
505
|
case "odp":
|
|
501
506
|
case "ods":
|
|
502
|
-
parseOpenOffice(
|
|
503
|
-
|
|
504
|
-
|
|
507
|
+
parseOpenOffice(filepath, internalCallback, config);
|
|
508
|
+
break;
|
|
509
|
+
case "pdf":
|
|
510
|
+
parsePdf(filepath, internalCallback, config);
|
|
511
|
+
break;
|
|
512
|
+
|
|
505
513
|
default:
|
|
506
|
-
|
|
507
|
-
|
|
514
|
+
throw ERRORMSG.extensionUnsupported(extension);
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
518
|
+
function internalCallback(data, err) {
|
|
519
|
+
if (err)
|
|
520
|
+
consoleError(err, config.outputErrorToConsole)
|
|
521
|
+
// Call the original callback
|
|
522
|
+
callback(data, err);
|
|
523
|
+
// Check if we need to preserve unzipped content files or delete them.
|
|
524
|
+
if (config.preserveTempFiles)
|
|
525
|
+
return;
|
|
526
|
+
// Delete decompress sublocation.
|
|
527
|
+
rimraf(decompressSubLocation, rimrafErr => consoleError(rimrafErr, config.outputErrorToConsole));
|
|
508
528
|
}
|
|
509
529
|
})
|
|
510
|
-
.catch(
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
530
|
+
.catch(error => {
|
|
531
|
+
consoleError(error, config.outputErrorToConsole);
|
|
532
|
+
callback(undefined, error);
|
|
533
|
+
});
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
/**
|
|
537
|
+
* Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
538
|
+
* @param {string | Buffer} file File path or file buffers
|
|
539
|
+
* @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
|
|
540
|
+
* @returns {Promise<string>}
|
|
541
|
+
*/
|
|
542
|
+
function parseOfficeAsync (file, config) {
|
|
543
|
+
return new Promise((res, rej) => {
|
|
544
|
+
parseOffice(file, function (data, err) {
|
|
545
|
+
if (err)
|
|
546
|
+
return rej(err);
|
|
547
|
+
return res(data);
|
|
548
|
+
}, config);
|
|
549
|
+
});
|
|
515
550
|
}
|
|
516
551
|
|
|
517
552
|
/**
|
|
@@ -522,147 +557,18 @@ function parseOffice(file, callback, deleteOfficeDist = true) {
|
|
|
522
557
|
function setDecompressionLocation(newLocation) {
|
|
523
558
|
if (newLocation != undefined) {
|
|
524
559
|
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
|
|
525
|
-
|
|
526
560
|
if (fs.existsSync(newLocation))
|
|
527
561
|
decompressSubLocation = newLocation;
|
|
528
562
|
return;
|
|
529
563
|
}
|
|
530
|
-
consoleError(ERRORMSG.locationNotFound(newLocation));
|
|
564
|
+
consoleError(ERRORMSG.locationNotFound(newLocation), config.outputErrorToConsole);
|
|
531
565
|
decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
532
566
|
}
|
|
533
567
|
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
function enableConsoleOutput() {
|
|
538
|
-
outputErrorToConsole = true;
|
|
539
|
-
}
|
|
540
|
-
|
|
541
|
-
/** Disabled console output
|
|
542
|
-
* @returns {void}
|
|
543
|
-
*/
|
|
544
|
-
function disableConsoleOutput() {
|
|
545
|
-
outputErrorToConsole = false;
|
|
546
|
-
}
|
|
547
|
-
|
|
548
|
-
|
|
549
|
-
// #region Promise versions of above functions
|
|
550
|
-
|
|
551
|
-
/** Async function that can be used with await to execute parseWord. Or it can be used with promises.
|
|
552
|
-
* @param {string} filename File path
|
|
553
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
554
|
-
* @returns {Promise<string>}
|
|
555
|
-
*/
|
|
556
|
-
var parseWordAsync = function (filename, deleteOfficeDist = true) {
|
|
557
|
-
return new Promise((resolve, reject) => {
|
|
558
|
-
try {
|
|
559
|
-
parseWord(filename, function (data, error) {
|
|
560
|
-
if (error)
|
|
561
|
-
return reject(error);
|
|
562
|
-
return resolve(data);
|
|
563
|
-
}, deleteOfficeDist);
|
|
564
|
-
}
|
|
565
|
-
catch (error) {
|
|
566
|
-
return reject(error);
|
|
567
|
-
}
|
|
568
|
-
})
|
|
569
|
-
}
|
|
570
|
-
|
|
571
|
-
/** Async function that can be used with await to execute parsePowerPoint. Or it can be used with promises.
|
|
572
|
-
* @param {string} filename File path
|
|
573
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
574
|
-
* @returns {Promise<string>}
|
|
575
|
-
*/
|
|
576
|
-
var parsePowerPointAsync = function (filename, deleteOfficeDist = true) {
|
|
577
|
-
return new Promise((resolve, reject) => {
|
|
578
|
-
try {
|
|
579
|
-
parsePowerPoint(filename, function (data, err) {
|
|
580
|
-
if (err)
|
|
581
|
-
return reject(err);
|
|
582
|
-
return resolve(data);
|
|
583
|
-
}, deleteOfficeDist);
|
|
584
|
-
}
|
|
585
|
-
catch (error) {
|
|
586
|
-
return reject(error);
|
|
587
|
-
}
|
|
588
|
-
})
|
|
589
|
-
}
|
|
590
|
-
|
|
591
|
-
/** Async function that can be used with await to execute parseExcel. Or it can be used with promises.
|
|
592
|
-
* @param {string} filename File path
|
|
593
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
594
|
-
* @returns {Promise<string>}
|
|
595
|
-
*/
|
|
596
|
-
var parseExcelAsync = function (filename, deleteOfficeDist = true) {
|
|
597
|
-
return new Promise((resolve, reject) => {
|
|
598
|
-
try {
|
|
599
|
-
parseExcel(filename, function (data, err) {
|
|
600
|
-
if (err)
|
|
601
|
-
return reject(err);
|
|
602
|
-
return resolve(data);
|
|
603
|
-
}, deleteOfficeDist);
|
|
604
|
-
}
|
|
605
|
-
catch (error) {
|
|
606
|
-
return reject(error);
|
|
607
|
-
}
|
|
608
|
-
})
|
|
609
|
-
}
|
|
610
|
-
|
|
611
|
-
/** Async function that can be used with await to execute parseOpenOffice. Or it can be used with promises.
|
|
612
|
-
* @param {string} filename File path
|
|
613
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
614
|
-
* @returns {Promise<string>}
|
|
615
|
-
*/
|
|
616
|
-
var parseOpenOfficeAsync = function (filename, deleteOfficeDist = true) {
|
|
617
|
-
return new Promise((resolve, reject) => {
|
|
618
|
-
try {
|
|
619
|
-
parseOpenOffice(filename, function (data, err) {
|
|
620
|
-
if (err)
|
|
621
|
-
return reject(err);
|
|
622
|
-
return resolve(data);
|
|
623
|
-
}, deleteOfficeDist);
|
|
624
|
-
}
|
|
625
|
-
catch (error) {
|
|
626
|
-
return reject(error);
|
|
627
|
-
}
|
|
628
|
-
})
|
|
629
|
-
}
|
|
630
|
-
|
|
631
|
-
/**
|
|
632
|
-
* Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
633
|
-
* @param {string | Buffer} file File path or file buffers
|
|
634
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
635
|
-
* @returns {Promise<string>}
|
|
636
|
-
*/
|
|
637
|
-
var parseOfficeAsync = function (file, deleteOfficeDist = true) {
|
|
638
|
-
return new Promise((resolve, reject) => {
|
|
639
|
-
try {
|
|
640
|
-
parseOffice(file, function (data, err) {
|
|
641
|
-
if (err)
|
|
642
|
-
return reject(err);
|
|
643
|
-
return resolve(data);
|
|
644
|
-
}, deleteOfficeDist);
|
|
645
|
-
}
|
|
646
|
-
catch (error) {
|
|
647
|
-
return reject(error);
|
|
648
|
-
}
|
|
649
|
-
})
|
|
650
|
-
}
|
|
651
|
-
// #endregion Async Versions
|
|
652
|
-
|
|
653
|
-
module.exports.parseWord = parseWord;
|
|
654
|
-
module.exports.parsePowerPoint = parsePowerPoint;
|
|
655
|
-
module.exports.parseExcel = parseExcel;
|
|
656
|
-
module.exports.parseOpenOffice = parseOpenOffice;
|
|
657
|
-
module.exports.parseOffice = parseOffice;
|
|
658
|
-
module.exports.parseWordAsync = parseWordAsync;
|
|
659
|
-
module.exports.parsePowerPointAsync = parsePowerPointAsync;
|
|
660
|
-
module.exports.parseExcelAsync = parseExcelAsync;
|
|
661
|
-
module.exports.parseOpenOfficeAsync = parseOpenOfficeAsync;
|
|
662
|
-
module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
568
|
+
// Export functions
|
|
569
|
+
module.exports.parseOffice = parseOffice;
|
|
570
|
+
module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
663
571
|
module.exports.setDecompressionLocation = setDecompressionLocation;
|
|
664
|
-
module.exports.enableConsoleOutput = enableConsoleOutput;
|
|
665
|
-
module.exports.disableConsoleOutput = disableConsoleOutput;
|
|
666
572
|
|
|
667
573
|
|
|
668
574
|
// Run this library on CLI
|
|
@@ -672,8 +578,8 @@ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').po
|
|
|
672
578
|
}
|
|
673
579
|
else if (process.argv.length == 3)
|
|
674
580
|
parseOfficeAsync(process.argv[2])
|
|
675
|
-
|
|
676
|
-
|
|
581
|
+
.then(text => console.log(text))
|
|
582
|
+
.catch(error => console.error(ERRORHEADER + error))
|
|
677
583
|
else
|
|
678
584
|
console.error(ERRORMSG.improperArguments)
|
|
679
585
|
}
|