officeparser 3.2.2 → 4.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +104 -228
- package/officeParser.js +472 -521
- package/package.json +10 -4
- package/typings/officeParser.d.ts +29 -67
package/officeParser.js
CHANGED
|
@@ -1,623 +1,574 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
|
-
const decompress
|
|
4
|
-
const
|
|
5
|
-
const
|
|
6
|
-
const
|
|
3
|
+
const decompress = require('decompress');
|
|
4
|
+
const fs = require('fs');
|
|
5
|
+
const rimraf = require('rimraf');
|
|
6
|
+
const fileType = require('file-type');
|
|
7
|
+
const pdfParse = require('pdf-parse');
|
|
8
|
+
const { DOMParser } = require('xmldom');
|
|
7
9
|
|
|
8
10
|
/** Header for error messages */
|
|
9
11
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
10
12
|
/** Error messages */
|
|
11
13
|
const ERRORMSG = {
|
|
12
|
-
extensionUnsupported: (ext) =>
|
|
13
|
-
fileCorrupted: (
|
|
14
|
-
fileDoesNotExist: (
|
|
15
|
-
locationNotFound: (location) =>
|
|
16
|
-
improperArguments:
|
|
14
|
+
extensionUnsupported: (ext) => `Sorry, OfficeParser currently support docx, pptx, xlsx, odt, odp, ods, pdf files only. Create a ticket in Issues on github to add support for ${ext} files. Stay tuned for further updates.`,
|
|
15
|
+
fileCorrupted: (filepath) => `Your file ${filepath} seems to be corrupted. If you are sure it is fine, please create a ticket in Issues on github with the file to reproduce error.`,
|
|
16
|
+
fileDoesNotExist: (filepath) => `File ${filepath} could not be found! Check if the file exists or verify if the relative path to the file is correct from your terminal's location.`,
|
|
17
|
+
locationNotFound: (location) => `Entered location ${location} is not valid! Check relative paths and reenter. OfficeParser will use root directory as decompress location.`,
|
|
18
|
+
improperArguments: `Improper arguments`,
|
|
19
|
+
improperBuffers: `Error occured while reading the file buffers`
|
|
17
20
|
}
|
|
18
21
|
/** Default sublocation for decompressing files under the current directory. */
|
|
19
22
|
const DEFAULTDECOMPRESSSUBLOCATION = "officeDist";
|
|
20
23
|
/** Location for decompressing files. Default is "officeDist" */
|
|
21
24
|
let decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
22
|
-
/** Flag to output errors to console other than normal error handling. Default is false as we anyway push the message for error handling. */
|
|
23
|
-
let outputErrorToConsole = false;
|
|
24
25
|
|
|
25
26
|
/** Console error if allowed
|
|
26
|
-
* @param {string} errorMessage
|
|
27
|
+
* @param {string} errorMessage Error message to show on the console
|
|
28
|
+
* @param {string} outputErrorToConsole Flag to show log on console. Ignore if not true.
|
|
27
29
|
* @returns {void}
|
|
28
30
|
*/
|
|
29
|
-
function consoleError(errorMessage) {
|
|
30
|
-
if (outputErrorToConsole)
|
|
31
|
-
|
|
31
|
+
function consoleError(errorMessage, outputErrorToConsole) {
|
|
32
|
+
if (!errorMessage || !outputErrorToConsole)
|
|
33
|
+
return;
|
|
34
|
+
console.error(ERRORHEADER + errorMessage);
|
|
32
35
|
}
|
|
33
36
|
|
|
34
|
-
/**
|
|
37
|
+
/** Returns parsed xml document for a given xml text.
|
|
35
38
|
* @param {string} xml The xml string from the doc file
|
|
36
|
-
* @
|
|
37
|
-
|
|
39
|
+
* @returns {XMLDocument}
|
|
40
|
+
*/
|
|
41
|
+
const parseString = (xml) => {
|
|
42
|
+
let parser = new DOMParser();
|
|
43
|
+
return parser.parseFromString(xml, "text/xml");
|
|
44
|
+
};
|
|
45
|
+
|
|
46
|
+
/** @typedef {Object} OfficeParserConfig
|
|
47
|
+
* @property {boolean} preserveTempFiles Flag to not delete the internal content files and the duplicate temp files that it uses after unzipping office files. Default is false. It deletes all of those files.
|
|
48
|
+
* @property {boolean} outputErrorToConsole Flag to show all the logs to console in case of an error irrespective of your own handling.
|
|
49
|
+
* @property {string} newlineDelimiter The delimiter used for every new line in places that allow multiline text like word. Default is \n.
|
|
50
|
+
* @property {boolean} ignoreNotes Flag to ignore notes from parsing in files like powerpoint. Default is false. It includes notes in the parsed text by default.
|
|
51
|
+
* @property {boolean} putNotesAtLast Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint. Default is false. It puts each notes right after its main slide content. If ignoreNotes is set to true, this flag is also ignored.
|
|
38
52
|
*/
|
|
39
|
-
const parseStringPromise = (xml, ignoreAttrs = true) => new Promise((resolve, reject) => {
|
|
40
|
-
xml2js.parseString(xml, { "ignoreAttrs": ignoreAttrs }, (err, result) => {
|
|
41
|
-
if (err)
|
|
42
|
-
reject(err);
|
|
43
|
-
resolve(result);
|
|
44
|
-
});
|
|
45
|
-
});
|
|
46
53
|
|
|
47
54
|
|
|
48
55
|
/** Main function for parsing text from word files
|
|
49
|
-
* @param {string}
|
|
50
|
-
* @param {function}
|
|
51
|
-
* @param {
|
|
56
|
+
* @param {string} filepath File path
|
|
57
|
+
* @param {function} callback Callback function that returns value or error
|
|
58
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
52
59
|
* @returns {void}
|
|
53
60
|
*/
|
|
54
|
-
function parseWord(
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
/** Store all the text content to respond */
|
|
66
|
-
let responseText = [];
|
|
67
|
-
|
|
68
|
-
/** Extracting text from Word files xml objects converted to js */
|
|
69
|
-
function extractTextFromWordXmlObjects(xmlObjects) {
|
|
70
|
-
// specifically for Arrays
|
|
71
|
-
if (Array.isArray(xmlObjects)) {
|
|
72
|
-
xmlObjects.forEach(item =>
|
|
73
|
-
(typeof item == "string") && (item != "")
|
|
74
|
-
? responseText.push(item)
|
|
75
|
-
: extractTextFromWordXmlObjects(item))
|
|
76
|
-
}
|
|
77
|
-
// for other JS Object
|
|
78
|
-
else if (typeof xmlObjects == "object") {
|
|
79
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
80
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
81
|
-
? (key == "w:t" || key == "_") && value != ""
|
|
82
|
-
? responseText.push(value)
|
|
83
|
-
: undefined
|
|
84
|
-
: extractTextFromWordXmlObjects(value);
|
|
85
|
-
}
|
|
86
|
-
}
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
const contentFile = 'word/document.xml';
|
|
90
|
-
decompress(filename,
|
|
91
|
-
decompressSubLocation,
|
|
92
|
-
{ filter: x => x.path == contentFile }
|
|
61
|
+
function parseWord(filepath, callback, config) {
|
|
62
|
+
/** The target content xml file for the docx file. */
|
|
63
|
+
const mainContentFile = 'word/document.xml';
|
|
64
|
+
const footnotesFile = 'word/footnotes.xml';
|
|
65
|
+
const endnotesFile = 'word/endnotes.xml';
|
|
66
|
+
/** The decompress location which contains the filename in it */
|
|
67
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
68
|
+
decompress(filepath,
|
|
69
|
+
decompressLocation,
|
|
70
|
+
{ filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
|
|
93
71
|
)
|
|
94
72
|
.then(files => {
|
|
95
|
-
if (files.length
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
73
|
+
if (files.length == 0)
|
|
74
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
75
|
+
|
|
76
|
+
return [...files.filter(file => file.path == mainContentFile),
|
|
77
|
+
...files.filter(file => file.path == footnotesFile),
|
|
78
|
+
...files.filter(file => file.path == endnotesFile)
|
|
79
|
+
]
|
|
80
|
+
.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
101
81
|
})
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
82
|
+
// ************************************* word xml files explanation *************************************
|
|
83
|
+
// Structure of xmlContent of a word file is simple.
|
|
84
|
+
// All text nodes are within w:t tags and each of the text nodes that belong in one paragraph are clubbed together within a w:p tag.
|
|
85
|
+
// So, we will filter out all the empty w:p tags and then combine all the w:t tag text inside for creating our response text.
|
|
86
|
+
// ******************************************************************************************************
|
|
87
|
+
.then(xmlContentArray => {
|
|
88
|
+
/** Store all the text content to respond */
|
|
89
|
+
let responseText = [];
|
|
90
|
+
|
|
91
|
+
xmlContentArray.forEach(xmlContent => {
|
|
92
|
+
/** Find text nodes with w:p tags */
|
|
93
|
+
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("w:p");
|
|
94
|
+
/** Store all the text content to respond */
|
|
95
|
+
responseText.push(
|
|
96
|
+
Array.from(xmlParagraphNodesList)
|
|
97
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by w:t tag
|
|
98
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("w:t").length != 0)
|
|
99
|
+
.map(paragraphNode => {
|
|
100
|
+
// Find text nodes with w:t tags
|
|
101
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
|
|
102
|
+
// Join the texts within this paragraph node without any spaces or delimiters.
|
|
103
|
+
return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
|
|
104
|
+
})
|
|
105
|
+
// Join each paragraph text with a new line delimiter.
|
|
106
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
107
|
+
);
|
|
114
108
|
});
|
|
115
109
|
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
110
|
+
// Join all responseText array
|
|
111
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
112
|
+
// Respond by calling the Callback function.
|
|
113
|
+
callback(responseText, undefined);
|
|
119
114
|
})
|
|
120
|
-
.catch(
|
|
121
|
-
consoleError(error)
|
|
122
|
-
return callback(undefined, error);
|
|
123
|
-
});
|
|
115
|
+
.catch(e => callback(undefined, e));
|
|
124
116
|
}
|
|
125
117
|
|
|
126
118
|
/** Main function for parsing text from PowerPoint files
|
|
127
|
-
* @param {string}
|
|
128
|
-
* @param {function}
|
|
129
|
-
* @param {
|
|
119
|
+
* @param {string} filepath File path
|
|
120
|
+
* @param {function} callback Callback function that returns value or error
|
|
121
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
130
122
|
* @returns {void}
|
|
131
123
|
*/
|
|
132
|
-
function parsePowerPoint(
|
|
133
|
-
if (!fs.existsSync(filename)) {
|
|
134
|
-
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
135
|
-
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
136
|
-
}
|
|
137
|
-
const ext = filename.split(".").pop().toLowerCase();
|
|
138
|
-
if (ext != 'pptx') {
|
|
139
|
-
consoleError(ERRORMSG.extensionUnsupported(extension));
|
|
140
|
-
return callback(undefined, ERRORMSG.extensionUnsupported(ext));
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
/** Store all the text content to respond */
|
|
144
|
-
let responseText = [];
|
|
145
|
-
|
|
146
|
-
/** Extracting text from powerpoint files xml objects converted to js */
|
|
147
|
-
function extractTextFromPowerPointXmlObjects(xmlObjects) {
|
|
148
|
-
// specifically for Arrays
|
|
149
|
-
if (Array.isArray(xmlObjects)) {
|
|
150
|
-
xmlObjects.forEach(item =>
|
|
151
|
-
(typeof item == "string") && (item != "")
|
|
152
|
-
? responseText.push(item)
|
|
153
|
-
: extractTextFromPowerPointXmlObjects(item))
|
|
154
|
-
}
|
|
155
|
-
// for other JS Object
|
|
156
|
-
else if (typeof xmlObjects == "object") {
|
|
157
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
158
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
159
|
-
? (key == "a:t" || key == "_") && value != ""
|
|
160
|
-
? responseText.push(value)
|
|
161
|
-
: undefined
|
|
162
|
-
: extractTextFromPowerPointXmlObjects(value);
|
|
163
|
-
}
|
|
164
|
-
}
|
|
165
|
-
}
|
|
166
|
-
|
|
124
|
+
function parsePowerPoint(filepath, callback, config) {
|
|
167
125
|
// Files regex that hold our content of interest
|
|
168
|
-
const
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
decompress(
|
|
174
|
-
|
|
175
|
-
{ filter: x =>
|
|
126
|
+
const allFilesRegex = /ppt\/(notesSlides|slides)\/(notesSlide|slide)\d+.xml/g;
|
|
127
|
+
const slidesRegex = /ppt\/slides\/slide\d+.xml/g;
|
|
128
|
+
|
|
129
|
+
/** The decompress location which contains the filename in it */
|
|
130
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
131
|
+
decompress(filepath,
|
|
132
|
+
decompressLocation,
|
|
133
|
+
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
176
134
|
)
|
|
177
135
|
.then(files => {
|
|
178
|
-
//
|
|
179
|
-
|
|
136
|
+
// Check if files is corrupted
|
|
137
|
+
if (files.length == 0)
|
|
138
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
180
139
|
|
|
181
|
-
if
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
140
|
+
// Check if any sorting is required.
|
|
141
|
+
if (!config.ignoreNotes && config.putNotesAtLast)
|
|
142
|
+
// Sort files according to previous order of taking text out of ppt/slides followed by ppt/notesSlides
|
|
143
|
+
// For this we are looking at the index of notes which results in -1 in the main slide file and exists at a certain index in notes file names.
|
|
144
|
+
files.sort((a,b) => a.path.indexOf("notes") - b.path.indexOf("notes"));
|
|
185
145
|
|
|
186
146
|
// Returning an array of all the xml contents read using fs.readFileSync
|
|
187
|
-
return files.map(file => fs.readFileSync(`${
|
|
147
|
+
return files.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
188
148
|
})
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
149
|
+
// ******************************** powerpoint xml files explanation ************************************
|
|
150
|
+
// Structure of xmlContent of a powerpoint file is simple.
|
|
151
|
+
// There are multiple xml files for each slide and correspondingly their notesSlide files.
|
|
152
|
+
// All text nodes are within a:t tags and each of the text nodes that belong in one paragraph are clubbed together within a a:p tag.
|
|
153
|
+
// So, we will filter out all the empty a:p tags and then combine all the a:t tag text inside for creating our response text.
|
|
154
|
+
// ******************************************************************************************************
|
|
155
|
+
.then(xmlContentArray => {
|
|
156
|
+
/** Store all the text content to respond */
|
|
157
|
+
let responseText = [];
|
|
158
|
+
|
|
159
|
+
xmlContentArray.forEach(xmlContent => {
|
|
160
|
+
/** Find text nodes with a:p tags */
|
|
161
|
+
const xmlParagraphNodesList = parseString(xmlContent).getElementsByTagName("a:p");
|
|
162
|
+
/** Store all the text content to respond */
|
|
163
|
+
responseText.push(
|
|
164
|
+
Array.from(xmlParagraphNodesList)
|
|
165
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
|
|
166
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
|
|
167
|
+
.map(paragraphNode => {
|
|
168
|
+
/** Find text nodes with a:t tags */
|
|
169
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
170
|
+
return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
|
|
171
|
+
})
|
|
172
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
173
|
+
);
|
|
202
174
|
});
|
|
203
175
|
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
176
|
+
// Join all responseText array
|
|
177
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
178
|
+
// Respond by calling the Callback function.
|
|
179
|
+
callback(responseText, undefined);
|
|
207
180
|
})
|
|
208
|
-
.catch(
|
|
209
|
-
consoleError(error)
|
|
210
|
-
return callback(undefined, error);
|
|
211
|
-
});
|
|
181
|
+
.catch(e => callback(undefined, e));
|
|
212
182
|
}
|
|
213
183
|
|
|
214
184
|
/** Main function for parsing text from Excel files
|
|
215
|
-
* @param {string}
|
|
216
|
-
* @param {function}
|
|
217
|
-
* @param {
|
|
185
|
+
* @param {string} filepath File path
|
|
186
|
+
* @param {function} callback Callback function that returns value or error
|
|
187
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
218
188
|
* @returns {void}
|
|
219
189
|
*/
|
|
220
|
-
function parseExcel(
|
|
221
|
-
if (!fs.existsSync(filename)) {
|
|
222
|
-
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
223
|
-
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
224
|
-
}
|
|
225
|
-
const ext = filename.split(".").pop().toLowerCase();
|
|
226
|
-
if (ext != 'xlsx') {
|
|
227
|
-
consoleError(ERRORMSG.extensionUnsupported(extension));
|
|
228
|
-
return callback(undefined, ERRORMSG.extensionUnsupported(ext));
|
|
229
|
-
}
|
|
230
|
-
|
|
231
|
-
/** Store all the text content to respond */
|
|
232
|
-
let responseText = [];
|
|
233
|
-
|
|
234
|
-
function extractTextFromExcelXmlObjects2dArray(xmlObjects2dArray) {
|
|
235
|
-
xmlObjects2dArray[0].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 0));
|
|
236
|
-
xmlObjects2dArray[1].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 1));
|
|
237
|
-
xmlObjects2dArray[2].map(xmlObjects => extractTextFromExcelXmlObjects(xmlObjects, 2));
|
|
238
|
-
}
|
|
239
|
-
|
|
240
|
-
/** Extracting text from Excel files xml objects converted to js */
|
|
241
|
-
function extractTextFromExcelXmlObjects(xmlObjects, contentFilesIndex) {
|
|
242
|
-
switch(contentFilesIndex) {
|
|
243
|
-
case 0: { // worksheet
|
|
244
|
-
// specifically for Arrays
|
|
245
|
-
if (Array.isArray(xmlObjects)) {
|
|
246
|
-
xmlObjects.forEach(item =>
|
|
247
|
-
item["v"]
|
|
248
|
-
? ((item["$"]["t"] != "s"))
|
|
249
|
-
? responseText.push(item["v"][0])
|
|
250
|
-
: undefined
|
|
251
|
-
: extractTextFromExcelXmlObjects(item, contentFilesIndex))
|
|
252
|
-
}
|
|
253
|
-
// for other JS Object
|
|
254
|
-
else if (typeof xmlObjects == "object") {
|
|
255
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
256
|
-
value["v"]
|
|
257
|
-
? ((value["$"]["t"] == "s"))
|
|
258
|
-
? responseText.push(value["v"][0])
|
|
259
|
-
: undefined
|
|
260
|
-
: extractTextFromExcelXmlObjects(value, contentFilesIndex);
|
|
261
|
-
}
|
|
262
|
-
}
|
|
263
|
-
break;
|
|
264
|
-
}
|
|
265
|
-
case 1: { // sharedStrings
|
|
266
|
-
// specifically for Arrays
|
|
267
|
-
if (Array.isArray(xmlObjects)) {
|
|
268
|
-
xmlObjects.forEach(item =>
|
|
269
|
-
(typeof item == "string") && (item != "")
|
|
270
|
-
? responseText.push(item)
|
|
271
|
-
: extractTextFromExcelXmlObjects(item, contentFilesIndex))
|
|
272
|
-
}
|
|
273
|
-
// for other JS Object
|
|
274
|
-
else if (typeof xmlObjects == "object") {
|
|
275
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
276
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
277
|
-
? (key == "t" || key == "_") && (value != "")
|
|
278
|
-
? responseText.push(value)
|
|
279
|
-
: undefined
|
|
280
|
-
: extractTextFromExcelXmlObjects(value, contentFilesIndex);
|
|
281
|
-
}
|
|
282
|
-
}
|
|
283
|
-
break;
|
|
284
|
-
}
|
|
285
|
-
case 2: { // drawings
|
|
286
|
-
// specifically for Arrays
|
|
287
|
-
if (Array.isArray(xmlObjects)) {
|
|
288
|
-
xmlObjects.forEach(item =>
|
|
289
|
-
(typeof item == "string") && (item != "")
|
|
290
|
-
? responseText.push(item)
|
|
291
|
-
: extractTextFromExcelXmlObjects(item, contentFilesIndex))
|
|
292
|
-
}
|
|
293
|
-
// for other JS Object
|
|
294
|
-
else if (typeof xmlObjects == "object") {
|
|
295
|
-
for (const [key, value] of Object.entries(xmlObjects)) {
|
|
296
|
-
(typeof value == "string") || (typeof value[0] == "string")
|
|
297
|
-
? (key == "a:t" || key == "_") && (value != "")
|
|
298
|
-
? responseText.push(value)
|
|
299
|
-
: undefined
|
|
300
|
-
: extractTextFromExcelXmlObjects(value, contentFilesIndex);
|
|
301
|
-
}
|
|
302
|
-
}
|
|
303
|
-
break;
|
|
304
|
-
}
|
|
305
|
-
}
|
|
306
|
-
}
|
|
307
|
-
|
|
190
|
+
function parseExcel(filepath, callback, config) {
|
|
308
191
|
// Files regex that hold our content of interest
|
|
309
|
-
const
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
192
|
+
const sheetsRegex = /xl\/worksheets\/sheet\d+.xml/g;
|
|
193
|
+
const drawingsRegex = /xl\/drawings\/drawing\d+.xml/g;
|
|
194
|
+
const chartsRegex = /xl\/charts\/chart\d+.xml/g;
|
|
195
|
+
const stringsFilePath = 'xl/sharedStrings.xml';
|
|
196
|
+
|
|
197
|
+
/** The decompress location which contains the filename in it */
|
|
198
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
199
|
+
decompress(filepath,
|
|
200
|
+
decompressLocation,
|
|
201
|
+
{ filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
|
|
318
202
|
)
|
|
319
203
|
.then(files => {
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
// Returning a 2dArray of all the xml contents read using fs.readFileSync and separated by array elements
|
|
330
|
-
return files2dArray.map(files => files.map(file => fs.readFileSync(`${decompressSubLocation}/${file.path}`, 'utf8')))
|
|
204
|
+
if (files.length == 0)
|
|
205
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
206
|
+
|
|
207
|
+
return {
|
|
208
|
+
sheetFiles: files.filter(file => file.path.match(sheetsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
209
|
+
drawingFiles: files.filter(file => file.path.match(drawingsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
210
|
+
chartFiles: files.filter(file => file.path.match(chartsRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
211
|
+
sharedStringsFile: files.filter(file => file.path == stringsFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
|
|
212
|
+
};
|
|
331
213
|
})
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
214
|
+
// ********************************** excel xml files explanation ***************************************
|
|
215
|
+
// Structure of xmlContent of an excel file is a bit complex.
|
|
216
|
+
// We have a sharedStrings.xml file which has strings inside t tags
|
|
217
|
+
// Each sheet has an individual sheet xml file which has numbers in v tags (probably value) inside c tags (probably cell)
|
|
218
|
+
// Each value of v tag is to be used as it is if the "t" attribute (probably type) of c tag is not "s" (probably shared string)
|
|
219
|
+
// If the "t" attribute of c tag is "s", then we use the value to select value from sharedStrings array with the value as its index.
|
|
220
|
+
// Drawing files contain all text for each drawing and have text nodes in a:t and paragraph nodes in a:p.
|
|
221
|
+
// ******************************************************************************************************
|
|
222
|
+
.then(xmlContentFilesObject => {
|
|
223
|
+
/** Store all the text content to respond */
|
|
224
|
+
let responseText = [];
|
|
225
|
+
|
|
226
|
+
/** Find text nodes with t tags in sharedStrings xml file */
|
|
227
|
+
const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
|
|
228
|
+
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
229
|
+
const sharedStrings = Array.from(sharedStringsXmlTNodesList).map(tNode => tNode.childNodes[0].nodeValue);
|
|
230
|
+
|
|
231
|
+
// Parse Sheet files
|
|
232
|
+
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
233
|
+
/** Find text nodes with c tags in sharedStrings xml file */
|
|
234
|
+
const sheetsXmlCNodesList = parseString(sheetXmlContent).getElementsByTagName("c");
|
|
235
|
+
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
236
|
+
responseText.push(
|
|
237
|
+
Array.from(sheetsXmlCNodesList)
|
|
238
|
+
// Filter c nodes than do not have any v nodes
|
|
239
|
+
.filter(cNode => cNode.getElementsByTagName("v").length != 0)
|
|
240
|
+
.map(cNode => {
|
|
241
|
+
/** Flag whether this node's value represents a string index */
|
|
242
|
+
const isString = cNode.getAttribute("t") == "s";
|
|
243
|
+
/** Find value nodes represented by v tags */
|
|
244
|
+
const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
|
|
245
|
+
// Validate text
|
|
246
|
+
if (isString && value >= sharedStrings.length)
|
|
247
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
248
|
+
|
|
249
|
+
return isString
|
|
250
|
+
? sharedStrings[value]
|
|
251
|
+
: value;
|
|
252
|
+
})
|
|
253
|
+
// Join each cell text within a sheet with a space.
|
|
254
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
255
|
+
);
|
|
345
256
|
});
|
|
346
257
|
|
|
347
|
-
|
|
348
|
-
.
|
|
258
|
+
// Parse Drawing files
|
|
259
|
+
xmlContentFilesObject.drawingFiles.forEach(drawingXmlContent => {
|
|
260
|
+
/** Find text nodes with a:p tags */
|
|
261
|
+
const drawingsXmlParagraphNodesList = parseString(drawingXmlContent).getElementsByTagName("a:p");
|
|
262
|
+
/** Store all the text content to respond */
|
|
263
|
+
responseText.push(
|
|
264
|
+
Array.from(drawingsXmlParagraphNodesList)
|
|
265
|
+
// Filter paragraph nodes than do not have any text nodes which are identifiable by a:t tag
|
|
266
|
+
.filter(paragraphNode => paragraphNode.getElementsByTagName("a:t").length != 0)
|
|
267
|
+
.map(paragraphNode => {
|
|
268
|
+
/** Find text nodes with a:t tags */
|
|
269
|
+
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
270
|
+
return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
|
|
271
|
+
})
|
|
272
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
273
|
+
);
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
// Parse Chart files
|
|
277
|
+
xmlContentFilesObject.chartFiles.forEach(chartXmlContent => {
|
|
278
|
+
/** Find text nodes with c:v tags */
|
|
279
|
+
const chartsXmlCVNodesList = parseString(chartXmlContent).getElementsByTagName("c:v");
|
|
280
|
+
/** Store all the text content to respond */
|
|
281
|
+
responseText.push(
|
|
282
|
+
Array.from(chartsXmlCVNodesList)
|
|
283
|
+
.map(cVNode => cVNode.childNodes[0].nodeValue)
|
|
284
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
285
|
+
);
|
|
286
|
+
});
|
|
349
287
|
|
|
288
|
+
// Join all responseText array
|
|
289
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
290
|
+
// Respond by calling the Callback function.
|
|
291
|
+
callback(responseText, undefined);
|
|
350
292
|
})
|
|
351
|
-
.catch(
|
|
352
|
-
consoleError(error)
|
|
353
|
-
return callback(undefined, error);
|
|
354
|
-
});
|
|
293
|
+
.catch(e => callback(undefined, e));
|
|
355
294
|
}
|
|
356
295
|
|
|
357
296
|
|
|
358
297
|
/** Main function for parsing text from open office files
|
|
359
|
-
* @param {string}
|
|
360
|
-
* @param {function}
|
|
361
|
-
* @param {
|
|
298
|
+
* @param {string} filepath File path
|
|
299
|
+
* @param {function} callback Callback function that returns value or error
|
|
300
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
362
301
|
* @returns {void}
|
|
363
302
|
*/
|
|
364
|
-
function parseOpenOffice(
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
303
|
+
function parseOpenOffice(filepath, callback, config) {
|
|
304
|
+
/** The target content xml file for the openoffice file. */
|
|
305
|
+
const mainContentFilePath = 'content.xml';
|
|
306
|
+
const objectContentFilesRegex = /Object \d+\/content.xml/g;
|
|
307
|
+
|
|
308
|
+
/** The decompress location which contains the filename in it */
|
|
309
|
+
const decompressLocation = `${decompressSubLocation}/${filepath.split("/").pop()}`;
|
|
310
|
+
decompress(filepath,
|
|
311
|
+
decompressLocation,
|
|
312
|
+
{ filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
|
|
313
|
+
)
|
|
314
|
+
.then(files => {
|
|
315
|
+
if (files.length == 0)
|
|
316
|
+
throw ERRORMSG.fileCorrupted(filepath);
|
|
374
317
|
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
function extractTextFromOpenOfficeXmlObjects(xmlObjects) {
|
|
379
|
-
// specifically for Arrays
|
|
380
|
-
if (Array.isArray(xmlObjects)) {
|
|
381
|
-
xmlObjects.forEach(item =>
|
|
382
|
-
(typeof item == "string") && (item != "")
|
|
383
|
-
? responseText.push(item)
|
|
384
|
-
: extractTextFromOpenOfficeXmlObjects(item))
|
|
318
|
+
return {
|
|
319
|
+
mainContentFile: files.filter(file => file.path == mainContentFilePath).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'))[0],
|
|
320
|
+
objectContentFiles: files.filter(file => file.path.match(objectContentFilesRegex)).map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8')),
|
|
385
321
|
}
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
322
|
+
})
|
|
323
|
+
// ********************************** openoffice xml files explanation **********************************
|
|
324
|
+
// Structure of xmlContent of openoffice files is simple.
|
|
325
|
+
// All text nodes are within text:h and text:p tags with all kinds of formatting within nested tags.
|
|
326
|
+
// All text in these tags are separated by new line delimiters.
|
|
327
|
+
// Objects like charts in ods files are in Object d+/content.xml with the same way as above.
|
|
328
|
+
// ******************************************************************************************************
|
|
329
|
+
.then(xmlContentFilesObject => {
|
|
330
|
+
/** Store all the notes text content to respond */
|
|
331
|
+
let notesText = [];
|
|
332
|
+
/** Store all the text content to respond */
|
|
333
|
+
let responseText = [];
|
|
334
|
+
|
|
335
|
+
/** List of allowed text tags */
|
|
336
|
+
const allowedTextTags = ["text:p", "text:h"];
|
|
337
|
+
/** List of notes tags */
|
|
338
|
+
const notesTag = "presentation:notes";
|
|
339
|
+
|
|
340
|
+
/** Main dfs traversal function that goes from one node to its children and returns the value out. */
|
|
341
|
+
function extractAllTextsFromNode(root) {
|
|
342
|
+
let xmlTextArray = []
|
|
343
|
+
for (let i = 0; i < root.childNodes.length; i++)
|
|
344
|
+
traversal(root.childNodes[i], xmlTextArray, true);
|
|
345
|
+
return xmlTextArray.join("");
|
|
346
|
+
}
|
|
347
|
+
/** Traversal function that gets recursive calling. */
|
|
348
|
+
function traversal(node, xmlTextArray, isFirstRecursion) {
|
|
349
|
+
if(!node.childNodes || node.childNodes.length == 0)
|
|
350
|
+
{
|
|
351
|
+
if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
|
|
352
|
+
if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
|
|
353
|
+
notesText.push(node.nodeValue);
|
|
354
|
+
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
355
|
+
notesText.push(config.newlineDelimiter ?? "\n");
|
|
356
|
+
}
|
|
357
|
+
else {
|
|
358
|
+
xmlTextArray.push(node.nodeValue);
|
|
359
|
+
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
360
|
+
xmlTextArray.push(config.newlineDelimiter ?? "\n");
|
|
361
|
+
}
|
|
362
|
+
}
|
|
363
|
+
return;
|
|
394
364
|
}
|
|
365
|
+
|
|
366
|
+
for (let i = 0; i < node.childNodes.length; i++)
|
|
367
|
+
traversal(node.childNodes[i], xmlTextArray, false);
|
|
395
368
|
}
|
|
396
|
-
}
|
|
397
369
|
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
consoleError(ERRORMSG.fileCorrupted(filename));
|
|
406
|
-
return callback(undefined, ERRORMSG.fileCorrupted(filename));
|
|
370
|
+
/** Checks if the given node has an ancestor which is a notes tag. We use this information to put the notes in the response text and its position. */
|
|
371
|
+
function isNotesNode(node) {
|
|
372
|
+
if (node.tagName == notesTag)
|
|
373
|
+
return true;
|
|
374
|
+
if (node.parentNode)
|
|
375
|
+
return isNotesNode(node.parentNode);
|
|
376
|
+
return false;
|
|
407
377
|
}
|
|
408
378
|
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
379
|
+
/** Checks if the given node has an ancestor which is also an allowed text tag. In that case, we ignore the child text tag. */
|
|
380
|
+
function isInvalidTextNode(node) {
|
|
381
|
+
if (allowedTextTags.includes(node.tagName))
|
|
382
|
+
return true;
|
|
383
|
+
if (node.parentNode)
|
|
384
|
+
return isInvalidTextNode(node.parentNode);
|
|
385
|
+
return false;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** The xml string parsed as xml array */
|
|
389
|
+
const xmlContentArray = [xmlContentFilesObject.mainContentFile, ...xmlContentFilesObject.objectContentFiles].map(xmlContent => parseString(xmlContent));
|
|
390
|
+
// Iterate over each xmlContent and extract text from them.
|
|
391
|
+
xmlContentArray.forEach(xmlContent => {
|
|
392
|
+
/** Find text nodes with text:h and text:p tags in xmlContent */
|
|
393
|
+
const xmlTextNodesList = [...Array.from(xmlContent
|
|
394
|
+
.getElementsByTagName("*"))
|
|
395
|
+
.filter(node => allowedTextTags.includes(node.tagName)
|
|
396
|
+
&& !isInvalidTextNode(node.parentNode))
|
|
397
|
+
];
|
|
398
|
+
/** Store all the text content to respond */
|
|
399
|
+
responseText.push(
|
|
400
|
+
xmlTextNodesList
|
|
401
|
+
// Add every text information from within this textNode and combine them together.
|
|
402
|
+
.map(textNode => extractAllTextsFromNode(textNode))
|
|
403
|
+
.filter(text => text != "")
|
|
404
|
+
.join(config.newlineDelimiter ?? "\n")
|
|
405
|
+
);
|
|
423
406
|
});
|
|
424
407
|
|
|
425
|
-
|
|
426
|
-
|
|
408
|
+
// Add notes text at the end if the user config says so.
|
|
409
|
+
// Note that we already have pushed the text content to notesText array while extracting all texts from the nodes.
|
|
410
|
+
if (!config.ignoreNotes && config.putNotesAtLast)
|
|
411
|
+
responseText = [...responseText, ...notesText];
|
|
427
412
|
|
|
413
|
+
// Join all responseText array
|
|
414
|
+
responseText = responseText.join(config.newlineDelimiter ?? "\n");
|
|
415
|
+
// Respond by calling the Callback function.
|
|
416
|
+
callback(responseText, undefined);
|
|
428
417
|
})
|
|
429
|
-
.catch(
|
|
430
|
-
consoleError(error)
|
|
431
|
-
return callback(undefined, error);
|
|
432
|
-
});
|
|
433
|
-
}
|
|
434
|
-
|
|
435
|
-
|
|
436
|
-
/** Main async function with callback to execute parseOffice for supported files
|
|
437
|
-
* @param {string} filename File path
|
|
438
|
-
* @param {function} callback Callback function that returns value or error
|
|
439
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
440
|
-
* @returns {void}
|
|
441
|
-
*/
|
|
442
|
-
function parseOffice(filename, callback, deleteOfficeDist = true) {
|
|
443
|
-
if (!fs.existsSync(filename)) {
|
|
444
|
-
consoleError(ERRORMSG.fileDoesNotExist(filename));
|
|
445
|
-
return callback(undefined, ERRORMSG.fileDoesNotExist(filename));
|
|
446
|
-
}
|
|
447
|
-
var extension = filename.split(".").pop().toLowerCase();
|
|
448
|
-
|
|
449
|
-
switch(extension)
|
|
450
|
-
{
|
|
451
|
-
case "docx":
|
|
452
|
-
parseWord(filename, (data, err) => callback(data, err), deleteOfficeDist);
|
|
453
|
-
return;
|
|
454
|
-
case "pptx":
|
|
455
|
-
parsePowerPoint(filename, (data, err) => callback(data, err), deleteOfficeDist);
|
|
456
|
-
return;
|
|
457
|
-
case "xlsx":
|
|
458
|
-
parseExcel(filename, (data, err) => callback(data, err), deleteOfficeDist);
|
|
459
|
-
return;
|
|
460
|
-
case "odt":
|
|
461
|
-
case "odp":
|
|
462
|
-
case "ods":
|
|
463
|
-
parseOpenOffice(filename, (data, err) => callback(data, err), deleteOfficeDist);
|
|
464
|
-
return;
|
|
465
|
-
|
|
466
|
-
default:
|
|
467
|
-
consoleError(ERRORMSG.extensionUnsupported(extension));
|
|
468
|
-
callback(undefined, ERRORMSG.extensionUnsupported(extension));
|
|
469
|
-
}
|
|
418
|
+
.catch(e => callback(undefined, e));
|
|
470
419
|
}
|
|
471
420
|
|
|
472
|
-
/**
|
|
473
|
-
|
|
474
|
-
* @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
|
|
475
|
-
* @returns {void}
|
|
476
|
-
*/
|
|
477
|
-
function setDecompressionLocation(newLocation) {
|
|
478
|
-
if (newLocation != undefined) {
|
|
479
|
-
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
|
|
480
|
-
|
|
481
|
-
if (fs.existsSync(newLocation))
|
|
482
|
-
decompressSubLocation = newLocation;
|
|
483
|
-
return;
|
|
484
|
-
}
|
|
485
|
-
consoleError(ERRORMSG.locationNotFound(newLocation));
|
|
486
|
-
decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
487
|
-
}
|
|
421
|
+
/** Header for error messages */
|
|
422
|
+
const PDFPARSEERRORHEADER = "[pdf-parse]: ";
|
|
488
423
|
|
|
489
|
-
/**
|
|
424
|
+
/** Main function for parsing text from pdf files
|
|
425
|
+
* @param {string} filepath File path
|
|
426
|
+
* @param {function} callback Callback function that returns value or error
|
|
427
|
+
* @param {OfficeParserConfig} config Config Object for officeParser
|
|
490
428
|
* @returns {void}
|
|
491
429
|
*/
|
|
492
|
-
function
|
|
493
|
-
|
|
430
|
+
function parsePdf(filepath, callback, config) {
|
|
431
|
+
// Get the data buffer for the given file path.
|
|
432
|
+
const dataBuffer = fs.readFileSync(filepath);
|
|
433
|
+
|
|
434
|
+
pdfParse(dataBuffer)
|
|
435
|
+
.then(data => {
|
|
436
|
+
let text = data.text;
|
|
437
|
+
if (!!config.newlineDelimiter && config.newlineDelimiter != "\n")
|
|
438
|
+
text = text.replaceAll("\n", config.newlineDelimiter)
|
|
439
|
+
callback(text, undefined);
|
|
440
|
+
})
|
|
441
|
+
.catch(e => callback(undefined, PDFPARSEERRORHEADER + e));
|
|
494
442
|
}
|
|
495
443
|
|
|
496
|
-
/**
|
|
444
|
+
/** Main async function with callback to execute parseOffice for supported files
|
|
445
|
+
* @param {string | Buffer} file File path or file buffers
|
|
446
|
+
* @param {function} callback Callback function that returns value or error
|
|
447
|
+
* @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
|
|
497
448
|
* @returns {void}
|
|
498
449
|
*/
|
|
499
|
-
function
|
|
500
|
-
|
|
501
|
-
|
|
450
|
+
function parseOffice(file, callback, config = {}) {
|
|
451
|
+
// Prepare file for processing
|
|
452
|
+
const filePreparedPromise = new Promise((res, rej) => {
|
|
453
|
+
// create temp file subdirectory if it does not exist
|
|
454
|
+
fs.mkdirSync(`${decompressSubLocation}/tempfiles`, { recursive: true });
|
|
455
|
+
|
|
456
|
+
// Check if buffer
|
|
457
|
+
if (Buffer.isBuffer(file)) {
|
|
458
|
+
// Guess file type from buffer
|
|
459
|
+
fileType.fromBuffer(file)
|
|
460
|
+
.then(data =>
|
|
461
|
+
{
|
|
462
|
+
// temp file name
|
|
463
|
+
const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${data.ext.toLowerCase()}`;
|
|
464
|
+
// write new file
|
|
465
|
+
fs.writeFileSync(newfilepath, file);
|
|
466
|
+
// resolve promise
|
|
467
|
+
res(newfilepath);
|
|
468
|
+
})
|
|
469
|
+
.catch(() => rej(ERRORMSG.improperBuffers));
|
|
470
|
+
return;
|
|
471
|
+
}
|
|
502
472
|
|
|
473
|
+
// Not buffers but real file path.
|
|
503
474
|
|
|
504
|
-
//
|
|
475
|
+
// Check if file exists
|
|
476
|
+
if (!fs.existsSync(file))
|
|
477
|
+
throw ERRORMSG.fileDoesNotExist(file);
|
|
505
478
|
|
|
506
|
-
|
|
507
|
-
|
|
508
|
-
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
try {
|
|
514
|
-
parseWord(filename, function (data, error) {
|
|
515
|
-
if (error)
|
|
516
|
-
return reject(error);
|
|
517
|
-
return resolve(data);
|
|
518
|
-
}, deleteOfficeDist);
|
|
519
|
-
}
|
|
520
|
-
catch (error) {
|
|
521
|
-
return reject(error);
|
|
522
|
-
}
|
|
523
|
-
})
|
|
524
|
-
}
|
|
479
|
+
// temp file name
|
|
480
|
+
const newfilepath = `${decompressSubLocation}/tempfiles/${new Date().getTime().toString()}.${file.split(".").pop().toLowerCase()}`;
|
|
481
|
+
// Copy the file into a temp location with the temp name
|
|
482
|
+
fs.copyFileSync(file, newfilepath)
|
|
483
|
+
// resolve promise
|
|
484
|
+
res(newfilepath);
|
|
485
|
+
});
|
|
525
486
|
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
|
|
544
|
-
|
|
487
|
+
// Process filePreparedPromise resolution.
|
|
488
|
+
filePreparedPromise
|
|
489
|
+
.then(filepath => {
|
|
490
|
+
// File extension. Already in lowercase when we prepared the temp file above.
|
|
491
|
+
const extension = filepath.split(".").pop();
|
|
492
|
+
|
|
493
|
+
// Switch between parsing functions depending on extension.
|
|
494
|
+
switch(extension) {
|
|
495
|
+
case "docx":
|
|
496
|
+
parseWord(filepath, internalCallback, config);
|
|
497
|
+
break;
|
|
498
|
+
case "pptx":
|
|
499
|
+
parsePowerPoint(filepath, internalCallback, config);
|
|
500
|
+
break;
|
|
501
|
+
case "xlsx":
|
|
502
|
+
parseExcel(filepath, internalCallback, config);
|
|
503
|
+
break;
|
|
504
|
+
case "odt":
|
|
505
|
+
case "odp":
|
|
506
|
+
case "ods":
|
|
507
|
+
parseOpenOffice(filepath, internalCallback, config);
|
|
508
|
+
break;
|
|
509
|
+
case "pdf":
|
|
510
|
+
parsePdf(filepath, internalCallback, config);
|
|
511
|
+
break;
|
|
512
|
+
|
|
513
|
+
default:
|
|
514
|
+
throw ERRORMSG.extensionUnsupported(extension);
|
|
515
|
+
}
|
|
545
516
|
|
|
546
|
-
/**
|
|
547
|
-
|
|
548
|
-
* @param {boolean} [deleteOfficeDist=true] Optional: Delete the officeDist directory created while unarchiving the doc file to get its content underneath. By default, we delete those files after we are done reading them.
|
|
549
|
-
* @returns {Promise<string>}
|
|
550
|
-
*/
|
|
551
|
-
var parseExcelAsync = function (filename, deleteOfficeDist = true) {
|
|
552
|
-
return new Promise((resolve, reject) => {
|
|
553
|
-
try {
|
|
554
|
-
parseExcel(filename, function (data, err) {
|
|
517
|
+
/** Internal callback function that calls the user's callback function passed in argument and removes the temp files if required */
|
|
518
|
+
function internalCallback(data, err) {
|
|
555
519
|
if (err)
|
|
556
|
-
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
520
|
+
consoleError(err, config.outputErrorToConsole)
|
|
521
|
+
// Call the original callback
|
|
522
|
+
callback(data, err);
|
|
523
|
+
// Check if we need to preserve unzipped content files or delete them.
|
|
524
|
+
if (config.preserveTempFiles)
|
|
525
|
+
return;
|
|
526
|
+
// Delete decompress sublocation.
|
|
527
|
+
rimraf(decompressSubLocation, rimrafErr => consoleError(rimrafErr, config.outputErrorToConsole));
|
|
528
|
+
}
|
|
529
|
+
})
|
|
530
|
+
.catch(error => {
|
|
531
|
+
consoleError(error, config.outputErrorToConsole);
|
|
532
|
+
callback(undefined, error);
|
|
533
|
+
});
|
|
564
534
|
}
|
|
565
535
|
|
|
566
|
-
/**
|
|
567
|
-
*
|
|
568
|
-
* @param {
|
|
536
|
+
/**
|
|
537
|
+
* Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
538
|
+
* @param {string | Buffer} file File path or file buffers
|
|
539
|
+
* @param {OfficeParserConfig} config [OPTIONAL]: Config Object for officeParser
|
|
569
540
|
* @returns {Promise<string>}
|
|
570
541
|
*/
|
|
571
|
-
|
|
572
|
-
return new Promise((
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
}
|
|
580
|
-
catch (error) {
|
|
581
|
-
return reject(error);
|
|
582
|
-
}
|
|
583
|
-
})
|
|
542
|
+
function parseOfficeAsync (file, config) {
|
|
543
|
+
return new Promise((res, rej) => {
|
|
544
|
+
parseOffice(file, function (data, err) {
|
|
545
|
+
if (err)
|
|
546
|
+
return rej(err);
|
|
547
|
+
return res(data);
|
|
548
|
+
}, config);
|
|
549
|
+
});
|
|
584
550
|
}
|
|
585
551
|
|
|
586
552
|
/**
|
|
587
|
-
*
|
|
588
|
-
* @param {string}
|
|
589
|
-
* @
|
|
590
|
-
* @returns {Promise<string>}
|
|
553
|
+
* Set decompression directory. The final decompressed data will be put inside officeDist folder within your directory
|
|
554
|
+
* @param {string} newLocation Relative path to the directory that will contain officeDist folder with decompressed data
|
|
555
|
+
* @returns {void}
|
|
591
556
|
*/
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
catch (error) {
|
|
602
|
-
return reject(error);
|
|
603
|
-
}
|
|
604
|
-
})
|
|
557
|
+
function setDecompressionLocation(newLocation) {
|
|
558
|
+
if (newLocation != undefined) {
|
|
559
|
+
newLocation = `${newLocation}${newLocation.endsWith('/') ? '' : '/'}${DEFAULTDECOMPRESSSUBLOCATION}`
|
|
560
|
+
if (fs.existsSync(newLocation))
|
|
561
|
+
decompressSubLocation = newLocation;
|
|
562
|
+
return;
|
|
563
|
+
}
|
|
564
|
+
consoleError(ERRORMSG.locationNotFound(newLocation), config.outputErrorToConsole);
|
|
565
|
+
decompressSubLocation = DEFAULTDECOMPRESSSUBLOCATION;
|
|
605
566
|
}
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
module.exports.
|
|
609
|
-
module.exports.
|
|
610
|
-
module.exports.parseExcel = parseExcel;
|
|
611
|
-
module.exports.parseOpenOffice = parseOpenOffice;
|
|
612
|
-
module.exports.parseOffice = parseOffice;
|
|
613
|
-
module.exports.parseWordAsync = parseWordAsync;
|
|
614
|
-
module.exports.parsePowerPointAsync = parsePowerPointAsync;
|
|
615
|
-
module.exports.parseExcelAsync = parseExcelAsync;
|
|
616
|
-
module.exports.parseOpenOfficeAsync = parseOpenOfficeAsync;
|
|
617
|
-
module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
567
|
+
|
|
568
|
+
// Export functions
|
|
569
|
+
module.exports.parseOffice = parseOffice;
|
|
570
|
+
module.exports.parseOfficeAsync = parseOfficeAsync;
|
|
618
571
|
module.exports.setDecompressionLocation = setDecompressionLocation;
|
|
619
|
-
module.exports.enableConsoleOutput = enableConsoleOutput;
|
|
620
|
-
module.exports.disableConsoleOutput = disableConsoleOutput;
|
|
621
572
|
|
|
622
573
|
|
|
623
574
|
// Run this library on CLI
|
|
@@ -627,8 +578,8 @@ if ((process.argv[0].split('/').pop() == "node" || process.argv[0].split('/').po
|
|
|
627
578
|
}
|
|
628
579
|
else if (process.argv.length == 3)
|
|
629
580
|
parseOfficeAsync(process.argv[2])
|
|
630
|
-
|
|
631
|
-
|
|
581
|
+
.then(text => console.log(text))
|
|
582
|
+
.catch(error => console.error(ERRORHEADER + error))
|
|
632
583
|
else
|
|
633
584
|
console.error(ERRORMSG.improperArguments)
|
|
634
585
|
}
|