officeparser 4.0.5 → 4.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +27 -11
- package/package.json +1 -1
package/officeParser.js
CHANGED
|
@@ -69,7 +69,8 @@ function parseWord(filepath, callback, config) {
|
|
|
69
69
|
{ filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
|
|
70
70
|
)
|
|
71
71
|
.then(files => {
|
|
72
|
-
if
|
|
72
|
+
// Verify if atleast the document xml file exists in the extracted files list.
|
|
73
|
+
if (!files.map(file => file.path).includes(mainContentFile))
|
|
73
74
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
74
75
|
|
|
75
76
|
return [...files.filter(file => file.path == mainContentFile),
|
|
@@ -99,7 +100,10 @@ function parseWord(filepath, callback, config) {
|
|
|
99
100
|
// Find text nodes with w:t tags
|
|
100
101
|
const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
|
|
101
102
|
// Join the texts within this paragraph node without any spaces or delimiters.
|
|
102
|
-
return Array.from(xmlTextNodeList)
|
|
103
|
+
return Array.from(xmlTextNodeList)
|
|
104
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
105
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
106
|
+
.join("");
|
|
103
107
|
})
|
|
104
108
|
// Join each paragraph text with a new line delimiter.
|
|
105
109
|
.join(config.newlineDelimiter ?? "\n")
|
|
@@ -132,8 +136,8 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
132
136
|
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
133
137
|
)
|
|
134
138
|
.then(files => {
|
|
135
|
-
//
|
|
136
|
-
if (files.length == 0)
|
|
139
|
+
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
140
|
+
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
|
|
137
141
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
138
142
|
|
|
139
143
|
// Check if any sorting is required.
|
|
@@ -166,7 +170,10 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
166
170
|
.map(paragraphNode => {
|
|
167
171
|
/** Find text nodes with a:t tags */
|
|
168
172
|
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
169
|
-
return Array.from(xmlTextNodeList)
|
|
173
|
+
return Array.from(xmlTextNodeList)
|
|
174
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
175
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
176
|
+
.join("");
|
|
170
177
|
})
|
|
171
178
|
.join(config.newlineDelimiter ?? "\n")
|
|
172
179
|
);
|
|
@@ -200,7 +207,8 @@ function parseExcel(filepath, callback, config) {
|
|
|
200
207
|
{ filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
|
|
201
208
|
)
|
|
202
209
|
.then(files => {
|
|
203
|
-
if
|
|
210
|
+
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
211
|
+
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
|
|
204
212
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
205
213
|
|
|
206
214
|
return {
|
|
@@ -225,7 +233,8 @@ function parseExcel(filepath, callback, config) {
|
|
|
225
233
|
/** Find text nodes with t tags in sharedStrings xml file */
|
|
226
234
|
const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
|
|
227
235
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
228
|
-
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
236
|
+
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
237
|
+
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
|
|
229
238
|
|
|
230
239
|
// Parse Sheet files
|
|
231
240
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
@@ -234,8 +243,10 @@ function parseExcel(filepath, callback, config) {
|
|
|
234
243
|
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
235
244
|
responseText.push(
|
|
236
245
|
Array.from(sheetsXmlCNodesList)
|
|
237
|
-
// Filter c nodes than do not have any v nodes
|
|
238
|
-
.filter(cNode => cNode.getElementsByTagName("v")
|
|
246
|
+
// Filter c nodes than do not have any valid v nodes
|
|
247
|
+
.filter(cNode => cNode.getElementsByTagName("v")[0]
|
|
248
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
249
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
|
|
239
250
|
.map(cNode => {
|
|
240
251
|
/** Flag whether this node's value represents a string index */
|
|
241
252
|
const isString = cNode.getAttribute("t") == "s";
|
|
@@ -266,7 +277,10 @@ function parseExcel(filepath, callback, config) {
|
|
|
266
277
|
.map(paragraphNode => {
|
|
267
278
|
/** Find text nodes with a:t tags */
|
|
268
279
|
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
269
|
-
return Array.from(xmlTextNodeList)
|
|
280
|
+
return Array.from(xmlTextNodeList)
|
|
281
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
282
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
283
|
+
.join("");
|
|
270
284
|
})
|
|
271
285
|
.join(config.newlineDelimiter ?? "\n")
|
|
272
286
|
);
|
|
@@ -279,6 +293,7 @@ function parseExcel(filepath, callback, config) {
|
|
|
279
293
|
/** Store all the text content to respond */
|
|
280
294
|
responseText.push(
|
|
281
295
|
Array.from(chartsXmlCVNodesList)
|
|
296
|
+
.filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
|
|
282
297
|
.map(cVNode => cVNode.childNodes[0].nodeValue)
|
|
283
298
|
.join(config.newlineDelimiter ?? "\n")
|
|
284
299
|
);
|
|
@@ -311,7 +326,8 @@ function parseOpenOffice(filepath, callback, config) {
|
|
|
311
326
|
{ filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
|
|
312
327
|
)
|
|
313
328
|
.then(files => {
|
|
314
|
-
if
|
|
329
|
+
// Verify if atleast the content xml file exists in the extracted files list.
|
|
330
|
+
if (!files.map(file => file.path).includes(mainContentFilePath))
|
|
315
331
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
316
332
|
|
|
317
333
|
return {
|
package/package.json
CHANGED