officeparser 4.0.5 → 4.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +28 -11
- package/package.json +1 -1
package/officeParser.js
CHANGED
|
@@ -69,7 +69,8 @@ function parseWord(filepath, callback, config) {
|
|
|
69
69
|
{ filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
|
|
70
70
|
)
|
|
71
71
|
.then(files => {
|
|
72
|
-
if
|
|
72
|
+
// Verify if atleast the document xml file exists in the extracted files list.
|
|
73
|
+
if (!files.map(file => file.path).includes(mainContentFile))
|
|
73
74
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
74
75
|
|
|
75
76
|
return [...files.filter(file => file.path == mainContentFile),
|
|
@@ -99,7 +100,10 @@ function parseWord(filepath, callback, config) {
|
|
|
99
100
|
// Find text nodes with w:t tags
|
|
100
101
|
const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
|
|
101
102
|
// Join the texts within this paragraph node without any spaces or delimiters.
|
|
102
|
-
return Array.from(xmlTextNodeList)
|
|
103
|
+
return Array.from(xmlTextNodeList)
|
|
104
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
105
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
106
|
+
.join("");
|
|
103
107
|
})
|
|
104
108
|
// Join each paragraph text with a new line delimiter.
|
|
105
109
|
.join(config.newlineDelimiter ?? "\n")
|
|
@@ -132,8 +136,8 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
132
136
|
{ filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
|
|
133
137
|
)
|
|
134
138
|
.then(files => {
|
|
135
|
-
//
|
|
136
|
-
if (files.length == 0)
|
|
139
|
+
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
140
|
+
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
|
|
137
141
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
138
142
|
|
|
139
143
|
// Check if any sorting is required.
|
|
@@ -166,7 +170,10 @@ function parsePowerPoint(filepath, callback, config) {
|
|
|
166
170
|
.map(paragraphNode => {
|
|
167
171
|
/** Find text nodes with a:t tags */
|
|
168
172
|
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
169
|
-
return Array.from(xmlTextNodeList)
|
|
173
|
+
return Array.from(xmlTextNodeList)
|
|
174
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
175
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
176
|
+
.join("");
|
|
170
177
|
})
|
|
171
178
|
.join(config.newlineDelimiter ?? "\n")
|
|
172
179
|
);
|
|
@@ -200,7 +207,8 @@ function parseExcel(filepath, callback, config) {
|
|
|
200
207
|
{ filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
|
|
201
208
|
)
|
|
202
209
|
.then(files => {
|
|
203
|
-
if
|
|
210
|
+
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
211
|
+
if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
|
|
204
212
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
205
213
|
|
|
206
214
|
return {
|
|
@@ -225,7 +233,9 @@ function parseExcel(filepath, callback, config) {
|
|
|
225
233
|
/** Find text nodes with t tags in sharedStrings xml file */
|
|
226
234
|
const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
|
|
227
235
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
228
|
-
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
236
|
+
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
237
|
+
.filter(tNode => tNode.childNodes[0] && tNode.childNodes[0].nodeValue)
|
|
238
|
+
.map(tNode => tNode.childNodes[0].nodeValue);
|
|
229
239
|
|
|
230
240
|
// Parse Sheet files
|
|
231
241
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
@@ -234,8 +244,10 @@ function parseExcel(filepath, callback, config) {
|
|
|
234
244
|
// Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
|
|
235
245
|
responseText.push(
|
|
236
246
|
Array.from(sheetsXmlCNodesList)
|
|
237
|
-
// Filter c nodes than do not have any v nodes
|
|
238
|
-
.filter(cNode => cNode.getElementsByTagName("v")
|
|
247
|
+
// Filter c nodes than do not have any valid v nodes
|
|
248
|
+
.filter(cNode => cNode.getElementsByTagName("v")[0]
|
|
249
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0]
|
|
250
|
+
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
|
|
239
251
|
.map(cNode => {
|
|
240
252
|
/** Flag whether this node's value represents a string index */
|
|
241
253
|
const isString = cNode.getAttribute("t") == "s";
|
|
@@ -266,7 +278,10 @@ function parseExcel(filepath, callback, config) {
|
|
|
266
278
|
.map(paragraphNode => {
|
|
267
279
|
/** Find text nodes with a:t tags */
|
|
268
280
|
const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
|
|
269
|
-
return Array.from(xmlTextNodeList)
|
|
281
|
+
return Array.from(xmlTextNodeList)
|
|
282
|
+
.filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
|
|
283
|
+
.map(textNode => textNode.childNodes[0].nodeValue)
|
|
284
|
+
.join("");
|
|
270
285
|
})
|
|
271
286
|
.join(config.newlineDelimiter ?? "\n")
|
|
272
287
|
);
|
|
@@ -279,6 +294,7 @@ function parseExcel(filepath, callback, config) {
|
|
|
279
294
|
/** Store all the text content to respond */
|
|
280
295
|
responseText.push(
|
|
281
296
|
Array.from(chartsXmlCVNodesList)
|
|
297
|
+
.filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
|
|
282
298
|
.map(cVNode => cVNode.childNodes[0].nodeValue)
|
|
283
299
|
.join(config.newlineDelimiter ?? "\n")
|
|
284
300
|
);
|
|
@@ -311,7 +327,8 @@ function parseOpenOffice(filepath, callback, config) {
|
|
|
311
327
|
{ filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
|
|
312
328
|
)
|
|
313
329
|
.then(files => {
|
|
314
|
-
if
|
|
330
|
+
// Verify if atleast the content xml file exists in the extracted files list.
|
|
331
|
+
if (!files.map(file => file.path).includes(mainContentFilePath))
|
|
315
332
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
316
333
|
|
|
317
334
|
return {
|
package/package.json
CHANGED