officeparser 4.0.5 → 4.0.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/officeParser.js +28 -11
  2. package/package.json +1 -1
package/officeParser.js CHANGED
@@ -69,7 +69,8 @@ function parseWord(filepath, callback, config) {
69
69
  { filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
70
70
  )
71
71
  .then(files => {
72
- if (files.length == 0)
72
+ // Verify if atleast the document xml file exists in the extracted files list.
73
+ if (!files.map(file => file.path).includes(mainContentFile))
73
74
  throw ERRORMSG.fileCorrupted(filepath);
74
75
 
75
76
  return [...files.filter(file => file.path == mainContentFile),
@@ -99,7 +100,10 @@ function parseWord(filepath, callback, config) {
99
100
  // Find text nodes with w:t tags
100
101
  const xmlTextNodeList = paragraphNode.getElementsByTagName("w:t");
101
102
  // Join the texts within this paragraph node without any spaces or delimiters.
102
- return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
103
+ return Array.from(xmlTextNodeList)
104
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
105
+ .map(textNode => textNode.childNodes[0].nodeValue)
106
+ .join("");
103
107
  })
104
108
  // Join each paragraph text with a new line delimiter.
105
109
  .join(config.newlineDelimiter ?? "\n")
@@ -132,8 +136,8 @@ function parsePowerPoint(filepath, callback, config) {
132
136
  { filter: x => x.path.match(config.ignoreNotes ? slidesRegex : allFilesRegex) }
133
137
  )
134
138
  .then(files => {
135
- // Check if files is corrupted
136
- if (files.length == 0)
139
+ // Verify if atleast the slides xml files exist in the extracted files list.
140
+ if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(slidesRegex)))
137
141
  throw ERRORMSG.fileCorrupted(filepath);
138
142
 
139
143
  // Check if any sorting is required.
@@ -166,7 +170,10 @@ function parsePowerPoint(filepath, callback, config) {
166
170
  .map(paragraphNode => {
167
171
  /** Find text nodes with a:t tags */
168
172
  const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
169
- return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
173
+ return Array.from(xmlTextNodeList)
174
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
175
+ .map(textNode => textNode.childNodes[0].nodeValue)
176
+ .join("");
170
177
  })
171
178
  .join(config.newlineDelimiter ?? "\n")
172
179
  );
@@ -200,7 +207,8 @@ function parseExcel(filepath, callback, config) {
200
207
  { filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
201
208
  )
202
209
  .then(files => {
203
- if (files.length == 0)
210
+ // Verify if atleast the slides xml files exist in the extracted files list.
211
+ if (files.length == 0 || !files.map(file => file.path).some(filename => filename.match(sheetsRegex)))
204
212
  throw ERRORMSG.fileCorrupted(filepath);
205
213
 
206
214
  return {
@@ -225,7 +233,9 @@ function parseExcel(filepath, callback, config) {
225
233
  /** Find text nodes with t tags in sharedStrings xml file */
226
234
  const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
227
235
  /** Create shared string array. This will be used as a map to get strings from within sheet files. */
228
- const sharedStrings = Array.from(sharedStringsXmlTNodesList).map(tNode => tNode.childNodes[0].nodeValue);
236
+ const sharedStrings = Array.from(sharedStringsXmlTNodesList)
237
+ .filter(tNode => tNode.childNodes[0] && tNode.childNodes[0].nodeValue)
238
+ .map(tNode => tNode.childNodes[0].nodeValue);
229
239
 
230
240
  // Parse Sheet files
231
241
  xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
@@ -234,8 +244,10 @@ function parseExcel(filepath, callback, config) {
234
244
  // Traverse through the nodes list and fill responseText with either the number value in its v node or find a mapped string from sharedStrings.
235
245
  responseText.push(
236
246
  Array.from(sheetsXmlCNodesList)
237
- // Filter c nodes than do not have any v nodes
238
- .filter(cNode => cNode.getElementsByTagName("v").length != 0)
247
+ // Filter c nodes than do not have any valid v nodes
248
+ .filter(cNode => cNode.getElementsByTagName("v")[0]
249
+ && cNode.getElementsByTagName("v")[0].childNodes[0]
250
+ && cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue)
239
251
  .map(cNode => {
240
252
  /** Flag whether this node's value represents a string index */
241
253
  const isString = cNode.getAttribute("t") == "s";
@@ -266,7 +278,10 @@ function parseExcel(filepath, callback, config) {
266
278
  .map(paragraphNode => {
267
279
  /** Find text nodes with a:t tags */
268
280
  const xmlTextNodeList = paragraphNode.getElementsByTagName("a:t");
269
- return Array.from(xmlTextNodeList).map(textNode => textNode.childNodes[0].nodeValue).join("");
281
+ return Array.from(xmlTextNodeList)
282
+ .filter(textNode => textNode.childNodes[0] && textNode.childNodes[0].nodeValue)
283
+ .map(textNode => textNode.childNodes[0].nodeValue)
284
+ .join("");
270
285
  })
271
286
  .join(config.newlineDelimiter ?? "\n")
272
287
  );
@@ -279,6 +294,7 @@ function parseExcel(filepath, callback, config) {
279
294
  /** Store all the text content to respond */
280
295
  responseText.push(
281
296
  Array.from(chartsXmlCVNodesList)
297
+ .filter(cVNode => cVNode.childNodes[0] && cVNode.childNodes[0].nodeValue)
282
298
  .map(cVNode => cVNode.childNodes[0].nodeValue)
283
299
  .join(config.newlineDelimiter ?? "\n")
284
300
  );
@@ -311,7 +327,8 @@ function parseOpenOffice(filepath, callback, config) {
311
327
  { filter: x => x.path == mainContentFilePath || x.path.match(objectContentFilesRegex) }
312
328
  )
313
329
  .then(files => {
314
- if (files.length == 0)
330
+ // Verify if atleast the content xml file exists in the extracted files list.
331
+ if (!files.map(file => file.path).includes(mainContentFilePath))
315
332
  throw ERRORMSG.fileCorrupted(filepath);
316
333
 
317
334
  return {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.0.5",
3
+ "version": "4.0.6",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [