officeparser 4.0.6 → 4.0.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/officeParser.js +10 -13
  2. package/package.json +1 -1
package/officeParser.js CHANGED
@@ -59,25 +59,23 @@ const parseString = (xml) => {
59
59
  */
60
60
  function parseWord(filepath, callback, config) {
61
61
  /** The target content xml file for the docx file. */
62
- const mainContentFile = 'word/document.xml';
63
- const footnotesFile = 'word/footnotes.xml';
64
- const endnotesFile = 'word/endnotes.xml';
62
+ const mainContentFileRegex = /word\/document[\d+]?.xml/g;
63
+ const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
64
+ const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
65
65
  /** The decompress location which contains the filename in it */
66
66
  const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
67
67
  decompress(filepath,
68
68
  decompressLocation,
69
- { filter: x => [mainContentFile, footnotesFile, endnotesFile].includes(x.path) }
69
+ { filter: x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.path.match(fileRegex)) }
70
70
  )
71
71
  .then(files => {
72
72
  // Verify if atleast the document xml file exists in the extracted files list.
73
- if (!files.map(file => file.path).includes(mainContentFile))
73
+ if (!files.some(file => file.path.match(mainContentFileRegex)))
74
74
  throw ERRORMSG.fileCorrupted(filepath);
75
75
 
76
- return [...files.filter(file => file.path == mainContentFile),
77
- ...files.filter(file => file.path == footnotesFile),
78
- ...files.filter(file => file.path == endnotesFile)
79
- ]
80
- .map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
76
+ return files
77
+ .filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
78
+ .map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
81
79
  })
82
80
  // ************************************* word xml files explanation *************************************
83
81
  // Structure of xmlContent of a word file is simple.
@@ -204,7 +202,7 @@ function parseExcel(filepath, callback, config) {
204
202
  const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
205
203
  decompress(filepath,
206
204
  decompressLocation,
207
- { filter: x => ([sheetsRegex, drawingsRegex, chartsRegex].findIndex(fileRegex => x.path.match(fileRegex)) > -1) || (x.path == stringsFilePath )}
205
+ { filter: x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.path.match(fileRegex)) || x.path == stringsFilePath }
208
206
  )
209
207
  .then(files => {
210
208
  // Verify if atleast the slides xml files exist in the extracted files list.
@@ -234,8 +232,7 @@ function parseExcel(filepath, callback, config) {
234
232
  const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
235
233
  /** Create shared string array. This will be used as a map to get strings from within sheet files. */
236
234
  const sharedStrings = Array.from(sharedStringsXmlTNodesList)
237
- .filter(tNode => tNode.childNodes[0] && tNode.childNodes[0].nodeValue)
238
- .map(tNode => tNode.childNodes[0].nodeValue);
235
+ .map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
239
236
 
240
237
  // Parse Sheet files
241
238
  xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "officeparser",
3
- "version": "4.0.6",
3
+ "version": "4.0.8",
4
4
  "description": "A Node.js library to parse text out of any office file. Currently supports docx, pptx, xlsx, odt, odp, ods, pdf files.",
5
5
  "main": "officeParser.js",
6
6
  "files": [