officeparser 4.0.6 → 4.0.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +10 -13
- package/package.json +1 -1
package/officeParser.js
CHANGED
|
@@ -59,25 +59,23 @@ const parseString = (xml) => {
|
|
|
59
59
|
*/
|
|
60
60
|
function parseWord(filepath, callback, config) {
|
|
61
61
|
/** The target content xml file for the docx file. */
|
|
62
|
-
const
|
|
63
|
-
const
|
|
64
|
-
const
|
|
62
|
+
const mainContentFileRegex = /word\/document[\d+]?.xml/g;
|
|
63
|
+
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/g;
|
|
64
|
+
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/g;
|
|
65
65
|
/** The decompress location which contains the filename in it */
|
|
66
66
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
67
67
|
decompress(filepath,
|
|
68
68
|
decompressLocation,
|
|
69
|
-
{ filter: x => [
|
|
69
|
+
{ filter: x => [mainContentFileRegex, footnotesFileRegex, endnotesFileRegex].some(fileRegex => x.path.match(fileRegex)) }
|
|
70
70
|
)
|
|
71
71
|
.then(files => {
|
|
72
72
|
// Verify if atleast the document xml file exists in the extracted files list.
|
|
73
|
-
if (!files.
|
|
73
|
+
if (!files.some(file => file.path.match(mainContentFileRegex)))
|
|
74
74
|
throw ERRORMSG.fileCorrupted(filepath);
|
|
75
75
|
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
]
|
|
80
|
-
.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
76
|
+
return files
|
|
77
|
+
.filter(file => file.path.match(mainContentFileRegex) || file.path.match(footnotesFileRegex) || file.path.match(endnotesFileRegex))
|
|
78
|
+
.map(file => fs.readFileSync(`${decompressLocation}/${file.path}`, 'utf8'));
|
|
81
79
|
})
|
|
82
80
|
// ************************************* word xml files explanation *************************************
|
|
83
81
|
// Structure of xmlContent of a word file is simple.
|
|
@@ -204,7 +202,7 @@ function parseExcel(filepath, callback, config) {
|
|
|
204
202
|
const decompressLocation = `${config.tempFilesLocation}/${filepath.split("/").pop()}`;
|
|
205
203
|
decompress(filepath,
|
|
206
204
|
decompressLocation,
|
|
207
|
-
{ filter: x =>
|
|
205
|
+
{ filter: x => [sheetsRegex, drawingsRegex, chartsRegex].some(fileRegex => x.path.match(fileRegex)) || x.path == stringsFilePath }
|
|
208
206
|
)
|
|
209
207
|
.then(files => {
|
|
210
208
|
// Verify if atleast the slides xml files exist in the extracted files list.
|
|
@@ -234,8 +232,7 @@ function parseExcel(filepath, callback, config) {
|
|
|
234
232
|
const sharedStringsXmlTNodesList = parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("t");
|
|
235
233
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
236
234
|
const sharedStrings = Array.from(sharedStringsXmlTNodesList)
|
|
237
|
-
.
|
|
238
|
-
.map(tNode => tNode.childNodes[0].nodeValue);
|
|
235
|
+
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '');
|
|
239
236
|
|
|
240
237
|
// Parse Sheet files
|
|
241
238
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
package/package.json
CHANGED