officeparser 5.2.0 → 5.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/officeParser.js +26 -14
- package/package.json +1 -1
- package/typings/officeParser.d.ts +7 -7
package/officeParser.js
CHANGED
|
@@ -8,9 +8,6 @@ const fileType = require('file-type');
|
|
|
8
8
|
const fs = require('fs');
|
|
9
9
|
const yauzl = require('yauzl');
|
|
10
10
|
|
|
11
|
-
/** Load pdfjs-dist once at module scope. This returns a Promise that resolves to the module. */
|
|
12
|
-
const pdfjsPromise = import('pdfjs-dist/legacy/build/pdf.mjs');
|
|
13
|
-
|
|
14
11
|
/** Header for error messages */
|
|
15
12
|
const ERRORHEADER = "[OfficeParser]: ";
|
|
16
13
|
/** Error messages */
|
|
@@ -238,12 +235,19 @@ function parseExcel(file, callback, config) {
|
|
|
238
235
|
&& cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue != ''
|
|
239
236
|
}
|
|
240
237
|
|
|
241
|
-
/** Find text nodes with t tags in sharedStrings
|
|
242
|
-
const
|
|
243
|
-
|
|
238
|
+
/** Find text nodes with t tags in sharedStrings.xml file. If the sharedStringsFile is not present, we return an empty array. */
|
|
239
|
+
const sharedStringsXmlSiNodesList = xmlContentFilesObject.sharedStringsFile != undefined
|
|
240
|
+
? parseString(xmlContentFilesObject.sharedStringsFile).getElementsByTagName("si")
|
|
241
|
+
: [];
|
|
242
|
+
|
|
244
243
|
/** Create shared string array. This will be used as a map to get strings from within sheet files. */
|
|
245
|
-
const sharedStrings = Array.from(
|
|
246
|
-
|
|
244
|
+
const sharedStrings = Array.from(sharedStringsXmlSiNodesList)
|
|
245
|
+
.map(siNode => {
|
|
246
|
+
// Concatenate all <t> nodes within the <si> node
|
|
247
|
+
return Array.from(siNode.getElementsByTagName("t"))
|
|
248
|
+
.map(tNode => tNode.childNodes[0]?.nodeValue ?? '') // Extract text content from each <t> node
|
|
249
|
+
.join(''); // Combine all <t> node text into a single string
|
|
250
|
+
});
|
|
247
251
|
|
|
248
252
|
// Parse Sheet files
|
|
249
253
|
xmlContentFilesObject.sheetFiles.forEach(sheetXmlContent => {
|
|
@@ -264,13 +268,14 @@ function parseExcel(file, callback, config) {
|
|
|
264
268
|
/** Flag whether this node's value represents an index in the shared string array */
|
|
265
269
|
const isIndexInSharedStrings = cNode.getAttribute("t") == "s";
|
|
266
270
|
/** Find value nodes represented by v tags */
|
|
267
|
-
const value =
|
|
271
|
+
const value = cNode.getElementsByTagName("v")[0].childNodes[0].nodeValue;
|
|
272
|
+
const valueAsIndex = Number(value);
|
|
268
273
|
// Validate text
|
|
269
|
-
if (isIndexInSharedStrings && value >= sharedStrings.length)
|
|
274
|
+
if (isIndexInSharedStrings && (valueAsIndex != parseInt(value, 10) || valueAsIndex >= sharedStrings.length))
|
|
270
275
|
throw ERRORMSG.fileCorrupted(file);
|
|
271
276
|
|
|
272
277
|
return isIndexInSharedStrings
|
|
273
|
-
? sharedStrings[
|
|
278
|
+
? sharedStrings[valueAsIndex]
|
|
274
279
|
: value;
|
|
275
280
|
}
|
|
276
281
|
// Should not reach here. If we do, it means we are not filtering out items that we are not ready to process.
|
|
@@ -375,13 +380,19 @@ function parseOpenOffice(file, callback, config) {
|
|
|
375
380
|
function traversal(node, xmlTextArray, isFirstRecursion) {
|
|
376
381
|
if (!node.childNodes || node.childNodes.length == 0) {
|
|
377
382
|
if (node.parentNode.tagName.indexOf('text') == 0 && node.nodeValue) {
|
|
383
|
+
// If the corresponding value is of type float, we take the value from office:value attribute.
|
|
384
|
+
// However, it is not on the parentNode but rather grandparentNode.
|
|
385
|
+
const value = node.parentNode.parentNode?.getAttribute('office:value-type') == 'float'
|
|
386
|
+
? Number(node.parentNode.parentNode.getAttribute('office:value'))
|
|
387
|
+
: node.nodeValue;
|
|
388
|
+
|
|
378
389
|
if (isNotesNode(node.parentNode) && (config.putNotesAtLast || config.ignoreNotes)) {
|
|
379
|
-
notesText.push(
|
|
390
|
+
notesText.push(value);
|
|
380
391
|
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
381
392
|
notesText.push(config.newlineDelimiter ?? "\n");
|
|
382
393
|
}
|
|
383
394
|
else {
|
|
384
|
-
xmlTextArray.push(
|
|
395
|
+
xmlTextArray.push(value);
|
|
385
396
|
if (allowedTextTags.includes(node.parentNode.tagName) && !isFirstRecursion)
|
|
386
397
|
xmlTextArray.push(config.newlineDelimiter ?? "\n");
|
|
387
398
|
}
|
|
@@ -450,7 +461,8 @@ function parseOpenOffice(file, callback, config) {
|
|
|
450
461
|
*/
|
|
451
462
|
async function parsePdf(file, callback, config) {
|
|
452
463
|
// Wait for pdfjs module to be loaded once
|
|
453
|
-
|
|
464
|
+
// Lazy import pdfjs to avoid Node startup issues for environments that don't use PDF parsing
|
|
465
|
+
const pdfjs = await import('pdfjs-dist/legacy/build/pdf.mjs');
|
|
454
466
|
|
|
455
467
|
// Get the pdfjs document for the filepath or Uint8Array buffers.
|
|
456
468
|
// pdfjs does not accept Buffers directly, so we convert them to Uint8Array.
|
package/package.json
CHANGED
|
@@ -18,16 +18,16 @@ export type OfficeParserConfig = {
|
|
|
18
18
|
putNotesAtLast?: boolean;
|
|
19
19
|
};
|
|
20
20
|
/** Main async function with callback to execute parseOffice for supported files
|
|
21
|
-
* @param {string | Buffer}
|
|
22
|
-
* @param {function}
|
|
23
|
-
* @param {OfficeParserConfig}
|
|
21
|
+
* @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
|
|
22
|
+
* @param {function} callback Callback function that returns value or error
|
|
23
|
+
* @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
|
|
24
24
|
* @returns {void}
|
|
25
25
|
*/
|
|
26
|
-
export function parseOffice(
|
|
26
|
+
export function parseOffice(srcFile: string | Buffer | ArrayBuffer, callback: Function, config?: OfficeParserConfig): void;
|
|
27
27
|
/** Main async function that can be used with await to execute parseOffice. Or it can be used with promises.
|
|
28
|
-
* @param {string | Buffer}
|
|
29
|
-
* @param {OfficeParserConfig}
|
|
28
|
+
* @param {string | Buffer | ArrayBuffer} srcFile File path or file buffers or Javascript ArrayBuffer
|
|
29
|
+
* @param {OfficeParserConfig} [config={}] [OPTIONAL]: Config Object for officeParser
|
|
30
30
|
* @returns {Promise<string>}
|
|
31
31
|
*/
|
|
32
|
-
export function parseOfficeAsync(
|
|
32
|
+
export function parseOfficeAsync(srcFile: string | Buffer | ArrayBuffer, config?: OfficeParserConfig): Promise<string>;
|
|
33
33
|
//# sourceMappingURL=officeParser.d.ts.map
|