officeparser 7.5.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +37 -0
- package/dist/OfficeParser.js +54 -6
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +153 -153
- package/dist/officeparser.browser.mjs +147 -147
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +176 -176
- package/dist/officeparser.browser.slim.mjs +176 -176
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/OpenOfficeParser.js +17 -3
- package/dist/parsers/PowerPointParser.js +21 -5
- package/dist/parsers/WordParser.js +12 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +1 -1
|
@@ -37,7 +37,7 @@ const parseEpub = async (buffer, config) => {
|
|
|
37
37
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, (path) => /META-INF\/container\.xml$/i.test(path)
|
|
38
38
|
|| /\.opf$/i.test(path)
|
|
39
39
|
|| /\.(xhtml|html|htm)$/i.test(path)
|
|
40
|
-
|| (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits);
|
|
40
|
+
|| (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits, config);
|
|
41
41
|
// The OPF path is authoritative via META-INF/container.xml; fall back to scanning
|
|
42
42
|
// for any .opf file for malformed archives that skip the container manifest.
|
|
43
43
|
let opfPath;
|
|
@@ -49,7 +49,7 @@ const parseEpub = async (buffer, config) => {
|
|
|
49
49
|
}
|
|
50
50
|
const opfFile = (opfPath && files.find(f => f.path === opfPath)) || files.find(f => /\.opf$/i.test(f.path));
|
|
51
51
|
if (!opfFile) {
|
|
52
|
-
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.
|
|
52
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.REQUIRED_PART_MISSING, config, { fileType: 'epub', part: 'OPF package document (.opf)' });
|
|
53
53
|
}
|
|
54
54
|
const opfDir = opfFile.path.includes('/') ? opfFile.path.substring(0, opfFile.path.lastIndexOf('/') + 1) : '';
|
|
55
55
|
const opfXml = (0, xmlUtils_js_1.parseXmlString)(opfFile.content.toString('utf-8'));
|
|
@@ -67,7 +67,16 @@ const parseExcel = async (buffer, config) => {
|
|
|
67
67
|
!!x.match(customPropsFileRegex) ||
|
|
68
68
|
!!x.match(appPropsFileRegex) ||
|
|
69
69
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
|
|
70
|
-
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits);
|
|
70
|
+
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits, config);
|
|
71
|
+
// Every workbook has xl/workbook.xml; without it the archive is not a spreadsheet.
|
|
72
|
+
// Resolved up front so a file that cannot be a workbook fails before any of the parsing
|
|
73
|
+
// work below, and read again further down for the sheet-name map.
|
|
74
|
+
const workbookFile = (0, zipUtils_js_1.findRequiredPart)(files, path => path === 'xl/workbook.xml', config, { fileType: 'xlsx', part: 'xl/workbook.xml' });
|
|
75
|
+
// Worksheets, by contrast, are not guaranteed: a workbook holding only chartsheets is
|
|
76
|
+
// valid and simply has no cell text to extract. Warn rather than fail, so the caller can
|
|
77
|
+
// tell "nothing to read here" from "we read nothing".
|
|
78
|
+
if (!files.some(file => !!file.path.match(sheetsRegex)))
|
|
79
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND, config);
|
|
71
80
|
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
72
81
|
// Updated to store structured content (rich text runs) or simple string
|
|
73
82
|
const sharedStrings = [];
|
|
@@ -372,9 +381,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
372
381
|
}
|
|
373
382
|
// Parse workbook.xml to get sheet names and map them to sheet files
|
|
374
383
|
const sheetNameMap = {};
|
|
375
|
-
const workbookFile = files.find(f => f.path === 'xl/workbook.xml');
|
|
376
384
|
const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
|
|
377
|
-
if (
|
|
385
|
+
if (workbookRelsFile) {
|
|
378
386
|
// Parse rels to get rId -> file mapping
|
|
379
387
|
const relsXml = (0, xmlUtils_js_1.parseXmlString)(workbookRelsFile.content.toString());
|
|
380
388
|
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
@@ -401,7 +409,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
401
409
|
}
|
|
402
410
|
}
|
|
403
411
|
const content = [];
|
|
404
|
-
const rawContents = [];
|
|
405
412
|
for (const file of files) {
|
|
406
413
|
if (file.path.match(mediaFileRegex))
|
|
407
414
|
continue;
|
|
@@ -418,9 +425,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
418
425
|
if (file.path.match(drawingRelsRegex))
|
|
419
426
|
continue;
|
|
420
427
|
if (file.path.match(sheetsRegex)) {
|
|
421
|
-
if (config.includeRawContent) {
|
|
422
|
-
rawContents.push(file.content.toString());
|
|
423
|
-
}
|
|
424
428
|
const sheetFilename = file.path.split('/').pop() || '';
|
|
425
429
|
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
426
430
|
const relsFile = files.find(f => f.path === relsFilename);
|
|
@@ -125,6 +125,8 @@ const cleanAttachmentName = (href) => {
|
|
|
125
125
|
const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
|
|
126
126
|
return cleaned.split('/').pop() || '';
|
|
127
127
|
};
|
|
128
|
+
/** The ODF document types this parser handles, used to validate a caller-supplied file type. */
|
|
129
|
+
const ODF_FILE_TYPES = ['odt', 'odp', 'ods'];
|
|
128
130
|
/**
|
|
129
131
|
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
130
132
|
*
|
|
@@ -148,10 +150,17 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
148
150
|
!!x.match(metaFileRegex) ||
|
|
149
151
|
!!x.match(stylesFileRegex) ||
|
|
150
152
|
!!x.match(mimetypeFileRegex) ||
|
|
151
|
-
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
|
|
153
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
|
|
152
154
|
// 1. Determine File Type
|
|
153
155
|
const mimetypeFile = files.find(f => f.path === 'mimetype');
|
|
154
|
-
|
|
156
|
+
// The archive's own mimetype entry is authoritative when present. When it is missing,
|
|
157
|
+
// fall back to the type the caller asked for (or that was derived from the extension)
|
|
158
|
+
// rather than assuming text: guessing 'odt' for a spreadsheet sends the parser down the
|
|
159
|
+
// office:text branch, which finds nothing in an office:spreadsheet body and yields an
|
|
160
|
+
// empty document for a perfectly valid file.
|
|
161
|
+
let fileType = ODF_FILE_TYPES.includes(config.fileType)
|
|
162
|
+
? config.fileType
|
|
163
|
+
: 'odt';
|
|
155
164
|
if (mimetypeFile) {
|
|
156
165
|
const mime = mimetypeFile.content.toString().trim();
|
|
157
166
|
if (mime.includes('spreadsheet'))
|
|
@@ -161,7 +170,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
161
170
|
else if (mime.includes('text'))
|
|
162
171
|
fileType = 'odt';
|
|
163
172
|
}
|
|
164
|
-
|
|
173
|
+
// The document body is the content.xml at the archive root. The fallback stays anchored
|
|
174
|
+
// and excludes embedded objects: an ODF file can carry Object N/content.xml for a chart
|
|
175
|
+
// or formula, and an unanchored match would promote one of those to the document body
|
|
176
|
+
// when the real one is missing, silently parsing a chart as if it were the whole file.
|
|
177
|
+
const mainContentFile = files.find(f => f.path === 'content.xml')
|
|
178
|
+
|| (0, zipUtils_js_1.findRequiredPart)(files, path => /(^|\/)content\.xml$/.test(path) && !objectContentFileRegex.test(path), config, { fileType, part: 'content.xml' });
|
|
165
179
|
const stylesFile = files.find(f => f.path === 'styles.xml');
|
|
166
180
|
const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
|
|
167
181
|
const content = [];
|
|
@@ -57,6 +57,7 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
57
57
|
const commentsFileRegex = /ppt\/comments\/comment\d+\.xml/;
|
|
58
58
|
const commentAuthorsRegex = /ppt\/commentAuthors\.xml/;
|
|
59
59
|
const slideMastersRegex = /ppt\/slideMasters\/slideMaster\d+\.xml/;
|
|
60
|
+
const presentationFileRegex = /ppt\/presentation\.xml/;
|
|
60
61
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, x => !!x.match(config.ignoreNotes ? slidesRegex : allFilesRegex) ||
|
|
61
62
|
!!x.match(corePropsFileRegex) ||
|
|
62
63
|
!!x.match(customPropsFileRegex) ||
|
|
@@ -64,7 +65,14 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
64
65
|
!!x.match(slideRelsRegex) ||
|
|
65
66
|
(!config.ignoreComments && (!!x.match(commentsFileRegex) || !!x.match(commentAuthorsRegex))) ||
|
|
66
67
|
(!config.ignoreSlideMasters && !!x.match(slideMastersRegex)) ||
|
|
67
|
-
|
|
68
|
+
!!x.match(presentationFileRegex) ||
|
|
69
|
+
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(chartFileRegex))), config.decompressionLimits, config);
|
|
70
|
+
// ppt/presentation.xml is the part that makes an archive a presentation, and unlike the
|
|
71
|
+
// slides it is always present: PowerPoint can save a deck with no slides at all, so an
|
|
72
|
+
// empty ppt/slides/ is a warning rather than a failure.
|
|
73
|
+
(0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(presentationFileRegex), config, { fileType: 'pptx', part: 'ppt/presentation.xml' });
|
|
74
|
+
if (!files.some(file => !!file.path.match(slidesRegex)))
|
|
75
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_SLIDES_FOUND, config);
|
|
68
76
|
// Extract metadata
|
|
69
77
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
70
78
|
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
@@ -89,7 +97,6 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
89
97
|
return aNum - bNum;
|
|
90
98
|
});
|
|
91
99
|
const content = [];
|
|
92
|
-
const rawContents = [];
|
|
93
100
|
const slideRelsMap = {};
|
|
94
101
|
const authorMap = {};
|
|
95
102
|
if (!config.ignoreComments) {
|
|
@@ -752,15 +759,24 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
752
759
|
continue;
|
|
753
760
|
if (file.path.match(slideRelsRegex))
|
|
754
761
|
continue;
|
|
762
|
+
// All three document-property parts are extracted for metadata and were read earlier;
|
|
763
|
+
// only the core one was skipped here, leaving the other two to be parsed again as if
|
|
764
|
+
// they might be slides.
|
|
755
765
|
if (file.path.match(corePropsFileRegex))
|
|
756
766
|
continue;
|
|
767
|
+
if (file.path.match(appPropsFileRegex))
|
|
768
|
+
continue;
|
|
769
|
+
if (file.path.match(customPropsFileRegex))
|
|
770
|
+
continue;
|
|
757
771
|
if (file.path.includes("comment"))
|
|
758
772
|
continue;
|
|
773
|
+
// This loop treats every remaining file as a slide or note, so the presentation part
|
|
774
|
+
// has to be skipped explicitly: it carries no slide number and would otherwise be
|
|
775
|
+
// added to the deck as an empty slide.
|
|
776
|
+
if (file.path.match(presentationFileRegex))
|
|
777
|
+
continue;
|
|
759
778
|
const xmlContentString = file.content.toString();
|
|
760
779
|
const xml = (0, xmlUtils_js_1.parseXmlString)(xmlContentString, { locator: config.includeRawContent });
|
|
761
|
-
if (config.includeRawContent) {
|
|
762
|
-
rawContents.push(xmlContentString);
|
|
763
|
-
}
|
|
764
780
|
const slideMatch = file.path.match(slideNumberRegex);
|
|
765
781
|
const slideNumber = slideMatch ? parseInt(slideMatch[1]) : 0;
|
|
766
782
|
const masterMatch = file.path.match(/slideMaster(\d+)\.xml/);
|
|
@@ -95,8 +95,12 @@ const parseWord = async (buffer, config) => {
|
|
|
95
95
|
const footnotesFileRegex = /word\/footnotes[\d+]?.xml/;
|
|
96
96
|
const endnotesFileRegex = /word\/endnotes[\d+]?.xml/;
|
|
97
97
|
const commentsFileRegex = /word\/comments[\d+]?.xml/;
|
|
98
|
-
|
|
99
|
-
|
|
98
|
+
// Headers and footers are the only parts a document can have many of: Word writes up to
|
|
99
|
+
// three per section (default, first page, even pages), so a handful of sections is enough
|
|
100
|
+
// to reach header10.xml. The single-character form the other parts use stops matching at
|
|
101
|
+
// nine, which would drop those later files as silently as not extracting them at all.
|
|
102
|
+
const headerFileRegex = /word\/header\d*\.xml/;
|
|
103
|
+
const footerFileRegex = /word\/footer\d*\.xml/;
|
|
100
104
|
const numberingFileRegex = /word\/numbering[\d+]?.xml/;
|
|
101
105
|
const mediaFileRegex = /(word\/)?media\/.*/;
|
|
102
106
|
const corePropsFileRegex = /docProps\/core[\d+]?.xml/;
|
|
@@ -244,7 +248,12 @@ const parseWord = async (buffer, config) => {
|
|
|
244
248
|
!!x.match(appPropsFileRegex) ||
|
|
245
249
|
!!x.match(relsFileRegex) ||
|
|
246
250
|
!!x.match(stylesFileRegex) ||
|
|
247
|
-
(
|
|
251
|
+
(!config.ignoreComments && !!x.match(commentsFileRegex)) ||
|
|
252
|
+
(!config.ignoreHeadersAndFooters && (!!x.match(headerFileRegex) || !!x.match(footerFileRegex))) ||
|
|
253
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
|
|
254
|
+
// A DOCX without its main document part is not a DOCX. Checked with the same regex the
|
|
255
|
+
// parse loop below uses to recognize it, so the two cannot fall out of step.
|
|
256
|
+
(0, zipUtils_js_1.findRequiredPart)(files, path => !!path.match(documentFileRegex), config, { fileType: 'docx', part: 'word/document.xml' });
|
|
248
257
|
// Extract metadata
|
|
249
258
|
const corePropsFile = files.find(f => f.path.match(corePropsFileRegex));
|
|
250
259
|
const metadata = corePropsFile ? (0, xmlUtils_js_1.parseOfficeMetadata)(corePropsFile.content.toString()) : {};
|
|
@@ -405,7 +414,6 @@ const parseWord = async (buffer, config) => {
|
|
|
405
414
|
}
|
|
406
415
|
}
|
|
407
416
|
const content = [];
|
|
408
|
-
const rawContents = [];
|
|
409
417
|
const numberingState = {};
|
|
410
418
|
const listCounters = {}; // Track item index per listId/level
|
|
411
419
|
// Helper to parse a paragraph node
|
|
@@ -1092,9 +1100,6 @@ const parseWord = async (buffer, config) => {
|
|
|
1092
1100
|
if (file.path.match(footerFileRegex))
|
|
1093
1101
|
continue;
|
|
1094
1102
|
const documentContent = file.content.toString();
|
|
1095
|
-
if (config.includeRawContent) {
|
|
1096
|
-
rawContents.push(documentContent);
|
|
1097
|
-
}
|
|
1098
1103
|
const doc = (0, xmlUtils_js_1.parseXmlString)(documentContent, { locator: config.includeRawContent });
|
|
1099
1104
|
const body = (0, xmlUtils_js_1.getFirstElementByTagName)(doc, "w:body");
|
|
1100
1105
|
if (body) {
|