officeparser 7.4.0 → 7.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +85 -4
- package/dist/OfficeParser.js +54 -6
- package/dist/generators/BaseGenerator.d.ts +15 -0
- package/dist/generators/BaseGenerator.js +31 -0
- package/dist/generators/HtmlGenerator.d.ts +9 -0
- package/dist/generators/HtmlGenerator.js +34 -4
- package/dist/generators/MarkdownGenerator.d.ts +13 -0
- package/dist/generators/MarkdownGenerator.js +115 -41
- package/dist/generators/RtfGenerator.d.ts +13 -0
- package/dist/generators/RtfGenerator.js +23 -2
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +34 -1
- package/dist/officeparser.browser.iife.js +160 -160
- package/dist/officeparser.browser.mjs +198 -198
- package/dist/officeparser.browser.slim.d.ts +34 -1
- package/dist/officeparser.browser.slim.iife.js +186 -186
- package/dist/officeparser.browser.slim.mjs +186 -186
- package/dist/parsers/EpubParser.js +2 -2
- package/dist/parsers/ExcelParser.js +11 -7
- package/dist/parsers/HtmlParser.js +39 -0
- package/dist/parsers/OpenOfficeParser.js +139 -167
- package/dist/parsers/PowerPointParser.js +48 -11
- package/dist/parsers/WordParser.js +33 -7
- package/dist/sbom.cdx.json +92 -92
- package/dist/types.d.ts +34 -1
- package/dist/types.js +10 -0
- package/dist/utils/configUtils.d.ts +15 -2
- package/dist/utils/configUtils.js +58 -13
- package/dist/utils/errorUtils.d.ts +8 -2
- package/dist/utils/errorUtils.js +23 -1
- package/dist/utils/mathUtils.d.ts +42 -0
- package/dist/utils/mathUtils.js +385 -0
- package/dist/utils/zipUtils.d.ts +64 -4
- package/dist/utils/zipUtils.js +188 -4
- package/package.json +9 -5
|
@@ -37,7 +37,7 @@ const parseEpub = async (buffer, config) => {
|
|
|
37
37
|
const files = await (0, zipUtils_js_1.extractFiles)(buffer, (path) => /META-INF\/container\.xml$/i.test(path)
|
|
38
38
|
|| /\.opf$/i.test(path)
|
|
39
39
|
|| /\.(xhtml|html|htm)$/i.test(path)
|
|
40
|
-
|| (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits);
|
|
40
|
+
|| (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits, config);
|
|
41
41
|
// The OPF path is authoritative via META-INF/container.xml; fall back to scanning
|
|
42
42
|
// for any .opf file for malformed archives that skip the container manifest.
|
|
43
43
|
let opfPath;
|
|
@@ -49,7 +49,7 @@ const parseEpub = async (buffer, config) => {
|
|
|
49
49
|
}
|
|
50
50
|
const opfFile = (opfPath && files.find(f => f.path === opfPath)) || files.find(f => /\.opf$/i.test(f.path));
|
|
51
51
|
if (!opfFile) {
|
|
52
|
-
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.
|
|
52
|
+
throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.REQUIRED_PART_MISSING, config, { fileType: 'epub', part: 'OPF package document (.opf)' });
|
|
53
53
|
}
|
|
54
54
|
const opfDir = opfFile.path.includes('/') ? opfFile.path.substring(0, opfFile.path.lastIndexOf('/') + 1) : '';
|
|
55
55
|
const opfXml = (0, xmlUtils_js_1.parseXmlString)(opfFile.content.toString('utf-8'));
|
|
@@ -67,7 +67,16 @@ const parseExcel = async (buffer, config) => {
|
|
|
67
67
|
!!x.match(customPropsFileRegex) ||
|
|
68
68
|
!!x.match(appPropsFileRegex) ||
|
|
69
69
|
(!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
|
|
70
|
-
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits);
|
|
70
|
+
((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits, config);
|
|
71
|
+
// Every workbook has xl/workbook.xml; without it the archive is not a spreadsheet.
|
|
72
|
+
// Resolved up front so a file that cannot be a workbook fails before any of the parsing
|
|
73
|
+
// work below, and read again further down for the sheet-name map.
|
|
74
|
+
const workbookFile = (0, zipUtils_js_1.findRequiredPart)(files, path => path === 'xl/workbook.xml', config, { fileType: 'xlsx', part: 'xl/workbook.xml' });
|
|
75
|
+
// Worksheets, by contrast, are not guaranteed: a workbook holding only chartsheets is
|
|
76
|
+
// valid and simply has no cell text to extract. Warn rather than fail, so the caller can
|
|
77
|
+
// tell "nothing to read here" from "we read nothing".
|
|
78
|
+
if (!files.some(file => !!file.path.match(sheetsRegex)))
|
|
79
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND, config);
|
|
71
80
|
const sharedStringsFile = files.find(f => f.path === stringsFilePath);
|
|
72
81
|
// Updated to store structured content (rich text runs) or simple string
|
|
73
82
|
const sharedStrings = [];
|
|
@@ -372,9 +381,8 @@ const parseExcel = async (buffer, config) => {
|
|
|
372
381
|
}
|
|
373
382
|
// Parse workbook.xml to get sheet names and map them to sheet files
|
|
374
383
|
const sheetNameMap = {};
|
|
375
|
-
const workbookFile = files.find(f => f.path === 'xl/workbook.xml');
|
|
376
384
|
const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
|
|
377
|
-
if (
|
|
385
|
+
if (workbookRelsFile) {
|
|
378
386
|
// Parse rels to get rId -> file mapping
|
|
379
387
|
const relsXml = (0, xmlUtils_js_1.parseXmlString)(workbookRelsFile.content.toString());
|
|
380
388
|
const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
|
|
@@ -401,7 +409,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
401
409
|
}
|
|
402
410
|
}
|
|
403
411
|
const content = [];
|
|
404
|
-
const rawContents = [];
|
|
405
412
|
for (const file of files) {
|
|
406
413
|
if (file.path.match(mediaFileRegex))
|
|
407
414
|
continue;
|
|
@@ -418,9 +425,6 @@ const parseExcel = async (buffer, config) => {
|
|
|
418
425
|
if (file.path.match(drawingRelsRegex))
|
|
419
426
|
continue;
|
|
420
427
|
if (file.path.match(sheetsRegex)) {
|
|
421
|
-
if (config.includeRawContent) {
|
|
422
|
-
rawContents.push(file.content.toString());
|
|
423
|
-
}
|
|
424
428
|
const sheetFilename = file.path.split('/').pop() || '';
|
|
425
429
|
const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
|
|
426
430
|
const relsFile = files.find(f => f.path === relsFilename);
|
|
@@ -4,6 +4,7 @@ exports.parseHtml = void 0;
|
|
|
4
4
|
const types_js_1 = require("../types.js");
|
|
5
5
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
6
6
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
7
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
7
8
|
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
8
9
|
/**
|
|
9
10
|
* Maximum element nesting depth accepted from an HTML/XHTML source before the parser gives up
|
|
@@ -11,6 +12,25 @@ const sanitize_js_1 = require("../utils/sanitize.js");
|
|
|
11
12
|
* `parseNode` for why this value and not a larger one.
|
|
12
13
|
*/
|
|
13
14
|
const MAX_HTML_NESTING_DEPTH = 256;
|
|
15
|
+
/**
|
|
16
|
+
* Presents an `HtmlNode` as a `MathNode` for the shared MathML converter.
|
|
17
|
+
*
|
|
18
|
+
* The shapes already line up field for field; the one thing that must happen here is entity
|
|
19
|
+
* decoding, since this parser keeps text nodes in their raw escaped form and `<` inside an
|
|
20
|
+
* `<mo>` is a less-than operator, not markup.
|
|
21
|
+
*/
|
|
22
|
+
const toMathNode = (node) => ({
|
|
23
|
+
tagName: node.tagName,
|
|
24
|
+
attributes: node.attributes,
|
|
25
|
+
text: node.text === undefined ? undefined : node.text
|
|
26
|
+
.replace(/ /g, ' ')
|
|
27
|
+
.replace(/</g, '<')
|
|
28
|
+
.replace(/>/g, '>')
|
|
29
|
+
.replace(/&/g, '&')
|
|
30
|
+
.replace(/"/g, '"')
|
|
31
|
+
.replace(/'/g, "'"),
|
|
32
|
+
children: (node.children || []).map(toMathNode),
|
|
33
|
+
});
|
|
14
34
|
const parseAttributes = (attrString) => {
|
|
15
35
|
const attrs = {};
|
|
16
36
|
// Attribute names follow the HTML5 rule - any character except whitespace and
|
|
@@ -572,6 +592,25 @@ const parseHtml = async (buffer, config) => {
|
|
|
572
592
|
metadata: { math: mathMode }
|
|
573
593
|
};
|
|
574
594
|
}
|
|
595
|
+
// Native MathML. This is what a real-world page and every EPUB3 uses (EpubParser
|
|
596
|
+
// routes each spine item through here), as opposed to the `data-math` round-trip
|
|
597
|
+
// contract above, which only ever appears in this library's own HTML output. Without
|
|
598
|
+
// it, a `<math>` element fell through to the generic element handling below, which
|
|
599
|
+
// concatenates descendant text: `<mfrac><mn>1</mn><mn>2</mn></mfrac>` became "12".
|
|
600
|
+
if (tagName === 'math' || tagName.endsWith(':math')) {
|
|
601
|
+
// `display="block"` is MathML's own attribute for a display equation; the legacy
|
|
602
|
+
// `mode="display"` means the same thing and is still emitted by older producers.
|
|
603
|
+
const isBlock = node.attributes?.['display'] === 'block'
|
|
604
|
+
|| node.attributes?.['mode'] === 'display';
|
|
605
|
+
const latex = (0, mathUtils_js_1.mathmlTreeToLatex)(toMathNode(node));
|
|
606
|
+
if ((0, mathUtils_js_1.isEmptyMath)(latex))
|
|
607
|
+
return null;
|
|
608
|
+
return {
|
|
609
|
+
type: 'code',
|
|
610
|
+
text: latex,
|
|
611
|
+
metadata: { math: isBlock ? 'block' : 'inline' }
|
|
612
|
+
};
|
|
613
|
+
}
|
|
575
614
|
// Admonition: inscript-editor's Admonition node renders
|
|
576
615
|
// <div class="admonition admonition-note" data-type="note">…children…</div>.
|
|
577
616
|
if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
|
|
@@ -27,6 +27,7 @@ const types_js_1 = require("../types.js");
|
|
|
27
27
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
28
28
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
29
29
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
30
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
30
31
|
/**
|
|
31
32
|
* Tracks how many table cells a single document has been allowed to materialize.
|
|
32
33
|
*
|
|
@@ -84,10 +85,28 @@ class CellBudget {
|
|
|
84
85
|
/** Resolves the configured cell budget, falling back to the documented default. */
|
|
85
86
|
const createCellBudget = (config) => new CellBudget(config.decompressionLimits?.maxTableCells ?? 1000000, config);
|
|
86
87
|
/**
|
|
87
|
-
*
|
|
88
|
-
*
|
|
89
|
-
*
|
|
88
|
+
* Merges a style's formatting over what it inherits, dropping any flag the style explicitly turns
|
|
89
|
+
* off rather than carrying a `false` forward.
|
|
90
|
+
*
|
|
91
|
+
* Generators all test these flags for truthiness, so a retained `false` would render the same - but
|
|
92
|
+
* it would not *compare* the same, and `MarkdownGenerator.optimizeNodes` merges adjacent text nodes
|
|
93
|
+
* only when their formatting objects are equal. Leaving `bold: false` on one node and nothing on
|
|
94
|
+
* its neighbour would silently stop that merge and fragment the output. Same reasoning, and same
|
|
95
|
+
* shape, as `WordParser`'s direct-run-property merge.
|
|
90
96
|
*/
|
|
97
|
+
const mergeFormatting = (inherited, override) => {
|
|
98
|
+
if (!override)
|
|
99
|
+
return { ...inherited };
|
|
100
|
+
const merged = { ...inherited };
|
|
101
|
+
for (const key of Object.keys(override)) {
|
|
102
|
+
const value = override[key];
|
|
103
|
+
if (value === false)
|
|
104
|
+
delete merged[key];
|
|
105
|
+
else if (value !== undefined)
|
|
106
|
+
merged[key] = value;
|
|
107
|
+
}
|
|
108
|
+
return merged;
|
|
109
|
+
};
|
|
91
110
|
const toRepeatCount = (attr) => {
|
|
92
111
|
const n = parseInt(attr || "1");
|
|
93
112
|
return Number.isFinite(n) && n > 0 ? n : 1;
|
|
@@ -106,6 +125,8 @@ const cleanAttachmentName = (href) => {
|
|
|
106
125
|
const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
|
|
107
126
|
return cleaned.split('/').pop() || '';
|
|
108
127
|
};
|
|
128
|
+
/** The ODF document types this parser handles, used to validate a caller-supplied file type. */
|
|
129
|
+
const ODF_FILE_TYPES = ['odt', 'odp', 'ods'];
|
|
109
130
|
/**
|
|
110
131
|
* Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
|
|
111
132
|
*
|
|
@@ -129,10 +150,17 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
129
150
|
!!x.match(metaFileRegex) ||
|
|
130
151
|
!!x.match(stylesFileRegex) ||
|
|
131
152
|
!!x.match(mimetypeFileRegex) ||
|
|
132
|
-
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
|
|
153
|
+
(!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
|
|
133
154
|
// 1. Determine File Type
|
|
134
155
|
const mimetypeFile = files.find(f => f.path === 'mimetype');
|
|
135
|
-
|
|
156
|
+
// The archive's own mimetype entry is authoritative when present. When it is missing,
|
|
157
|
+
// fall back to the type the caller asked for (or that was derived from the extension)
|
|
158
|
+
// rather than assuming text: guessing 'odt' for a spreadsheet sends the parser down the
|
|
159
|
+
// office:text branch, which finds nothing in an office:spreadsheet body and yields an
|
|
160
|
+
// empty document for a perfectly valid file.
|
|
161
|
+
let fileType = ODF_FILE_TYPES.includes(config.fileType)
|
|
162
|
+
? config.fileType
|
|
163
|
+
: 'odt';
|
|
136
164
|
if (mimetypeFile) {
|
|
137
165
|
const mime = mimetypeFile.content.toString().trim();
|
|
138
166
|
if (mime.includes('spreadsheet'))
|
|
@@ -142,7 +170,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
142
170
|
else if (mime.includes('text'))
|
|
143
171
|
fileType = 'odt';
|
|
144
172
|
}
|
|
145
|
-
|
|
173
|
+
// The document body is the content.xml at the archive root. The fallback stays anchored
|
|
174
|
+
// and excludes embedded objects: an ODF file can carry Object N/content.xml for a chart
|
|
175
|
+
// or formula, and an unanchored match would promote one of those to the document body
|
|
176
|
+
// when the real one is missing, silently parsing a chart as if it were the whole file.
|
|
177
|
+
const mainContentFile = files.find(f => f.path === 'content.xml')
|
|
178
|
+
|| (0, zipUtils_js_1.findRequiredPart)(files, path => /(^|\/)content\.xml$/.test(path) && !objectContentFileRegex.test(path), config, { fileType, part: 'content.xml' });
|
|
146
179
|
const stylesFile = files.find(f => f.path === 'styles.xml');
|
|
147
180
|
const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
|
|
148
181
|
const content = [];
|
|
@@ -159,8 +192,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
159
192
|
let lastWasList = false;
|
|
160
193
|
let traverse;
|
|
161
194
|
// Helper to parse styles
|
|
162
|
-
const parseStyles = (
|
|
163
|
-
const styles = (0, xmlUtils_js_1.getElementsByTagName)(
|
|
195
|
+
const parseStyles = (scope) => {
|
|
196
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(scope, "style:style");
|
|
164
197
|
for (const style of styles) {
|
|
165
198
|
const name = style.getAttribute("style:name");
|
|
166
199
|
if (!name)
|
|
@@ -188,6 +221,17 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
188
221
|
if (dropCap) {
|
|
189
222
|
styleInfo.dropCap = true;
|
|
190
223
|
}
|
|
224
|
+
// Page/column breaks. ODF attaches these to the paragraph style rather than
|
|
225
|
+
// writing an inline element the way DOCX's `<w:br w:type="page"/>` does, which is
|
|
226
|
+
// why `includeBreakNodes` produced nothing at all for ODF: there was no inline
|
|
227
|
+
// element to find. Only the two break kinds that map onto a BreakMetadata type
|
|
228
|
+
// are carried; `auto` and `even-page`/`odd-page` have no equivalent.
|
|
229
|
+
const breakBefore = paraProps.getAttribute("fo:break-before");
|
|
230
|
+
if (breakBefore === 'page' || breakBefore === 'column')
|
|
231
|
+
styleInfo.breakBefore = breakBefore;
|
|
232
|
+
const breakAfter = paraProps.getAttribute("fo:break-after");
|
|
233
|
+
if (breakAfter === 'page' || breakAfter === 'column')
|
|
234
|
+
styleInfo.breakAfter = breakAfter;
|
|
191
235
|
}
|
|
192
236
|
if (Object.keys(styleInfo).length > 0) {
|
|
193
237
|
paragraphStyleMap[name] = styleInfo;
|
|
@@ -203,14 +247,28 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
203
247
|
formatting.backgroundColor = bgColor;
|
|
204
248
|
}
|
|
205
249
|
if (textProps) {
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
250
|
+
// Record the *off* states as an explicit `false`, not as an absent key.
|
|
251
|
+
//
|
|
252
|
+
// Now that a paragraph style's text properties are inherited by the runs inside it,
|
|
253
|
+
// a span has to be able to turn one back off: LibreOffice writes
|
|
254
|
+
// `fo:font-weight="normal"` on the span whenever a user un-bolds part of a
|
|
255
|
+
// bold-styled paragraph. With only the `true` side recorded, that span had nothing
|
|
256
|
+
// to override the inherited value with and came out bold - wrong in the opposite
|
|
257
|
+
// direction from the bug the inheritance fixed. `TextFormatting`'s flags are
|
|
258
|
+
// `boolean | undefined` precisely so "explicitly off" is expressible.
|
|
259
|
+
const fontWeight = textProps.getAttribute("fo:font-weight") || textProps.getAttribute("style:font-weight-asian");
|
|
260
|
+
// Numeric weights are the same axis: 600+ is bold, below that is not.
|
|
261
|
+
if (fontWeight)
|
|
262
|
+
formatting.bold = fontWeight === "bold" || /^[6-9]00$/.test(fontWeight);
|
|
263
|
+
const fontStyle = textProps.getAttribute("fo:font-style") || textProps.getAttribute("style:font-style-asian");
|
|
264
|
+
if (fontStyle)
|
|
265
|
+
formatting.italic = fontStyle === "italic" || fontStyle === "oblique";
|
|
266
|
+
const underline = textProps.getAttribute("style:text-underline-style");
|
|
267
|
+
if (underline)
|
|
268
|
+
formatting.underline = underline !== "none";
|
|
269
|
+
const lineThrough = textProps.getAttribute("style:text-line-through-style");
|
|
270
|
+
if (lineThrough)
|
|
271
|
+
formatting.strikethrough = lineThrough !== "none";
|
|
214
272
|
const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
|
|
215
273
|
if (size)
|
|
216
274
|
formatting.size = size;
|
|
@@ -247,62 +305,6 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
247
305
|
*
|
|
248
306
|
* @param node - The paragraph element to parse
|
|
249
307
|
*/
|
|
250
|
-
const parseMathML = (node) => {
|
|
251
|
-
if (!node)
|
|
252
|
-
return '';
|
|
253
|
-
if (node.nodeType === 3) { // Text node
|
|
254
|
-
return node.textContent || '';
|
|
255
|
-
}
|
|
256
|
-
if (node.nodeType !== 1) { // Not an element
|
|
257
|
-
return '';
|
|
258
|
-
}
|
|
259
|
-
const element = node;
|
|
260
|
-
const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
|
|
261
|
-
switch (tagName) {
|
|
262
|
-
case 'math':
|
|
263
|
-
case 'mrow':
|
|
264
|
-
case 'semantics':
|
|
265
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
266
|
-
case 'mfrac': {
|
|
267
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
268
|
-
if (children.length >= 2) {
|
|
269
|
-
return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
|
|
270
|
-
}
|
|
271
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
272
|
-
}
|
|
273
|
-
case 'msub': {
|
|
274
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
275
|
-
if (children.length >= 2) {
|
|
276
|
-
return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
|
|
277
|
-
}
|
|
278
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
279
|
-
}
|
|
280
|
-
case 'msup': {
|
|
281
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
282
|
-
if (children.length >= 2) {
|
|
283
|
-
return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
|
|
284
|
-
}
|
|
285
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
286
|
-
}
|
|
287
|
-
case 'msubsup': {
|
|
288
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
289
|
-
if (children.length >= 3) {
|
|
290
|
-
return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
|
|
291
|
-
}
|
|
292
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
293
|
-
}
|
|
294
|
-
case 'mi':
|
|
295
|
-
case 'mn':
|
|
296
|
-
case 'mo':
|
|
297
|
-
case 'mtext':
|
|
298
|
-
case 'ms':
|
|
299
|
-
return element.textContent || '';
|
|
300
|
-
case 'annotation':
|
|
301
|
-
return '';
|
|
302
|
-
default:
|
|
303
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
304
|
-
}
|
|
305
|
-
};
|
|
306
308
|
/**
|
|
307
309
|
* Helper to parse inline content (text, spans, links, notes, etc.) recursively.
|
|
308
310
|
*
|
|
@@ -365,6 +367,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
365
367
|
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
366
368
|
});
|
|
367
369
|
}
|
|
370
|
+
else if (tagName === 'text:soft-page-break') {
|
|
371
|
+
// The page boundary the editor recorded at its last save. DOCX's equivalent
|
|
372
|
+
// is `w:lastRenderedPageBreak`, so it maps onto the same break type rather
|
|
373
|
+
// than onto 'page', which is reserved for a break the author asked for.
|
|
374
|
+
if (config.includeBreakNodes) {
|
|
375
|
+
children.push({ type: 'break', metadata: { breakType: 'lastRenderedPage' } });
|
|
376
|
+
}
|
|
377
|
+
}
|
|
368
378
|
else if (tagName === 'text:line-break') {
|
|
369
379
|
// Line break
|
|
370
380
|
fullText += '\n';
|
|
@@ -378,7 +388,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
378
388
|
else if (tagName === 'text:span') {
|
|
379
389
|
// Formatted text span
|
|
380
390
|
const styleName = element.getAttribute("text:style-name");
|
|
381
|
-
const formatting = styleName ?
|
|
391
|
+
const formatting = styleName ? mergeFormatting(parentFormatting, styleMap[styleName]) : parentFormatting;
|
|
382
392
|
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
|
|
383
393
|
fullText += spanContent.text;
|
|
384
394
|
children.push(...spanContent.children);
|
|
@@ -483,22 +493,25 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
483
493
|
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
484
494
|
if (mathNode) {
|
|
485
495
|
isFormula = true;
|
|
486
|
-
formulaText =
|
|
496
|
+
formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
|
|
487
497
|
}
|
|
488
498
|
}
|
|
489
499
|
}
|
|
490
500
|
if (isFormula) {
|
|
491
501
|
fullText += formulaText;
|
|
492
|
-
|
|
493
|
-
|
|
502
|
+
// A `code` node carrying `math`, not a plain `text` node: the formula
|
|
503
|
+
// is LaTeX, and marking it as such is what lets generators render it
|
|
504
|
+
// as maths rather than emit it as prose that happens to contain
|
|
505
|
+
// backslashes. Same node shape DOCX, PPTX, HTML and Markdown produce.
|
|
506
|
+
const formulaNode = {
|
|
507
|
+
type: 'code',
|
|
494
508
|
text: formulaText,
|
|
495
|
-
|
|
496
|
-
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
509
|
+
metadata: { math: 'inline', ...(linkMetadata ?? {}) }
|
|
497
510
|
};
|
|
498
511
|
if (config.includeRawContent) {
|
|
499
|
-
|
|
512
|
+
formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
|
|
500
513
|
}
|
|
501
|
-
children.push(
|
|
514
|
+
children.push(formulaNode);
|
|
502
515
|
}
|
|
503
516
|
else {
|
|
504
517
|
// Standard inline image extraction fallback if object is not a formula
|
|
@@ -588,8 +601,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
588
601
|
const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
|
|
589
602
|
const alignment = styleInfo?.alignment;
|
|
590
603
|
const dropCap = styleInfo?.dropCap;
|
|
604
|
+
const formatting = mergeFormatting({}, paraStyle ? styleMap[paraStyle] : undefined);
|
|
591
605
|
// Parse content recursively using the new helper
|
|
592
|
-
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap,
|
|
606
|
+
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, formatting, undefined, sourceXml);
|
|
593
607
|
// Add style name to metadata of children if they don't have one
|
|
594
608
|
if (paraStyle) {
|
|
595
609
|
content.children.forEach(child => {
|
|
@@ -800,7 +814,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
800
814
|
// whole cell array, so rows x cols is what actually exhausts memory; charge those
|
|
801
815
|
// copies against the same budget.
|
|
802
816
|
const allowedRows = cells.length === 0
|
|
803
|
-
? rowsRepeated
|
|
817
|
+
? (rowsRepeated > 0 ? 1 + cellBudget.take(rowsRepeated - 1) : 0)
|
|
804
818
|
: Math.min(rowsRepeated, 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
|
|
805
819
|
for (let k = 0; k < allowedRows; k++) {
|
|
806
820
|
if ((k & 255) === 0)
|
|
@@ -839,82 +853,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
839
853
|
// splitting a huge repeat expansion across many small tables. `traverse` and the
|
|
840
854
|
// spreadsheet branch below both close over this; `parseTable` receives it explicitly.
|
|
841
855
|
const cellBudget = createCellBudget(config);
|
|
842
|
-
//
|
|
856
|
+
// Automatic styles are local to content.xml, but their definitions have exactly the
|
|
857
|
+
// shape styles.xml uses, so they go through the same reader rather than a second copy of
|
|
858
|
+
// it - the copy is how `fo:break-before` came to be read in neither place.
|
|
843
859
|
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
|
|
844
860
|
if (automaticStyles) {
|
|
845
|
-
|
|
846
|
-
for (const style of styles) {
|
|
847
|
-
const name = style.getAttribute("style:name");
|
|
848
|
-
if (!name)
|
|
849
|
-
continue;
|
|
850
|
-
// Parse paragraph properties for alignment
|
|
851
|
-
const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
|
|
852
|
-
const styleInfo = {};
|
|
853
|
-
if (paraProps) {
|
|
854
|
-
const textAlign = paraProps.getAttribute("fo:text-align");
|
|
855
|
-
if (textAlign) {
|
|
856
|
-
const alignMap = {
|
|
857
|
-
'start': 'left',
|
|
858
|
-
'left': 'left',
|
|
859
|
-
'center': 'center',
|
|
860
|
-
'end': 'right',
|
|
861
|
-
'right': 'right',
|
|
862
|
-
'justify': 'justify'
|
|
863
|
-
};
|
|
864
|
-
if (alignMap[textAlign]) {
|
|
865
|
-
styleInfo.alignment = alignMap[textAlign];
|
|
866
|
-
}
|
|
867
|
-
}
|
|
868
|
-
const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
|
|
869
|
-
if (dropCap)
|
|
870
|
-
styleInfo.dropCap = true;
|
|
871
|
-
}
|
|
872
|
-
if (Object.keys(styleInfo).length > 0) {
|
|
873
|
-
paragraphStyleMap[name] = styleInfo;
|
|
874
|
-
}
|
|
875
|
-
const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
|
|
876
|
-
const formatting = {};
|
|
877
|
-
if (cellProps) {
|
|
878
|
-
const bgColor = cellProps.getAttribute("fo:background-color");
|
|
879
|
-
if (bgColor && bgColor !== 'transparent')
|
|
880
|
-
formatting.backgroundColor = bgColor;
|
|
881
|
-
}
|
|
882
|
-
const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
|
|
883
|
-
if (textProps) {
|
|
884
|
-
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
885
|
-
formatting.bold = true;
|
|
886
|
-
if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
|
|
887
|
-
formatting.italic = true;
|
|
888
|
-
if (textProps.getAttribute("style:text-underline-style") === "solid")
|
|
889
|
-
formatting.underline = true;
|
|
890
|
-
if (textProps.getAttribute("style:text-line-through-style") === "solid")
|
|
891
|
-
formatting.strikethrough = true;
|
|
892
|
-
const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
|
|
893
|
-
if (size)
|
|
894
|
-
formatting.size = size;
|
|
895
|
-
const color = textProps.getAttribute("fo:color");
|
|
896
|
-
if (color)
|
|
897
|
-
formatting.color = color;
|
|
898
|
-
// Background color
|
|
899
|
-
const bgColor = textProps.getAttribute("fo:background-color");
|
|
900
|
-
if (bgColor && bgColor !== 'transparent')
|
|
901
|
-
formatting.backgroundColor = bgColor;
|
|
902
|
-
// Font family
|
|
903
|
-
const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
|
|
904
|
-
if (fontName)
|
|
905
|
-
formatting.font = fontName;
|
|
906
|
-
// Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
|
|
907
|
-
const textPosition = textProps.getAttribute("style:text-position");
|
|
908
|
-
if (textPosition) {
|
|
909
|
-
if (textPosition.startsWith("sub"))
|
|
910
|
-
formatting.subscript = true;
|
|
911
|
-
if (textPosition.startsWith("super"))
|
|
912
|
-
formatting.superscript = true;
|
|
913
|
-
}
|
|
914
|
-
}
|
|
915
|
-
if (Object.keys(formatting).length > 0)
|
|
916
|
-
styleMap[name] = formatting;
|
|
917
|
-
}
|
|
861
|
+
parseStyles(automaticStyles);
|
|
918
862
|
}
|
|
919
863
|
// Start traversal
|
|
920
864
|
const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
|
|
@@ -940,8 +884,26 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
940
884
|
* @param sourceXml - The source XML string for raw content extraction
|
|
941
885
|
* @param asSheet - If true, treats tables as sheets (for ODS)
|
|
942
886
|
*/
|
|
887
|
+
/**
|
|
888
|
+
* Emits the break a paragraph style asks for, on the given side of that paragraph.
|
|
889
|
+
*
|
|
890
|
+
* ODF has no inline break element for these - `fo:break-before="page"` sits on the style,
|
|
891
|
+
* so the break is a property of the paragraph rather than a run inside it. That makes it a
|
|
892
|
+
* sibling emitted around the paragraph node, not a child of it, which is the one structural
|
|
893
|
+
* difference from how DOCX's `<w:br w:type="page"/>` lands.
|
|
894
|
+
*/
|
|
895
|
+
const pushStyleBreak = (styleName, targetArray, edge) => {
|
|
896
|
+
if (!config.includeBreakNodes || !styleName)
|
|
897
|
+
return;
|
|
898
|
+
const info = paragraphStyleMap[styleName];
|
|
899
|
+
const breakType = edge === 'before' ? info?.breakBefore : info?.breakAfter;
|
|
900
|
+
if (!breakType)
|
|
901
|
+
return;
|
|
902
|
+
targetArray.push({ type: 'break', metadata: { breakType } });
|
|
903
|
+
};
|
|
943
904
|
traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
|
|
944
905
|
if (node.tagName === "text:p") {
|
|
906
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
|
|
945
907
|
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
946
908
|
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
947
909
|
const metadata = {
|
|
@@ -971,9 +933,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
971
933
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
972
934
|
}
|
|
973
935
|
targetArray.push(pNode);
|
|
936
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
|
|
974
937
|
lastWasList = false;
|
|
975
938
|
}
|
|
976
939
|
else if (node.tagName === "text:h") {
|
|
940
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
|
|
977
941
|
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
978
942
|
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
979
943
|
const metadata = {
|
|
@@ -998,6 +962,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
998
962
|
hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
999
963
|
}
|
|
1000
964
|
targetArray.push(hNode);
|
|
965
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
|
|
1001
966
|
lastWasList = false;
|
|
1002
967
|
}
|
|
1003
968
|
else if (node.tagName === "table:table") {
|
|
@@ -1270,15 +1235,18 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1270
1235
|
const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
|
|
1271
1236
|
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
1272
1237
|
if (mathNode) {
|
|
1273
|
-
// Math formula object at block level
|
|
1274
|
-
|
|
1238
|
+
// Math formula object at block level - a display equation, so the
|
|
1239
|
+
// inner node is `math: 'block'` where the inline site above emits
|
|
1240
|
+
// `math: 'inline'`.
|
|
1241
|
+
const formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
|
|
1275
1242
|
const formulaNode = {
|
|
1276
1243
|
type: 'paragraph',
|
|
1277
1244
|
text: formulaText,
|
|
1278
1245
|
children: [
|
|
1279
1246
|
{
|
|
1280
|
-
type: '
|
|
1281
|
-
text: formulaText
|
|
1247
|
+
type: 'code',
|
|
1248
|
+
text: formulaText,
|
|
1249
|
+
metadata: { math: 'block' }
|
|
1282
1250
|
}
|
|
1283
1251
|
]
|
|
1284
1252
|
};
|
|
@@ -1358,7 +1326,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1358
1326
|
if (spans.length > 0) {
|
|
1359
1327
|
for (const span of spans) {
|
|
1360
1328
|
const styleName = span.getAttribute("text:style-name");
|
|
1361
|
-
|
|
1329
|
+
// Through `mergeFormatting` like every other span site, so
|
|
1330
|
+
// an explicit `false` is dropped rather than written onto
|
|
1331
|
+
// the node - and so the node gets its own object instead of
|
|
1332
|
+
// aliasing the shared style-table entry.
|
|
1333
|
+
const formatting = mergeFormatting({}, styleName ? styleMap[styleName] : undefined);
|
|
1362
1334
|
const text = span.textContent || '';
|
|
1363
1335
|
cellText += text;
|
|
1364
1336
|
const textNode = {
|
|
@@ -1423,22 +1395,22 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1423
1395
|
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
1424
1396
|
if (mathNode) {
|
|
1425
1397
|
isFormula = true;
|
|
1426
|
-
formulaText =
|
|
1398
|
+
formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
|
|
1427
1399
|
}
|
|
1428
1400
|
}
|
|
1429
1401
|
}
|
|
1430
1402
|
}
|
|
1431
1403
|
if (isFormula) {
|
|
1432
1404
|
cellText += formulaText;
|
|
1433
|
-
const
|
|
1434
|
-
type: '
|
|
1405
|
+
const formulaNode = {
|
|
1406
|
+
type: 'code',
|
|
1435
1407
|
text: formulaText,
|
|
1436
|
-
|
|
1408
|
+
metadata: { math: 'inline' }
|
|
1437
1409
|
};
|
|
1438
1410
|
if (config.includeRawContent) {
|
|
1439
|
-
|
|
1411
|
+
formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
|
|
1440
1412
|
}
|
|
1441
|
-
children.push(
|
|
1413
|
+
children.push(formulaNode);
|
|
1442
1414
|
}
|
|
1443
1415
|
else if (drawImages.length > 0) {
|
|
1444
1416
|
// logic for image node
|