officeparser 7.4.0 → 7.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +48 -4
- package/dist/generators/BaseGenerator.d.ts +15 -0
- package/dist/generators/BaseGenerator.js +31 -0
- package/dist/generators/HtmlGenerator.d.ts +9 -0
- package/dist/generators/HtmlGenerator.js +34 -4
- package/dist/generators/MarkdownGenerator.d.ts +13 -0
- package/dist/generators/MarkdownGenerator.js +115 -41
- package/dist/generators/RtfGenerator.d.ts +13 -0
- package/dist/generators/RtfGenerator.js +23 -2
- package/dist/officeparser.browser.iife.js +148 -148
- package/dist/officeparser.browser.mjs +186 -186
- package/dist/officeparser.browser.slim.iife.js +153 -153
- package/dist/officeparser.browser.slim.mjs +153 -153
- package/dist/parsers/HtmlParser.js +39 -0
- package/dist/parsers/OpenOfficeParser.js +122 -164
- package/dist/parsers/PowerPointParser.js +27 -6
- package/dist/parsers/WordParser.js +21 -0
- package/dist/sbom.cdx.json +92 -92
- package/dist/utils/mathUtils.d.ts +42 -0
- package/dist/utils/mathUtils.js +385 -0
- package/package.json +9 -5
|
@@ -4,6 +4,7 @@ exports.parseHtml = void 0;
|
|
|
4
4
|
const types_js_1 = require("../types.js");
|
|
5
5
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
6
6
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
7
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
7
8
|
const sanitize_js_1 = require("../utils/sanitize.js");
|
|
8
9
|
/**
|
|
9
10
|
* Maximum element nesting depth accepted from an HTML/XHTML source before the parser gives up
|
|
@@ -11,6 +12,25 @@ const sanitize_js_1 = require("../utils/sanitize.js");
|
|
|
11
12
|
* `parseNode` for why this value and not a larger one.
|
|
12
13
|
*/
|
|
13
14
|
const MAX_HTML_NESTING_DEPTH = 256;
|
|
15
|
+
/**
|
|
16
|
+
* Presents an `HtmlNode` as a `MathNode` for the shared MathML converter.
|
|
17
|
+
*
|
|
18
|
+
* The shapes already line up field for field; the one thing that must happen here is entity
|
|
19
|
+
* decoding, since this parser keeps text nodes in their raw escaped form and `<` inside an
|
|
20
|
+
* `<mo>` is a less-than operator, not markup.
|
|
21
|
+
*/
|
|
22
|
+
const toMathNode = (node) => ({
|
|
23
|
+
tagName: node.tagName,
|
|
24
|
+
attributes: node.attributes,
|
|
25
|
+
text: node.text === undefined ? undefined : node.text
|
|
26
|
+
.replace(/ /g, ' ')
|
|
27
|
+
.replace(/</g, '<')
|
|
28
|
+
.replace(/>/g, '>')
|
|
29
|
+
.replace(/&/g, '&')
|
|
30
|
+
.replace(/"/g, '"')
|
|
31
|
+
.replace(/'/g, "'"),
|
|
32
|
+
children: (node.children || []).map(toMathNode),
|
|
33
|
+
});
|
|
14
34
|
const parseAttributes = (attrString) => {
|
|
15
35
|
const attrs = {};
|
|
16
36
|
// Attribute names follow the HTML5 rule - any character except whitespace and
|
|
@@ -572,6 +592,25 @@ const parseHtml = async (buffer, config) => {
|
|
|
572
592
|
metadata: { math: mathMode }
|
|
573
593
|
};
|
|
574
594
|
}
|
|
595
|
+
// Native MathML. This is what a real-world page and every EPUB3 uses (EpubParser
|
|
596
|
+
// routes each spine item through here), as opposed to the `data-math` round-trip
|
|
597
|
+
// contract above, which only ever appears in this library's own HTML output. Without
|
|
598
|
+
// it, a `<math>` element fell through to the generic element handling below, which
|
|
599
|
+
// concatenates descendant text: `<mfrac><mn>1</mn><mn>2</mn></mfrac>` became "12".
|
|
600
|
+
if (tagName === 'math' || tagName.endsWith(':math')) {
|
|
601
|
+
// `display="block"` is MathML's own attribute for a display equation; the legacy
|
|
602
|
+
// `mode="display"` means the same thing and is still emitted by older producers.
|
|
603
|
+
const isBlock = node.attributes?.['display'] === 'block'
|
|
604
|
+
|| node.attributes?.['mode'] === 'display';
|
|
605
|
+
const latex = (0, mathUtils_js_1.mathmlTreeToLatex)(toMathNode(node));
|
|
606
|
+
if ((0, mathUtils_js_1.isEmptyMath)(latex))
|
|
607
|
+
return null;
|
|
608
|
+
return {
|
|
609
|
+
type: 'code',
|
|
610
|
+
text: latex,
|
|
611
|
+
metadata: { math: isBlock ? 'block' : 'inline' }
|
|
612
|
+
};
|
|
613
|
+
}
|
|
575
614
|
// Admonition: inscript-editor's Admonition node renders
|
|
576
615
|
// <div class="admonition admonition-note" data-type="note">…children…</div>.
|
|
577
616
|
if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
|
|
@@ -27,6 +27,7 @@ const types_js_1 = require("../types.js");
|
|
|
27
27
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
28
28
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
29
29
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
30
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
30
31
|
/**
|
|
31
32
|
* Tracks how many table cells a single document has been allowed to materialize.
|
|
32
33
|
*
|
|
@@ -84,10 +85,28 @@ class CellBudget {
|
|
|
84
85
|
/** Resolves the configured cell budget, falling back to the documented default. */
|
|
85
86
|
const createCellBudget = (config) => new CellBudget(config.decompressionLimits?.maxTableCells ?? 1000000, config);
|
|
86
87
|
/**
|
|
87
|
-
*
|
|
88
|
-
*
|
|
89
|
-
*
|
|
88
|
+
* Merges a style's formatting over what it inherits, dropping any flag the style explicitly turns
|
|
89
|
+
* off rather than carrying a `false` forward.
|
|
90
|
+
*
|
|
91
|
+
* Generators all test these flags for truthiness, so a retained `false` would render the same - but
|
|
92
|
+
* it would not *compare* the same, and `MarkdownGenerator.optimizeNodes` merges adjacent text nodes
|
|
93
|
+
* only when their formatting objects are equal. Leaving `bold: false` on one node and nothing on
|
|
94
|
+
* its neighbour would silently stop that merge and fragment the output. Same reasoning, and same
|
|
95
|
+
* shape, as `WordParser`'s direct-run-property merge.
|
|
90
96
|
*/
|
|
97
|
+
const mergeFormatting = (inherited, override) => {
|
|
98
|
+
if (!override)
|
|
99
|
+
return { ...inherited };
|
|
100
|
+
const merged = { ...inherited };
|
|
101
|
+
for (const key of Object.keys(override)) {
|
|
102
|
+
const value = override[key];
|
|
103
|
+
if (value === false)
|
|
104
|
+
delete merged[key];
|
|
105
|
+
else if (value !== undefined)
|
|
106
|
+
merged[key] = value;
|
|
107
|
+
}
|
|
108
|
+
return merged;
|
|
109
|
+
};
|
|
91
110
|
const toRepeatCount = (attr) => {
|
|
92
111
|
const n = parseInt(attr || "1");
|
|
93
112
|
return Number.isFinite(n) && n > 0 ? n : 1;
|
|
@@ -159,8 +178,8 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
159
178
|
let lastWasList = false;
|
|
160
179
|
let traverse;
|
|
161
180
|
// Helper to parse styles
|
|
162
|
-
const parseStyles = (
|
|
163
|
-
const styles = (0, xmlUtils_js_1.getElementsByTagName)(
|
|
181
|
+
const parseStyles = (scope) => {
|
|
182
|
+
const styles = (0, xmlUtils_js_1.getElementsByTagName)(scope, "style:style");
|
|
164
183
|
for (const style of styles) {
|
|
165
184
|
const name = style.getAttribute("style:name");
|
|
166
185
|
if (!name)
|
|
@@ -188,6 +207,17 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
188
207
|
if (dropCap) {
|
|
189
208
|
styleInfo.dropCap = true;
|
|
190
209
|
}
|
|
210
|
+
// Page/column breaks. ODF attaches these to the paragraph style rather than
|
|
211
|
+
// writing an inline element the way DOCX's `<w:br w:type="page"/>` does, which is
|
|
212
|
+
// why `includeBreakNodes` produced nothing at all for ODF: there was no inline
|
|
213
|
+
// element to find. Only the two break kinds that map onto a BreakMetadata type
|
|
214
|
+
// are carried; `auto` and `even-page`/`odd-page` have no equivalent.
|
|
215
|
+
const breakBefore = paraProps.getAttribute("fo:break-before");
|
|
216
|
+
if (breakBefore === 'page' || breakBefore === 'column')
|
|
217
|
+
styleInfo.breakBefore = breakBefore;
|
|
218
|
+
const breakAfter = paraProps.getAttribute("fo:break-after");
|
|
219
|
+
if (breakAfter === 'page' || breakAfter === 'column')
|
|
220
|
+
styleInfo.breakAfter = breakAfter;
|
|
191
221
|
}
|
|
192
222
|
if (Object.keys(styleInfo).length > 0) {
|
|
193
223
|
paragraphStyleMap[name] = styleInfo;
|
|
@@ -203,14 +233,28 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
203
233
|
formatting.backgroundColor = bgColor;
|
|
204
234
|
}
|
|
205
235
|
if (textProps) {
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
236
|
+
// Record the *off* states as an explicit `false`, not as an absent key.
|
|
237
|
+
//
|
|
238
|
+
// Now that a paragraph style's text properties are inherited by the runs inside it,
|
|
239
|
+
// a span has to be able to turn one back off: LibreOffice writes
|
|
240
|
+
// `fo:font-weight="normal"` on the span whenever a user un-bolds part of a
|
|
241
|
+
// bold-styled paragraph. With only the `true` side recorded, that span had nothing
|
|
242
|
+
// to override the inherited value with and came out bold - wrong in the opposite
|
|
243
|
+
// direction from the bug the inheritance fixed. `TextFormatting`'s flags are
|
|
244
|
+
// `boolean | undefined` precisely so "explicitly off" is expressible.
|
|
245
|
+
const fontWeight = textProps.getAttribute("fo:font-weight") || textProps.getAttribute("style:font-weight-asian");
|
|
246
|
+
// Numeric weights are the same axis: 600+ is bold, below that is not.
|
|
247
|
+
if (fontWeight)
|
|
248
|
+
formatting.bold = fontWeight === "bold" || /^[6-9]00$/.test(fontWeight);
|
|
249
|
+
const fontStyle = textProps.getAttribute("fo:font-style") || textProps.getAttribute("style:font-style-asian");
|
|
250
|
+
if (fontStyle)
|
|
251
|
+
formatting.italic = fontStyle === "italic" || fontStyle === "oblique";
|
|
252
|
+
const underline = textProps.getAttribute("style:text-underline-style");
|
|
253
|
+
if (underline)
|
|
254
|
+
formatting.underline = underline !== "none";
|
|
255
|
+
const lineThrough = textProps.getAttribute("style:text-line-through-style");
|
|
256
|
+
if (lineThrough)
|
|
257
|
+
formatting.strikethrough = lineThrough !== "none";
|
|
214
258
|
const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
|
|
215
259
|
if (size)
|
|
216
260
|
formatting.size = size;
|
|
@@ -247,62 +291,6 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
247
291
|
*
|
|
248
292
|
* @param node - The paragraph element to parse
|
|
249
293
|
*/
|
|
250
|
-
const parseMathML = (node) => {
|
|
251
|
-
if (!node)
|
|
252
|
-
return '';
|
|
253
|
-
if (node.nodeType === 3) { // Text node
|
|
254
|
-
return node.textContent || '';
|
|
255
|
-
}
|
|
256
|
-
if (node.nodeType !== 1) { // Not an element
|
|
257
|
-
return '';
|
|
258
|
-
}
|
|
259
|
-
const element = node;
|
|
260
|
-
const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
|
|
261
|
-
switch (tagName) {
|
|
262
|
-
case 'math':
|
|
263
|
-
case 'mrow':
|
|
264
|
-
case 'semantics':
|
|
265
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
266
|
-
case 'mfrac': {
|
|
267
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
268
|
-
if (children.length >= 2) {
|
|
269
|
-
return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
|
|
270
|
-
}
|
|
271
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
272
|
-
}
|
|
273
|
-
case 'msub': {
|
|
274
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
275
|
-
if (children.length >= 2) {
|
|
276
|
-
return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
|
|
277
|
-
}
|
|
278
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
279
|
-
}
|
|
280
|
-
case 'msup': {
|
|
281
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
282
|
-
if (children.length >= 2) {
|
|
283
|
-
return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
|
|
284
|
-
}
|
|
285
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
286
|
-
}
|
|
287
|
-
case 'msubsup': {
|
|
288
|
-
const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
|
|
289
|
-
if (children.length >= 3) {
|
|
290
|
-
return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
|
|
291
|
-
}
|
|
292
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
293
|
-
}
|
|
294
|
-
case 'mi':
|
|
295
|
-
case 'mn':
|
|
296
|
-
case 'mo':
|
|
297
|
-
case 'mtext':
|
|
298
|
-
case 'ms':
|
|
299
|
-
return element.textContent || '';
|
|
300
|
-
case 'annotation':
|
|
301
|
-
return '';
|
|
302
|
-
default:
|
|
303
|
-
return Array.from(element.childNodes).map(parseMathML).join('');
|
|
304
|
-
}
|
|
305
|
-
};
|
|
306
294
|
/**
|
|
307
295
|
* Helper to parse inline content (text, spans, links, notes, etc.) recursively.
|
|
308
296
|
*
|
|
@@ -365,6 +353,14 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
365
353
|
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
366
354
|
});
|
|
367
355
|
}
|
|
356
|
+
else if (tagName === 'text:soft-page-break') {
|
|
357
|
+
// The page boundary the editor recorded at its last save. DOCX's equivalent
|
|
358
|
+
// is `w:lastRenderedPageBreak`, so it maps onto the same break type rather
|
|
359
|
+
// than onto 'page', which is reserved for a break the author asked for.
|
|
360
|
+
if (config.includeBreakNodes) {
|
|
361
|
+
children.push({ type: 'break', metadata: { breakType: 'lastRenderedPage' } });
|
|
362
|
+
}
|
|
363
|
+
}
|
|
368
364
|
else if (tagName === 'text:line-break') {
|
|
369
365
|
// Line break
|
|
370
366
|
fullText += '\n';
|
|
@@ -378,7 +374,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
378
374
|
else if (tagName === 'text:span') {
|
|
379
375
|
// Formatted text span
|
|
380
376
|
const styleName = element.getAttribute("text:style-name");
|
|
381
|
-
const formatting = styleName ?
|
|
377
|
+
const formatting = styleName ? mergeFormatting(parentFormatting, styleMap[styleName]) : parentFormatting;
|
|
382
378
|
const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
|
|
383
379
|
fullText += spanContent.text;
|
|
384
380
|
children.push(...spanContent.children);
|
|
@@ -483,22 +479,25 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
483
479
|
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
484
480
|
if (mathNode) {
|
|
485
481
|
isFormula = true;
|
|
486
|
-
formulaText =
|
|
482
|
+
formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
|
|
487
483
|
}
|
|
488
484
|
}
|
|
489
485
|
}
|
|
490
486
|
if (isFormula) {
|
|
491
487
|
fullText += formulaText;
|
|
492
|
-
|
|
493
|
-
|
|
488
|
+
// A `code` node carrying `math`, not a plain `text` node: the formula
|
|
489
|
+
// is LaTeX, and marking it as such is what lets generators render it
|
|
490
|
+
// as maths rather than emit it as prose that happens to contain
|
|
491
|
+
// backslashes. Same node shape DOCX, PPTX, HTML and Markdown produce.
|
|
492
|
+
const formulaNode = {
|
|
493
|
+
type: 'code',
|
|
494
494
|
text: formulaText,
|
|
495
|
-
|
|
496
|
-
metadata: linkMetadata ? { ...linkMetadata } : undefined
|
|
495
|
+
metadata: { math: 'inline', ...(linkMetadata ?? {}) }
|
|
497
496
|
};
|
|
498
497
|
if (config.includeRawContent) {
|
|
499
|
-
|
|
498
|
+
formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
|
|
500
499
|
}
|
|
501
|
-
children.push(
|
|
500
|
+
children.push(formulaNode);
|
|
502
501
|
}
|
|
503
502
|
else {
|
|
504
503
|
// Standard inline image extraction fallback if object is not a formula
|
|
@@ -588,8 +587,9 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
588
587
|
const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
|
|
589
588
|
const alignment = styleInfo?.alignment;
|
|
590
589
|
const dropCap = styleInfo?.dropCap;
|
|
590
|
+
const formatting = mergeFormatting({}, paraStyle ? styleMap[paraStyle] : undefined);
|
|
591
591
|
// Parse content recursively using the new helper
|
|
592
|
-
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap,
|
|
592
|
+
const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, formatting, undefined, sourceXml);
|
|
593
593
|
// Add style name to metadata of children if they don't have one
|
|
594
594
|
if (paraStyle) {
|
|
595
595
|
content.children.forEach(child => {
|
|
@@ -800,7 +800,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
800
800
|
// whole cell array, so rows x cols is what actually exhausts memory; charge those
|
|
801
801
|
// copies against the same budget.
|
|
802
802
|
const allowedRows = cells.length === 0
|
|
803
|
-
? rowsRepeated
|
|
803
|
+
? (rowsRepeated > 0 ? 1 + cellBudget.take(rowsRepeated - 1) : 0)
|
|
804
804
|
: Math.min(rowsRepeated, 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
|
|
805
805
|
for (let k = 0; k < allowedRows; k++) {
|
|
806
806
|
if ((k & 255) === 0)
|
|
@@ -839,82 +839,12 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
839
839
|
// splitting a huge repeat expansion across many small tables. `traverse` and the
|
|
840
840
|
// spreadsheet branch below both close over this; `parseTable` receives it explicitly.
|
|
841
841
|
const cellBudget = createCellBudget(config);
|
|
842
|
-
//
|
|
842
|
+
// Automatic styles are local to content.xml, but their definitions have exactly the
|
|
843
|
+
// shape styles.xml uses, so they go through the same reader rather than a second copy of
|
|
844
|
+
// it - the copy is how `fo:break-before` came to be read in neither place.
|
|
843
845
|
const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
|
|
844
846
|
if (automaticStyles) {
|
|
845
|
-
|
|
846
|
-
for (const style of styles) {
|
|
847
|
-
const name = style.getAttribute("style:name");
|
|
848
|
-
if (!name)
|
|
849
|
-
continue;
|
|
850
|
-
// Parse paragraph properties for alignment
|
|
851
|
-
const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
|
|
852
|
-
const styleInfo = {};
|
|
853
|
-
if (paraProps) {
|
|
854
|
-
const textAlign = paraProps.getAttribute("fo:text-align");
|
|
855
|
-
if (textAlign) {
|
|
856
|
-
const alignMap = {
|
|
857
|
-
'start': 'left',
|
|
858
|
-
'left': 'left',
|
|
859
|
-
'center': 'center',
|
|
860
|
-
'end': 'right',
|
|
861
|
-
'right': 'right',
|
|
862
|
-
'justify': 'justify'
|
|
863
|
-
};
|
|
864
|
-
if (alignMap[textAlign]) {
|
|
865
|
-
styleInfo.alignment = alignMap[textAlign];
|
|
866
|
-
}
|
|
867
|
-
}
|
|
868
|
-
const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
|
|
869
|
-
if (dropCap)
|
|
870
|
-
styleInfo.dropCap = true;
|
|
871
|
-
}
|
|
872
|
-
if (Object.keys(styleInfo).length > 0) {
|
|
873
|
-
paragraphStyleMap[name] = styleInfo;
|
|
874
|
-
}
|
|
875
|
-
const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
|
|
876
|
-
const formatting = {};
|
|
877
|
-
if (cellProps) {
|
|
878
|
-
const bgColor = cellProps.getAttribute("fo:background-color");
|
|
879
|
-
if (bgColor && bgColor !== 'transparent')
|
|
880
|
-
formatting.backgroundColor = bgColor;
|
|
881
|
-
}
|
|
882
|
-
const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
|
|
883
|
-
if (textProps) {
|
|
884
|
-
if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
|
|
885
|
-
formatting.bold = true;
|
|
886
|
-
if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
|
|
887
|
-
formatting.italic = true;
|
|
888
|
-
if (textProps.getAttribute("style:text-underline-style") === "solid")
|
|
889
|
-
formatting.underline = true;
|
|
890
|
-
if (textProps.getAttribute("style:text-line-through-style") === "solid")
|
|
891
|
-
formatting.strikethrough = true;
|
|
892
|
-
const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
|
|
893
|
-
if (size)
|
|
894
|
-
formatting.size = size;
|
|
895
|
-
const color = textProps.getAttribute("fo:color");
|
|
896
|
-
if (color)
|
|
897
|
-
formatting.color = color;
|
|
898
|
-
// Background color
|
|
899
|
-
const bgColor = textProps.getAttribute("fo:background-color");
|
|
900
|
-
if (bgColor && bgColor !== 'transparent')
|
|
901
|
-
formatting.backgroundColor = bgColor;
|
|
902
|
-
// Font family
|
|
903
|
-
const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
|
|
904
|
-
if (fontName)
|
|
905
|
-
formatting.font = fontName;
|
|
906
|
-
// Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
|
|
907
|
-
const textPosition = textProps.getAttribute("style:text-position");
|
|
908
|
-
if (textPosition) {
|
|
909
|
-
if (textPosition.startsWith("sub"))
|
|
910
|
-
formatting.subscript = true;
|
|
911
|
-
if (textPosition.startsWith("super"))
|
|
912
|
-
formatting.superscript = true;
|
|
913
|
-
}
|
|
914
|
-
}
|
|
915
|
-
if (Object.keys(formatting).length > 0)
|
|
916
|
-
styleMap[name] = formatting;
|
|
917
|
-
}
|
|
847
|
+
parseStyles(automaticStyles);
|
|
918
848
|
}
|
|
919
849
|
// Start traversal
|
|
920
850
|
const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
|
|
@@ -940,8 +870,26 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
940
870
|
* @param sourceXml - The source XML string for raw content extraction
|
|
941
871
|
* @param asSheet - If true, treats tables as sheets (for ODS)
|
|
942
872
|
*/
|
|
873
|
+
/**
|
|
874
|
+
* Emits the break a paragraph style asks for, on the given side of that paragraph.
|
|
875
|
+
*
|
|
876
|
+
* ODF has no inline break element for these - `fo:break-before="page"` sits on the style,
|
|
877
|
+
* so the break is a property of the paragraph rather than a run inside it. That makes it a
|
|
878
|
+
* sibling emitted around the paragraph node, not a child of it, which is the one structural
|
|
879
|
+
* difference from how DOCX's `<w:br w:type="page"/>` lands.
|
|
880
|
+
*/
|
|
881
|
+
const pushStyleBreak = (styleName, targetArray, edge) => {
|
|
882
|
+
if (!config.includeBreakNodes || !styleName)
|
|
883
|
+
return;
|
|
884
|
+
const info = paragraphStyleMap[styleName];
|
|
885
|
+
const breakType = edge === 'before' ? info?.breakBefore : info?.breakAfter;
|
|
886
|
+
if (!breakType)
|
|
887
|
+
return;
|
|
888
|
+
targetArray.push({ type: 'break', metadata: { breakType } });
|
|
889
|
+
};
|
|
943
890
|
traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
|
|
944
891
|
if (node.tagName === "text:p") {
|
|
892
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
|
|
945
893
|
const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
946
894
|
const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
|
|
947
895
|
const metadata = {
|
|
@@ -971,9 +919,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
971
919
|
pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
972
920
|
}
|
|
973
921
|
targetArray.push(pNode);
|
|
922
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
|
|
974
923
|
lastWasList = false;
|
|
975
924
|
}
|
|
976
925
|
else if (node.tagName === "text:h") {
|
|
926
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
|
|
977
927
|
const level = parseInt(node.getAttribute("text:outline-level") || "1");
|
|
978
928
|
const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
|
|
979
929
|
const metadata = {
|
|
@@ -998,6 +948,7 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
998
948
|
hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
|
|
999
949
|
}
|
|
1000
950
|
targetArray.push(hNode);
|
|
951
|
+
pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
|
|
1001
952
|
lastWasList = false;
|
|
1002
953
|
}
|
|
1003
954
|
else if (node.tagName === "table:table") {
|
|
@@ -1270,15 +1221,18 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1270
1221
|
const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
|
|
1271
1222
|
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
1272
1223
|
if (mathNode) {
|
|
1273
|
-
// Math formula object at block level
|
|
1274
|
-
|
|
1224
|
+
// Math formula object at block level - a display equation, so the
|
|
1225
|
+
// inner node is `math: 'block'` where the inline site above emits
|
|
1226
|
+
// `math: 'inline'`.
|
|
1227
|
+
const formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
|
|
1275
1228
|
const formulaNode = {
|
|
1276
1229
|
type: 'paragraph',
|
|
1277
1230
|
text: formulaText,
|
|
1278
1231
|
children: [
|
|
1279
1232
|
{
|
|
1280
|
-
type: '
|
|
1281
|
-
text: formulaText
|
|
1233
|
+
type: 'code',
|
|
1234
|
+
text: formulaText,
|
|
1235
|
+
metadata: { math: 'block' }
|
|
1282
1236
|
}
|
|
1283
1237
|
]
|
|
1284
1238
|
};
|
|
@@ -1358,7 +1312,11 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1358
1312
|
if (spans.length > 0) {
|
|
1359
1313
|
for (const span of spans) {
|
|
1360
1314
|
const styleName = span.getAttribute("text:style-name");
|
|
1361
|
-
|
|
1315
|
+
// Through `mergeFormatting` like every other span site, so
|
|
1316
|
+
// an explicit `false` is dropped rather than written onto
|
|
1317
|
+
// the node - and so the node gets its own object instead of
|
|
1318
|
+
// aliasing the shared style-table entry.
|
|
1319
|
+
const formatting = mergeFormatting({}, styleName ? styleMap[styleName] : undefined);
|
|
1362
1320
|
const text = span.textContent || '';
|
|
1363
1321
|
cellText += text;
|
|
1364
1322
|
const textNode = {
|
|
@@ -1423,22 +1381,22 @@ const parseOpenOffice = async (buffer, config) => {
|
|
|
1423
1381
|
const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
|
|
1424
1382
|
if (mathNode) {
|
|
1425
1383
|
isFormula = true;
|
|
1426
|
-
formulaText =
|
|
1384
|
+
formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
|
|
1427
1385
|
}
|
|
1428
1386
|
}
|
|
1429
1387
|
}
|
|
1430
1388
|
}
|
|
1431
1389
|
if (isFormula) {
|
|
1432
1390
|
cellText += formulaText;
|
|
1433
|
-
const
|
|
1434
|
-
type: '
|
|
1391
|
+
const formulaNode = {
|
|
1392
|
+
type: 'code',
|
|
1435
1393
|
text: formulaText,
|
|
1436
|
-
|
|
1394
|
+
metadata: { math: 'inline' }
|
|
1437
1395
|
};
|
|
1438
1396
|
if (config.includeRawContent) {
|
|
1439
|
-
|
|
1397
|
+
formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
|
|
1440
1398
|
}
|
|
1441
|
-
children.push(
|
|
1399
|
+
children.push(formulaNode);
|
|
1442
1400
|
}
|
|
1443
1401
|
else if (drawImages.length > 0) {
|
|
1444
1402
|
// logic for image node
|
|
@@ -29,6 +29,7 @@ const astUtils_js_1 = require("../utils/astUtils.js");
|
|
|
29
29
|
const chartUtils_js_1 = require("../utils/chartUtils.js");
|
|
30
30
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
31
31
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
32
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
32
33
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
33
34
|
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
34
35
|
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
@@ -576,6 +577,30 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
576
577
|
activeNode.children?.push({ type: 'text', text: "\n" });
|
|
577
578
|
}
|
|
578
579
|
}
|
|
580
|
+
else {
|
|
581
|
+
// Equations. This loop dispatches on `a:r`/`a:fld`, so an `m:oMath` -
|
|
582
|
+
// which is a sibling of the runs, not one of them - was never visited at
|
|
583
|
+
// all and the formula vanished from the slide without a warning.
|
|
584
|
+
//
|
|
585
|
+
// PowerPoint writes the equation either directly in the paragraph or
|
|
586
|
+
// wrapped in `mc:AlternateContent`/`a14:m` for pre-2010 readers, so take
|
|
587
|
+
// the element itself when it is the equation and search inside it
|
|
588
|
+
// otherwise. `getElementsByTagName` returns document order, which is the
|
|
589
|
+
// order the equations are read in.
|
|
590
|
+
const isMath = tag === "m:oMath" || tag === "m:oMathPara";
|
|
591
|
+
const equations = isMath ? [element] : (0, xmlUtils_js_1.getElementsByTagName)(element, "m:oMath");
|
|
592
|
+
for (const equation of equations) {
|
|
593
|
+
const latex = (0, mathUtils_js_1.ommlToLatex)(equation);
|
|
594
|
+
if ((0, mathUtils_js_1.isEmptyMath)(latex))
|
|
595
|
+
continue;
|
|
596
|
+
activeNode.text += latex;
|
|
597
|
+
activeNode.children?.push({
|
|
598
|
+
type: 'code',
|
|
599
|
+
text: latex,
|
|
600
|
+
metadata: { math: tag === "m:oMathPara" ? 'block' : 'inline' }
|
|
601
|
+
});
|
|
602
|
+
}
|
|
603
|
+
}
|
|
579
604
|
}
|
|
580
605
|
}
|
|
581
606
|
}
|
|
@@ -623,12 +648,8 @@ const parsePowerPoint = async (buffer, config) => {
|
|
|
623
648
|
}
|
|
624
649
|
// Case 4: Grouped shape (recursive!)
|
|
625
650
|
else if (tag === "p:grpSp") {
|
|
626
|
-
//
|
|
627
|
-
|
|
628
|
-
// Recurse into the nested tree
|
|
629
|
-
if (nestedTree) {
|
|
630
|
-
nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
|
|
631
|
-
}
|
|
651
|
+
// Recurse into the group element itself which holds the child shapes
|
|
652
|
+
nodes.push(...traverseSpTree(element, slideNumber, xmlContentString));
|
|
632
653
|
}
|
|
633
654
|
}
|
|
634
655
|
return nodes;
|
|
@@ -66,6 +66,7 @@ const types_js_1 = require("../types.js");
|
|
|
66
66
|
const astUtils_js_1 = require("../utils/astUtils.js");
|
|
67
67
|
const errorUtils_js_1 = require("../utils/errorUtils.js");
|
|
68
68
|
const imageUtils_js_1 = require("../utils/imageUtils.js");
|
|
69
|
+
const mathUtils_js_1 = require("../utils/mathUtils.js");
|
|
69
70
|
const ocrUtils_js_1 = require("../utils/ocrUtils.js");
|
|
70
71
|
const xmlUtils_js_1 = require("../utils/xmlUtils.js");
|
|
71
72
|
const zipUtils_js_1 = require("../utils/zipUtils.js");
|
|
@@ -751,6 +752,26 @@ const parseWord = async (buffer, config) => {
|
|
|
751
752
|
}
|
|
752
753
|
}
|
|
753
754
|
}
|
|
755
|
+
else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'm:oMath' || node.nodeName === 'oMath'
|
|
756
|
+
|| node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara')) {
|
|
757
|
+
// Equations. Without this branch they reach the generic fallback below, which
|
|
758
|
+
// recurses into every child and concatenates the `m:t` runs with no separators -
|
|
759
|
+
// so `<m:num>1</m:num><m:den>2</m:den>` came out as "12". That is worse than
|
|
760
|
+
// dropping the formula: the result still reads as a number, so nothing downstream
|
|
761
|
+
// can tell it is wrong.
|
|
762
|
+
//
|
|
763
|
+
// `m:oMathPara` is a display equation on its own line; a bare `m:oMath` is inline.
|
|
764
|
+
const isBlock = node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara';
|
|
765
|
+
const latex = (0, mathUtils_js_1.ommlToLatex)(node);
|
|
766
|
+
if (!(0, mathUtils_js_1.isEmptyMath)(latex)) {
|
|
767
|
+
text += latex;
|
|
768
|
+
children.push({
|
|
769
|
+
type: 'code',
|
|
770
|
+
text: latex,
|
|
771
|
+
metadata: { math: isBlock ? 'block' : 'inline' }
|
|
772
|
+
});
|
|
773
|
+
}
|
|
774
|
+
}
|
|
754
775
|
else if (node.childNodes.length > 0) {
|
|
755
776
|
// Generic fallback for unknown elements that might contain content
|
|
756
777
|
for (const child of Array.from(node.childNodes))
|