officeparser 7.4.0 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -4,6 +4,7 @@ exports.parseHtml = void 0;
4
4
  const types_js_1 = require("../types.js");
5
5
  const astUtils_js_1 = require("../utils/astUtils.js");
6
6
  const errorUtils_js_1 = require("../utils/errorUtils.js");
7
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
7
8
  const sanitize_js_1 = require("../utils/sanitize.js");
8
9
  /**
9
10
  * Maximum element nesting depth accepted from an HTML/XHTML source before the parser gives up
@@ -11,6 +12,25 @@ const sanitize_js_1 = require("../utils/sanitize.js");
11
12
  * `parseNode` for why this value and not a larger one.
12
13
  */
13
14
  const MAX_HTML_NESTING_DEPTH = 256;
15
+ /**
16
+ * Presents an `HtmlNode` as a `MathNode` for the shared MathML converter.
17
+ *
18
+ * The shapes already line up field for field; the one thing that must happen here is entity
19
+ * decoding, since this parser keeps text nodes in their raw escaped form and `<` inside an
20
+ * `<mo>` is a less-than operator, not markup.
21
+ */
22
+ const toMathNode = (node) => ({
23
+ tagName: node.tagName,
24
+ attributes: node.attributes,
25
+ text: node.text === undefined ? undefined : node.text
26
+ .replace(/&nbsp;/g, ' ')
27
+ .replace(/&lt;/g, '<')
28
+ .replace(/&gt;/g, '>')
29
+ .replace(/&amp;/g, '&')
30
+ .replace(/&quot;/g, '"')
31
+ .replace(/&#39;/g, "'"),
32
+ children: (node.children || []).map(toMathNode),
33
+ });
14
34
  const parseAttributes = (attrString) => {
15
35
  const attrs = {};
16
36
  // Attribute names follow the HTML5 rule - any character except whitespace and
@@ -572,6 +592,25 @@ const parseHtml = async (buffer, config) => {
572
592
  metadata: { math: mathMode }
573
593
  };
574
594
  }
595
+ // Native MathML. This is what a real-world page and every EPUB3 uses (EpubParser
596
+ // routes each spine item through here), as opposed to the `data-math` round-trip
597
+ // contract above, which only ever appears in this library's own HTML output. Without
598
+ // it, a `<math>` element fell through to the generic element handling below, which
599
+ // concatenates descendant text: `<mfrac><mn>1</mn><mn>2</mn></mfrac>` became "12".
600
+ if (tagName === 'math' || tagName.endsWith(':math')) {
601
+ // `display="block"` is MathML's own attribute for a display equation; the legacy
602
+ // `mode="display"` means the same thing and is still emitted by older producers.
603
+ const isBlock = node.attributes?.['display'] === 'block'
604
+ || node.attributes?.['mode'] === 'display';
605
+ const latex = (0, mathUtils_js_1.mathmlTreeToLatex)(toMathNode(node));
606
+ if ((0, mathUtils_js_1.isEmptyMath)(latex))
607
+ return null;
608
+ return {
609
+ type: 'code',
610
+ text: latex,
611
+ metadata: { math: isBlock ? 'block' : 'inline' }
612
+ };
613
+ }
575
614
  // Admonition: inscript-editor's Admonition node renders
576
615
  // <div class="admonition admonition-note" data-type="note">…children…</div>.
577
616
  if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
@@ -27,6 +27,7 @@ const types_js_1 = require("../types.js");
27
27
  const astUtils_js_1 = require("../utils/astUtils.js");
28
28
  const chartUtils_js_1 = require("../utils/chartUtils.js");
29
29
  const errorUtils_js_1 = require("../utils/errorUtils.js");
30
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
30
31
  /**
31
32
  * Tracks how many table cells a single document has been allowed to materialize.
32
33
  *
@@ -84,10 +85,28 @@ class CellBudget {
84
85
  /** Resolves the configured cell budget, falling back to the documented default. */
85
86
  const createCellBudget = (config) => new CellBudget(config.decompressionLimits?.maxTableCells ?? 1000000, config);
86
87
  /**
87
- * Coerces a `table:number-*-repeated` attribute to a usable repeat count. A missing, zero,
88
- * negative or non-numeric value becomes 1 (the element renders once), so a garbage attribute
89
- * can neither drop a cell nor poison arithmetic downstream (`colIndex += NaN`).
88
+ * Merges a style's formatting over what it inherits, dropping any flag the style explicitly turns
89
+ * off rather than carrying a `false` forward.
90
+ *
91
+ * Generators all test these flags for truthiness, so a retained `false` would render the same - but
92
+ * it would not *compare* the same, and `MarkdownGenerator.optimizeNodes` merges adjacent text nodes
93
+ * only when their formatting objects are equal. Leaving `bold: false` on one node and nothing on
94
+ * its neighbour would silently stop that merge and fragment the output. Same reasoning, and same
95
+ * shape, as `WordParser`'s direct-run-property merge.
90
96
  */
97
+ const mergeFormatting = (inherited, override) => {
98
+ if (!override)
99
+ return { ...inherited };
100
+ const merged = { ...inherited };
101
+ for (const key of Object.keys(override)) {
102
+ const value = override[key];
103
+ if (value === false)
104
+ delete merged[key];
105
+ else if (value !== undefined)
106
+ merged[key] = value;
107
+ }
108
+ return merged;
109
+ };
91
110
  const toRepeatCount = (attr) => {
92
111
  const n = parseInt(attr || "1");
93
112
  return Number.isFinite(n) && n > 0 ? n : 1;
@@ -159,8 +178,8 @@ const parseOpenOffice = async (buffer, config) => {
159
178
  let lastWasList = false;
160
179
  let traverse;
161
180
  // Helper to parse styles
162
- const parseStyles = (xml) => {
163
- const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
181
+ const parseStyles = (scope) => {
182
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(scope, "style:style");
164
183
  for (const style of styles) {
165
184
  const name = style.getAttribute("style:name");
166
185
  if (!name)
@@ -188,6 +207,17 @@ const parseOpenOffice = async (buffer, config) => {
188
207
  if (dropCap) {
189
208
  styleInfo.dropCap = true;
190
209
  }
210
+ // Page/column breaks. ODF attaches these to the paragraph style rather than
211
+ // writing an inline element the way DOCX's `<w:br w:type="page"/>` does, which is
212
+ // why `includeBreakNodes` produced nothing at all for ODF: there was no inline
213
+ // element to find. Only the two break kinds that map onto a BreakMetadata type
214
+ // are carried; `auto` and `even-page`/`odd-page` have no equivalent.
215
+ const breakBefore = paraProps.getAttribute("fo:break-before");
216
+ if (breakBefore === 'page' || breakBefore === 'column')
217
+ styleInfo.breakBefore = breakBefore;
218
+ const breakAfter = paraProps.getAttribute("fo:break-after");
219
+ if (breakAfter === 'page' || breakAfter === 'column')
220
+ styleInfo.breakAfter = breakAfter;
191
221
  }
192
222
  if (Object.keys(styleInfo).length > 0) {
193
223
  paragraphStyleMap[name] = styleInfo;
@@ -203,14 +233,28 @@ const parseOpenOffice = async (buffer, config) => {
203
233
  formatting.backgroundColor = bgColor;
204
234
  }
205
235
  if (textProps) {
206
- if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
207
- formatting.bold = true;
208
- if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
209
- formatting.italic = true;
210
- if (textProps.getAttribute("style:text-underline-style") === "solid")
211
- formatting.underline = true;
212
- if (textProps.getAttribute("style:text-line-through-style") === "solid")
213
- formatting.strikethrough = true;
236
+ // Record the *off* states as an explicit `false`, not as an absent key.
237
+ //
238
+ // Now that a paragraph style's text properties are inherited by the runs inside it,
239
+ // a span has to be able to turn one back off: LibreOffice writes
240
+ // `fo:font-weight="normal"` on the span whenever a user un-bolds part of a
241
+ // bold-styled paragraph. With only the `true` side recorded, that span had nothing
242
+ // to override the inherited value with and came out bold - wrong in the opposite
243
+ // direction from the bug the inheritance fixed. `TextFormatting`'s flags are
244
+ // `boolean | undefined` precisely so "explicitly off" is expressible.
245
+ const fontWeight = textProps.getAttribute("fo:font-weight") || textProps.getAttribute("style:font-weight-asian");
246
+ // Numeric weights are the same axis: 600+ is bold, below that is not.
247
+ if (fontWeight)
248
+ formatting.bold = fontWeight === "bold" || /^[6-9]00$/.test(fontWeight);
249
+ const fontStyle = textProps.getAttribute("fo:font-style") || textProps.getAttribute("style:font-style-asian");
250
+ if (fontStyle)
251
+ formatting.italic = fontStyle === "italic" || fontStyle === "oblique";
252
+ const underline = textProps.getAttribute("style:text-underline-style");
253
+ if (underline)
254
+ formatting.underline = underline !== "none";
255
+ const lineThrough = textProps.getAttribute("style:text-line-through-style");
256
+ if (lineThrough)
257
+ formatting.strikethrough = lineThrough !== "none";
214
258
  const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
215
259
  if (size)
216
260
  formatting.size = size;
@@ -247,62 +291,6 @@ const parseOpenOffice = async (buffer, config) => {
247
291
  *
248
292
  * @param node - The paragraph element to parse
249
293
  */
250
- const parseMathML = (node) => {
251
- if (!node)
252
- return '';
253
- if (node.nodeType === 3) { // Text node
254
- return node.textContent || '';
255
- }
256
- if (node.nodeType !== 1) { // Not an element
257
- return '';
258
- }
259
- const element = node;
260
- const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
261
- switch (tagName) {
262
- case 'math':
263
- case 'mrow':
264
- case 'semantics':
265
- return Array.from(element.childNodes).map(parseMathML).join('');
266
- case 'mfrac': {
267
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
268
- if (children.length >= 2) {
269
- return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
270
- }
271
- return Array.from(element.childNodes).map(parseMathML).join('');
272
- }
273
- case 'msub': {
274
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
275
- if (children.length >= 2) {
276
- return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
277
- }
278
- return Array.from(element.childNodes).map(parseMathML).join('');
279
- }
280
- case 'msup': {
281
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
282
- if (children.length >= 2) {
283
- return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
284
- }
285
- return Array.from(element.childNodes).map(parseMathML).join('');
286
- }
287
- case 'msubsup': {
288
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
289
- if (children.length >= 3) {
290
- return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
291
- }
292
- return Array.from(element.childNodes).map(parseMathML).join('');
293
- }
294
- case 'mi':
295
- case 'mn':
296
- case 'mo':
297
- case 'mtext':
298
- case 'ms':
299
- return element.textContent || '';
300
- case 'annotation':
301
- return '';
302
- default:
303
- return Array.from(element.childNodes).map(parseMathML).join('');
304
- }
305
- };
306
294
  /**
307
295
  * Helper to parse inline content (text, spans, links, notes, etc.) recursively.
308
296
  *
@@ -365,6 +353,14 @@ const parseOpenOffice = async (buffer, config) => {
365
353
  metadata: linkMetadata ? { ...linkMetadata } : undefined
366
354
  });
367
355
  }
356
+ else if (tagName === 'text:soft-page-break') {
357
+ // The page boundary the editor recorded at its last save. DOCX's equivalent
358
+ // is `w:lastRenderedPageBreak`, so it maps onto the same break type rather
359
+ // than onto 'page', which is reserved for a break the author asked for.
360
+ if (config.includeBreakNodes) {
361
+ children.push({ type: 'break', metadata: { breakType: 'lastRenderedPage' } });
362
+ }
363
+ }
368
364
  else if (tagName === 'text:line-break') {
369
365
  // Line break
370
366
  fullText += '\n';
@@ -378,7 +374,7 @@ const parseOpenOffice = async (buffer, config) => {
378
374
  else if (tagName === 'text:span') {
379
375
  // Formatted text span
380
376
  const styleName = element.getAttribute("text:style-name");
381
- const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
377
+ const formatting = styleName ? mergeFormatting(parentFormatting, styleMap[styleName]) : parentFormatting;
382
378
  const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
383
379
  fullText += spanContent.text;
384
380
  children.push(...spanContent.children);
@@ -483,22 +479,25 @@ const parseOpenOffice = async (buffer, config) => {
483
479
  const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
484
480
  if (mathNode) {
485
481
  isFormula = true;
486
- formulaText = parseMathML(mathNode).trim();
482
+ formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
487
483
  }
488
484
  }
489
485
  }
490
486
  if (isFormula) {
491
487
  fullText += formulaText;
492
- const textNode = {
493
- type: 'text',
488
+ // A `code` node carrying `math`, not a plain `text` node: the formula
489
+ // is LaTeX, and marking it as such is what lets generators render it
490
+ // as maths rather than emit it as prose that happens to contain
491
+ // backslashes. Same node shape DOCX, PPTX, HTML and Markdown produce.
492
+ const formulaNode = {
493
+ type: 'code',
494
494
  text: formulaText,
495
- formatting: parentFormatting,
496
- metadata: linkMetadata ? { ...linkMetadata } : undefined
495
+ metadata: { math: 'inline', ...(linkMetadata ?? {}) }
497
496
  };
498
497
  if (config.includeRawContent) {
499
- textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
498
+ formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
500
499
  }
501
- children.push(textNode);
500
+ children.push(formulaNode);
502
501
  }
503
502
  else {
504
503
  // Standard inline image extraction fallback if object is not a formula
@@ -588,8 +587,9 @@ const parseOpenOffice = async (buffer, config) => {
588
587
  const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
589
588
  const alignment = styleInfo?.alignment;
590
589
  const dropCap = styleInfo?.dropCap;
590
+ const formatting = mergeFormatting({}, paraStyle ? styleMap[paraStyle] : undefined);
591
591
  // Parse content recursively using the new helper
592
- const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, {}, undefined, sourceXml);
592
+ const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, formatting, undefined, sourceXml);
593
593
  // Add style name to metadata of children if they don't have one
594
594
  if (paraStyle) {
595
595
  content.children.forEach(child => {
@@ -800,7 +800,7 @@ const parseOpenOffice = async (buffer, config) => {
800
800
  // whole cell array, so rows x cols is what actually exhausts memory; charge those
801
801
  // copies against the same budget.
802
802
  const allowedRows = cells.length === 0
803
- ? rowsRepeated
803
+ ? (rowsRepeated > 0 ? 1 + cellBudget.take(rowsRepeated - 1) : 0)
804
804
  : Math.min(rowsRepeated, 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
805
805
  for (let k = 0; k < allowedRows; k++) {
806
806
  if ((k & 255) === 0)
@@ -839,82 +839,12 @@ const parseOpenOffice = async (buffer, config) => {
839
839
  // splitting a huge repeat expansion across many small tables. `traverse` and the
840
840
  // spreadsheet branch below both close over this; `parseTable` receives it explicitly.
841
841
  const cellBudget = createCellBudget(config);
842
- // Parse automatic styles (local to content.xml)
842
+ // Automatic styles are local to content.xml, but their definitions have exactly the
843
+ // shape styles.xml uses, so they go through the same reader rather than a second copy of
844
+ // it - the copy is how `fo:break-before` came to be read in neither place.
843
845
  const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
844
846
  if (automaticStyles) {
845
- const styles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "style:style");
846
- for (const style of styles) {
847
- const name = style.getAttribute("style:name");
848
- if (!name)
849
- continue;
850
- // Parse paragraph properties for alignment
851
- const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
852
- const styleInfo = {};
853
- if (paraProps) {
854
- const textAlign = paraProps.getAttribute("fo:text-align");
855
- if (textAlign) {
856
- const alignMap = {
857
- 'start': 'left',
858
- 'left': 'left',
859
- 'center': 'center',
860
- 'end': 'right',
861
- 'right': 'right',
862
- 'justify': 'justify'
863
- };
864
- if (alignMap[textAlign]) {
865
- styleInfo.alignment = alignMap[textAlign];
866
- }
867
- }
868
- const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
869
- if (dropCap)
870
- styleInfo.dropCap = true;
871
- }
872
- if (Object.keys(styleInfo).length > 0) {
873
- paragraphStyleMap[name] = styleInfo;
874
- }
875
- const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
876
- const formatting = {};
877
- if (cellProps) {
878
- const bgColor = cellProps.getAttribute("fo:background-color");
879
- if (bgColor && bgColor !== 'transparent')
880
- formatting.backgroundColor = bgColor;
881
- }
882
- const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
883
- if (textProps) {
884
- if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
885
- formatting.bold = true;
886
- if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
887
- formatting.italic = true;
888
- if (textProps.getAttribute("style:text-underline-style") === "solid")
889
- formatting.underline = true;
890
- if (textProps.getAttribute("style:text-line-through-style") === "solid")
891
- formatting.strikethrough = true;
892
- const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
893
- if (size)
894
- formatting.size = size;
895
- const color = textProps.getAttribute("fo:color");
896
- if (color)
897
- formatting.color = color;
898
- // Background color
899
- const bgColor = textProps.getAttribute("fo:background-color");
900
- if (bgColor && bgColor !== 'transparent')
901
- formatting.backgroundColor = bgColor;
902
- // Font family
903
- const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
904
- if (fontName)
905
- formatting.font = fontName;
906
- // Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
907
- const textPosition = textProps.getAttribute("style:text-position");
908
- if (textPosition) {
909
- if (textPosition.startsWith("sub"))
910
- formatting.subscript = true;
911
- if (textPosition.startsWith("super"))
912
- formatting.superscript = true;
913
- }
914
- }
915
- if (Object.keys(formatting).length > 0)
916
- styleMap[name] = formatting;
917
- }
847
+ parseStyles(automaticStyles);
918
848
  }
919
849
  // Start traversal
920
850
  const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
@@ -940,8 +870,26 @@ const parseOpenOffice = async (buffer, config) => {
940
870
  * @param sourceXml - The source XML string for raw content extraction
941
871
  * @param asSheet - If true, treats tables as sheets (for ODS)
942
872
  */
873
+ /**
874
+ * Emits the break a paragraph style asks for, on the given side of that paragraph.
875
+ *
876
+ * ODF has no inline break element for these - `fo:break-before="page"` sits on the style,
877
+ * so the break is a property of the paragraph rather than a run inside it. That makes it a
878
+ * sibling emitted around the paragraph node, not a child of it, which is the one structural
879
+ * difference from how DOCX's `<w:br w:type="page"/>` lands.
880
+ */
881
+ const pushStyleBreak = (styleName, targetArray, edge) => {
882
+ if (!config.includeBreakNodes || !styleName)
883
+ return;
884
+ const info = paragraphStyleMap[styleName];
885
+ const breakType = edge === 'before' ? info?.breakBefore : info?.breakAfter;
886
+ if (!breakType)
887
+ return;
888
+ targetArray.push({ type: 'break', metadata: { breakType } });
889
+ };
943
890
  traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
944
891
  if (node.tagName === "text:p") {
892
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
945
893
  const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
946
894
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
947
895
  const metadata = {
@@ -971,9 +919,11 @@ const parseOpenOffice = async (buffer, config) => {
971
919
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
972
920
  }
973
921
  targetArray.push(pNode);
922
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
974
923
  lastWasList = false;
975
924
  }
976
925
  else if (node.tagName === "text:h") {
926
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
977
927
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
978
928
  const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
979
929
  const metadata = {
@@ -998,6 +948,7 @@ const parseOpenOffice = async (buffer, config) => {
998
948
  hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
999
949
  }
1000
950
  targetArray.push(hNode);
951
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
1001
952
  lastWasList = false;
1002
953
  }
1003
954
  else if (node.tagName === "table:table") {
@@ -1270,15 +1221,18 @@ const parseOpenOffice = async (buffer, config) => {
1270
1221
  const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
1271
1222
  const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1272
1223
  if (mathNode) {
1273
- // Math formula object at block level
1274
- const formulaText = parseMathML(mathNode).trim();
1224
+ // Math formula object at block level - a display equation, so the
1225
+ // inner node is `math: 'block'` where the inline site above emits
1226
+ // `math: 'inline'`.
1227
+ const formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
1275
1228
  const formulaNode = {
1276
1229
  type: 'paragraph',
1277
1230
  text: formulaText,
1278
1231
  children: [
1279
1232
  {
1280
- type: 'text',
1281
- text: formulaText
1233
+ type: 'code',
1234
+ text: formulaText,
1235
+ metadata: { math: 'block' }
1282
1236
  }
1283
1237
  ]
1284
1238
  };
@@ -1358,7 +1312,11 @@ const parseOpenOffice = async (buffer, config) => {
1358
1312
  if (spans.length > 0) {
1359
1313
  for (const span of spans) {
1360
1314
  const styleName = span.getAttribute("text:style-name");
1361
- const formatting = styleName ? styleMap[styleName] : {};
1315
+ // Through `mergeFormatting` like every other span site, so
1316
+ // an explicit `false` is dropped rather than written onto
1317
+ // the node - and so the node gets its own object instead of
1318
+ // aliasing the shared style-table entry.
1319
+ const formatting = mergeFormatting({}, styleName ? styleMap[styleName] : undefined);
1362
1320
  const text = span.textContent || '';
1363
1321
  cellText += text;
1364
1322
  const textNode = {
@@ -1423,22 +1381,22 @@ const parseOpenOffice = async (buffer, config) => {
1423
1381
  const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1424
1382
  if (mathNode) {
1425
1383
  isFormula = true;
1426
- formulaText = parseMathML(mathNode).trim();
1384
+ formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
1427
1385
  }
1428
1386
  }
1429
1387
  }
1430
1388
  }
1431
1389
  if (isFormula) {
1432
1390
  cellText += formulaText;
1433
- const textNode = {
1434
- type: 'text',
1391
+ const formulaNode = {
1392
+ type: 'code',
1435
1393
  text: formulaText,
1436
- formatting: {}
1394
+ metadata: { math: 'inline' }
1437
1395
  };
1438
1396
  if (config.includeRawContent) {
1439
- textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1397
+ formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1440
1398
  }
1441
- children.push(textNode);
1399
+ children.push(formulaNode);
1442
1400
  }
1443
1401
  else if (drawImages.length > 0) {
1444
1402
  // logic for image node
@@ -29,6 +29,7 @@ const astUtils_js_1 = require("../utils/astUtils.js");
29
29
  const chartUtils_js_1 = require("../utils/chartUtils.js");
30
30
  const errorUtils_js_1 = require("../utils/errorUtils.js");
31
31
  const imageUtils_js_1 = require("../utils/imageUtils.js");
32
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
32
33
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
33
34
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
34
35
  const zipUtils_js_1 = require("../utils/zipUtils.js");
@@ -576,6 +577,30 @@ const parsePowerPoint = async (buffer, config) => {
576
577
  activeNode.children?.push({ type: 'text', text: "\n" });
577
578
  }
578
579
  }
580
+ else {
581
+ // Equations. This loop dispatches on `a:r`/`a:fld`, so an `m:oMath` -
582
+ // which is a sibling of the runs, not one of them - was never visited at
583
+ // all and the formula vanished from the slide without a warning.
584
+ //
585
+ // PowerPoint writes the equation either directly in the paragraph or
586
+ // wrapped in `mc:AlternateContent`/`a14:m` for pre-2010 readers, so take
587
+ // the element itself when it is the equation and search inside it
588
+ // otherwise. `getElementsByTagName` returns document order, which is the
589
+ // order the equations are read in.
590
+ const isMath = tag === "m:oMath" || tag === "m:oMathPara";
591
+ const equations = isMath ? [element] : (0, xmlUtils_js_1.getElementsByTagName)(element, "m:oMath");
592
+ for (const equation of equations) {
593
+ const latex = (0, mathUtils_js_1.ommlToLatex)(equation);
594
+ if ((0, mathUtils_js_1.isEmptyMath)(latex))
595
+ continue;
596
+ activeNode.text += latex;
597
+ activeNode.children?.push({
598
+ type: 'code',
599
+ text: latex,
600
+ metadata: { math: tag === "m:oMathPara" ? 'block' : 'inline' }
601
+ });
602
+ }
603
+ }
579
604
  }
580
605
  }
581
606
  }
@@ -623,12 +648,8 @@ const parsePowerPoint = async (buffer, config) => {
623
648
  }
624
649
  // Case 4: Grouped shape (recursive!)
625
650
  else if (tag === "p:grpSp") {
626
- // Extract the nested <p:spTree> inside the group
627
- const nestedTree = (0, xmlUtils_js_1.getFirstElementByTagName)(element, "p:spTree");
628
- // Recurse into the nested tree
629
- if (nestedTree) {
630
- nodes.push(...traverseSpTree(nestedTree, slideNumber, xmlContentString));
631
- }
651
+ // Recurse into the group element itself which holds the child shapes
652
+ nodes.push(...traverseSpTree(element, slideNumber, xmlContentString));
632
653
  }
633
654
  }
634
655
  return nodes;
@@ -66,6 +66,7 @@ const types_js_1 = require("../types.js");
66
66
  const astUtils_js_1 = require("../utils/astUtils.js");
67
67
  const errorUtils_js_1 = require("../utils/errorUtils.js");
68
68
  const imageUtils_js_1 = require("../utils/imageUtils.js");
69
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
69
70
  const ocrUtils_js_1 = require("../utils/ocrUtils.js");
70
71
  const xmlUtils_js_1 = require("../utils/xmlUtils.js");
71
72
  const zipUtils_js_1 = require("../utils/zipUtils.js");
@@ -751,6 +752,26 @@ const parseWord = async (buffer, config) => {
751
752
  }
752
753
  }
753
754
  }
755
+ else if ((0, xmlUtils_js_1.isElement)(node) && (node.nodeName === 'm:oMath' || node.nodeName === 'oMath'
756
+ || node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara')) {
757
+ // Equations. Without this branch they reach the generic fallback below, which
758
+ // recurses into every child and concatenates the `m:t` runs with no separators -
759
+ // so `<m:num>1</m:num><m:den>2</m:den>` came out as "12". That is worse than
760
+ // dropping the formula: the result still reads as a number, so nothing downstream
761
+ // can tell it is wrong.
762
+ //
763
+ // `m:oMathPara` is a display equation on its own line; a bare `m:oMath` is inline.
764
+ const isBlock = node.nodeName === 'm:oMathPara' || node.nodeName === 'oMathPara';
765
+ const latex = (0, mathUtils_js_1.ommlToLatex)(node);
766
+ if (!(0, mathUtils_js_1.isEmptyMath)(latex)) {
767
+ text += latex;
768
+ children.push({
769
+ type: 'code',
770
+ text: latex,
771
+ metadata: { math: isBlock ? 'block' : 'inline' }
772
+ });
773
+ }
774
+ }
754
775
  else if (node.childNodes.length > 0) {
755
776
  // Generic fallback for unknown elements that might contain content
756
777
  for (const child of Array.from(node.childNodes))