officeparser 7.4.0 → 7.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +85 -4
  2. package/dist/OfficeParser.js +54 -6
  3. package/dist/generators/BaseGenerator.d.ts +15 -0
  4. package/dist/generators/BaseGenerator.js +31 -0
  5. package/dist/generators/HtmlGenerator.d.ts +9 -0
  6. package/dist/generators/HtmlGenerator.js +34 -4
  7. package/dist/generators/MarkdownGenerator.d.ts +13 -0
  8. package/dist/generators/MarkdownGenerator.js +115 -41
  9. package/dist/generators/RtfGenerator.d.ts +13 -0
  10. package/dist/generators/RtfGenerator.js +23 -2
  11. package/dist/index.d.ts +2 -2
  12. package/dist/officeparser.browser.d.ts +34 -1
  13. package/dist/officeparser.browser.iife.js +160 -160
  14. package/dist/officeparser.browser.mjs +198 -198
  15. package/dist/officeparser.browser.slim.d.ts +34 -1
  16. package/dist/officeparser.browser.slim.iife.js +186 -186
  17. package/dist/officeparser.browser.slim.mjs +186 -186
  18. package/dist/parsers/EpubParser.js +2 -2
  19. package/dist/parsers/ExcelParser.js +11 -7
  20. package/dist/parsers/HtmlParser.js +39 -0
  21. package/dist/parsers/OpenOfficeParser.js +139 -167
  22. package/dist/parsers/PowerPointParser.js +48 -11
  23. package/dist/parsers/WordParser.js +33 -7
  24. package/dist/sbom.cdx.json +92 -92
  25. package/dist/types.d.ts +34 -1
  26. package/dist/types.js +10 -0
  27. package/dist/utils/configUtils.d.ts +15 -2
  28. package/dist/utils/configUtils.js +58 -13
  29. package/dist/utils/errorUtils.d.ts +8 -2
  30. package/dist/utils/errorUtils.js +23 -1
  31. package/dist/utils/mathUtils.d.ts +42 -0
  32. package/dist/utils/mathUtils.js +385 -0
  33. package/dist/utils/zipUtils.d.ts +64 -4
  34. package/dist/utils/zipUtils.js +188 -4
  35. package/package.json +9 -5
@@ -37,7 +37,7 @@ const parseEpub = async (buffer, config) => {
37
37
  const files = await (0, zipUtils_js_1.extractFiles)(buffer, (path) => /META-INF\/container\.xml$/i.test(path)
38
38
  || /\.opf$/i.test(path)
39
39
  || /\.(xhtml|html|htm)$/i.test(path)
40
- || (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits);
40
+ || (!!config.extractAttachments && /\.(png|jpe?g|gif|svg|webp)$/i.test(path)), config.decompressionLimits, config);
41
41
  // The OPF path is authoritative via META-INF/container.xml; fall back to scanning
42
42
  // for any .opf file for malformed archives that skip the container manifest.
43
43
  let opfPath;
@@ -49,7 +49,7 @@ const parseEpub = async (buffer, config) => {
49
49
  }
50
50
  const opfFile = (opfPath && files.find(f => f.path === opfPath)) || files.find(f => /\.opf$/i.test(f.path));
51
51
  if (!opfFile) {
52
- throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.FILE_CORRUPTED, config, 'epub (no OPF manifest found)');
52
+ throw (0, errorUtils_js_1.getOfficeError)(types_js_1.OfficeErrorType.REQUIRED_PART_MISSING, config, { fileType: 'epub', part: 'OPF package document (.opf)' });
53
53
  }
54
54
  const opfDir = opfFile.path.includes('/') ? opfFile.path.substring(0, opfFile.path.lastIndexOf('/') + 1) : '';
55
55
  const opfXml = (0, xmlUtils_js_1.parseXmlString)(opfFile.content.toString('utf-8'));
@@ -67,7 +67,16 @@ const parseExcel = async (buffer, config) => {
67
67
  !!x.match(customPropsFileRegex) ||
68
68
  !!x.match(appPropsFileRegex) ||
69
69
  (!!config.extractAttachments && (!!x.match(mediaFileRegex) || !!x.match(drawingRelsRegex))) ||
70
- ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits);
70
+ ((!!config.extractAttachments || !config.ignoreComments) && !!x.match(relsRegex)), config.decompressionLimits, config);
71
+ // Every workbook has xl/workbook.xml; without it the archive is not a spreadsheet.
72
+ // Resolved up front so a file that cannot be a workbook fails before any of the parsing
73
+ // work below, and read again further down for the sheet-name map.
74
+ const workbookFile = (0, zipUtils_js_1.findRequiredPart)(files, path => path === 'xl/workbook.xml', config, { fileType: 'xlsx', part: 'xl/workbook.xml' });
75
+ // Worksheets, by contrast, are not guaranteed: a workbook holding only chartsheets is
76
+ // valid and simply has no cell text to extract. Warn rather than fail, so the caller can
77
+ // tell "nothing to read here" from "we read nothing".
78
+ if (!files.some(file => !!file.path.match(sheetsRegex)))
79
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.NO_WORKSHEETS_FOUND, config);
71
80
  const sharedStringsFile = files.find(f => f.path === stringsFilePath);
72
81
  // Updated to store structured content (rich text runs) or simple string
73
82
  const sharedStrings = [];
@@ -372,9 +381,8 @@ const parseExcel = async (buffer, config) => {
372
381
  }
373
382
  // Parse workbook.xml to get sheet names and map them to sheet files
374
383
  const sheetNameMap = {};
375
- const workbookFile = files.find(f => f.path === 'xl/workbook.xml');
376
384
  const workbookRelsFile = files.find(f => f.path === 'xl/_rels/workbook.xml.rels');
377
- if (workbookFile && workbookRelsFile) {
385
+ if (workbookRelsFile) {
378
386
  // Parse rels to get rId -> file mapping
379
387
  const relsXml = (0, xmlUtils_js_1.parseXmlString)(workbookRelsFile.content.toString());
380
388
  const relationships = (0, xmlUtils_js_1.getElementsByTagName)(relsXml, "Relationship");
@@ -401,7 +409,6 @@ const parseExcel = async (buffer, config) => {
401
409
  }
402
410
  }
403
411
  const content = [];
404
- const rawContents = [];
405
412
  for (const file of files) {
406
413
  if (file.path.match(mediaFileRegex))
407
414
  continue;
@@ -418,9 +425,6 @@ const parseExcel = async (buffer, config) => {
418
425
  if (file.path.match(drawingRelsRegex))
419
426
  continue;
420
427
  if (file.path.match(sheetsRegex)) {
421
- if (config.includeRawContent) {
422
- rawContents.push(file.content.toString());
423
- }
424
428
  const sheetFilename = file.path.split('/').pop() || '';
425
429
  const relsFilename = `xl/worksheets/_rels/${sheetFilename}.rels`;
426
430
  const relsFile = files.find(f => f.path === relsFilename);
@@ -4,6 +4,7 @@ exports.parseHtml = void 0;
4
4
  const types_js_1 = require("../types.js");
5
5
  const astUtils_js_1 = require("../utils/astUtils.js");
6
6
  const errorUtils_js_1 = require("../utils/errorUtils.js");
7
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
7
8
  const sanitize_js_1 = require("../utils/sanitize.js");
8
9
  /**
9
10
  * Maximum element nesting depth accepted from an HTML/XHTML source before the parser gives up
@@ -11,6 +12,25 @@ const sanitize_js_1 = require("../utils/sanitize.js");
11
12
  * `parseNode` for why this value and not a larger one.
12
13
  */
13
14
  const MAX_HTML_NESTING_DEPTH = 256;
15
+ /**
16
+ * Presents an `HtmlNode` as a `MathNode` for the shared MathML converter.
17
+ *
18
+ * The shapes already line up field for field; the one thing that must happen here is entity
19
+ * decoding, since this parser keeps text nodes in their raw escaped form and `<` inside an
20
+ * `<mo>` is a less-than operator, not markup.
21
+ */
22
+ const toMathNode = (node) => ({
23
+ tagName: node.tagName,
24
+ attributes: node.attributes,
25
+ text: node.text === undefined ? undefined : node.text
26
+ .replace(/&nbsp;/g, ' ')
27
+ .replace(/&lt;/g, '<')
28
+ .replace(/&gt;/g, '>')
29
+ .replace(/&amp;/g, '&')
30
+ .replace(/&quot;/g, '"')
31
+ .replace(/&#39;/g, "'"),
32
+ children: (node.children || []).map(toMathNode),
33
+ });
14
34
  const parseAttributes = (attrString) => {
15
35
  const attrs = {};
16
36
  // Attribute names follow the HTML5 rule - any character except whitespace and
@@ -572,6 +592,25 @@ const parseHtml = async (buffer, config) => {
572
592
  metadata: { math: mathMode }
573
593
  };
574
594
  }
595
+ // Native MathML. This is what a real-world page and every EPUB3 uses (EpubParser
596
+ // routes each spine item through here), as opposed to the `data-math` round-trip
597
+ // contract above, which only ever appears in this library's own HTML output. Without
598
+ // it, a `<math>` element fell through to the generic element handling below, which
599
+ // concatenates descendant text: `<mfrac><mn>1</mn><mn>2</mn></mfrac>` became "12".
600
+ if (tagName === 'math' || tagName.endsWith(':math')) {
601
+ // `display="block"` is MathML's own attribute for a display equation; the legacy
602
+ // `mode="display"` means the same thing and is still emitted by older producers.
603
+ const isBlock = node.attributes?.['display'] === 'block'
604
+ || node.attributes?.['mode'] === 'display';
605
+ const latex = (0, mathUtils_js_1.mathmlTreeToLatex)(toMathNode(node));
606
+ if ((0, mathUtils_js_1.isEmptyMath)(latex))
607
+ return null;
608
+ return {
609
+ type: 'code',
610
+ text: latex,
611
+ metadata: { math: isBlock ? 'block' : 'inline' }
612
+ };
613
+ }
575
614
  // Admonition: inscript-editor's Admonition node renders
576
615
  // <div class="admonition admonition-note" data-type="note">…children…</div>.
577
616
  if (tagName === 'div' && (node.attributes?.class || '').split(/\s+/).includes('admonition')) {
@@ -27,6 +27,7 @@ const types_js_1 = require("../types.js");
27
27
  const astUtils_js_1 = require("../utils/astUtils.js");
28
28
  const chartUtils_js_1 = require("../utils/chartUtils.js");
29
29
  const errorUtils_js_1 = require("../utils/errorUtils.js");
30
+ const mathUtils_js_1 = require("../utils/mathUtils.js");
30
31
  /**
31
32
  * Tracks how many table cells a single document has been allowed to materialize.
32
33
  *
@@ -84,10 +85,28 @@ class CellBudget {
84
85
  /** Resolves the configured cell budget, falling back to the documented default. */
85
86
  const createCellBudget = (config) => new CellBudget(config.decompressionLimits?.maxTableCells ?? 1000000, config);
86
87
  /**
87
- * Coerces a `table:number-*-repeated` attribute to a usable repeat count. A missing, zero,
88
- * negative or non-numeric value becomes 1 (the element renders once), so a garbage attribute
89
- * can neither drop a cell nor poison arithmetic downstream (`colIndex += NaN`).
88
+ * Merges a style's formatting over what it inherits, dropping any flag the style explicitly turns
89
+ * off rather than carrying a `false` forward.
90
+ *
91
+ * Generators all test these flags for truthiness, so a retained `false` would render the same - but
92
+ * it would not *compare* the same, and `MarkdownGenerator.optimizeNodes` merges adjacent text nodes
93
+ * only when their formatting objects are equal. Leaving `bold: false` on one node and nothing on
94
+ * its neighbour would silently stop that merge and fragment the output. Same reasoning, and same
95
+ * shape, as `WordParser`'s direct-run-property merge.
90
96
  */
97
+ const mergeFormatting = (inherited, override) => {
98
+ if (!override)
99
+ return { ...inherited };
100
+ const merged = { ...inherited };
101
+ for (const key of Object.keys(override)) {
102
+ const value = override[key];
103
+ if (value === false)
104
+ delete merged[key];
105
+ else if (value !== undefined)
106
+ merged[key] = value;
107
+ }
108
+ return merged;
109
+ };
91
110
  const toRepeatCount = (attr) => {
92
111
  const n = parseInt(attr || "1");
93
112
  return Number.isFinite(n) && n > 0 ? n : 1;
@@ -106,6 +125,8 @@ const cleanAttachmentName = (href) => {
106
125
  const cleaned = href.replace(/^\.\//, '').replace(/\/$/, '');
107
126
  return cleaned.split('/').pop() || '';
108
127
  };
128
+ /** The ODF document types this parser handles, used to validate a caller-supplied file type. */
129
+ const ODF_FILE_TYPES = ['odt', 'odp', 'ods'];
109
130
  /**
110
131
  * Parses an OpenOffice document (.odt, .odp, .ods) and extracts content.
111
132
  *
@@ -129,10 +150,17 @@ const parseOpenOffice = async (buffer, config) => {
129
150
  !!x.match(metaFileRegex) ||
130
151
  !!x.match(stylesFileRegex) ||
131
152
  !!x.match(mimetypeFileRegex) ||
132
- (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits);
153
+ (!!config.extractAttachments && !!x.match(mediaFileRegex)), config.decompressionLimits, config);
133
154
  // 1. Determine File Type
134
155
  const mimetypeFile = files.find(f => f.path === 'mimetype');
135
- let fileType = 'odt'; // Default
156
+ // The archive's own mimetype entry is authoritative when present. When it is missing,
157
+ // fall back to the type the caller asked for (or that was derived from the extension)
158
+ // rather than assuming text: guessing 'odt' for a spreadsheet sends the parser down the
159
+ // office:text branch, which finds nothing in an office:spreadsheet body and yields an
160
+ // empty document for a perfectly valid file.
161
+ let fileType = ODF_FILE_TYPES.includes(config.fileType)
162
+ ? config.fileType
163
+ : 'odt';
136
164
  if (mimetypeFile) {
137
165
  const mime = mimetypeFile.content.toString().trim();
138
166
  if (mime.includes('spreadsheet'))
@@ -142,7 +170,12 @@ const parseOpenOffice = async (buffer, config) => {
142
170
  else if (mime.includes('text'))
143
171
  fileType = 'odt';
144
172
  }
145
- const mainContentFile = files.find(f => f.path === 'content.xml') || files.find(f => f.path.match(contentFileRegex));
173
+ // The document body is the content.xml at the archive root. The fallback stays anchored
174
+ // and excludes embedded objects: an ODF file can carry Object N/content.xml for a chart
175
+ // or formula, and an unanchored match would promote one of those to the document body
176
+ // when the real one is missing, silently parsing a chart as if it were the whole file.
177
+ const mainContentFile = files.find(f => f.path === 'content.xml')
178
+ || (0, zipUtils_js_1.findRequiredPart)(files, path => /(^|\/)content\.xml$/.test(path) && !objectContentFileRegex.test(path), config, { fileType, part: 'content.xml' });
146
179
  const stylesFile = files.find(f => f.path === 'styles.xml');
147
180
  const stylesDom = stylesFile ? (0, xmlUtils_js_1.parseXmlString)(stylesFile.content.toString()) : undefined;
148
181
  const content = [];
@@ -159,8 +192,8 @@ const parseOpenOffice = async (buffer, config) => {
159
192
  let lastWasList = false;
160
193
  let traverse;
161
194
  // Helper to parse styles
162
- const parseStyles = (xml) => {
163
- const styles = (0, xmlUtils_js_1.getElementsByTagName)(xml, "style:style");
195
+ const parseStyles = (scope) => {
196
+ const styles = (0, xmlUtils_js_1.getElementsByTagName)(scope, "style:style");
164
197
  for (const style of styles) {
165
198
  const name = style.getAttribute("style:name");
166
199
  if (!name)
@@ -188,6 +221,17 @@ const parseOpenOffice = async (buffer, config) => {
188
221
  if (dropCap) {
189
222
  styleInfo.dropCap = true;
190
223
  }
224
+ // Page/column breaks. ODF attaches these to the paragraph style rather than
225
+ // writing an inline element the way DOCX's `<w:br w:type="page"/>` does, which is
226
+ // why `includeBreakNodes` produced nothing at all for ODF: there was no inline
227
+ // element to find. Only the two break kinds that map onto a BreakMetadata type
228
+ // are carried; `auto` and `even-page`/`odd-page` have no equivalent.
229
+ const breakBefore = paraProps.getAttribute("fo:break-before");
230
+ if (breakBefore === 'page' || breakBefore === 'column')
231
+ styleInfo.breakBefore = breakBefore;
232
+ const breakAfter = paraProps.getAttribute("fo:break-after");
233
+ if (breakAfter === 'page' || breakAfter === 'column')
234
+ styleInfo.breakAfter = breakAfter;
191
235
  }
192
236
  if (Object.keys(styleInfo).length > 0) {
193
237
  paragraphStyleMap[name] = styleInfo;
@@ -203,14 +247,28 @@ const parseOpenOffice = async (buffer, config) => {
203
247
  formatting.backgroundColor = bgColor;
204
248
  }
205
249
  if (textProps) {
206
- if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
207
- formatting.bold = true;
208
- if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
209
- formatting.italic = true;
210
- if (textProps.getAttribute("style:text-underline-style") === "solid")
211
- formatting.underline = true;
212
- if (textProps.getAttribute("style:text-line-through-style") === "solid")
213
- formatting.strikethrough = true;
250
+ // Record the *off* states as an explicit `false`, not as an absent key.
251
+ //
252
+ // Now that a paragraph style's text properties are inherited by the runs inside it,
253
+ // a span has to be able to turn one back off: LibreOffice writes
254
+ // `fo:font-weight="normal"` on the span whenever a user un-bolds part of a
255
+ // bold-styled paragraph. With only the `true` side recorded, that span had nothing
256
+ // to override the inherited value with and came out bold - wrong in the opposite
257
+ // direction from the bug the inheritance fixed. `TextFormatting`'s flags are
258
+ // `boolean | undefined` precisely so "explicitly off" is expressible.
259
+ const fontWeight = textProps.getAttribute("fo:font-weight") || textProps.getAttribute("style:font-weight-asian");
260
+ // Numeric weights are the same axis: 600+ is bold, below that is not.
261
+ if (fontWeight)
262
+ formatting.bold = fontWeight === "bold" || /^[6-9]00$/.test(fontWeight);
263
+ const fontStyle = textProps.getAttribute("fo:font-style") || textProps.getAttribute("style:font-style-asian");
264
+ if (fontStyle)
265
+ formatting.italic = fontStyle === "italic" || fontStyle === "oblique";
266
+ const underline = textProps.getAttribute("style:text-underline-style");
267
+ if (underline)
268
+ formatting.underline = underline !== "none";
269
+ const lineThrough = textProps.getAttribute("style:text-line-through-style");
270
+ if (lineThrough)
271
+ formatting.strikethrough = lineThrough !== "none";
214
272
  const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
215
273
  if (size)
216
274
  formatting.size = size;
@@ -247,62 +305,6 @@ const parseOpenOffice = async (buffer, config) => {
247
305
  *
248
306
  * @param node - The paragraph element to parse
249
307
  */
250
- const parseMathML = (node) => {
251
- if (!node)
252
- return '';
253
- if (node.nodeType === 3) { // Text node
254
- return node.textContent || '';
255
- }
256
- if (node.nodeType !== 1) { // Not an element
257
- return '';
258
- }
259
- const element = node;
260
- const tagName = element.tagName.toLowerCase().replace(/^.*:/, ''); // strip namespace prefix
261
- switch (tagName) {
262
- case 'math':
263
- case 'mrow':
264
- case 'semantics':
265
- return Array.from(element.childNodes).map(parseMathML).join('');
266
- case 'mfrac': {
267
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
268
- if (children.length >= 2) {
269
- return `(${parseMathML(children[0])})/(${parseMathML(children[1])})`;
270
- }
271
- return Array.from(element.childNodes).map(parseMathML).join('');
272
- }
273
- case 'msub': {
274
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
275
- if (children.length >= 2) {
276
- return `${parseMathML(children[0])}_${parseMathML(children[1])}`;
277
- }
278
- return Array.from(element.childNodes).map(parseMathML).join('');
279
- }
280
- case 'msup': {
281
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
282
- if (children.length >= 2) {
283
- return `${parseMathML(children[0])}^${parseMathML(children[1])}`;
284
- }
285
- return Array.from(element.childNodes).map(parseMathML).join('');
286
- }
287
- case 'msubsup': {
288
- const children = Array.from(element.childNodes).filter((n) => n.nodeType === 1);
289
- if (children.length >= 3) {
290
- return `${parseMathML(children[0])}_${parseMathML(children[1])}^${parseMathML(children[2])}`;
291
- }
292
- return Array.from(element.childNodes).map(parseMathML).join('');
293
- }
294
- case 'mi':
295
- case 'mn':
296
- case 'mo':
297
- case 'mtext':
298
- case 'ms':
299
- return element.textContent || '';
300
- case 'annotation':
301
- return '';
302
- default:
303
- return Array.from(element.childNodes).map(parseMathML).join('');
304
- }
305
- };
306
308
  /**
307
309
  * Helper to parse inline content (text, spans, links, notes, etc.) recursively.
308
310
  *
@@ -365,6 +367,14 @@ const parseOpenOffice = async (buffer, config) => {
365
367
  metadata: linkMetadata ? { ...linkMetadata } : undefined
366
368
  });
367
369
  }
370
+ else if (tagName === 'text:soft-page-break') {
371
+ // The page boundary the editor recorded at its last save. DOCX's equivalent
372
+ // is `w:lastRenderedPageBreak`, so it maps onto the same break type rather
373
+ // than onto 'page', which is reserved for a break the author asked for.
374
+ if (config.includeBreakNodes) {
375
+ children.push({ type: 'break', metadata: { breakType: 'lastRenderedPage' } });
376
+ }
377
+ }
368
378
  else if (tagName === 'text:line-break') {
369
379
  // Line break
370
380
  fullText += '\n';
@@ -378,7 +388,7 @@ const parseOpenOffice = async (buffer, config) => {
378
388
  else if (tagName === 'text:span') {
379
389
  // Formatted text span
380
390
  const styleName = element.getAttribute("text:style-name");
381
- const formatting = styleName ? { ...parentFormatting, ...styleMap[styleName] } : parentFormatting;
391
+ const formatting = styleName ? mergeFormatting(parentFormatting, styleMap[styleName]) : parentFormatting;
382
392
  const spanContent = parseInlineContent(element, styleMap, config, notes, paragraphStyleMap, formatting, linkMetadata, sourceXml);
383
393
  fullText += spanContent.text;
384
394
  children.push(...spanContent.children);
@@ -483,22 +493,25 @@ const parseOpenOffice = async (buffer, config) => {
483
493
  const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
484
494
  if (mathNode) {
485
495
  isFormula = true;
486
- formulaText = parseMathML(mathNode).trim();
496
+ formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
487
497
  }
488
498
  }
489
499
  }
490
500
  if (isFormula) {
491
501
  fullText += formulaText;
492
- const textNode = {
493
- type: 'text',
502
+ // A `code` node carrying `math`, not a plain `text` node: the formula
503
+ // is LaTeX, and marking it as such is what lets generators render it
504
+ // as maths rather than emit it as prose that happens to contain
505
+ // backslashes. Same node shape DOCX, PPTX, HTML and Markdown produce.
506
+ const formulaNode = {
507
+ type: 'code',
494
508
  text: formulaText,
495
- formatting: parentFormatting,
496
- metadata: linkMetadata ? { ...linkMetadata } : undefined
509
+ metadata: { math: 'inline', ...(linkMetadata ?? {}) }
497
510
  };
498
511
  if (config.includeRawContent) {
499
- textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
512
+ formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, sourceXml, config);
500
513
  }
501
- children.push(textNode);
514
+ children.push(formulaNode);
502
515
  }
503
516
  else {
504
517
  // Standard inline image extraction fallback if object is not a formula
@@ -588,8 +601,9 @@ const parseOpenOffice = async (buffer, config) => {
588
601
  const styleInfo = paraStyle ? paraStyleMap[paraStyle] : undefined;
589
602
  const alignment = styleInfo?.alignment;
590
603
  const dropCap = styleInfo?.dropCap;
604
+ const formatting = mergeFormatting({}, paraStyle ? styleMap[paraStyle] : undefined);
591
605
  // Parse content recursively using the new helper
592
- const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, {}, undefined, sourceXml);
606
+ const content = parseInlineContent(node, styleMap, config, notes, paraStyleMap, formatting, undefined, sourceXml);
593
607
  // Add style name to metadata of children if they don't have one
594
608
  if (paraStyle) {
595
609
  content.children.forEach(child => {
@@ -800,7 +814,7 @@ const parseOpenOffice = async (buffer, config) => {
800
814
  // whole cell array, so rows x cols is what actually exhausts memory; charge those
801
815
  // copies against the same budget.
802
816
  const allowedRows = cells.length === 0
803
- ? rowsRepeated
817
+ ? (rowsRepeated > 0 ? 1 + cellBudget.take(rowsRepeated - 1) : 0)
804
818
  : Math.min(rowsRepeated, 1 + Math.floor(cellBudget.take(Math.max(0, (rowsRepeated - 1) * cells.length)) / cells.length));
805
819
  for (let k = 0; k < allowedRows; k++) {
806
820
  if ((k & 255) === 0)
@@ -839,82 +853,12 @@ const parseOpenOffice = async (buffer, config) => {
839
853
  // splitting a huge repeat expansion across many small tables. `traverse` and the
840
854
  // spreadsheet branch below both close over this; `parseTable` receives it explicitly.
841
855
  const cellBudget = createCellBudget(config);
842
- // Parse automatic styles (local to content.xml)
856
+ // Automatic styles are local to content.xml, but their definitions have exactly the
857
+ // shape styles.xml uses, so they go through the same reader rather than a second copy of
858
+ // it - the copy is how `fo:break-before` came to be read in neither place.
843
859
  const automaticStyles = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:automatic-styles");
844
860
  if (automaticStyles) {
845
- const styles = (0, xmlUtils_js_1.getElementsByTagName)(automaticStyles, "style:style");
846
- for (const style of styles) {
847
- const name = style.getAttribute("style:name");
848
- if (!name)
849
- continue;
850
- // Parse paragraph properties for alignment
851
- const paraProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:paragraph-properties");
852
- const styleInfo = {};
853
- if (paraProps) {
854
- const textAlign = paraProps.getAttribute("fo:text-align");
855
- if (textAlign) {
856
- const alignMap = {
857
- 'start': 'left',
858
- 'left': 'left',
859
- 'center': 'center',
860
- 'end': 'right',
861
- 'right': 'right',
862
- 'justify': 'justify'
863
- };
864
- if (alignMap[textAlign]) {
865
- styleInfo.alignment = alignMap[textAlign];
866
- }
867
- }
868
- const dropCap = (0, xmlUtils_js_1.getFirstElementByTagName)(paraProps, "style:drop-cap");
869
- if (dropCap)
870
- styleInfo.dropCap = true;
871
- }
872
- if (Object.keys(styleInfo).length > 0) {
873
- paragraphStyleMap[name] = styleInfo;
874
- }
875
- const cellProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:table-cell-properties");
876
- const formatting = {};
877
- if (cellProps) {
878
- const bgColor = cellProps.getAttribute("fo:background-color");
879
- if (bgColor && bgColor !== 'transparent')
880
- formatting.backgroundColor = bgColor;
881
- }
882
- const textProps = (0, xmlUtils_js_1.getFirstElementByTagName)(style, "style:text-properties");
883
- if (textProps) {
884
- if (textProps.getAttribute("fo:font-weight") === "bold" || textProps.getAttribute("style:font-weight-asian") === "bold")
885
- formatting.bold = true;
886
- if (textProps.getAttribute("fo:font-style") === "italic" || textProps.getAttribute("style:font-style-asian") === "italic")
887
- formatting.italic = true;
888
- if (textProps.getAttribute("style:text-underline-style") === "solid")
889
- formatting.underline = true;
890
- if (textProps.getAttribute("style:text-line-through-style") === "solid")
891
- formatting.strikethrough = true;
892
- const size = textProps.getAttribute("fo:font-size") || textProps.getAttribute("style:font-size-asian");
893
- if (size)
894
- formatting.size = size;
895
- const color = textProps.getAttribute("fo:color");
896
- if (color)
897
- formatting.color = color;
898
- // Background color
899
- const bgColor = textProps.getAttribute("fo:background-color");
900
- if (bgColor && bgColor !== 'transparent')
901
- formatting.backgroundColor = bgColor;
902
- // Font family
903
- const fontName = textProps.getAttribute("style:font-name") || textProps.getAttribute("fo:font-family");
904
- if (fontName)
905
- formatting.font = fontName;
906
- // Subscript/Superscript from text-position (e.g., "sub 58%" or "super 58%")
907
- const textPosition = textProps.getAttribute("style:text-position");
908
- if (textPosition) {
909
- if (textPosition.startsWith("sub"))
910
- formatting.subscript = true;
911
- if (textPosition.startsWith("super"))
912
- formatting.superscript = true;
913
- }
914
- }
915
- if (Object.keys(formatting).length > 0)
916
- styleMap[name] = formatting;
917
- }
861
+ parseStyles(automaticStyles);
918
862
  }
919
863
  // Start traversal
920
864
  const officeBody = (0, xmlUtils_js_1.getFirstElementByTagName)(xml, "office:body");
@@ -940,8 +884,26 @@ const parseOpenOffice = async (buffer, config) => {
940
884
  * @param sourceXml - The source XML string for raw content extraction
941
885
  * @param asSheet - If true, treats tables as sheets (for ODS)
942
886
  */
887
+ /**
888
+ * Emits the break a paragraph style asks for, on the given side of that paragraph.
889
+ *
890
+ * ODF has no inline break element for these - `fo:break-before="page"` sits on the style,
891
+ * so the break is a property of the paragraph rather than a run inside it. That makes it a
892
+ * sibling emitted around the paragraph node, not a child of it, which is the one structural
893
+ * difference from how DOCX's `<w:br w:type="page"/>` lands.
894
+ */
895
+ const pushStyleBreak = (styleName, targetArray, edge) => {
896
+ if (!config.includeBreakNodes || !styleName)
897
+ return;
898
+ const info = paragraphStyleMap[styleName];
899
+ const breakType = edge === 'before' ? info?.breakBefore : info?.breakAfter;
900
+ if (!breakType)
901
+ return;
902
+ targetArray.push({ type: 'break', metadata: { breakType } });
903
+ };
943
904
  traverse = (node, targetArray, forceHeading = false, sourceXml = '', asSheet = false) => {
944
905
  if (node.tagName === "text:p") {
906
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
945
907
  const pContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
946
908
  const type = (forceHeading || (node.getAttribute("text:style-name") || '').toLowerCase().includes('title')) ? 'heading' : 'paragraph';
947
909
  const metadata = {
@@ -971,9 +933,11 @@ const parseOpenOffice = async (buffer, config) => {
971
933
  pNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
972
934
  }
973
935
  targetArray.push(pNode);
936
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
974
937
  lastWasList = false;
975
938
  }
976
939
  else if (node.tagName === "text:h") {
940
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'before');
977
941
  const level = parseInt(node.getAttribute("text:outline-level") || "1");
978
942
  const hContent = parseParagraphContent(node, paragraphStyleMap, styleMap, config, sourceXml);
979
943
  const metadata = {
@@ -998,6 +962,7 @@ const parseOpenOffice = async (buffer, config) => {
998
962
  hNode.rawContent = (0, xmlUtils_js_1.getRawContent)(node, sourceXml, config);
999
963
  }
1000
964
  targetArray.push(hNode);
965
+ pushStyleBreak(node.getAttribute("text:style-name"), targetArray, 'after');
1001
966
  lastWasList = false;
1002
967
  }
1003
968
  else if (node.tagName === "table:table") {
@@ -1270,15 +1235,18 @@ const parseOpenOffice = async (buffer, config) => {
1270
1235
  const objXml = (0, xmlUtils_js_1.parseXmlString)(objectFile.content.toString());
1271
1236
  const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1272
1237
  if (mathNode) {
1273
- // Math formula object at block level
1274
- const formulaText = parseMathML(mathNode).trim();
1238
+ // Math formula object at block level - a display equation, so the
1239
+ // inner node is `math: 'block'` where the inline site above emits
1240
+ // `math: 'inline'`.
1241
+ const formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
1275
1242
  const formulaNode = {
1276
1243
  type: 'paragraph',
1277
1244
  text: formulaText,
1278
1245
  children: [
1279
1246
  {
1280
- type: 'text',
1281
- text: formulaText
1247
+ type: 'code',
1248
+ text: formulaText,
1249
+ metadata: { math: 'block' }
1282
1250
  }
1283
1251
  ]
1284
1252
  };
@@ -1358,7 +1326,11 @@ const parseOpenOffice = async (buffer, config) => {
1358
1326
  if (spans.length > 0) {
1359
1327
  for (const span of spans) {
1360
1328
  const styleName = span.getAttribute("text:style-name");
1361
- const formatting = styleName ? styleMap[styleName] : {};
1329
+ // Through `mergeFormatting` like every other span site, so
1330
+ // an explicit `false` is dropped rather than written onto
1331
+ // the node - and so the node gets its own object instead of
1332
+ // aliasing the shared style-table entry.
1333
+ const formatting = mergeFormatting({}, styleName ? styleMap[styleName] : undefined);
1362
1334
  const text = span.textContent || '';
1363
1335
  cellText += text;
1364
1336
  const textNode = {
@@ -1423,22 +1395,22 @@ const parseOpenOffice = async (buffer, config) => {
1423
1395
  const mathNode = (0, xmlUtils_js_1.getFirstElementByTagName)(objXml, "math");
1424
1396
  if (mathNode) {
1425
1397
  isFormula = true;
1426
- formulaText = parseMathML(mathNode).trim();
1398
+ formulaText = (0, mathUtils_js_1.mathmlToLatex)(mathNode).trim();
1427
1399
  }
1428
1400
  }
1429
1401
  }
1430
1402
  }
1431
1403
  if (isFormula) {
1432
1404
  cellText += formulaText;
1433
- const textNode = {
1434
- type: 'text',
1405
+ const formulaNode = {
1406
+ type: 'code',
1435
1407
  text: formulaText,
1436
- formatting: {}
1408
+ metadata: { math: 'inline' }
1437
1409
  };
1438
1410
  if (config.includeRawContent) {
1439
- textNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1411
+ formulaNode.rawContent = (0, xmlUtils_js_1.getRawContent)(frame, xmlString, config);
1440
1412
  }
1441
- children.push(textNode);
1413
+ children.push(formulaNode);
1442
1414
  }
1443
1415
  else if (drawImages.length > 0) {
1444
1416
  // logic for image node