officeparser 7.4.0 → 7.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +85 -4
  2. package/dist/OfficeParser.js +54 -6
  3. package/dist/generators/BaseGenerator.d.ts +15 -0
  4. package/dist/generators/BaseGenerator.js +31 -0
  5. package/dist/generators/HtmlGenerator.d.ts +9 -0
  6. package/dist/generators/HtmlGenerator.js +34 -4
  7. package/dist/generators/MarkdownGenerator.d.ts +13 -0
  8. package/dist/generators/MarkdownGenerator.js +115 -41
  9. package/dist/generators/RtfGenerator.d.ts +13 -0
  10. package/dist/generators/RtfGenerator.js +23 -2
  11. package/dist/index.d.ts +2 -2
  12. package/dist/officeparser.browser.d.ts +34 -1
  13. package/dist/officeparser.browser.iife.js +160 -160
  14. package/dist/officeparser.browser.mjs +198 -198
  15. package/dist/officeparser.browser.slim.d.ts +34 -1
  16. package/dist/officeparser.browser.slim.iife.js +186 -186
  17. package/dist/officeparser.browser.slim.mjs +186 -186
  18. package/dist/parsers/EpubParser.js +2 -2
  19. package/dist/parsers/ExcelParser.js +11 -7
  20. package/dist/parsers/HtmlParser.js +39 -0
  21. package/dist/parsers/OpenOfficeParser.js +139 -167
  22. package/dist/parsers/PowerPointParser.js +48 -11
  23. package/dist/parsers/WordParser.js +33 -7
  24. package/dist/sbom.cdx.json +92 -92
  25. package/dist/types.d.ts +34 -1
  26. package/dist/types.js +10 -0
  27. package/dist/utils/configUtils.d.ts +15 -2
  28. package/dist/utils/configUtils.js +58 -13
  29. package/dist/utils/errorUtils.d.ts +8 -2
  30. package/dist/utils/errorUtils.js +23 -1
  31. package/dist/utils/mathUtils.d.ts +42 -0
  32. package/dist/utils/mathUtils.js +385 -0
  33. package/dist/utils/zipUtils.d.ts +64 -4
  34. package/dist/utils/zipUtils.js +188 -4
  35. package/package.json +9 -5
package/README.md CHANGED
@@ -128,7 +128,7 @@ npx officeparser my_document --fileType=docx --to=json
128
128
  | `--includeRawContent` | boolean | `false` | Include raw XML/RTF in nodes |
129
129
  | `--serializeRawContent` | boolean | `true` | Include stringified XML in metadata |
130
130
  | `--preserveXmlWhitespace` | boolean | `false` | Keep raw formatting space |
131
- | `--includeBreakNodes` | boolean | `false` | Include break nodes (DOCX only) |
131
+ | `--includeBreakNodes` | boolean | `false` | Include break nodes (DOCX and ODF) |
132
132
  | `--verbose` | boolean | `false` | Show full error stack traces and warning logs |
133
133
  | `--includeFormatting` | boolean | `true` | Include formatting style map matching |
134
134
  | `--renderMetadata` | boolean | `false` | Render metadata as visible content in the generated output |
@@ -206,6 +206,15 @@ const ast = await officeParser.parseOffice(buffer);
206
206
  > const ast = await officeParser.parseOffice(markdownBuffer, { fileType: 'md' });
207
207
  > ```
208
208
 
209
+ > [!NOTE]
210
+ > **ZIP-backed formats are identified from inside the archive.** DOCX, XLSX, PPTX, ODT, ODS, ODP
211
+ > and EPUB are all ZIP files, and telling them apart from the first bytes alone is unreliable for
212
+ > archives written by streaming producers or holding very many parts. When the byte signature is
213
+ > inconclusive, the archive is opened and the format is read from its own declaration
214
+ > (`[Content_Types].xml`, or the `mimetype` entry), so these parse from a buffer without a hint.
215
+ > Supplying `fileType` remains the fastest and most certain route: it decides which parser runs,
216
+ > and for these formats no archive inspection is done at all.
217
+
209
218
  ### Cancellation with AbortSignal
210
219
 
211
220
  You can pass a standard `AbortSignal` (e.g. from an `AbortController`) to cancel an active parse operation. This is especially useful for setting request-level timeouts or canceling long-running parses (like large PDFs with OCR).
@@ -533,6 +542,34 @@ interface OfficeIssue {
533
542
  }
534
543
  ```
535
544
 
545
+ Thrown errors carry the same object on `error.officeIssue`, so a failed parse is identified by
546
+ the same stable `code` you would branch on for a warning, rather than by matching message text:
547
+
548
+ ```js
549
+ try {
550
+ const ast = await officeParser.parseOffice(buffer, { fileType: 'docx' });
551
+ } catch (err) {
552
+ switch (err.officeIssue?.code) {
553
+ case 'ZIP_NO_ENTRIES_FOUND': // not a ZIP archive at all
554
+ case 'ZIP_TRUNCATED': // cut off in transfer, entries incomplete
555
+ case 'REQUIRED_PART_MISSING': // readable ZIP, but not the format it claims
556
+ console.error('Unusable file:', err.officeIssue.message);
557
+ break;
558
+ default:
559
+ throw err;
560
+ }
561
+ }
562
+ ```
563
+
564
+ > [!IMPORTANT]
565
+ > **A corrupt file throws; it does not parse as an empty document.** If an archive is not
566
+ > readable, is truncated, or is missing the part its format requires (`word/document.xml`,
567
+ > `xl/workbook.xml`, `ppt/presentation.xml`, ODF `content.xml`, the EPUB OPF), parsing rejects
568
+ > with one of the codes above. An empty result therefore means the document really is empty.
569
+ > Files that are legitimately empty still parse, and say so through `onWarning` /
570
+ > `ast.warnings` (`NO_WORKSHEETS_FOUND` for a chartsheet-only workbook, `NO_SLIDES_FOUND` for a
571
+ > presentation with no slides).
572
+
536
573
  ---
537
574
 
538
575
  ## Deep Dive: Document Components
@@ -608,7 +645,14 @@ formatting: {
608
645
  }
609
646
  ```
610
647
 
611
- ### 6. Break Nodes (DOCX only)
648
+ > [!NOTE]
649
+ > On a **content node**, an absent flag and `false` mean the same thing — the flag is simply not
650
+ > applied. On **`ast.metadata.styleMap`**, they differ: an absent flag means the style says nothing
651
+ > about that property (so it inherits), while `false` means the style explicitly turns it off
652
+ > (ODF's `fo:font-weight="normal"`, DOCX's `<w:b w:val="0"/>`). Code resolving inheritance itself
653
+ > must test `=== undefined`, not truthiness, or it will treat "explicitly off" as "unspecified".
654
+
655
+ ### 6. Break Nodes (DOCX and ODF)
612
656
 
613
657
  When `includeBreakNodes: true`, break elements appear as nodes:
614
658
 
@@ -623,6 +667,43 @@ Break Node (type: 'break')
623
667
  > [!NOTE]
624
668
  > Break nodes have no `text` property, but `ast.toText()` and `ast.to('text')` automatically convert them to the configured newline delimiter.
625
669
 
670
+ > [!NOTE]
671
+ > DOCX writes breaks inline (`w:br`/`w:cr`), so they land as children of the paragraph. ODF instead
672
+ > carries page and column breaks on the paragraph *style* (`fo:break-before`/`fo:break-after`), so those
673
+ > are emitted as siblings around the paragraph rather than inside it. `<text:soft-page-break/>` maps onto
674
+ > `lastRenderedPage`, the same type as DOCX's `w:lastRenderedPageBreak`.
675
+
676
+ ### 6b. Equations
677
+
678
+ Equations are extracted from every format that can carry them and normalized to **LaTeX**, so a
679
+ formula means the same thing whichever format it arrived in:
680
+
681
+ | Source format | Markup in the file |
682
+ |---|---|
683
+ | DOCX, PPTX | OOXML `<m:oMath>` / `<m:oMathPara>` |
684
+ | ODT, ODP, ODS | MathML inside the embedded formula object |
685
+ | HTML, EPUB | native MathML `<math>` |
686
+ | Markdown | `$inline$` / `$$block$$` |
687
+
688
+ They all land as the same node:
689
+
690
+ ```text
691
+ Code Node (type: 'code')
692
+ ├── text: '\\frac{1}{2}' // LaTeX, whatever the source markup was
693
+ └── metadata: { math: 'inline' | 'block' }
694
+ ```
695
+
696
+ Fractions, sub/superscripts, radicals, delimiters, n-ary operators (sums, integrals), named
697
+ functions, accents, bars, matrices and math alphabets (`ℝ`, `𝒜`, …) are all preserved. When a
698
+ document supplies its own TeX source in an `<annotation encoding="application/x-tex">`, that is
699
+ used verbatim in preference to anything reconstructed from the presentation markup.
700
+
701
+ > [!NOTE]
702
+ > Equation text is *structure*, not prose: a fraction whose numerator and denominator are simply
703
+ > concatenated reads as a different number rather than as obviously-missing content. Consumers that
704
+ > index document text should treat `code` nodes carrying `math` as opaque LaTeX rather than
705
+ > splitting them as words.
706
+
626
707
  ### 7. Document Metadata
627
708
 
628
709
  ```ts
@@ -688,7 +769,7 @@ idempotent and `.md → AST → HTML → AST → .md` survives unchanged.
688
769
  | Attribute lists | `![alt](img.png){width=50% .centered}` | `ImageMetadata.width` / `.align`, `TableMetadata.align` |
689
770
  | Citations | `[@smith2024]` | `TextMetadata.citationKey` |
690
771
  | Wikilinks | `[[Page]]` / `[[Page\|Alias]]` | `TextMetadata.wikilink`, `.link`, `.linkType` |
691
- | Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `TextMetadata.math` (`'inline' \| 'block'`) |
772
+ | Inline/block math | `$E=mc^2$` / `` $$...$$ `` | `type: 'code'`, `CodeMetadata.math` (`'inline' \| 'block'`) |
692
773
  | Frontmatter arrays | `tags: [a, b]` or `tags: ["a","b"]` | Real array in `metadata.customProperties`/`nativeProperties` |
693
774
  | MDX components (import-only) | `<Component prop="x">...</Component>` | Stripped; inner Markdown is kept. Never generated back. |
694
775
 
@@ -888,7 +969,7 @@ Pass as the second argument to `parseOffice(file, config)`.
888
969
  | `includeRawContent` | `boolean` | `false` | Attach raw XML/RTF source to each node |
889
970
  | `serializeRawContent` | `boolean` | `true` | Re-serialize XML to clean strings (only if `includeRawContent: true`) |
890
971
  | `preserveXmlWhitespace` | `boolean` | `false` | Preserve original XML whitespace during serialization |
891
- | `includeBreakNodes` | `boolean` | `false` | Include `w:br` / `w:cr` as typed break nodes (DOCX only) |
972
+ | `includeBreakNodes` | `boolean` | `false` | Include typed break nodes: DOCX `w:br`/`w:cr`, ODF `fo:break-before`/`fo:break-after` and `text:soft-page-break` |
892
973
  | `ignoreInternalLinks` | `boolean` | `false` | Strip bookmarks and internal cross-references from AST |
893
974
  | `fileType` | `SupportedFileType \| null` | `null` | **Required for text-based binary data** (`'md'`, `'html'`, `'csv'`) as these lack magic bytes. |
894
975
  | `csvDelimiter` | `string` | `','` | Input delimiter when parsing CSV files |
@@ -55,6 +55,32 @@ const envUtils_js_1 = require("./utils/envUtils.js");
55
55
  const errorUtils_js_1 = require("./utils/errorUtils.js");
56
56
  const moduleLoader_js_1 = require("./utils/moduleLoader.js");
57
57
  const ocrUtils_js_1 = require("./utils/ocrUtils.js");
58
+ const zipUtils_js_1 = require("./utils/zipUtils.js");
59
+ /** What magic-byte sniffing reports for an archive it could not identify further. */
60
+ const GENERIC_ZIP_EXTENSION = 'zip';
61
+ /** The formats that are ZIP archives, and so cannot be contradicted by a bare `zip` result. */
62
+ const ZIP_BACKED_FILE_TYPES = new Set(['docx', 'xlsx', 'pptx', 'odt', 'ods', 'odp', 'epub']);
63
+ /**
64
+ * Upgrades a magic-byte result of `zip` (or none at all) into the specific office format the
65
+ * archive declares, by reading that declaration from inside the archive.
66
+ *
67
+ * Byte sniffing identifies an OOXML package by parsing `[Content_Types].xml`, but it walks the
68
+ * archive under fixed budgets and reports a plain `zip` when it runs out before finding that
69
+ * part. Since `zip` is not a format this library parses, a valid document then failed as an
70
+ * unsupported file type. Our own reader has no such budget, so it settles the question whenever
71
+ * sniffing is inconclusive.
72
+ *
73
+ * @param detected - What magic-byte sniffing reported, if anything
74
+ * @param buffer - The file content
75
+ * @param config - Resolved parser configuration, for its decompression limits
76
+ * @returns The resolved type, the original detection when nothing better is found, or undefined
77
+ */
78
+ const resolveZipBackedType = async (detected, buffer, config) => {
79
+ if (detected && detected !== GENERIC_ZIP_EXTENSION)
80
+ return detected;
81
+ const resolved = await (0, zipUtils_js_1.detectOfficeTypeFromZip)(buffer, config.decompressionLimits ?? {});
82
+ return resolved ?? detected;
83
+ };
58
84
  /**
59
85
  * Main parser class providing office document parsing functionality.
60
86
  *
@@ -168,15 +194,16 @@ class OfficeParser {
168
194
  // This matches v6 behavior and prevents crashes in older Node environments
169
195
  // where file-type 22.x might be incompatible.
170
196
  if (buffer.length > 0 && !ext) {
197
+ let detected;
171
198
  try {
172
199
  const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
173
200
  const type = await fileTypeFromBuffer(buffer);
174
201
  if (type) {
175
- ext = type.ext;
202
+ detected = type.ext;
176
203
  }
177
204
  else {
178
205
  // If no extension could be detected and none was provided,
179
- // it might be a text-based format (csv, md, html) which
206
+ // it might be a text-based format (csv, md, html) which
180
207
  // lack magic bytes. We'll let the switch default handle it.
181
208
  }
182
209
  }
@@ -184,16 +211,28 @@ class OfficeParser {
184
211
  // Log warning but don't crash; the switch below will handle unsupported/missing ext
185
212
  (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED, internalConfig, { error });
186
213
  }
214
+ ext = await resolveZipBackedType(detected, buffer, internalConfig) ?? '';
187
215
  }
188
216
  else if (buffer.length > 0 && ext) {
189
- // If extension is known, we can optionally verify it, but we wrap it
217
+ // If extension is known, we can optionally verify it, but we wrap it
190
218
  // in a try-catch to avoid breaking Node 18 if file-type fails to load.
191
219
  try {
192
220
  const { fileTypeFromBuffer } = await (0, moduleLoader_js_1.loadFileType)();
193
221
  const type = await fileTypeFromBuffer(buffer);
194
- if (type && type.ext.toLowerCase() !== ext.toLowerCase()) {
222
+ // A bare `zip` cannot contradict a caller who already said "this is a
223
+ // docx", so there is nothing a closer look could add. Skipping it keeps an
224
+ // explicit fileType the cheapest route, rather than making it pay for an
225
+ // archive scan that exists only to decide whether to warn.
226
+ const worthResolving = !(type?.ext === GENERIC_ZIP_EXTENSION && ZIP_BACKED_FILE_TYPES.has(ext.toLowerCase()));
227
+ const detected = worthResolving
228
+ ? await resolveZipBackedType(type?.ext, buffer, internalConfig)
229
+ : type?.ext;
230
+ // A bare `zip` says only that the bytes are an archive, which every format
231
+ // on this path already is. Reporting it as a mismatch against the caller's
232
+ // own extension is noise, so only a resolved format is worth comparing.
233
+ if (detected && detected !== GENERIC_ZIP_EXTENSION && detected.toLowerCase() !== ext.toLowerCase()) {
195
234
  // Mismatch found between authoritative extension and detected content
196
- (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected: type.ext, expected: ext });
235
+ (0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH, internalConfig, { detected, expected: ext });
197
236
  }
198
237
  }
199
238
  catch (error) {
@@ -218,7 +257,16 @@ class OfficeParser {
218
257
  case 'odt':
219
258
  case 'odp':
220
259
  case 'ods':
221
- result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, internalConfig);
260
+ // The three ODF types share one parser, which needs to know which of them
261
+ // it is looking at. It normally reads that from the archive's mimetype
262
+ // entry; passing the resolved type along gives it something accurate to
263
+ // fall back on when that entry is missing.
264
+ //
265
+ // Overridden on a copy rather than on internalConfig: resolveParserConfig
266
+ // returns an already-complete config by reference, so writing to it would
267
+ // pin the caller's own object to this file's type and misroute every later
268
+ // parse that reused it.
269
+ result = await (0, OpenOfficeParser_js_1.parseOpenOffice)(buffer, { ...internalConfig, fileType: ext.toLowerCase() });
222
270
  break;
223
271
  case 'pdf':
224
272
  result = await (0, PdfParser_js_1.parsePdf)(buffer, internalConfig);
@@ -81,6 +81,21 @@ export declare abstract class BaseGenerator<D extends UniversalGeneratorFormat =
81
81
  * hasn't already been claimed; otherwise a sequential counter guarantees uniqueness.
82
82
  */
83
83
  protected getFootnoteKey(note: OfficeContentNode): string;
84
+ /**
85
+ * True when every content-bearing text descendant satisfies `test` - i.e. the property is
86
+ * uniform across the whole node and therefore says nothing the node type does not already say.
87
+ *
88
+ * Used to decide whether a heading's or header row's inherited formatting can be dropped. The
89
+ * distinction matters: an ODF heading whose paragraph style is bold and 14pt yields a heading
90
+ * where *every* run is bold and 14pt, and re-emitting that gives `# **Heading**` in Markdown
91
+ * and, worse in RTF/HTML, an inner font-size that overrides the heading's own and visibly
92
+ * shrinks it. But `# Normal **Bold** Normal` is an author contrasting one word against the
93
+ * rest, and dropping that would discard real meaning. Only the uniform case is safe.
94
+ *
95
+ * Returns false when there is no text to judge, so an empty or image-only node never triggers
96
+ * suppression.
97
+ */
98
+ protected hasUniformFormatting(node: OfficeContentNode, test: (formatting: OfficeContentNode['formatting']) => boolean): boolean;
84
99
  /**
85
100
  * Recursively extracts plain text from a node and its children.
86
101
  */
@@ -187,6 +187,37 @@ class BaseGenerator {
187
187
  this.noteFootnoteKeys.set(note, key);
188
188
  return key;
189
189
  }
190
+ /**
191
+ * True when every content-bearing text descendant satisfies `test` - i.e. the property is
192
+ * uniform across the whole node and therefore says nothing the node type does not already say.
193
+ *
194
+ * Used to decide whether a heading's or header row's inherited formatting can be dropped. The
195
+ * distinction matters: an ODF heading whose paragraph style is bold and 14pt yields a heading
196
+ * where *every* run is bold and 14pt, and re-emitting that gives `# **Heading**` in Markdown
197
+ * and, worse in RTF/HTML, an inner font-size that overrides the heading's own and visibly
198
+ * shrinks it. But `# Normal **Bold** Normal` is an author contrasting one word against the
199
+ * rest, and dropping that would discard real meaning. Only the uniform case is safe.
200
+ *
201
+ * Returns false when there is no text to judge, so an empty or image-only node never triggers
202
+ * suppression.
203
+ */
204
+ hasUniformFormatting(node, test) {
205
+ let sawText = false;
206
+ const walk = (n) => {
207
+ if (n.type === 'text') {
208
+ // Whitespace-only runs carry no visible formatting either way, so they neither
209
+ // count as evidence nor veto - otherwise a stray unformatted space between two
210
+ // bold runs would defeat the check on almost every real heading.
211
+ if (!(n.text || '').trim())
212
+ return true;
213
+ sawText = true;
214
+ return test(n.formatting);
215
+ }
216
+ return (n.children ?? []).every(walk);
217
+ };
218
+ const uniform = (node.children ?? []).every(walk);
219
+ return sawText && uniform;
220
+ }
190
221
  /**
191
222
  * Recursively extracts plain text from a node and its children.
192
223
  */
@@ -6,6 +6,14 @@ import { BaseGenerator } from './BaseGenerator.js';
6
6
  export declare class HtmlGenerator extends BaseGenerator<'html'> {
7
7
  private chartCounter;
8
8
  private isSpreadsheetMode;
9
+ /**
10
+ * Set while rendering a heading's children, so `formatText` can drop the run-level bold and
11
+ * font-size the `<hN>` already establishes. See the note there, and the identical flag in
12
+ * `RtfGenerator`, where the same inherited size actively shrinks the heading.
13
+ */
14
+ private inHeading;
15
+ /** As `inHeading`, but for the inherited font size - see `hasUniformFormatting`. */
16
+ private headingUniformSize;
9
17
  constructor(ast: OfficeParserAST, config?: GeneratorConfig<'html'>);
10
18
  /**
11
19
  * Generates HTML string from the provided AST.
@@ -24,6 +32,7 @@ export declare class HtmlGenerator extends BaseGenerator<'html'> {
24
32
  */
25
33
  private tableNestingLevel;
26
34
  protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => string | Promise<string>, override?: string | boolean | void): Promise<string>;
35
+ private processNodeRecursiveInner;
27
36
  /**
28
37
  * Internal processor for individual nodes.
29
38
  */
@@ -88,6 +88,14 @@ function resolveStandalone(standalone) {
88
88
  class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
89
89
  chartCounter = 0;
90
90
  isSpreadsheetMode = false;
91
+ /**
92
+ * Set while rendering a heading's children, so `formatText` can drop the run-level bold and
93
+ * font-size the `<hN>` already establishes. See the note there, and the identical flag in
94
+ * `RtfGenerator`, where the same inherited size actively shrinks the heading.
95
+ */
96
+ inHeading = false;
97
+ /** As `inHeading`, but for the inherited font size - see `hasUniformFormatting`. */
98
+ headingUniformSize = false;
91
99
  constructor(ast, config) {
92
100
  super('html', ast, config);
93
101
  }
@@ -502,6 +510,21 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
502
510
  // method entirely, so without repeating the check here the signal would be silently
503
511
  // inert for this generator - which is exactly how it was missed.
504
512
  (0, errorUtils_js_1.checkAbortSignal)(this.config.abortSignal);
513
+ const wasInHeading = this.inHeading;
514
+ const wasHeadingSize = this.headingUniformSize;
515
+ if (node.type === 'heading') {
516
+ this.inHeading = this.hasUniformFormatting(node, f => f?.bold === true);
517
+ this.headingUniformSize = this.hasUniformFormatting(node, f => !!f?.size);
518
+ }
519
+ try {
520
+ return await this.processNodeRecursiveInner(node, processor, override);
521
+ }
522
+ finally {
523
+ this.inHeading = wasInHeading;
524
+ this.headingUniformSize = wasHeadingSize;
525
+ }
526
+ }
527
+ async processNodeRecursiveInner(node, processor, override) {
505
528
  // Use pre-evaluated override if provided, otherwise call handleOnNode
506
529
  const actualOverride = override !== undefined ? override : await this.handleOnNode(node);
507
530
  // Returning false skips the node and its children
@@ -1076,7 +1099,12 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
1076
1099
  let result = this.escape(text);
1077
1100
  const f = node.formatting;
1078
1101
  if (this.config.includeFormatting && f) {
1079
- if (f.bold)
1102
+ // Inside an `<hN>`, the heading's own styling is authoritative. A run that also carries
1103
+ // bold and a font size - the normal case for ODF, where a heading's paragraph style is
1104
+ // inherited by its runs - would wrap the text in `<b>` the heading already implies and,
1105
+ // worse, in a `<span style="font-size: 14pt">` that *shrinks* the heading to the size
1106
+ // its paragraph style happened to name. See the same suppression in RtfGenerator.
1107
+ if (f.bold && !this.inHeading)
1080
1108
  result = `<b>${result}</b>`;
1081
1109
  if (f.italic)
1082
1110
  result = `<i>${result}</i>`;
@@ -1088,7 +1116,9 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
1088
1116
  result = `<sub>${result}</sub>`;
1089
1117
  if (f.superscript)
1090
1118
  result = `<sup>${result}</sup>`;
1091
- const styles = this.getInlineStyles(node);
1119
+ const styles = this.headingUniformSize
1120
+ ? this.getInlineStyles(node, { skipFontSize: true })
1121
+ : this.getInlineStyles(node);
1092
1122
  if (styles) {
1093
1123
  result = `<span style="${styles}">${result}</span>`;
1094
1124
  }
@@ -1120,7 +1150,7 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
1120
1150
  }
1121
1151
  return result;
1122
1152
  }
1123
- getInlineStyles(node) {
1153
+ getInlineStyles(node, options = {}) {
1124
1154
  const styles = [];
1125
1155
  // Colors/sizes/fonts/alignments are free strings from an untrusted document;
1126
1156
  // run each through sanitizeCssValue so it can't break out of the style="" attribute
@@ -1154,7 +1184,7 @@ class HtmlGenerator extends BaseGenerator_js_1.BaseGenerator {
1154
1184
  pushSafe('color', f.color);
1155
1185
  if (f.backgroundColor)
1156
1186
  pushSafe('background-color', f.backgroundColor);
1157
- if (f.size)
1187
+ if (f.size && !options.skipFontSize)
1158
1188
  pushSafe('font-size', f.size);
1159
1189
  if (f.font) {
1160
1190
  const safeFont = (0, sanitize_js_1.sanitizeCssValue)(f.font);
@@ -31,6 +31,18 @@ import { BaseGenerator } from './BaseGenerator.js';
31
31
  */
32
32
  export declare class MarkdownGenerator extends BaseGenerator<'md'> {
33
33
  private isInsideTable;
34
+ /**
35
+ * Set while rendering the children of a heading, or the cells of a table's header row.
36
+ *
37
+ * Markdown already conveys "this is a heading" with `#` and "this is a header row" with the
38
+ * separator line, so a run inside one that also carries bold - the normal case for ODF, whose
39
+ * heading and header-row paragraph styles are bold and are now inherited by their runs - would
40
+ * render as `# **Heading**` and `| **ITEM** |`. That is redundant rather than wrong, but it
41
+ * also round-trips back into bold text nodes nested inside a heading, so the noise compounds
42
+ * on every parse/generate cycle. Emphasis the node type already implies is dropped; every
43
+ * other formatting flag still comes through.
44
+ */
45
+ private inImplicitBold;
34
46
  private hoistedContent;
35
47
  private collectedAbbreviations;
36
48
  private resolvedDialect;
@@ -72,6 +84,7 @@ export declare class MarkdownGenerator extends BaseGenerator<'md'> {
72
84
  private optimizeNodes;
73
85
  private areFormattingEqual;
74
86
  private renderMarkdownTable;
87
+ private collectNotesFrom;
75
88
  private renderMarkdownTableInternal;
76
89
  private hasNestedTable;
77
90
  private hasColspanOrRowspan;