officeparser 7.4.0 → 7.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/README.md +85 -4
  2. package/dist/OfficeParser.js +54 -6
  3. package/dist/generators/BaseGenerator.d.ts +15 -0
  4. package/dist/generators/BaseGenerator.js +31 -0
  5. package/dist/generators/HtmlGenerator.d.ts +9 -0
  6. package/dist/generators/HtmlGenerator.js +34 -4
  7. package/dist/generators/MarkdownGenerator.d.ts +13 -0
  8. package/dist/generators/MarkdownGenerator.js +115 -41
  9. package/dist/generators/RtfGenerator.d.ts +13 -0
  10. package/dist/generators/RtfGenerator.js +23 -2
  11. package/dist/index.d.ts +2 -2
  12. package/dist/officeparser.browser.d.ts +34 -1
  13. package/dist/officeparser.browser.iife.js +160 -160
  14. package/dist/officeparser.browser.mjs +198 -198
  15. package/dist/officeparser.browser.slim.d.ts +34 -1
  16. package/dist/officeparser.browser.slim.iife.js +186 -186
  17. package/dist/officeparser.browser.slim.mjs +186 -186
  18. package/dist/parsers/EpubParser.js +2 -2
  19. package/dist/parsers/ExcelParser.js +11 -7
  20. package/dist/parsers/HtmlParser.js +39 -0
  21. package/dist/parsers/OpenOfficeParser.js +139 -167
  22. package/dist/parsers/PowerPointParser.js +48 -11
  23. package/dist/parsers/WordParser.js +33 -7
  24. package/dist/sbom.cdx.json +92 -92
  25. package/dist/types.d.ts +34 -1
  26. package/dist/types.js +10 -0
  27. package/dist/utils/configUtils.d.ts +15 -2
  28. package/dist/utils/configUtils.js +58 -13
  29. package/dist/utils/errorUtils.d.ts +8 -2
  30. package/dist/utils/errorUtils.js +23 -1
  31. package/dist/utils/mathUtils.d.ts +42 -0
  32. package/dist/utils/mathUtils.js +385 -0
  33. package/dist/utils/zipUtils.d.ts +64 -4
  34. package/dist/utils/zipUtils.js +188 -4
  35. package/package.json +9 -5
@@ -113,6 +113,18 @@ function resolveFallbackToHtml(fallbackToHtml) {
113
113
  */
114
114
  class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
115
115
  isInsideTable = false;
116
+ /**
117
+ * Set while rendering the children of a heading, or the cells of a table's header row.
118
+ *
119
+ * Markdown already conveys "this is a heading" with `#` and "this is a header row" with the
120
+ * separator line, so a run inside one that also carries bold - the normal case for ODF, whose
121
+ * heading and header-row paragraph styles are bold and are now inherited by their runs - would
122
+ * render as `# **Heading**` and `| **ITEM** |`. That is redundant rather than wrong, but it
123
+ * also round-trips back into bold text nodes nested inside a heading, so the noise compounds
124
+ * on every parse/generate cycle. Emphasis the node type already implies is dropped; every
125
+ * other formatting flag still comes through.
126
+ */
127
+ inImplicitBold = false;
116
128
  hoistedContent = [];
117
129
  collectedAbbreviations = new Map();
118
130
  resolvedDialect;
@@ -245,7 +257,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
245
257
  let text = (0, sanitize_js_1.markdownEscapeText)(node.text || '');
246
258
  if (this.config.includeFormatting && node.formatting) {
247
259
  const emphasisAsterisk = this.resolvedDialect.emphasisMarker === 'asterisk';
248
- if (node.formatting.bold)
260
+ if (node.formatting.bold && !this.inImplicitBold)
249
261
  text = emphasisAsterisk ? `**${text}**` : `__${text}__`;
250
262
  if (node.formatting.italic)
251
263
  text = emphasisAsterisk ? `*${text}*` : `_${text}_`;
@@ -544,7 +556,12 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
544
556
  if (!this.resolvedDialect.definitionLists)
545
557
  return `${childrenOutput}\n\n`;
546
558
  return `: ${childrenOutput}\n`;
547
- default:
559
+ case 'chart':
560
+ case 'drawing':
561
+ case 'comment':
562
+ case 'header':
563
+ case 'footer':
564
+ case 'slideMaster':
548
565
  return childrenOutput;
549
566
  }
550
567
  };
@@ -612,14 +629,19 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
612
629
  if (typeof override === 'string') {
613
630
  return override;
614
631
  }
632
+ const walkedByProcessor = node.type === 'table' || node.type === 'sheet';
633
+ const wasInImplicitBold = this.inImplicitBold;
634
+ if (node.type === 'heading' && this.hasUniformFormatting(node, f => f?.bold === true))
635
+ this.inImplicitBold = true;
615
636
  let childrenOutput = '';
616
- if (node.children && node.children.length > 0) {
637
+ if (!walkedByProcessor && node.children && node.children.length > 0) {
617
638
  // Optimization: Merge adjacent text nodes with identical formatting
618
639
  const optimizedChildren = this.optimizeNodes(node.children);
619
640
  for (const child of optimizedChildren) {
620
641
  childrenOutput += await this.processNodeRecursive(child, processor);
621
642
  }
622
643
  }
644
+ this.inImplicitBold = wasInImplicitBold;
623
645
  // When the dialect has no footnote syntax, a footnote/endnote is inlined right at its
624
646
  // reference point instead (see below) - so it must not also be collected into the
625
647
  // end-of-document "### Notes" section, or its content would be duplicated.
@@ -627,11 +649,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
627
649
  const meta = note.metadata;
628
650
  return (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') && !this.resolvedDialect.footnotes;
629
651
  };
630
- if (node.notes && node.notes.length > 0) {
631
- if (node.type !== 'slide') {
632
- this.collectedNotes.push(...node.notes.filter(note => !isInlinedFootnote(note)));
633
- }
634
- }
652
+ this.collectNotesFrom(node);
635
653
  let result = await processor(node, childrenOutput);
636
654
  if (node.type === 'slide' && node.notes && node.notes.length > 0) {
637
655
  for (const note of node.notes) {
@@ -732,46 +750,73 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
732
750
  this.isInsideTable = false;
733
751
  return result;
734
752
  }
753
+ collectNotesFrom(node) {
754
+ if (!node.notes || node.notes.length === 0)
755
+ return;
756
+ if (node.type === 'slide')
757
+ return;
758
+ const isInlinedFootnote = (note) => {
759
+ const meta = note.metadata;
760
+ return (meta?.noteType === 'footnote' || meta?.noteType === 'endnote') && !this.resolvedDialect.footnotes;
761
+ };
762
+ this.collectedNotes.push(...node.notes.filter(note => !isInlinedFootnote(note)));
763
+ }
735
764
  async renderMarkdownTableInternal(node, processor) {
736
765
  let tableOutput = '';
737
766
  let maxCols = 0;
738
767
  // First pass: Process rows and determine max columns (accounting for colspans)
739
768
  const processedRows = [];
740
769
  for (const rowNode of (node.children ?? [])) {
741
- const override = await this.handleOnNode(rowNode);
742
- if (override === false)
743
- continue;
744
- if (typeof override === 'string') {
745
- processedRows.push([override]);
746
- continue;
747
- }
748
- const rowCells = [];
749
- let lastCol = -1;
750
- if (rowNode.children) {
751
- const cellNodes = rowNode.children.filter(c => c.type === 'cell');
752
- for (const cellNode of cellNodes) {
753
- const currentCol = cellNode.metadata?.col ?? (lastCol + 1);
754
- // Fill gaps with empty cells
755
- while (lastCol < currentCol - 1) {
756
- rowCells.push(' ');
757
- lastCol++;
758
- }
759
- // Process cell content
760
- let cellContent = await this.processNodeRecursive(cellNode, processor);
761
- // Use <br> fallback only if allowed, otherwise space
762
- const br = this.resolvedFallbackToHtml.cellLineBreaks ? '<br>' : ' ';
763
- cellContent = cellContent.trim().replace(/\n+/g, br).replace(/\|/g, '\\|');
764
- rowCells.push(cellContent);
765
- // Handle colspan by adding empty cells
766
- const colSpan = cellNode.metadata?.colSpan || 1;
767
- for (let i = 1; i < colSpan; i++) {
768
- rowCells.push(' ');
770
+ // The first row becomes the header row - the `| --- |` separator emitted below marks
771
+ // it as such - so bold inside it is already implied.
772
+ const wasInImplicitBold = this.inImplicitBold;
773
+ if (processedRows.length === 0 && this.hasUniformFormatting(rowNode, f => f?.bold === true))
774
+ this.inImplicitBold = true;
775
+ try {
776
+ const override = await this.handleOnNode(rowNode);
777
+ if (override === false)
778
+ continue;
779
+ if (typeof override === 'string') {
780
+ processedRows.push([override]);
781
+ continue;
782
+ }
783
+ // After the override checks, not before: a row the caller skipped via `onNode` must not
784
+ // still contribute its footnote to the end-of-document Notes section, where it would
785
+ // appear with no `[^id]` marker anywhere in the document pointing at it.
786
+ // `renderTableAsHtml` gets this right by returning early, so collecting here keeps the
787
+ // pipe and HTML paths agreeing on what a skipped row means.
788
+ this.collectNotesFrom(rowNode);
789
+ const rowCells = [];
790
+ let lastCol = -1;
791
+ if (rowNode.children) {
792
+ const cellNodes = rowNode.children.filter(c => c.type === 'cell');
793
+ for (const cellNode of cellNodes) {
794
+ const currentCol = cellNode.metadata?.col ?? (lastCol + 1);
795
+ // Fill gaps with empty cells
796
+ while (lastCol < currentCol - 1) {
797
+ rowCells.push(' ');
798
+ lastCol++;
799
+ }
800
+ // Process cell content
801
+ let cellContent = await this.processNodeRecursive(cellNode, processor);
802
+ // Use <br> fallback only if allowed, otherwise space
803
+ const br = this.resolvedFallbackToHtml.cellLineBreaks ? '<br>' : ' ';
804
+ cellContent = cellContent.trim().replace(/\n+/g, br).replace(/\|/g, '\\|');
805
+ rowCells.push(cellContent);
806
+ // Handle colspan by adding empty cells
807
+ const colSpan = cellNode.metadata?.colSpan || 1;
808
+ for (let i = 1; i < colSpan; i++) {
809
+ rowCells.push(' ');
810
+ }
811
+ lastCol = currentCol + colSpan - 1;
769
812
  }
770
- lastCol = currentCol + colSpan - 1;
771
813
  }
814
+ processedRows.push(rowCells);
815
+ maxCols = Math.max(maxCols, rowCells.length);
816
+ }
817
+ finally {
818
+ this.inImplicitBold = wasInImplicitBold;
772
819
  }
773
- processedRows.push(rowCells);
774
- maxCols = Math.max(maxCols, rowCells.length);
775
820
  }
776
821
  // Second pass: Build table string with separator
777
822
  for (let i = 0; i < processedRows.length; i++) {
@@ -842,6 +887,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
842
887
  return `<table${alignAttr}>\n${rows}</table>\n`;
843
888
  }
844
889
  else if (node.type === 'row') {
890
+ this.collectNotesFrom(node);
845
891
  let cells = '';
846
892
  if (node.children) {
847
893
  for (const cell of node.children) {
@@ -851,6 +897,7 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
851
897
  return ` <tr>\n${cells} </tr>\n`;
852
898
  }
853
899
  else if (node.type === 'cell') {
900
+ this.collectNotesFrom(node);
854
901
  const meta = node.metadata;
855
902
  const rs = meta?.rowSpan > 1 ? ` rowspan="${meta.rowSpan}"` : '';
856
903
  const cs = meta?.colSpan > 1 ? ` colspan="${meta.colSpan}"` : '';
@@ -861,10 +908,16 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
861
908
  content += await this.processNodeRecursive(child, async (n, co) => {
862
909
  switch (n.type) {
863
910
  case 'text': {
911
+ if (n.metadata) {
912
+ const m = n.metadata;
913
+ if (m.abbreviationTitle) {
914
+ this.collectedAbbreviations.set(n.text || '', m.abbreviationTitle);
915
+ }
916
+ }
864
917
  // Inside HTML table cells, entity-encode angle brackets so cell
865
918
  // text can't inject a raw tag (e.g. </td><script>).
866
919
  let text = (0, sanitize_js_1.markdownEscapeText)(n.text || '');
867
- if (n.formatting?.bold)
920
+ if (n.formatting?.bold && !this.inImplicitBold)
868
921
  text = `<b>${text}</b>`;
869
922
  if (n.formatting?.italic)
870
923
  text = `<i>${text}</i>`;
@@ -882,7 +935,28 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
882
935
  return `<h${level}>${co}</h${level}>`;
883
936
  }
884
937
  case 'table': return await this.renderTableAsHtml(n);
885
- default: return co;
938
+ case 'list':
939
+ case 'image':
940
+ case 'chart':
941
+ case 'drawing':
942
+ case 'slide':
943
+ case 'note':
944
+ case 'sheet':
945
+ case 'row':
946
+ case 'cell':
947
+ case 'page':
948
+ case 'break':
949
+ case 'code':
950
+ case 'comment':
951
+ case 'header':
952
+ case 'footer':
953
+ case 'slideMaster':
954
+ case 'embed':
955
+ case 'admonition':
956
+ case 'definitionList':
957
+ case 'definitionTerm':
958
+ case 'definitionDescription':
959
+ return co;
886
960
  }
887
961
  });
888
962
  }
@@ -6,6 +6,19 @@ import { BaseGenerator } from './BaseGenerator.js';
6
6
  export declare class RtfGenerator extends BaseGenerator<'rtf'> {
7
7
  private colorTable;
8
8
  private inTable;
9
+ /**
10
+ * Set while rendering a heading's children.
11
+ *
12
+ * A heading emits its own `{\\b\\fs44 ...}` wrapper, so a run inside it that also carries bold
13
+ * and a size - which is now the normal case for ODF, where the heading's paragraph style is
14
+ * inherited by its runs - would emit a nested `\\fs28` that *overrides* the outer `\\fs44`.
15
+ * The heading then renders at the body-text size it was styled with rather than at heading
16
+ * size. Suppressing the inherited weight and size inside a heading keeps the heading's own
17
+ * wrapper authoritative; every other property (colour, font) still comes through.
18
+ */
19
+ private inHeading;
20
+ /** As `inHeading`, but for the inherited font size - see `hasUniformFormatting`. */
21
+ private headingUniformSize;
9
22
  constructor(ast: OfficeParserAST, config?: GeneratorConfig<'rtf'>);
10
23
  generate(): Promise<ConversionResult<'rtf'>>;
11
24
  protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => Promise<string>): Promise<string>;
@@ -10,6 +10,19 @@ const errorUtils_js_1 = require("../utils/errorUtils.js");
10
10
  class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
11
11
  colorTable = [];
12
12
  inTable = false;
13
+ /**
14
+ * Set while rendering a heading's children.
15
+ *
16
+ * A heading emits its own `{\\b\\fs44 ...}` wrapper, so a run inside it that also carries bold
17
+ * and a size - which is now the normal case for ODF, where the heading's paragraph style is
18
+ * inherited by its runs - would emit a nested `\\fs28` that *overrides* the outer `\\fs44`.
19
+ * The heading then renders at the body-text size it was styled with rather than at heading
20
+ * size. Suppressing the inherited weight and size inside a heading keeps the heading's own
21
+ * wrapper authoritative; every other property (colour, font) still comes through.
22
+ */
23
+ inHeading = false;
24
+ /** As `inHeading`, but for the inherited font size - see `hasUniformFormatting`. */
25
+ headingUniformSize = false;
13
26
  constructor(ast, config) {
14
27
  super('rtf', ast, config);
15
28
  }
@@ -68,8 +81,16 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
68
81
  const wasInTable = this.inTable;
69
82
  if (node.type === 'table')
70
83
  this.inTable = true;
84
+ const wasInHeading = this.inHeading;
85
+ const wasHeadingSize = this.headingUniformSize;
86
+ if (node.type === 'heading') {
87
+ this.inHeading = this.hasUniformFormatting(node, f => f?.bold === true);
88
+ this.headingUniformSize = this.hasUniformFormatting(node, f => !!f?.size);
89
+ }
71
90
  const result = await super.processNodeRecursive(node, processor);
72
91
  this.inTable = wasInTable;
92
+ this.inHeading = wasInHeading;
93
+ this.headingUniformSize = wasHeadingSize;
73
94
  return result;
74
95
  }
75
96
  async renderBody(ast) {
@@ -98,7 +119,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
98
119
  if (this.config.includeFormatting && f) {
99
120
  let prefix = '';
100
121
  let suffix = '';
101
- if (f.bold) {
122
+ if (f.bold && !this.inHeading) {
102
123
  prefix += '\\b ';
103
124
  suffix = '\\b0 ' + suffix;
104
125
  }
@@ -127,7 +148,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
127
148
  const idx = this.getColorIndex(f.backgroundColor);
128
149
  prefix += `\\highlight${idx + 1} `;
129
150
  }
130
- if (f.size) {
151
+ if (f.size && !this.headingUniformSize) {
131
152
  let pt = 12; // default
132
153
  const val = parseFloat(f.size);
133
154
  if (!isNaN(val)) {
package/dist/index.d.ts CHANGED
@@ -51,10 +51,10 @@
51
51
  import { OfficeParser } from './OfficeParser.js';
52
52
  import { OfficeGenerator } from './OfficeGenerator.js';
53
53
  import { OfficeConverter } from './OfficeConverter.js';
54
- import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, ConversionResult } from './types.js';
54
+ import { OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverterConfig, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult } from './types.js';
55
55
  declare const parseOffice: typeof OfficeParser.parseOffice;
56
56
  declare const terminateOcr: typeof OfficeParser.terminateOcr;
57
57
  declare const convert: typeof OfficeConverter.convert;
58
58
  declare const generate: typeof OfficeGenerator.generate;
59
- export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, ConversionResult, };
59
+ export { OfficeParser, parseOffice, terminateOcr, OfficeParserConfig, OfficeParserAST, OfficeContentNode, OfficeAttachment, OfficeMetadata, TextFormatting, SupportedFileType, OfficeContentNodeType, OfficeMimeType, SlideMetadata, SheetMetadata, HeadingMetadata, ListMetadata, CellMetadata, ImageMetadata, PageMetadata, ContentMetadata, BreakMetadata, OfficeGenerator, GeneratorConfig, SupportedDestination, UniversalGeneratorFormat, ChunkingConfig, ChunkingStrategy, FixedSizeChunkingConfig, DocumentStructureChunkingConfig, SemanticChunkingConfig, OfficeChunk, OfficeConverter, OfficeConverterConfig, convert, generate, OfficeErrorType, OfficeWarningType, OfficeError, ConversionResult, };
60
60
  export default OfficeParser;
@@ -41,6 +41,12 @@ export declare enum OfficeErrorType {
41
41
  ZIP_ENTRY_INVALID_SIZE = "ZIP_ENTRY_INVALID_SIZE",
42
42
  /** ZIP uncompressed size limit exceeded */
43
43
  ZIP_SIZE_LIMIT_EXCEEDED = "ZIP_SIZE_LIMIT_EXCEEDED",
44
+ /** ZIP data yielded no readable entries (corrupt, truncated, or not a ZIP archive) */
45
+ ZIP_NO_ENTRIES_FOUND = "ZIP_NO_ENTRIES_FOUND",
46
+ /** ZIP data is truncated: the End of Central Directory record is absent */
47
+ ZIP_TRUNCATED = "ZIP_TRUNCATED",
48
+ /** A readable ZIP archive is missing the part its document format requires */
49
+ REQUIRED_PART_MISSING = "REQUIRED_PART_MISSING",
44
50
  /** Document element/structure nesting exceeded the safe recursion depth */
45
51
  MAX_NESTING_DEPTH_EXCEEDED = "MAX_NESTING_DEPTH_EXCEEDED",
46
52
  /** Embedding call timed out */
@@ -90,7 +96,11 @@ export declare enum OfficeWarningType {
90
96
  /** A metadata override could not be represented in the destination format's vocabulary */
91
97
  METADATA_NOT_REPRESENTABLE = "METADATA_NOT_REPRESENTABLE",
92
98
  /** A styleMap output.tag was not an allowed element name and was ignored */
93
- INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG"
99
+ INVALID_STYLE_MAP_TAG = "INVALID_STYLE_MAP_TAG",
100
+ /** A workbook archive contains no worksheet parts (chartsheet-only workbooks are legitimate) */
101
+ NO_WORKSHEETS_FOUND = "NO_WORKSHEETS_FOUND",
102
+ /** A presentation archive contains no slides (a zero-slide presentation is legitimate) */
103
+ NO_SLIDES_FOUND = "NO_SLIDES_FOUND"
94
104
  }
95
105
  /**
96
106
  * Consolidated timeout settings for OCR operations.
@@ -463,6 +473,29 @@ export interface OfficeIssue {
463
473
  /** Optional additional context or original error object. */
464
474
  details?: any;
465
475
  }
476
+ /**
477
+ * An Error thrown by OfficeParser, carrying the structured issue that produced it.
478
+ *
479
+ * Catching code can branch on `error.officeIssue.code`, the same stable enum used for warnings,
480
+ * instead of matching against message text. Errors that originate outside the library (and
481
+ * `AbortError`, which is deliberately re-thrown untouched so cancellation stays detectable via
482
+ * `error.name`) do not carry this property, hence the optional marker.
483
+ *
484
+ * @example
485
+ * ```typescript
486
+ * try {
487
+ * await parseOffice(buffer, { fileType: 'docx' });
488
+ * } catch (err) {
489
+ * if ((err as OfficeError).officeIssue?.code === OfficeErrorType.REQUIRED_PART_MISSING) {
490
+ * // the archive is readable, but it is not a docx
491
+ * }
492
+ * }
493
+ * ```
494
+ */
495
+ export interface OfficeError extends Error {
496
+ /** The structured issue this error was created from. */
497
+ officeIssue?: OfficeIssue;
498
+ }
466
499
  /**
467
500
  * The result of a document conversion operation.
468
501
  */