officeparser 7.1.0 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. package/README.md +93 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/cli.d.ts +4 -0
  5. package/dist/cli.js +12 -3
  6. package/dist/defaults.js +11 -0
  7. package/dist/generators/BaseGenerator.d.ts +3 -3
  8. package/dist/generators/ChunkingGenerator.js +8 -1
  9. package/dist/generators/CsvGenerator.d.ts +1 -1
  10. package/dist/generators/HtmlGenerator.d.ts +2 -1
  11. package/dist/generators/HtmlGenerator.js +462 -40
  12. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  13. package/dist/generators/MarkdownGenerator.js +3 -1
  14. package/dist/generators/PdfGenerator.d.ts +1 -1
  15. package/dist/generators/PdfGenerator.js +0 -6
  16. package/dist/generators/RtfGenerator.d.ts +2 -1
  17. package/dist/generators/RtfGenerator.js +43 -6
  18. package/dist/generators/TextGenerator.d.ts +1 -1
  19. package/dist/officeparser.browser.d.ts +259 -52
  20. package/dist/officeparser.browser.iife.js +380 -93
  21. package/dist/officeparser.browser.mjs +380 -93
  22. package/dist/parsers/CsvParser.js +1 -1
  23. package/dist/parsers/ExcelParser.js +63 -19
  24. package/dist/parsers/HtmlParser.js +10 -1
  25. package/dist/parsers/MarkdownParser.js +13 -10
  26. package/dist/parsers/OpenOfficeParser.js +57 -34
  27. package/dist/parsers/PdfParser.js +23 -1
  28. package/dist/parsers/PowerPointParser.js +164 -40
  29. package/dist/parsers/RtfParser.js +28 -24
  30. package/dist/parsers/WordParser.js +154 -11
  31. package/dist/sbom.cdx.json +100 -100
  32. package/dist/types.d.ts +265 -52
  33. package/dist/types.js +2 -0
  34. package/dist/utils/astUtils.d.ts +2 -2
  35. package/dist/utils/astUtils.js +2 -1
  36. package/dist/utils/configUtils.d.ts +5 -0
  37. package/dist/utils/configUtils.js +55 -1
  38. package/dist/utils/errorUtils.js +2 -1
  39. package/dist/utils/xmlUtils.d.ts +9 -0
  40. package/dist/utils/xmlUtils.js +53 -1
  41. package/package.json +1 -1
@@ -37,7 +37,7 @@ export declare class MarkdownGenerator extends BaseGenerator<'md'> {
37
37
  *
38
38
  * @returns A Markdown string
39
39
  */
40
- generate(): Promise<ConversionResult>;
40
+ generate(): Promise<ConversionResult<'md'>>;
41
41
  /**
42
42
  * Recursively processes nodes and builds output.
43
43
  * Overridden to provide AST optimization (merging adjacent text nodes).
@@ -136,7 +136,9 @@ class MarkdownGenerator extends BaseGenerator_js_1.BaseGenerator {
136
136
  else if (this.config.generateIds) {
137
137
  id = ` {#${this.slugify(this.getNodeText(node))}}`;
138
138
  }
139
- const anchors = remainingAnchors.map(aid => `<a name="${aid}"></a>`).join('');
139
+ const anchors = this.config.mdConfig.fallbackToHtml
140
+ ? remainingAnchors.map(aid => `<a name="${aid}"></a>`).join('')
141
+ : '';
140
142
  let content = `${prefix}${childrenOutput}${id}`;
141
143
  // Alignment fallback via HTML div/p
142
144
  if (this.config.mdConfig.fallbackToHtml && meta?.alignment && meta.alignment !== 'left') {
@@ -9,7 +9,7 @@ import { BaseGenerator } from './BaseGenerator.js';
9
9
  */
10
10
  export declare class PdfGenerator extends BaseGenerator<'pdf'> {
11
11
  constructor(ast: OfficeParserAST, config?: GeneratorConfig<'pdf'>);
12
- generate(): Promise<ConversionResult>;
12
+ generate(): Promise<ConversionResult<'pdf'>>;
13
13
  /**
14
14
  * Node.js implementation using Puppeteer.
15
15
  * Uses dynamic import to avoid bundling puppeteer into the library core.
@@ -23,12 +23,6 @@ class PdfGenerator extends BaseGenerator_js_1.BaseGenerator {
23
23
  const htmlGenerator = new HtmlGenerator_js_1.HtmlGenerator(this.ast, {
24
24
  ...this.config,
25
25
  htmlConfig: { ...this.config.htmlConfig, standalone: true },
26
- mdConfig: undefined,
27
- chunksConfig: undefined,
28
- csvConfig: undefined,
29
- textConfig: undefined,
30
- pdfConfig: undefined,
31
- rtfConfig: undefined,
32
26
  });
33
27
  const htmlResult = await htmlGenerator.generate();
34
28
  const html = typeof htmlResult.value === 'string' ? htmlResult.value : '';
@@ -7,9 +7,10 @@ export declare class RtfGenerator extends BaseGenerator<'rtf'> {
7
7
  private colorTable;
8
8
  private inTable;
9
9
  constructor(ast: OfficeParserAST, config?: GeneratorConfig<'rtf'>);
10
- generate(): Promise<ConversionResult>;
10
+ generate(): Promise<ConversionResult<'rtf'>>;
11
11
  protected processNodeRecursive(node: OfficeContentNode, processor: (node: OfficeContentNode, childrenOutput: string) => Promise<string>): Promise<string>;
12
12
  private renderBody;
13
13
  private getColorIndex;
14
+ private isLightColor;
14
15
  private escapeRtf;
15
16
  }
@@ -15,7 +15,7 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
15
15
  this.colorTable = [];
16
16
  // We first process all nodes to collect colors and analyze structure
17
17
  const bodyContent = await this.renderBody(this.ast);
18
- let output = '{\\rtf1\\ansi\\deff0\n';
18
+ let output = '{\\rtf1\\ansi\\uc1\\deff0\n';
19
19
  // 1. Info Group (Metadata)
20
20
  if (this.config.renderMetadata && this.ast.metadata) {
21
21
  output += '{\\info';
@@ -100,16 +100,34 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
100
100
  suffix = '\\strike0 ' + suffix;
101
101
  }
102
102
  if (f.color) {
103
- const idx = this.getColorIndex(f.color);
104
- prefix += `\\cf${idx + 1} `;
103
+ // RTF default background is white. Ensure light text without a dark background remains readable.
104
+ const isTextLight = this.isLightColor(f.color);
105
+ const isBgLight = !f.backgroundColor || this.isLightColor(f.backgroundColor);
106
+ if (!(isTextLight && isBgLight)) {
107
+ const idx = this.getColorIndex(f.color);
108
+ prefix += `\\cf${idx + 1} `;
109
+ }
105
110
  }
106
111
  if (f.backgroundColor) {
107
112
  const idx = this.getColorIndex(f.backgroundColor);
108
113
  prefix += `\\highlight${idx + 1} `;
109
114
  }
110
115
  if (f.size) {
111
- const pt = parseInt(f.size);
112
- prefix += `\\fs${pt * 2} `;
116
+ let pt = 12; // default
117
+ const val = parseFloat(f.size);
118
+ if (!isNaN(val)) {
119
+ if (f.size.includes('in'))
120
+ pt = val * 72;
121
+ else if (f.size.includes('cm'))
122
+ pt = val * 28.3465;
123
+ else if (f.size.includes('mm'))
124
+ pt = val * 2.83465;
125
+ else if (f.size.includes('px'))
126
+ pt = val * 0.75;
127
+ else
128
+ pt = val;
129
+ }
130
+ prefix += `\\fs${Math.round(pt * 2)} `;
113
131
  }
114
132
  text = `{\\f0 ${prefix}${text}${suffix}}`;
115
133
  }
@@ -214,13 +232,32 @@ class RtfGenerator extends BaseGenerator_js_1.BaseGenerator {
214
232
  }
215
233
  return idx;
216
234
  }
235
+ isLightColor(hex) {
236
+ if (!hex || hex.length !== 7 || !hex.startsWith('#'))
237
+ return false;
238
+ const r = parseInt(hex.substring(1, 3), 16);
239
+ const g = parseInt(hex.substring(3, 5), 16);
240
+ const b = parseInt(hex.substring(5, 7), 16);
241
+ if (isNaN(r) || isNaN(g) || isNaN(b))
242
+ return false;
243
+ // Simple luminance calculation
244
+ const luminance = (0.299 * r + 0.587 * g + 0.114 * b) / 255;
245
+ return luminance > 0.8;
246
+ }
217
247
  escapeRtf(text) {
218
248
  return text
219
249
  .replace(/\\/g, '\\\\')
220
250
  .replace(/{/g, '\\{')
221
251
  .replace(/}/g, '\\}')
222
252
  .replace(/[^\x00-\x7F]/g, (match) => {
223
- return `\\u${match.charCodeAt(0)}?`;
253
+ let code = match.charCodeAt(0);
254
+ if (code < 256) {
255
+ return `\\'${code.toString(16).padStart(2, '0')}`;
256
+ }
257
+ if (code > 32767) {
258
+ code -= 65536;
259
+ }
260
+ return `{\\uc0\\u${code}}`;
224
261
  });
225
262
  }
226
263
  }
@@ -8,6 +8,6 @@ export declare class TextGenerator extends BaseGenerator<'text'> {
8
8
  /**
9
9
  * Generates plain text by concatenating text content from nodes.
10
10
  */
11
- generate(): Promise<ConversionResult>;
11
+ generate(): Promise<ConversionResult<'text'>>;
12
12
  private renderTable;
13
13
  }
@@ -70,7 +70,9 @@ export declare enum OfficeWarningType {
70
70
  /** No chunks were generated for the document given the current strategy */
71
71
  EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
72
72
  /** A node was skipped because it only contained whitespace */
73
- WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
73
+ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
74
+ /** The HTML generator containerWidth option is invalid */
75
+ INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
74
76
  }
75
77
  /**
76
78
  * Consolidated timeout settings for OCR operations.
@@ -223,10 +225,23 @@ export interface OfficeParserConfig {
223
225
  */
224
226
  ignoreNotes?: boolean;
225
227
  /**
226
- * Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint.
227
- * Default is false. It puts each notes right after its main slide content.
228
- * If ignoreNotes is set to true, this flag is also ignored.
229
- * @note This flag currently does not affect RTF files; RTF footnotes/endnotes are always collected and appended at the end of the content.
228
+ * Flag to ignore comments from parsing.
229
+ * Default is false.
230
+ */
231
+ ignoreComments?: boolean;
232
+ /**
233
+ * Flag to ignore headers and footers from parsing.
234
+ * Default is false.
235
+ */
236
+ ignoreHeadersAndFooters?: boolean;
237
+ /**
238
+ * Flag to ignore slide masters from parsing in PowerPoint.
239
+ * Default is false.
240
+ */
241
+ ignoreSlideMasters?: boolean;
242
+ /**
243
+ * @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`.
244
+ * This option is now completely ignored by all parsers.
230
245
  */
231
246
  putNotesAtLast?: boolean;
232
247
  /**
@@ -351,9 +366,10 @@ export interface OfficeIssue {
351
366
  /**
352
367
  * The result of a document conversion operation.
353
368
  */
354
- export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
369
+ export type ConversionValue<D extends UniversalGeneratorFormat> = D extends "pdf" ? Uint8Array | string : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : string;
370
+ export interface ConversionResult<D extends UniversalGeneratorFormat> {
355
371
  /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
356
- value: D extends "pdf" ? Uint8Array : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
372
+ value: ConversionValue<D>;
357
373
  /** A collection of issues (warnings/infos) generated during the process. */
358
374
  messages: OfficeIssue[];
359
375
  }
@@ -475,17 +491,31 @@ export interface CommonGeneratorConfig {
475
491
  * Restricts format-specific configurations to their respective destinations.
476
492
  */
477
493
  /**
478
- * Mapping of destination formats to their specific configuration interfaces.
494
+ * Maps a destination format string to its corresponding specific configuration object type.
479
495
  */
480
- export interface GeneratorSubConfigMap {
481
- html: HtmlGeneratorConfig;
482
- md: MdGeneratorConfig;
483
- pdf: PdfGeneratorConfig;
484
- csv: CsvGeneratorConfig;
485
- text: TextGeneratorConfig;
486
- rtf: RtfGeneratorConfig;
487
- chunks: ChunkingConfig;
488
- }
496
+ export type GeneratorSpecificConfig<D extends string> = D extends "html" ? {
497
+ htmlConfig?: HtmlGeneratorConfig;
498
+ } : D extends "md" ? {
499
+ mdConfig?: MdGeneratorConfig;
500
+ } : D extends "pdf" ? {
501
+ pdfConfig?: PdfGeneratorConfig;
502
+ } : D extends "csv" ? {
503
+ csvConfig?: CsvGeneratorConfig;
504
+ } : D extends "text" ? {
505
+ textConfig?: TextGeneratorConfig;
506
+ } : D extends "rtf" ? {
507
+ rtfConfig?: RtfGeneratorConfig;
508
+ } : D extends "chunks" ? {
509
+ chunksConfig?: ChunkingConfig;
510
+ } : Partial<{
511
+ htmlConfig: HtmlGeneratorConfig;
512
+ mdConfig: MdGeneratorConfig;
513
+ pdfConfig: PdfGeneratorConfig;
514
+ csvConfig: CsvGeneratorConfig;
515
+ textConfig: TextGeneratorConfig;
516
+ rtfConfig: RtfGeneratorConfig;
517
+ chunksConfig: ChunkingConfig;
518
+ }>;
489
519
  /**
490
520
  * Configuration options for document generators.
491
521
  *
@@ -496,9 +526,7 @@ export interface GeneratorSubConfigMap {
496
526
  *
497
527
  * @template D The destination format string. Defaults to `string` for a general configuration.
498
528
  */
499
- export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
500
- [K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
501
- };
529
+ export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & GeneratorSpecificConfig<D>;
502
530
  /**
503
531
  * Configuration options for the OfficeConverter.
504
532
  * Combines relevant parser and generator settings for a seamless one-step conversion.
@@ -530,6 +558,19 @@ export type OfficeConverterConfig<D extends string = string, T extends Supported
530
558
  */
531
559
  onWarning?: (issue: OfficeIssue) => void;
532
560
  };
561
+ /**
562
+ * Configuration options for granular raw HTML injections.
563
+ */
564
+ export interface HtmlInjectionConfig {
565
+ /** Raw HTML injected immediately after the opening <head> tag */
566
+ headStart?: string;
567
+ /** Raw HTML injected immediately before the closing </head> tag */
568
+ headEnd?: string;
569
+ /** Raw HTML injected immediately after the opening <body> tag */
570
+ bodyStart?: string;
571
+ /** Raw HTML injected immediately before the closing </body> tag */
572
+ bodyEnd?: string;
573
+ }
533
574
  /**
534
575
  * Configuration options for HTML generation.
535
576
  */
@@ -544,6 +585,25 @@ export interface HtmlGeneratorConfig {
544
585
  * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
545
586
  */
546
587
  chartJsSrc?: string;
588
+ /**
589
+ * Custom container width for the generated HTML.
590
+ * Can be a number (pixels) or string (e.g., '900px', '100%').
591
+ * If not specified or set to 'auto', it defaults based on the content type:
592
+ * - Spreadsheet: '100%'
593
+ * - Presentation/Slides: '1100px'
594
+ * - Standard Document (PDF/DOCX/RTF/etc.): '900px'
595
+ */
596
+ containerWidth?: string | number;
597
+ /**
598
+ * Custom CSS to append to the generated HTML document.
599
+ * This CSS will be included in the `<style>` block and can be used to style
600
+ * custom classes added during AST manipulation or override default styles.
601
+ */
602
+ customCss?: string;
603
+ /**
604
+ * Granular injection points for custom HTML, scripts, and styles.
605
+ */
606
+ injections?: HtmlInjectionConfig;
547
607
  }
548
608
  /**
549
609
  * Configuration options for PDF generation.
@@ -619,7 +679,7 @@ export interface StructuredStyleMapping {
619
679
  * The structural type of the node (e.g., 'paragraph', 'heading', 'text').
620
680
  * Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
621
681
  */
622
- nodeType?: string;
682
+ nodeType?: OfficeContentNodeType;
623
683
  /**
624
684
  * A dictionary of attributes to match on the node.
625
685
  *
@@ -685,7 +745,7 @@ export interface CsvGeneratorConfig {
685
745
  /**
686
746
  * Whether to merge all selected sheets into a single CSV.
687
747
  * If false, returns a ZIP archive containing individual CSV files.
688
- * Defaults to false.
748
+ * Defaults to true.
689
749
  */
690
750
  mergeSheets?: boolean;
691
751
  /**
@@ -944,11 +1004,16 @@ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods"
944
1004
  /**
945
1005
  * Types of content nodes in the AST.
946
1006
  */
947
- export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment";
1007
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment" | "header" | "footer" | "slideMaster";
948
1008
  /**
949
1009
  * Supported MIME types for attachments.
950
1010
  */
951
1011
  export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" | "application/vnd.openxmlformats-officedocument.presentationml.presentation" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation" | "application/rtf" | "text/csv" | "text/markdown" | "text/html";
1012
+ /**
1013
+ * Text alignment options.
1014
+ * Common in spreadsheet cells, paragraph styles, and text elements.
1015
+ */
1016
+ export type TextAlignment = "left" | "center" | "right" | "justify";
952
1017
  /**
953
1018
  * Text formatting options available for text content.
954
1019
  * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
@@ -1022,7 +1087,7 @@ export interface TextFormatting {
1022
1087
  * Common in spreadsheet cells or paragraph styles.
1023
1088
  * @example "center", "right"
1024
1089
  */
1025
- alignment?: "left" | "center" | "right" | "justify";
1090
+ alignment?: TextAlignment;
1026
1091
  }
1027
1092
  /**
1028
1093
  * Metadata for a slide in PowerPoint.
@@ -1072,7 +1137,7 @@ export interface HeadingMetadata {
1072
1137
  /** The heading level (e.g., 1 for H1). */
1073
1138
  level: number;
1074
1139
  /** The alignment of the heading. */
1075
- alignment?: "left" | "center" | "right" | "justify";
1140
+ alignment?: TextAlignment;
1076
1141
  /** The style of the heading. */
1077
1142
  style?: string;
1078
1143
  /** Detailed indentation information. */
@@ -1085,7 +1150,7 @@ export interface HeadingMetadata {
1085
1150
  */
1086
1151
  export interface ParagraphMetadata {
1087
1152
  /** The alignment of the paragraph. */
1088
- alignment?: "left" | "center" | "right" | "justify";
1153
+ alignment?: TextAlignment;
1089
1154
  /** The style of the paragraph. */
1090
1155
  style?: string;
1091
1156
  /** Detailed indentation information. */
@@ -1113,7 +1178,7 @@ export interface ListMetadata {
1113
1178
  * Text alignment of the list item.
1114
1179
  * @example 'left', 'center', 'right', 'justify'
1115
1180
  */
1116
- alignment: "left" | "center" | "right" | "justify";
1181
+ alignment: TextAlignment;
1117
1182
  /**
1118
1183
  * The list ID from the Word document's numbering definition.
1119
1184
  * Used to identify which list definition this item belongs to.
@@ -1163,6 +1228,15 @@ export interface CellMetadata {
1163
1228
  style?: string;
1164
1229
  /** Unique anchor IDs for internal linking. */
1165
1230
  anchorIds?: string[];
1231
+ /** Background color for this cell in hex format (e.g. #FFFFFF). */
1232
+ backgroundColor?: string;
1233
+ }
1234
+ /**
1235
+ * Metadata for a table.
1236
+ */
1237
+ export interface TableMetadata {
1238
+ /** Unique anchor IDs for internal linking. */
1239
+ anchorIds?: string[];
1166
1240
  }
1167
1241
  /**
1168
1242
  * Metadata for a chart node in the document.
@@ -1250,6 +1324,8 @@ export interface NoteMetadata {
1250
1324
  noteId?: string;
1251
1325
  /** Unique anchor IDs for internal linking. */
1252
1326
  anchorIds?: string[];
1327
+ /** The slide number this note is associated with (used in PowerPoint). */
1328
+ slideNumber?: number;
1253
1329
  }
1254
1330
  /**
1255
1331
  * Metadata for break nodes.
@@ -1285,10 +1361,25 @@ export interface CodeMetadata {
1285
1361
  /** Unique anchor IDs for internal linking. */
1286
1362
  anchorIds?: string[];
1287
1363
  }
1364
+ /**
1365
+ * Metadata for a comment/annotation.
1366
+ */
1367
+ export interface CommentMetadata {
1368
+ author?: string;
1369
+ initials?: string;
1370
+ date?: string;
1371
+ commentId?: string;
1372
+ }
1373
+ /**
1374
+ * Metadata for a header or footer.
1375
+ */
1376
+ export interface HeaderFooterMetadata {
1377
+ type: "default" | "first" | "even" | string;
1378
+ }
1288
1379
  /**
1289
1380
  * Union type for content metadata.
1290
1381
  */
1291
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
1382
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
1292
1383
  /**
1293
1384
  * Represents a node in the document content tree.
1294
1385
  * This is the core building block of the parsed document structure.
@@ -1315,13 +1406,10 @@ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata |
1315
1406
  * children: [...]
1316
1407
  * }
1317
1408
  */
1318
- export interface OfficeContentNode {
1319
- /**
1320
- * The type of the node.
1321
- * Determines how the node should be interpreted and rendered.
1322
- * Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
1323
- */
1324
- type: OfficeContentNodeType;
1409
+ /**
1410
+ * Shared properties available on all document content nodes.
1411
+ */
1412
+ export interface BaseContentNode {
1325
1413
  /**
1326
1414
  * The complete text content of the node and all its children combined.
1327
1415
  * For container nodes (paragraph, heading), this is the concatenation of all child text.
@@ -1339,6 +1427,16 @@ export interface OfficeContentNode {
1339
1427
  * @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
1340
1428
  */
1341
1429
  children?: OfficeContentNode[];
1430
+ /**
1431
+ * Comments attached to this specific node.
1432
+ * Keeps annotations completely separate from the actual content flow.
1433
+ */
1434
+ comments?: OfficeContentNode[];
1435
+ /**
1436
+ * Notes (like footnotes or slide notes) attached to this specific node.
1437
+ * Keeps notes separate from the actual structural children.
1438
+ */
1439
+ notes?: OfficeContentNode[];
1342
1440
  /**
1343
1441
  * Text formatting applied to this node.
1344
1442
  * Only applicable to text-containing nodes.
@@ -1346,16 +1444,6 @@ export interface OfficeContentNode {
1346
1444
  * @example { bold: true, size: "12", font: "Arial" }
1347
1445
  */
1348
1446
  formatting?: TextFormatting;
1349
- /**
1350
- * Type-specific metadata providing additional context about the node.
1351
- * The metadata structure depends on the node type:
1352
- * - Headings: { level: 1 }
1353
- * - Lists: { listType: 'ordered', indentation: 0 }
1354
- * - Cells: { row: 0, col: 0 }
1355
- * - Slides: { slideNumber: 1 }
1356
- * @example { level: 1 } for a heading
1357
- */
1358
- metadata?: ContentMetadata;
1359
1447
  /**
1360
1448
  * The raw source content for this node.
1361
1449
  * - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
@@ -1367,6 +1455,93 @@ export interface OfficeContentNode {
1367
1455
  */
1368
1456
  rawContent?: string;
1369
1457
  }
1458
+ /**
1459
+ * Represents a node in the document content tree.
1460
+ * This is the core building block of the parsed document structure.
1461
+ * Content nodes can be nested to represent hierarchical document structures
1462
+ * (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
1463
+ *
1464
+ * @example
1465
+ * // A simple paragraph with formatted text
1466
+ * {
1467
+ * type: 'paragraph',
1468
+ * text: 'Hello world',
1469
+ * children: [
1470
+ * { type: 'text', text: 'Hello ', formatting: { bold: true } },
1471
+ * { type: 'text', text: 'world', formatting: { italic: true } }
1472
+ * ]
1473
+ * }
1474
+ *
1475
+ * @example
1476
+ * // A heading with metadata
1477
+ * {
1478
+ * type: 'heading',
1479
+ * text: 'Chapter 1',
1480
+ * metadata: { level: 1 },
1481
+ * children: [...]
1482
+ * }
1483
+ */
1484
+ export type OfficeContentNode = BaseContentNode & ({
1485
+ type: "slide";
1486
+ metadata?: SlideMetadata;
1487
+ } | {
1488
+ type: "sheet";
1489
+ metadata?: SheetMetadata;
1490
+ } | {
1491
+ type: "heading";
1492
+ metadata?: HeadingMetadata;
1493
+ } | {
1494
+ type: "list";
1495
+ metadata?: ListMetadata;
1496
+ } | {
1497
+ type: "cell";
1498
+ metadata?: CellMetadata;
1499
+ } | {
1500
+ type: "image";
1501
+ metadata?: ImageMetadata;
1502
+ } | {
1503
+ type: "chart";
1504
+ metadata?: ChartMetadata;
1505
+ } | {
1506
+ type: "page";
1507
+ metadata?: PageMetadata;
1508
+ } | {
1509
+ type: "paragraph";
1510
+ metadata?: ParagraphMetadata;
1511
+ } | {
1512
+ type: "text";
1513
+ metadata?: TextMetadata;
1514
+ } | {
1515
+ type: "note";
1516
+ metadata?: NoteMetadata;
1517
+ } | {
1518
+ type: "break";
1519
+ metadata?: BreakMetadata;
1520
+ } | {
1521
+ type: "code";
1522
+ metadata?: CodeMetadata;
1523
+ } | {
1524
+ type: "comment";
1525
+ metadata?: CommentMetadata;
1526
+ } | {
1527
+ type: "header";
1528
+ metadata?: HeaderFooterMetadata;
1529
+ } | {
1530
+ type: "footer";
1531
+ metadata?: HeaderFooterMetadata;
1532
+ } | {
1533
+ type: "table";
1534
+ metadata?: TableMetadata;
1535
+ } | {
1536
+ type: "row";
1537
+ metadata?: undefined;
1538
+ } | {
1539
+ type: "drawing";
1540
+ metadata?: undefined;
1541
+ } | {
1542
+ type: "slideMaster";
1543
+ metadata?: SlideMetadata;
1544
+ });
1370
1545
  /**
1371
1546
  * Structured information extracted from a chart.
1372
1547
  */
@@ -1506,14 +1681,40 @@ export interface OfficeMetadata {
1506
1681
  * Values are typed as string, number, boolean, or Date where the source format provides type information.
1507
1682
  */
1508
1683
  customProperties?: Record<string, string | number | boolean | Date>;
1684
+ /** Keywords associated with the document. */
1685
+ keywords?: string;
1686
+ /**
1687
+ * Contains all format-specific metadata fields extracted verbatim.
1688
+ * Consumers can use this to access properties not mapped to the standard OfficeMetadata fields.
1689
+ * Examples: all <meta> tags in HTML, app.xml properties in DOCX, XMP dicts in PDF.
1690
+ */
1691
+ nativeProperties?: Record<string, any>;
1692
+ }
1693
+ /**
1694
+ * Contains out-of-band layout elements and templates that are not part of the main document flow.
1695
+ */
1696
+ export interface OfficeAuxiliaryContent {
1697
+ /** Headers extracted from the document. */
1698
+ headers?: OfficeContentNode[];
1699
+ /** Footers extracted from the document. */
1700
+ footers?: OfficeContentNode[];
1701
+ /** Slide Masters extracted from presentations. */
1702
+ slideMasters?: OfficeContentNode[];
1509
1703
  }
1510
1704
  /**
1511
- * The Abstract Syntax Tree (AST) returned by the parser.
1512
- * This is the root data structure representing the entire parsed document.
1705
+ * The Root Abstract Syntax Tree (AST) representing a parsed Office Document.
1706
+ * This is the ultimate output of `OfficeParser.parseOffice()`.
1707
+ *
1708
+ * DESIGN PHILOSOPHY:
1709
+ * The AST is designed to be a universal, format-agnostic representation of document content.
1710
+ * Whether the input was a PDF, DOCX, XLSX, Markdown, or HTML file, the resulting AST
1711
+ * uses the same consistent structure (`OfficeContentNode` trees).
1513
1712
  *
1514
- * The AST provides a format-agnostic representation of the document that can be easily
1515
- * processed, transformed, or converted to other formats. It preserves the document's
1516
- * structure, content, formatting, and metadata while abstracting away format-specific details.
1713
+ * ### Key Top-Level Properties:
1714
+ * - `metadata`: Document-level properties (author, title, stats).
1715
+ * - `content`: The main sequential flow of the document (paragraphs, tables, slides, sheets).
1716
+ * - `attachments`: Extracted binary assets (images, embedded files).
1717
+ * - `auxiliary`: Out-of-band layout/template elements (headers, footers, slide masters).
1517
1718
  *
1518
1719
  * @example
1519
1720
  * ```typescript
@@ -1563,6 +1764,12 @@ export interface OfficeParserAST {
1563
1764
  * @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
1564
1765
  */
1565
1766
  content: OfficeContentNode[];
1767
+ /**
1768
+ * Out-of-band layout and template elements that are not part of the main text flow.
1769
+ * Extracted only if the respective `ignore...` config flags are false.
1770
+ * Contains elements like `headers`, `footers`, and `slideMasters`.
1771
+ */
1772
+ auxiliary?: OfficeAuxiliaryContent;
1566
1773
  /**
1567
1774
  * Attachments extracted from the document (images, charts, embedded files).
1568
1775
  * Only populated when `config.extractAttachments` is true.
@@ -1606,7 +1813,7 @@ export interface OfficeParserAST {
1606
1813
  * const md = await ast.to('md');
1607
1814
  * ```
1608
1815
  */
1609
- to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1816
+ to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1610
1817
  }
1611
1818
  /**
1612
1819
  * Main parser class providing office document parsing functionality.
@@ -1690,7 +1897,7 @@ export declare class OfficeGenerator {
1690
1897
  */
1691
1898
  static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
1692
1899
  type: T;
1693
- }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1900
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1694
1901
  }
1695
1902
  /**
1696
1903
  * Utility type to infer the file type from a file path string literal.