officeparser 7.1.0 → 7.2.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +152 -56
  2. package/dist/OfficeGenerator.d.ts +6 -2
  3. package/dist/OfficeGenerator.js +30 -9
  4. package/dist/OfficeParser.d.ts +1 -1
  5. package/dist/OfficeParser.js +1 -1
  6. package/dist/cli.d.ts +18 -12
  7. package/dist/cli.js +255 -81
  8. package/dist/defaults.js +12 -1
  9. package/dist/generators/BaseGenerator.d.ts +4 -3
  10. package/dist/generators/BaseGenerator.js +13 -1
  11. package/dist/generators/ChunkingGenerator.js +32 -5
  12. package/dist/generators/CsvGenerator.d.ts +1 -1
  13. package/dist/generators/HtmlGenerator.d.ts +2 -1
  14. package/dist/generators/HtmlGenerator.js +481 -42
  15. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  16. package/dist/generators/MarkdownGenerator.js +35 -2
  17. package/dist/generators/PdfGenerator.d.ts +1 -1
  18. package/dist/generators/PdfGenerator.js +0 -6
  19. package/dist/generators/RtfGenerator.d.ts +2 -1
  20. package/dist/generators/RtfGenerator.js +49 -6
  21. package/dist/generators/TextGenerator.d.ts +1 -1
  22. package/dist/generators/TextGenerator.js +6 -0
  23. package/dist/officeparser.browser.d.ts +267 -54
  24. package/dist/officeparser.browser.iife.js +599 -187
  25. package/dist/officeparser.browser.mjs +599 -187
  26. package/dist/parsers/CsvParser.js +1 -1
  27. package/dist/parsers/ExcelParser.js +63 -19
  28. package/dist/parsers/HtmlParser.js +10 -1
  29. package/dist/parsers/MarkdownParser.js +13 -10
  30. package/dist/parsers/OpenOfficeParser.js +57 -34
  31. package/dist/parsers/PdfParser.js +28 -3
  32. package/dist/parsers/PowerPointParser.js +164 -40
  33. package/dist/parsers/RtfParser.js +28 -24
  34. package/dist/parsers/WordParser.js +154 -11
  35. package/dist/sbom.cdx.json +100 -100
  36. package/dist/types.d.ts +268 -53
  37. package/dist/types.js +4 -0
  38. package/dist/utils/astUtils.d.ts +2 -2
  39. package/dist/utils/astUtils.js +2 -1
  40. package/dist/utils/configUtils.d.ts +5 -0
  41. package/dist/utils/configUtils.js +55 -1
  42. package/dist/utils/errorUtils.js +3 -1
  43. package/dist/utils/moduleLoader.js +55 -11
  44. package/dist/utils/xmlUtils.d.ts +9 -0
  45. package/dist/utils/xmlUtils.js +53 -1
  46. package/package.json +6 -3
@@ -7,6 +7,8 @@
7
7
  export declare enum OfficeErrorType {
8
8
  /** Unsupported file extension */
9
9
  EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
10
+ /** Unsupported output generator format */
11
+ FORMAT_UNSUPPORTED = "FORMAT_UNSUPPORTED",
10
12
  /** File appears to be corrupted or malformed */
11
13
  FILE_CORRUPTED = "FILE_CORRUPTED",
12
14
  /** File could not be found at the specified path */
@@ -70,7 +72,9 @@ export declare enum OfficeWarningType {
70
72
  /** No chunks were generated for the document given the current strategy */
71
73
  EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
72
74
  /** A node was skipped because it only contained whitespace */
73
- WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
75
+ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
76
+ /** The HTML generator containerWidth option is invalid */
77
+ INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
74
78
  }
75
79
  /**
76
80
  * Consolidated timeout settings for OCR operations.
@@ -223,10 +227,23 @@ export interface OfficeParserConfig {
223
227
  */
224
228
  ignoreNotes?: boolean;
225
229
  /**
226
- * Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint.
227
- * Default is false. It puts each notes right after its main slide content.
228
- * If ignoreNotes is set to true, this flag is also ignored.
229
- * @note This flag currently does not affect RTF files; RTF footnotes/endnotes are always collected and appended at the end of the content.
230
+ * Flag to ignore comments from parsing.
231
+ * Default is false.
232
+ */
233
+ ignoreComments?: boolean;
234
+ /**
235
+ * Flag to ignore headers and footers from parsing.
236
+ * Default is false.
237
+ */
238
+ ignoreHeadersAndFooters?: boolean;
239
+ /**
240
+ * Flag to ignore slide masters from parsing in PowerPoint.
241
+ * Default is false.
242
+ */
243
+ ignoreSlideMasters?: boolean;
244
+ /**
245
+ * @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`.
246
+ * This option is now completely ignored by all parsers.
230
247
  */
231
248
  putNotesAtLast?: boolean;
232
249
  /**
@@ -351,9 +368,10 @@ export interface OfficeIssue {
351
368
  /**
352
369
  * The result of a document conversion operation.
353
370
  */
354
- export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
371
+ export type ConversionValue<D extends UniversalGeneratorFormat> = D extends "pdf" ? Uint8Array | string : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : string;
372
+ export interface ConversionResult<D extends UniversalGeneratorFormat> {
355
373
  /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
356
- value: D extends "pdf" ? Uint8Array : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
374
+ value: ConversionValue<D>;
357
375
  /** A collection of issues (warnings/infos) generated during the process. */
358
376
  messages: OfficeIssue[];
359
377
  }
@@ -381,7 +399,7 @@ export interface CommonGeneratorConfig {
381
399
  * 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
382
400
  * 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
383
401
  * 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
384
- * 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
402
+ * 4. **Async Support**: The callback can be `async`, allowing you to load external data or perform complex logic during generation.
385
403
  */
386
404
  onNode?: (node: OfficeContentNode) => string | false | Promise<string | false | void> | void;
387
405
  /**
@@ -475,17 +493,31 @@ export interface CommonGeneratorConfig {
475
493
  * Restricts format-specific configurations to their respective destinations.
476
494
  */
477
495
  /**
478
- * Mapping of destination formats to their specific configuration interfaces.
496
+ * Maps a destination format string to its corresponding specific configuration object type.
479
497
  */
480
- export interface GeneratorSubConfigMap {
481
- html: HtmlGeneratorConfig;
482
- md: MdGeneratorConfig;
483
- pdf: PdfGeneratorConfig;
484
- csv: CsvGeneratorConfig;
485
- text: TextGeneratorConfig;
486
- rtf: RtfGeneratorConfig;
487
- chunks: ChunkingConfig;
488
- }
498
+ export type GeneratorSpecificConfig<D extends string> = D extends "html" ? {
499
+ htmlConfig?: HtmlGeneratorConfig;
500
+ } : D extends "md" ? {
501
+ mdConfig?: MdGeneratorConfig;
502
+ } : D extends "pdf" ? {
503
+ pdfConfig?: PdfGeneratorConfig;
504
+ } : D extends "csv" ? {
505
+ csvConfig?: CsvGeneratorConfig;
506
+ } : D extends "text" ? {
507
+ textConfig?: TextGeneratorConfig;
508
+ } : D extends "rtf" ? {
509
+ rtfConfig?: RtfGeneratorConfig;
510
+ } : D extends "chunks" ? {
511
+ chunksConfig?: ChunkingConfig;
512
+ } : Partial<{
513
+ htmlConfig: HtmlGeneratorConfig;
514
+ mdConfig: MdGeneratorConfig;
515
+ pdfConfig: PdfGeneratorConfig;
516
+ csvConfig: CsvGeneratorConfig;
517
+ textConfig: TextGeneratorConfig;
518
+ rtfConfig: RtfGeneratorConfig;
519
+ chunksConfig: ChunkingConfig;
520
+ }>;
489
521
  /**
490
522
  * Configuration options for document generators.
491
523
  *
@@ -496,9 +528,7 @@ export interface GeneratorSubConfigMap {
496
528
  *
497
529
  * @template D The destination format string. Defaults to `string` for a general configuration.
498
530
  */
499
- export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
500
- [K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
501
- };
531
+ export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & GeneratorSpecificConfig<D>;
502
532
  /**
503
533
  * Configuration options for the OfficeConverter.
504
534
  * Combines relevant parser and generator settings for a seamless one-step conversion.
@@ -530,6 +560,19 @@ export type OfficeConverterConfig<D extends string = string, T extends Supported
530
560
  */
531
561
  onWarning?: (issue: OfficeIssue) => void;
532
562
  };
563
+ /**
564
+ * Configuration options for granular raw HTML injections.
565
+ */
566
+ export interface HtmlInjectionConfig {
567
+ /** Raw HTML injected immediately after the opening <head> tag */
568
+ headStart?: string;
569
+ /** Raw HTML injected immediately before the closing </head> tag */
570
+ headEnd?: string;
571
+ /** Raw HTML injected immediately after the opening <body> tag */
572
+ bodyStart?: string;
573
+ /** Raw HTML injected immediately before the closing </body> tag */
574
+ bodyEnd?: string;
575
+ }
533
576
  /**
534
577
  * Configuration options for HTML generation.
535
578
  */
@@ -544,6 +587,25 @@ export interface HtmlGeneratorConfig {
544
587
  * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
545
588
  */
546
589
  chartJsSrc?: string;
590
+ /**
591
+ * Custom container width for the generated HTML.
592
+ * Can be a number (pixels) or string (e.g., '900px', '100%').
593
+ * If not specified or set to 'auto', it defaults based on the content type:
594
+ * - Spreadsheet: '100%'
595
+ * - Presentation/Slides: '1100px'
596
+ * - Standard Document (PDF/DOCX/RTF/etc.): '900px'
597
+ */
598
+ containerWidth?: string | number;
599
+ /**
600
+ * Custom CSS to append to the generated HTML document.
601
+ * This CSS will be included in the `<style>` block and can be used to style
602
+ * custom classes added during AST manipulation or override default styles.
603
+ */
604
+ customCss?: string;
605
+ /**
606
+ * Granular injection points for custom HTML, scripts, and styles.
607
+ */
608
+ injections?: HtmlInjectionConfig;
547
609
  }
548
610
  /**
549
611
  * Configuration options for PDF generation.
@@ -619,7 +681,7 @@ export interface StructuredStyleMapping {
619
681
  * The structural type of the node (e.g., 'paragraph', 'heading', 'text').
620
682
  * Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
621
683
  */
622
- nodeType?: string;
684
+ nodeType?: OfficeContentNodeType;
623
685
  /**
624
686
  * A dictionary of attributes to match on the node.
625
687
  *
@@ -685,7 +747,7 @@ export interface CsvGeneratorConfig {
685
747
  /**
686
748
  * Whether to merge all selected sheets into a single CSV.
687
749
  * If false, returns a ZIP archive containing individual CSV files.
688
- * Defaults to false.
750
+ * Defaults to true.
689
751
  */
690
752
  mergeSheets?: boolean;
691
753
  /**
@@ -944,11 +1006,16 @@ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods"
944
1006
  /**
945
1007
  * Types of content nodes in the AST.
946
1008
  */
947
- export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment";
1009
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment" | "header" | "footer" | "slideMaster";
948
1010
  /**
949
1011
  * Supported MIME types for attachments.
950
1012
  */
951
1013
  export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" | "application/vnd.openxmlformats-officedocument.presentationml.presentation" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation" | "application/rtf" | "text/csv" | "text/markdown" | "text/html";
1014
+ /**
1015
+ * Text alignment options.
1016
+ * Common in spreadsheet cells, paragraph styles, and text elements.
1017
+ */
1018
+ export type TextAlignment = "left" | "center" | "right" | "justify";
952
1019
  /**
953
1020
  * Text formatting options available for text content.
954
1021
  * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
@@ -1022,7 +1089,7 @@ export interface TextFormatting {
1022
1089
  * Common in spreadsheet cells or paragraph styles.
1023
1090
  * @example "center", "right"
1024
1091
  */
1025
- alignment?: "left" | "center" | "right" | "justify";
1092
+ alignment?: TextAlignment;
1026
1093
  }
1027
1094
  /**
1028
1095
  * Metadata for a slide in PowerPoint.
@@ -1072,7 +1139,7 @@ export interface HeadingMetadata {
1072
1139
  /** The heading level (e.g., 1 for H1). */
1073
1140
  level: number;
1074
1141
  /** The alignment of the heading. */
1075
- alignment?: "left" | "center" | "right" | "justify";
1142
+ alignment?: TextAlignment;
1076
1143
  /** The style of the heading. */
1077
1144
  style?: string;
1078
1145
  /** Detailed indentation information. */
@@ -1085,7 +1152,7 @@ export interface HeadingMetadata {
1085
1152
  */
1086
1153
  export interface ParagraphMetadata {
1087
1154
  /** The alignment of the paragraph. */
1088
- alignment?: "left" | "center" | "right" | "justify";
1155
+ alignment?: TextAlignment;
1089
1156
  /** The style of the paragraph. */
1090
1157
  style?: string;
1091
1158
  /** Detailed indentation information. */
@@ -1113,7 +1180,7 @@ export interface ListMetadata {
1113
1180
  * Text alignment of the list item.
1114
1181
  * @example 'left', 'center', 'right', 'justify'
1115
1182
  */
1116
- alignment: "left" | "center" | "right" | "justify";
1183
+ alignment: TextAlignment;
1117
1184
  /**
1118
1185
  * The list ID from the Word document's numbering definition.
1119
1186
  * Used to identify which list definition this item belongs to.
@@ -1163,6 +1230,15 @@ export interface CellMetadata {
1163
1230
  style?: string;
1164
1231
  /** Unique anchor IDs for internal linking. */
1165
1232
  anchorIds?: string[];
1233
+ /** Background color for this cell in hex format (e.g. #FFFFFF). */
1234
+ backgroundColor?: string;
1235
+ }
1236
+ /**
1237
+ * Metadata for a table.
1238
+ */
1239
+ export interface TableMetadata {
1240
+ /** Unique anchor IDs for internal linking. */
1241
+ anchorIds?: string[];
1166
1242
  }
1167
1243
  /**
1168
1244
  * Metadata for a chart node in the document.
@@ -1250,6 +1326,8 @@ export interface NoteMetadata {
1250
1326
  noteId?: string;
1251
1327
  /** Unique anchor IDs for internal linking. */
1252
1328
  anchorIds?: string[];
1329
+ /** The slide number this note is associated with (used in PowerPoint). */
1330
+ slideNumber?: number;
1253
1331
  }
1254
1332
  /**
1255
1333
  * Metadata for break nodes.
@@ -1285,10 +1363,25 @@ export interface CodeMetadata {
1285
1363
  /** Unique anchor IDs for internal linking. */
1286
1364
  anchorIds?: string[];
1287
1365
  }
1366
+ /**
1367
+ * Metadata for a comment/annotation.
1368
+ */
1369
+ export interface CommentMetadata {
1370
+ author?: string;
1371
+ initials?: string;
1372
+ date?: string;
1373
+ commentId?: string;
1374
+ }
1375
+ /**
1376
+ * Metadata for a header or footer.
1377
+ */
1378
+ export interface HeaderFooterMetadata {
1379
+ type: "default" | "first" | "even" | string;
1380
+ }
1288
1381
  /**
1289
1382
  * Union type for content metadata.
1290
1383
  */
1291
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
1384
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
1292
1385
  /**
1293
1386
  * Represents a node in the document content tree.
1294
1387
  * This is the core building block of the parsed document structure.
@@ -1315,13 +1408,10 @@ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata |
1315
1408
  * children: [...]
1316
1409
  * }
1317
1410
  */
1318
- export interface OfficeContentNode {
1319
- /**
1320
- * The type of the node.
1321
- * Determines how the node should be interpreted and rendered.
1322
- * Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
1323
- */
1324
- type: OfficeContentNodeType;
1411
+ /**
1412
+ * Shared properties available on all document content nodes.
1413
+ */
1414
+ export interface BaseContentNode {
1325
1415
  /**
1326
1416
  * The complete text content of the node and all its children combined.
1327
1417
  * For container nodes (paragraph, heading), this is the concatenation of all child text.
@@ -1339,6 +1429,16 @@ export interface OfficeContentNode {
1339
1429
  * @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
1340
1430
  */
1341
1431
  children?: OfficeContentNode[];
1432
+ /**
1433
+ * Comments attached to this specific node.
1434
+ * Keeps annotations completely separate from the actual content flow.
1435
+ */
1436
+ comments?: OfficeContentNode[];
1437
+ /**
1438
+ * Notes (like footnotes or slide notes) attached to this specific node.
1439
+ * Keeps notes separate from the actual structural children.
1440
+ */
1441
+ notes?: OfficeContentNode[];
1342
1442
  /**
1343
1443
  * Text formatting applied to this node.
1344
1444
  * Only applicable to text-containing nodes.
@@ -1346,16 +1446,6 @@ export interface OfficeContentNode {
1346
1446
  * @example { bold: true, size: "12", font: "Arial" }
1347
1447
  */
1348
1448
  formatting?: TextFormatting;
1349
- /**
1350
- * Type-specific metadata providing additional context about the node.
1351
- * The metadata structure depends on the node type:
1352
- * - Headings: { level: 1 }
1353
- * - Lists: { listType: 'ordered', indentation: 0 }
1354
- * - Cells: { row: 0, col: 0 }
1355
- * - Slides: { slideNumber: 1 }
1356
- * @example { level: 1 } for a heading
1357
- */
1358
- metadata?: ContentMetadata;
1359
1449
  /**
1360
1450
  * The raw source content for this node.
1361
1451
  * - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
@@ -1367,6 +1457,93 @@ export interface OfficeContentNode {
1367
1457
  */
1368
1458
  rawContent?: string;
1369
1459
  }
1460
+ /**
1461
+ * Represents a node in the document content tree.
1462
+ * This is the core building block of the parsed document structure.
1463
+ * Content nodes can be nested to represent hierarchical document structures
1464
+ * (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
1465
+ *
1466
+ * @example
1467
+ * // A simple paragraph with formatted text
1468
+ * {
1469
+ * type: 'paragraph',
1470
+ * text: 'Hello world',
1471
+ * children: [
1472
+ * { type: 'text', text: 'Hello ', formatting: { bold: true } },
1473
+ * { type: 'text', text: 'world', formatting: { italic: true } }
1474
+ * ]
1475
+ * }
1476
+ *
1477
+ * @example
1478
+ * // A heading with metadata
1479
+ * {
1480
+ * type: 'heading',
1481
+ * text: 'Chapter 1',
1482
+ * metadata: { level: 1 },
1483
+ * children: [...]
1484
+ * }
1485
+ */
1486
+ export type OfficeContentNode = BaseContentNode & ({
1487
+ type: "slide";
1488
+ metadata?: SlideMetadata;
1489
+ } | {
1490
+ type: "sheet";
1491
+ metadata?: SheetMetadata;
1492
+ } | {
1493
+ type: "heading";
1494
+ metadata?: HeadingMetadata;
1495
+ } | {
1496
+ type: "list";
1497
+ metadata?: ListMetadata;
1498
+ } | {
1499
+ type: "cell";
1500
+ metadata?: CellMetadata;
1501
+ } | {
1502
+ type: "image";
1503
+ metadata?: ImageMetadata;
1504
+ } | {
1505
+ type: "chart";
1506
+ metadata?: ChartMetadata;
1507
+ } | {
1508
+ type: "page";
1509
+ metadata?: PageMetadata;
1510
+ } | {
1511
+ type: "paragraph";
1512
+ metadata?: ParagraphMetadata;
1513
+ } | {
1514
+ type: "text";
1515
+ metadata?: TextMetadata;
1516
+ } | {
1517
+ type: "note";
1518
+ metadata?: NoteMetadata;
1519
+ } | {
1520
+ type: "break";
1521
+ metadata?: BreakMetadata;
1522
+ } | {
1523
+ type: "code";
1524
+ metadata?: CodeMetadata;
1525
+ } | {
1526
+ type: "comment";
1527
+ metadata?: CommentMetadata;
1528
+ } | {
1529
+ type: "header";
1530
+ metadata?: HeaderFooterMetadata;
1531
+ } | {
1532
+ type: "footer";
1533
+ metadata?: HeaderFooterMetadata;
1534
+ } | {
1535
+ type: "table";
1536
+ metadata?: TableMetadata;
1537
+ } | {
1538
+ type: "row";
1539
+ metadata?: undefined;
1540
+ } | {
1541
+ type: "drawing";
1542
+ metadata?: undefined;
1543
+ } | {
1544
+ type: "slideMaster";
1545
+ metadata?: SlideMetadata;
1546
+ });
1370
1547
  /**
1371
1548
  * Structured information extracted from a chart.
1372
1549
  */
@@ -1506,14 +1683,40 @@ export interface OfficeMetadata {
1506
1683
  * Values are typed as string, number, boolean, or Date where the source format provides type information.
1507
1684
  */
1508
1685
  customProperties?: Record<string, string | number | boolean | Date>;
1686
+ /** Keywords associated with the document. */
1687
+ keywords?: string;
1688
+ /**
1689
+ * Contains all format-specific metadata fields extracted verbatim.
1690
+ * Consumers can use this to access properties not mapped to the standard OfficeMetadata fields.
1691
+ * Examples: all <meta> tags in HTML, app.xml properties in DOCX, XMP dicts in PDF.
1692
+ */
1693
+ nativeProperties?: Record<string, any>;
1694
+ }
1695
+ /**
1696
+ * Contains out-of-band layout elements and templates that are not part of the main document flow.
1697
+ */
1698
+ export interface OfficeAuxiliaryContent {
1699
+ /** Headers extracted from the document. */
1700
+ headers?: OfficeContentNode[];
1701
+ /** Footers extracted from the document. */
1702
+ footers?: OfficeContentNode[];
1703
+ /** Slide Masters extracted from presentations. */
1704
+ slideMasters?: OfficeContentNode[];
1509
1705
  }
1510
1706
  /**
1511
- * The Abstract Syntax Tree (AST) returned by the parser.
1512
- * This is the root data structure representing the entire parsed document.
1707
+ * The Root Abstract Syntax Tree (AST) representing a parsed Office Document.
1708
+ * This is the ultimate output of `OfficeParser.parseOffice()`.
1513
1709
  *
1514
- * The AST provides a format-agnostic representation of the document that can be easily
1515
- * processed, transformed, or converted to other formats. It preserves the document's
1516
- * structure, content, formatting, and metadata while abstracting away format-specific details.
1710
+ * DESIGN PHILOSOPHY:
1711
+ * The AST is designed to be a universal, format-agnostic representation of document content.
1712
+ * Whether the input was a PDF, DOCX, XLSX, Markdown, or HTML file, the resulting AST
1713
+ * uses the same consistent structure (`OfficeContentNode` trees).
1714
+ *
1715
+ * ### Key Top-Level Properties:
1716
+ * - `metadata`: Document-level properties (author, title, stats).
1717
+ * - `content`: The main sequential flow of the document (paragraphs, tables, slides, sheets).
1718
+ * - `attachments`: Extracted binary assets (images, embedded files).
1719
+ * - `auxiliary`: Out-of-band layout/template elements (headers, footers, slide masters).
1517
1720
  *
1518
1721
  * @example
1519
1722
  * ```typescript
@@ -1563,6 +1766,12 @@ export interface OfficeParserAST {
1563
1766
  * @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
1564
1767
  */
1565
1768
  content: OfficeContentNode[];
1769
+ /**
1770
+ * Out-of-band layout and template elements that are not part of the main text flow.
1771
+ * Extracted only if the respective `ignore...` config flags are false.
1772
+ * Contains elements like `headers`, `footers`, and `slideMasters`.
1773
+ */
1774
+ auxiliary?: OfficeAuxiliaryContent;
1566
1775
  /**
1567
1776
  * Attachments extracted from the document (images, charts, embedded files).
1568
1777
  * Only populated when `config.extractAttachments` is true.
@@ -1606,7 +1815,7 @@ export interface OfficeParserAST {
1606
1815
  * const md = await ast.to('md');
1607
1816
  * ```
1608
1817
  */
1609
- to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1818
+ to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1610
1819
  }
1611
1820
  /**
1612
1821
  * Main parser class providing office document parsing functionality.
@@ -1653,7 +1862,7 @@ export declare class OfficeParser {
1653
1862
  * });
1654
1863
  *
1655
1864
  * // Parse a Buffer with OCR enabled
1656
- * const buffer = await fetch('document.pdf').then(r => r.arrayBuffer());
1865
+ * const buffer = await retrieveData('document.pdf').then(r => r.arrayBuffer());
1657
1866
  * const ast = await OfficeParser.parseOffice(buffer, {
1658
1867
  * ocr: true,
1659
1868
  * ocrLanguage: 'eng+fra'
@@ -1678,6 +1887,10 @@ export declare class OfficeParser {
1678
1887
  * Main generator class providing document conversion functionality.
1679
1888
  */
1680
1889
  export declare class OfficeGenerator {
1890
+ /**
1891
+ * Normalizes format aliases (e.g., 'txt' to 'text', 'markdown' to 'md') to standard internal formats.
1892
+ */
1893
+ static normalizeDestination(dest: string): UniversalGeneratorFormat;
1681
1894
  /**
1682
1895
  * Generates a file of the specified type from an AST.
1683
1896
  * This is the single source of truth for generation logic.
@@ -1690,7 +1903,7 @@ export declare class OfficeGenerator {
1690
1903
  */
1691
1904
  static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
1692
1905
  type: T;
1693
- }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1906
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1694
1907
  }
1695
1908
  /**
1696
1909
  * Utility type to infer the file type from a file path string literal.