officeparser 7.0.3 → 7.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +152 -18
  2. package/dist/OfficeGenerator.d.ts +1 -1
  3. package/dist/OfficeGenerator.js +16 -7
  4. package/dist/OfficeParser.js +6 -0
  5. package/dist/cli.d.ts +4 -0
  6. package/dist/cli.js +12 -3
  7. package/dist/defaults.js +27 -1
  8. package/dist/generators/BaseGenerator.d.ts +3 -3
  9. package/dist/generators/ChunkingGenerator.js +31 -4
  10. package/dist/generators/CsvGenerator.d.ts +1 -1
  11. package/dist/generators/HtmlGenerator.d.ts +2 -1
  12. package/dist/generators/HtmlGenerator.js +462 -40
  13. package/dist/generators/MarkdownGenerator.d.ts +1 -1
  14. package/dist/generators/MarkdownGenerator.js +3 -1
  15. package/dist/generators/PdfGenerator.d.ts +1 -1
  16. package/dist/generators/PdfGenerator.js +51 -10
  17. package/dist/generators/RtfGenerator.d.ts +2 -1
  18. package/dist/generators/RtfGenerator.js +43 -6
  19. package/dist/generators/TextGenerator.d.ts +1 -1
  20. package/dist/officeparser.browser.d.ts +377 -53
  21. package/dist/officeparser.browser.iife.js +380 -93
  22. package/dist/officeparser.browser.mjs +380 -93
  23. package/dist/parsers/CsvParser.js +6 -1
  24. package/dist/parsers/ExcelParser.js +69 -21
  25. package/dist/parsers/HtmlParser.js +15 -1
  26. package/dist/parsers/MarkdownParser.js +18 -10
  27. package/dist/parsers/OpenOfficeParser.js +61 -34
  28. package/dist/parsers/PdfParser.js +26 -1
  29. package/dist/parsers/PowerPointParser.js +168 -40
  30. package/dist/parsers/RtfParser.js +30 -24
  31. package/dist/parsers/WordParser.js +158 -11
  32. package/dist/sbom.cdx.json +100 -100
  33. package/dist/types.d.ts +383 -53
  34. package/dist/types.js +4 -0
  35. package/dist/utils/astUtils.d.ts +2 -2
  36. package/dist/utils/astUtils.js +2 -1
  37. package/dist/utils/configUtils.d.ts +5 -0
  38. package/dist/utils/configUtils.js +69 -2
  39. package/dist/utils/errorUtils.d.ts +20 -0
  40. package/dist/utils/errorUtils.js +39 -3
  41. package/dist/utils/moduleLoader.js +3 -3
  42. package/dist/utils/ocrUtils.js +271 -66
  43. package/dist/utils/xmlUtils.d.ts +17 -0
  44. package/dist/utils/xmlUtils.js +85 -1
  45. package/package.json +3 -2
package/dist/types.d.ts CHANGED
@@ -28,7 +28,9 @@ export declare enum OfficeErrorType {
28
28
  /** Output mapping in style mapping is invalid */
29
29
  INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
30
30
  /** Semantic chunking strategy is selected but no embedding function is provided */
31
- MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
31
+ MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
32
+ /** The operation was aborted */
33
+ OPERATION_ABORTED = "OPERATION_ABORTED"
32
34
  }
33
35
  /**
34
36
  * Standard warning types for OfficeParser.
@@ -66,7 +68,71 @@ export declare enum OfficeWarningType {
66
68
  /** No chunks were generated for the document given the current strategy */
67
69
  EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
68
70
  /** A node was skipped because it only contained whitespace */
69
- WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
71
+ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
72
+ /** The HTML generator containerWidth option is invalid */
73
+ INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
74
+ }
75
+ /**
76
+ * Consolidated timeout settings for OCR operations.
77
+ * Preferred over the individual flat timeout properties on {@link OcrConfig},
78
+ * which are now deprecated.
79
+ *
80
+ * If a key is present here, it takes priority over the corresponding deprecated
81
+ * flat property (e.g. `timeout.autoTerminate` wins over `autoTerminateTimeout`).
82
+ * Set any value to `0` to disable that specific timeout.
83
+ */
84
+ export interface OcrTimeoutConfig {
85
+ /**
86
+ * Timeout in milliseconds of inactivity before the OCR worker pool is
87
+ * automatically terminated and freed.
88
+ *
89
+ * The timer resets every time a new OCR job is enqueued. When the last
90
+ * job completes and this duration passes without a new one, the entire
91
+ * worker pool is torn down so that no background threads keep the Node.js
92
+ * process alive unnecessarily.
93
+ *
94
+ * Set to `0` to keep workers alive indefinitely (useful when you want to
95
+ * call {@link terminateOcr} manually at shutdown time).
96
+ * Default is 10,000 ms (10 seconds).
97
+ */
98
+ autoTerminate?: number;
99
+ /**
100
+ * Timeout in milliseconds for initializing a Tesseract worker
101
+ * (loading the JS runtime, downloading or loading the `.traineddata`
102
+ * language file) or for re-initializing an existing worker with a
103
+ * different language.
104
+ *
105
+ * Multi-language combinations (e.g. `'por+eng+spa'`) must download a
106
+ * separate `.traineddata` file for each language and are therefore
107
+ * particularly susceptible to slow networks. Tune this value upward if
108
+ * your OCR environment has high network latency or if you are loading
109
+ * languages from disk in a large container image.
110
+ *
111
+ * When the timeout fires, the failed job is rejected with a non-fatal
112
+ * {@link OfficeWarningType.OCR_FAILED} warning and parsing continues
113
+ * without OCR output for that image. The stalled worker is terminated
114
+ * and removed from the pool to prevent thread leaks.
115
+ *
116
+ * Set to `0` to wait indefinitely (not recommended for production; a hung
117
+ * network request will block the entire OCR queue for that language).
118
+ * Default is 60,000 ms (60 seconds).
119
+ */
120
+ workerLoad?: number;
121
+ /**
122
+ * Timeout in milliseconds for the actual OCR text-recognition call
123
+ * (`worker.recognize(image)`) on an already-initialized Tesseract worker.
124
+ *
125
+ * Recognition time scales with image resolution and the number of active
126
+ * languages. Very high-resolution scans or unusual character sets can
127
+ * exceed the default. If this timeout fires, the job is rejected with a
128
+ * non-fatal {@link OfficeWarningType.OCR_FAILED} warning; the worker is
129
+ * terminated and evicted from the pool because its internal state after a
130
+ * mid-recognition timeout is undefined.
131
+ *
132
+ * Set to `0` to wait indefinitely.
133
+ * Default is 30,000 ms (30 seconds).
134
+ */
135
+ recognition?: number;
70
136
  }
71
137
  /**
72
138
  * Configuration options for OCR.
@@ -102,11 +168,33 @@ export interface OcrConfig {
102
168
  */
103
169
  langPath?: string;
104
170
  /**
171
+ * Consolidated timeout settings for all OCR operations.
172
+ *
173
+ * Prefer this over the deprecated flat timeout properties.
174
+ * If `timeout.autoTerminate` is set, it takes priority over the deprecated `autoTerminateTimeout`.
175
+ */
176
+ timeout?: OcrTimeoutConfig;
177
+ /**
178
+ * @deprecated Use `timeout.autoTerminate` instead.
179
+ *
105
180
  * Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
106
181
  * Set to 0 to disable auto-termination.
107
182
  * Default is 10,000 (10 seconds).
183
+ *
184
+ * If `timeout.autoTerminate` is also set, that value takes priority over this one.
108
185
  */
109
186
  autoTerminateTimeout?: number;
187
+ /**
188
+ * An optional AbortSignal propagated from the main parser configuration to abort active OCR jobs.
189
+ * If the signal is aborted:
190
+ * 1. Any pending OCR jobs in the scheduler queue are rejected immediately.
191
+ * 2. Any active OCR job running on a Tesseract worker will reject, the worker will be
192
+ * terminated, and it will be removed from the pool to avoid hanging worker threads.
193
+ *
194
+ * Developers should prefer passing this at the top level of `parseOffice` (as `config.abortSignal`),
195
+ * which automatically propagates here.
196
+ */
197
+ abortSignal?: AbortSignal | null;
110
198
  }
111
199
  /**
112
200
  * Configuration options for the OfficeParser.
@@ -135,10 +223,23 @@ export interface OfficeParserConfig {
135
223
  */
136
224
  ignoreNotes?: boolean;
137
225
  /**
138
- * Flag, if set to true, will collectively put all the parsed text from notes at last in files like powerpoint.
139
- * Default is false. It puts each notes right after its main slide content.
140
- * If ignoreNotes is set to true, this flag is also ignored.
141
- * @note This flag currently does not affect RTF files; RTF footnotes/endnotes are always collected and appended at the end of the content.
226
+ * Flag to ignore comments from parsing.
227
+ * Default is false.
228
+ */
229
+ ignoreComments?: boolean;
230
+ /**
231
+ * Flag to ignore headers and footers from parsing.
232
+ * Default is false.
233
+ */
234
+ ignoreHeadersAndFooters?: boolean;
235
+ /**
236
+ * Flag to ignore slide masters from parsing in PowerPoint.
237
+ * Default is false.
238
+ */
239
+ ignoreSlideMasters?: boolean;
240
+ /**
241
+ * @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`.
242
+ * This option is now completely ignored by all parsers.
142
243
  */
143
244
  putNotesAtLast?: boolean;
144
245
  /**
@@ -173,6 +274,20 @@ export interface OfficeParserConfig {
173
274
  * If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
174
275
  */
175
276
  ocrConfig?: OcrConfig;
277
+ /**
278
+ * An optional AbortSignal to cancel the parsing operation.
279
+ * When aborted, the parser immediately rejects with a standard AbortError (DOMException).
280
+ *
281
+ * ### Format-Specific Abort Behavior:
282
+ * - **PDF**: Checked between page loads and before individual image OCR operations.
283
+ * - **RTF**: Checked before parsing/traversal and before running OCR on image attachments.
284
+ * - **DOCX/XLSX/PPTX/ODF**: Checked during zip decompression before loading and parsing XML files.
285
+ * - **CSV/MD/HTML**: Checked at the start of the parsing phase.
286
+ *
287
+ * Note: If an OCR operation is currently running on a Tesseract worker when aborted,
288
+ * the worker will be terminated and removed from the worker pool automatically to prevent leaks.
289
+ */
290
+ abortSignal?: AbortSignal | null;
176
291
  /**
177
292
  * Flag to serialize raw content (XML) as clean, formatted strings.
178
293
  * Only relevant when `includeRawContent` is true.
@@ -254,9 +369,10 @@ export interface OfficeIssue {
254
369
  /**
255
370
  * The result of a document conversion operation.
256
371
  */
257
- export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
372
+ type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : string;
373
+ export interface ConversionResult<D extends UniversalGeneratorFormat> {
258
374
  /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
259
- value: D extends 'pdf' ? Uint8Array : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
375
+ value: ConversionValue<D>;
260
376
  /** A collection of issues (warnings/infos) generated during the process. */
261
377
  messages: OfficeIssue[];
262
378
  }
@@ -366,23 +482,43 @@ export interface CommonGeneratorConfig {
366
482
  * Defaults to false.
367
483
  */
368
484
  ignoreInternalLinks?: boolean;
485
+ /**
486
+ * An optional AbortSignal to cancel the generation operation.
487
+ * When aborted, the generator immediately rejects with a standard AbortError.
488
+ * Currently supported by PdfGenerator and ChunkingGenerator.
489
+ */
490
+ abortSignal?: AbortSignal | null;
369
491
  }
370
492
  /**
371
493
  * Destination-aware generator configuration.
372
494
  * Restricts format-specific configurations to their respective destinations.
373
495
  */
374
496
  /**
375
- * Mapping of destination formats to their specific configuration interfaces.
497
+ * Maps a destination format string to its corresponding specific configuration object type.
376
498
  */
377
- export interface GeneratorSubConfigMap {
378
- html: HtmlGeneratorConfig;
379
- md: MdGeneratorConfig;
380
- pdf: PdfGeneratorConfig;
381
- csv: CsvGeneratorConfig;
382
- text: TextGeneratorConfig;
383
- rtf: RtfGeneratorConfig;
384
- chunks: ChunkingConfig;
385
- }
499
+ type GeneratorSpecificConfig<D extends string> = D extends 'html' ? {
500
+ htmlConfig?: HtmlGeneratorConfig;
501
+ } : D extends 'md' ? {
502
+ mdConfig?: MdGeneratorConfig;
503
+ } : D extends 'pdf' ? {
504
+ pdfConfig?: PdfGeneratorConfig;
505
+ } : D extends 'csv' ? {
506
+ csvConfig?: CsvGeneratorConfig;
507
+ } : D extends 'text' ? {
508
+ textConfig?: TextGeneratorConfig;
509
+ } : D extends 'rtf' ? {
510
+ rtfConfig?: RtfGeneratorConfig;
511
+ } : D extends 'chunks' ? {
512
+ chunksConfig?: ChunkingConfig;
513
+ } : Partial<{
514
+ htmlConfig: HtmlGeneratorConfig;
515
+ mdConfig: MdGeneratorConfig;
516
+ pdfConfig: PdfGeneratorConfig;
517
+ csvConfig: CsvGeneratorConfig;
518
+ textConfig: TextGeneratorConfig;
519
+ rtfConfig: RtfGeneratorConfig;
520
+ chunksConfig: ChunkingConfig;
521
+ }>;
386
522
  /**
387
523
  * Configuration options for document generators.
388
524
  *
@@ -393,9 +529,7 @@ export interface GeneratorSubConfigMap {
393
529
  *
394
530
  * @template D The destination format string. Defaults to `string` for a general configuration.
395
531
  */
396
- export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
397
- [K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
398
- };
532
+ export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & GeneratorSpecificConfig<D>;
399
533
  /**
400
534
  * Configuration options for the OfficeConverter.
401
535
  * Combines relevant parser and generator settings for a seamless one-step conversion.
@@ -440,10 +574,28 @@ export type DeepRequired<T> = T extends Function | Date | Buffer | RegExp ? T :
440
574
  * it is a discriminated union whose members cannot be uniformly deep-required.
441
575
  */
442
576
  export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
443
- [K in keyof Omit<GeneratorSubConfigMap, 'chunks'> as `${K}Config`]: GeneratorSubConfigMap[K];
577
+ htmlConfig: HtmlGeneratorConfig;
578
+ mdConfig: MdGeneratorConfig;
579
+ pdfConfig: PdfGeneratorConfig;
580
+ csvConfig: CsvGeneratorConfig;
581
+ textConfig: TextGeneratorConfig;
582
+ rtfConfig: RtfGeneratorConfig;
444
583
  }> & {
445
584
  chunksConfig: ChunkingConfig;
446
585
  };
586
+ /**
587
+ * Configuration options for granular raw HTML injections.
588
+ */
589
+ export interface HtmlInjectionConfig {
590
+ /** Raw HTML injected immediately after the opening <head> tag */
591
+ headStart?: string;
592
+ /** Raw HTML injected immediately before the closing </head> tag */
593
+ headEnd?: string;
594
+ /** Raw HTML injected immediately after the opening <body> tag */
595
+ bodyStart?: string;
596
+ /** Raw HTML injected immediately before the closing </body> tag */
597
+ bodyEnd?: string;
598
+ }
447
599
  /**
448
600
  * Configuration options for HTML generation.
449
601
  */
@@ -458,6 +610,25 @@ export interface HtmlGeneratorConfig {
458
610
  * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
459
611
  */
460
612
  chartJsSrc?: string;
613
+ /**
614
+ * Custom container width for the generated HTML.
615
+ * Can be a number (pixels) or string (e.g., '900px', '100%').
616
+ * If not specified or set to 'auto', it defaults based on the content type:
617
+ * - Spreadsheet: '100%'
618
+ * - Presentation/Slides: '1100px'
619
+ * - Standard Document (PDF/DOCX/RTF/etc.): '900px'
620
+ */
621
+ containerWidth?: string | number;
622
+ /**
623
+ * Custom CSS to append to the generated HTML document.
624
+ * This CSS will be included in the `<style>` block and can be used to style
625
+ * custom classes added during AST manipulation or override default styles.
626
+ */
627
+ customCss?: string;
628
+ /**
629
+ * Granular injection points for custom HTML, scripts, and styles.
630
+ */
631
+ injections?: HtmlInjectionConfig;
461
632
  }
462
633
  /**
463
634
  * Configuration options for PDF generation.
@@ -494,6 +665,12 @@ export interface PdfGeneratorConfig {
494
665
  * Useful for setting custom executable paths or args in CI/CD.
495
666
  */
496
667
  launchOptions?: any;
668
+ /**
669
+ * Timeout in milliseconds for PDF generation.
670
+ * Limits the time spent waiting for Puppeteer to launch, load content, and render PDF.
671
+ * Defaults to 30000 ms (30 seconds). Set to 0 to disable.
672
+ */
673
+ timeout?: number;
497
674
  }
498
675
  /**
499
676
  * Structured style mapping definition for the StyleMapper.
@@ -527,7 +704,7 @@ export interface StructuredStyleMapping {
527
704
  * The structural type of the node (e.g., 'paragraph', 'heading', 'text').
528
705
  * Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
529
706
  */
530
- nodeType?: string;
707
+ nodeType?: OfficeContentNodeType;
531
708
  /**
532
709
  * A dictionary of attributes to match on the node.
533
710
  *
@@ -593,7 +770,7 @@ export interface CsvGeneratorConfig {
593
770
  /**
594
771
  * Whether to merge all selected sheets into a single CSV.
595
772
  * If false, returns a ZIP archive containing individual CSV files.
596
- * Defaults to false.
773
+ * Defaults to true.
597
774
  */
598
775
  mergeSheets?: boolean;
599
776
  /**
@@ -799,6 +976,11 @@ export interface SemanticChunkingConfig extends BaseChunkingConfig {
799
976
  * Default is 50.
800
977
  */
801
978
  embeddingBatchSize?: number;
979
+ /**
980
+ * Timeout in milliseconds for individual embedding API calls.
981
+ * Defaults to 10000 ms (10 seconds). Set to 0 to disable.
982
+ */
983
+ timeout?: number;
802
984
  }
803
985
  /**
804
986
  * Discriminated union of all chunking strategy configurations.
@@ -847,11 +1029,16 @@ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods'
847
1029
  /**
848
1030
  * Types of content nodes in the AST.
849
1031
  */
850
- export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment';
1032
+ export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster';
851
1033
  /**
852
1034
  * Supported MIME types for attachments.
853
1035
  */
854
1036
  export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet' | 'application/vnd.openxmlformats-officedocument.presentationml.presentation' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation' | 'application/rtf' | 'text/csv' | 'text/markdown' | 'text/html';
1037
+ /**
1038
+ * Text alignment options.
1039
+ * Common in spreadsheet cells, paragraph styles, and text elements.
1040
+ */
1041
+ export type TextAlignment = 'left' | 'center' | 'right' | 'justify';
855
1042
  /**
856
1043
  * Text formatting options available for text content.
857
1044
  * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
@@ -925,7 +1112,7 @@ export interface TextFormatting {
925
1112
  * Common in spreadsheet cells or paragraph styles.
926
1113
  * @example "center", "right"
927
1114
  */
928
- alignment?: 'left' | 'center' | 'right' | 'justify';
1115
+ alignment?: TextAlignment;
929
1116
  }
930
1117
  /**
931
1118
  * Metadata for a slide in PowerPoint.
@@ -975,7 +1162,7 @@ export interface HeadingMetadata {
975
1162
  /** The heading level (e.g., 1 for H1). */
976
1163
  level: number;
977
1164
  /** The alignment of the heading. */
978
- alignment?: 'left' | 'center' | 'right' | 'justify';
1165
+ alignment?: TextAlignment;
979
1166
  /** The style of the heading. */
980
1167
  style?: string;
981
1168
  /** Detailed indentation information. */
@@ -988,7 +1175,7 @@ export interface HeadingMetadata {
988
1175
  */
989
1176
  export interface ParagraphMetadata {
990
1177
  /** The alignment of the paragraph. */
991
- alignment?: 'left' | 'center' | 'right' | 'justify';
1178
+ alignment?: TextAlignment;
992
1179
  /** The style of the paragraph. */
993
1180
  style?: string;
994
1181
  /** Detailed indentation information. */
@@ -1016,7 +1203,7 @@ export interface ListMetadata {
1016
1203
  * Text alignment of the list item.
1017
1204
  * @example 'left', 'center', 'right', 'justify'
1018
1205
  */
1019
- alignment: 'left' | 'center' | 'right' | 'justify';
1206
+ alignment: TextAlignment;
1020
1207
  /**
1021
1208
  * The list ID from the Word document's numbering definition.
1022
1209
  * Used to identify which list definition this item belongs to.
@@ -1066,6 +1253,15 @@ export interface CellMetadata {
1066
1253
  style?: string;
1067
1254
  /** Unique anchor IDs for internal linking. */
1068
1255
  anchorIds?: string[];
1256
+ /** Background color for this cell in hex format (e.g. #FFFFFF). */
1257
+ backgroundColor?: string;
1258
+ }
1259
+ /**
1260
+ * Metadata for a table.
1261
+ */
1262
+ export interface TableMetadata {
1263
+ /** Unique anchor IDs for internal linking. */
1264
+ anchorIds?: string[];
1069
1265
  }
1070
1266
  /**
1071
1267
  * Metadata for a chart node in the document.
@@ -1153,6 +1349,8 @@ export interface NoteMetadata {
1153
1349
  noteId?: string;
1154
1350
  /** Unique anchor IDs for internal linking. */
1155
1351
  anchorIds?: string[];
1352
+ /** The slide number this note is associated with (used in PowerPoint). */
1353
+ slideNumber?: number;
1156
1354
  }
1157
1355
  /**
1158
1356
  * Metadata for break nodes.
@@ -1188,10 +1386,25 @@ export interface CodeMetadata {
1188
1386
  /** Unique anchor IDs for internal linking. */
1189
1387
  anchorIds?: string[];
1190
1388
  }
1389
+ /**
1390
+ * Metadata for a comment/annotation.
1391
+ */
1392
+ export interface CommentMetadata {
1393
+ author?: string;
1394
+ initials?: string;
1395
+ date?: string;
1396
+ commentId?: string;
1397
+ }
1398
+ /**
1399
+ * Metadata for a header or footer.
1400
+ */
1401
+ export interface HeaderFooterMetadata {
1402
+ type: 'default' | 'first' | 'even' | string;
1403
+ }
1191
1404
  /**
1192
1405
  * Union type for content metadata.
1193
1406
  */
1194
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
1407
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
1195
1408
  /**
1196
1409
  * Represents a node in the document content tree.
1197
1410
  * This is the core building block of the parsed document structure.
@@ -1218,13 +1431,10 @@ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata |
1218
1431
  * children: [...]
1219
1432
  * }
1220
1433
  */
1221
- export interface OfficeContentNode {
1222
- /**
1223
- * The type of the node.
1224
- * Determines how the node should be interpreted and rendered.
1225
- * Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
1226
- */
1227
- type: OfficeContentNodeType;
1434
+ /**
1435
+ * Shared properties available on all document content nodes.
1436
+ */
1437
+ export interface BaseContentNode {
1228
1438
  /**
1229
1439
  * The complete text content of the node and all its children combined.
1230
1440
  * For container nodes (paragraph, heading), this is the concatenation of all child text.
@@ -1242,6 +1452,16 @@ export interface OfficeContentNode {
1242
1452
  * @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
1243
1453
  */
1244
1454
  children?: OfficeContentNode[];
1455
+ /**
1456
+ * Comments attached to this specific node.
1457
+ * Keeps annotations completely separate from the actual content flow.
1458
+ */
1459
+ comments?: OfficeContentNode[];
1460
+ /**
1461
+ * Notes (like footnotes or slide notes) attached to this specific node.
1462
+ * Keeps notes separate from the actual structural children.
1463
+ */
1464
+ notes?: OfficeContentNode[];
1245
1465
  /**
1246
1466
  * Text formatting applied to this node.
1247
1467
  * Only applicable to text-containing nodes.
@@ -1249,16 +1469,6 @@ export interface OfficeContentNode {
1249
1469
  * @example { bold: true, size: "12", font: "Arial" }
1250
1470
  */
1251
1471
  formatting?: TextFormatting;
1252
- /**
1253
- * Type-specific metadata providing additional context about the node.
1254
- * The metadata structure depends on the node type:
1255
- * - Headings: { level: 1 }
1256
- * - Lists: { listType: 'ordered', indentation: 0 }
1257
- * - Cells: { row: 0, col: 0 }
1258
- * - Slides: { slideNumber: 1 }
1259
- * @example { level: 1 } for a heading
1260
- */
1261
- metadata?: ContentMetadata;
1262
1472
  /**
1263
1473
  * The raw source content for this node.
1264
1474
  * - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
@@ -1270,6 +1480,93 @@ export interface OfficeContentNode {
1270
1480
  */
1271
1481
  rawContent?: string;
1272
1482
  }
1483
+ /**
1484
+ * Represents a node in the document content tree.
1485
+ * This is the core building block of the parsed document structure.
1486
+ * Content nodes can be nested to represent hierarchical document structures
1487
+ * (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
1488
+ *
1489
+ * @example
1490
+ * // A simple paragraph with formatted text
1491
+ * {
1492
+ * type: 'paragraph',
1493
+ * text: 'Hello world',
1494
+ * children: [
1495
+ * { type: 'text', text: 'Hello ', formatting: { bold: true } },
1496
+ * { type: 'text', text: 'world', formatting: { italic: true } }
1497
+ * ]
1498
+ * }
1499
+ *
1500
+ * @example
1501
+ * // A heading with metadata
1502
+ * {
1503
+ * type: 'heading',
1504
+ * text: 'Chapter 1',
1505
+ * metadata: { level: 1 },
1506
+ * children: [...]
1507
+ * }
1508
+ */
1509
+ export type OfficeContentNode = BaseContentNode & ({
1510
+ type: 'slide';
1511
+ metadata?: SlideMetadata;
1512
+ } | {
1513
+ type: 'sheet';
1514
+ metadata?: SheetMetadata;
1515
+ } | {
1516
+ type: 'heading';
1517
+ metadata?: HeadingMetadata;
1518
+ } | {
1519
+ type: 'list';
1520
+ metadata?: ListMetadata;
1521
+ } | {
1522
+ type: 'cell';
1523
+ metadata?: CellMetadata;
1524
+ } | {
1525
+ type: 'image';
1526
+ metadata?: ImageMetadata;
1527
+ } | {
1528
+ type: 'chart';
1529
+ metadata?: ChartMetadata;
1530
+ } | {
1531
+ type: 'page';
1532
+ metadata?: PageMetadata;
1533
+ } | {
1534
+ type: 'paragraph';
1535
+ metadata?: ParagraphMetadata;
1536
+ } | {
1537
+ type: 'text';
1538
+ metadata?: TextMetadata;
1539
+ } | {
1540
+ type: 'note';
1541
+ metadata?: NoteMetadata;
1542
+ } | {
1543
+ type: 'break';
1544
+ metadata?: BreakMetadata;
1545
+ } | {
1546
+ type: 'code';
1547
+ metadata?: CodeMetadata;
1548
+ } | {
1549
+ type: 'comment';
1550
+ metadata?: CommentMetadata;
1551
+ } | {
1552
+ type: 'header';
1553
+ metadata?: HeaderFooterMetadata;
1554
+ } | {
1555
+ type: 'footer';
1556
+ metadata?: HeaderFooterMetadata;
1557
+ } | {
1558
+ type: 'table';
1559
+ metadata?: TableMetadata;
1560
+ } | {
1561
+ type: 'row';
1562
+ metadata?: undefined;
1563
+ } | {
1564
+ type: 'drawing';
1565
+ metadata?: undefined;
1566
+ } | {
1567
+ type: 'slideMaster';
1568
+ metadata?: SlideMetadata;
1569
+ });
1273
1570
  /**
1274
1571
  * Structured information extracted from a chart.
1275
1572
  */
@@ -1409,14 +1706,40 @@ export interface OfficeMetadata {
1409
1706
  * Values are typed as string, number, boolean, or Date where the source format provides type information.
1410
1707
  */
1411
1708
  customProperties?: Record<string, string | number | boolean | Date>;
1709
+ /** Keywords associated with the document. */
1710
+ keywords?: string;
1711
+ /**
1712
+ * Contains all format-specific metadata fields extracted verbatim.
1713
+ * Consumers can use this to access properties not mapped to the standard OfficeMetadata fields.
1714
+ * Examples: all <meta> tags in HTML, app.xml properties in DOCX, XMP dicts in PDF.
1715
+ */
1716
+ nativeProperties?: Record<string, any>;
1717
+ }
1718
+ /**
1719
+ * Contains out-of-band layout elements and templates that are not part of the main document flow.
1720
+ */
1721
+ export interface OfficeAuxiliaryContent {
1722
+ /** Headers extracted from the document. */
1723
+ headers?: OfficeContentNode[];
1724
+ /** Footers extracted from the document. */
1725
+ footers?: OfficeContentNode[];
1726
+ /** Slide Masters extracted from presentations. */
1727
+ slideMasters?: OfficeContentNode[];
1412
1728
  }
1413
1729
  /**
1414
- * The Abstract Syntax Tree (AST) returned by the parser.
1415
- * This is the root data structure representing the entire parsed document.
1730
+ * The Root Abstract Syntax Tree (AST) representing a parsed Office Document.
1731
+ * This is the ultimate output of `OfficeParser.parseOffice()`.
1416
1732
  *
1417
- * The AST provides a format-agnostic representation of the document that can be easily
1418
- * processed, transformed, or converted to other formats. It preserves the document's
1419
- * structure, content, formatting, and metadata while abstracting away format-specific details.
1733
+ * DESIGN PHILOSOPHY:
1734
+ * The AST is designed to be a universal, format-agnostic representation of document content.
1735
+ * Whether the input was a PDF, DOCX, XLSX, Markdown, or HTML file, the resulting AST
1736
+ * uses the same consistent structure (`OfficeContentNode` trees).
1737
+ *
1738
+ * ### Key Top-Level Properties:
1739
+ * - `metadata`: Document-level properties (author, title, stats).
1740
+ * - `content`: The main sequential flow of the document (paragraphs, tables, slides, sheets).
1741
+ * - `attachments`: Extracted binary assets (images, embedded files).
1742
+ * - `auxiliary`: Out-of-band layout/template elements (headers, footers, slide masters).
1420
1743
  *
1421
1744
  * @example
1422
1745
  * ```typescript
@@ -1466,6 +1789,12 @@ export interface OfficeParserAST {
1466
1789
  * @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
1467
1790
  */
1468
1791
  content: OfficeContentNode[];
1792
+ /**
1793
+ * Out-of-band layout and template elements that are not part of the main text flow.
1794
+ * Extracted only if the respective `ignore...` config flags are false.
1795
+ * Contains elements like `headers`, `footers`, and `slideMasters`.
1796
+ */
1797
+ auxiliary?: OfficeAuxiliaryContent;
1469
1798
  /**
1470
1799
  * Attachments extracted from the document (images, charts, embedded files).
1471
1800
  * Only populated when `config.extractAttachments` is true.
@@ -1509,5 +1838,6 @@ export interface OfficeParserAST {
1509
1838
  * const md = await ast.to('md');
1510
1839
  * ```
1511
1840
  */
1512
- to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1841
+ to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
1513
1842
  }
1843
+ export {};
package/dist/types.js CHANGED
@@ -33,6 +33,8 @@ var OfficeErrorType;
33
33
  OfficeErrorType["INVALID_OUTPUT_MAPPING"] = "INVALID_OUTPUT_MAPPING";
34
34
  /** Semantic chunking strategy is selected but no embedding function is provided */
35
35
  OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
36
+ /** The operation was aborted */
37
+ OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
36
38
  })(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
37
39
  /**
38
40
  * Standard warning types for OfficeParser.
@@ -72,4 +74,6 @@ var OfficeWarningType;
72
74
  OfficeWarningType["EMPTY_CHUNK_GENERATED"] = "EMPTY_CHUNK_GENERATED";
73
75
  /** A node was skipped because it only contained whitespace */
74
76
  OfficeWarningType["WHITESPACE_NODE_SKIPPED"] = "WHITESPACE_NODE_SKIPPED";
77
+ /** The HTML generator containerWidth option is invalid */
78
+ OfficeWarningType["INVALID_CONTAINER_WIDTH"] = "INVALID_CONTAINER_WIDTH";
75
79
  })(OfficeWarningType || (exports.OfficeWarningType = OfficeWarningType = {}));
@@ -1,4 +1,4 @@
1
- import { OfficeParserAST, OfficeContentNode, OfficeMetadata, OfficeAttachment, SupportedFileType, OfficeParserConfig } from '../types.js';
1
+ import { OfficeAttachment, OfficeAuxiliaryContent, OfficeContentNode, OfficeMetadata, OfficeParserAST, OfficeParserConfig, SupportedFileType } from '../types.js';
2
2
  /**
3
3
  * Creates a fully-featured OfficeParserAST object with conversion methods.
4
4
  *
@@ -13,4 +13,4 @@ import { OfficeParserAST, OfficeContentNode, OfficeMetadata, OfficeAttachment, S
13
13
  * @param toTextSync - Synchronous text extraction logic (for backward compatibility)
14
14
  * @returns An object conforming to OfficeParserAST
15
15
  */
16
- export declare function createAST(type: SupportedFileType, metadata: OfficeMetadata, content: OfficeContentNode[], attachments: OfficeAttachment[], config: OfficeParserConfig, toTextSync: () => string): OfficeParserAST;
16
+ export declare function createAST(type: SupportedFileType, metadata: OfficeMetadata, content: OfficeContentNode[], attachments: OfficeAttachment[], config: OfficeParserConfig, auxiliary: OfficeAuxiliaryContent | undefined, toTextSync: () => string): OfficeParserAST;