officeparser 6.1.1 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +301 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +74 -31
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +828 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +783 -5
  54. package/dist/types.js +73 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.d.ts +8 -3
  60. package/dist/utils/envUtils.js +117 -34
  61. package/dist/utils/errorUtils.d.ts +17 -29
  62. package/dist/utils/errorUtils.js +110 -52
  63. package/dist/utils/moduleLoader.js +19 -11
  64. package/dist/utils/ocrUtils.js +2 -1
  65. package/dist/utils/sheetUtils.d.ts +7 -0
  66. package/dist/utils/sheetUtils.js +35 -0
  67. package/dist/utils/styleMapper.d.ts +36 -0
  68. package/dist/utils/styleMapper.js +224 -0
  69. package/dist/utils/xmlUtils.d.ts +0 -8
  70. package/dist/utils/xmlUtils.js +2 -1
  71. package/package.json +26 -7
package/dist/types.d.ts CHANGED
@@ -1,3 +1,73 @@
1
+ /**
2
+ * Standard error types for OfficeParser.
3
+ * Use these to identify the kind of error being reported.
4
+ */
5
+ export declare enum OfficeErrorType {
6
+ /** Unsupported file extension */
7
+ EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
8
+ /** File appears to be corrupted or malformed */
9
+ FILE_CORRUPTED = "FILE_CORRUPTED",
10
+ /** File could not be found at the specified path */
11
+ FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST",
12
+ /** Specified location/directory is not reachable or is a directory */
13
+ LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND",
14
+ /** Arguments passed to the function are missing or invalid */
15
+ IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
16
+ /** Error occurred while reading or processing file buffers */
17
+ IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
18
+ /** Input type is not a supported type (string, Buffer, ArrayBuffer) */
19
+ INVALID_INPUT = "INVALID_INPUT",
20
+ /** PDF worker source is missing (required in browser) */
21
+ PDF_WORKER_MISSING = "PDF_WORKER_MISSING",
22
+ /** Attempted to use Node.js-only features in a browser environment */
23
+ FEATURE_NOT_SUPPORTED_IN_BROWSER = "FEATURE_NOT_SUPPORTED_IN_BROWSER",
24
+ /** Style mapping string is malformed */
25
+ INVALID_STYLE_MAPPING = "INVALID_STYLE_MAPPING",
26
+ /** Selector in style mapping is invalid */
27
+ INVALID_SELECTOR = "INVALID_SELECTOR",
28
+ /** Output mapping in style mapping is invalid */
29
+ INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
30
+ /** Semantic chunking strategy is selected but no embedding function is provided */
31
+ MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
32
+ }
33
+ /**
34
+ * Standard warning types for OfficeParser.
35
+ * Use these for reporting non-fatal issues or performance tips.
36
+ */
37
+ export declare enum OfficeWarningType {
38
+ /** Performance advice (e.g., Rosetta translation on Mac) */
39
+ PERFORMANCE_TIP = "PERFORMANCE_TIP",
40
+ /** OCR processing failed for an attachment */
41
+ OCR_FAILED = "OCR_FAILED",
42
+ /** Extraction of structured chart data failed */
43
+ CHART_DATA_EXTRACTION_FAILED = "CHART_DATA_EXTRACTION_FAILED",
44
+ /** Automatic worker path failed, falling back to CDN */
45
+ PDF_WORKER_FALLBACK = "PDF_WORKER_FALLBACK",
46
+ /** General attachment extraction failure */
47
+ ATTACHMENT_EXTRACTION_FAILED = "ATTACHMENT_EXTRACTION_FAILED",
48
+ /** Failed to load a specific page in a multi-page document */
49
+ PAGE_LOAD_FAILED = "PAGE_LOAD_FAILED",
50
+ /** Failed to load a required dynamic dependency */
51
+ DEPENDENCY_LOAD_FAILED = "DEPENDENCY_LOAD_FAILED",
52
+ /** Failed to extract images from a source */
53
+ IMAGE_EXTRACTION_FAILED = "IMAGE_EXTRACTION_FAILED",
54
+ /** Failed to extract annotations from a document */
55
+ ANNOTATION_EXTRACTION_FAILED = "ANNOTATION_EXTRACTION_FAILED",
56
+ /** Failed to process an extracted image bitmap */
57
+ IMAGE_PROCESSING_FAILED = "IMAGE_PROCESSING_FAILED",
58
+ /** Warning about limitations of browser-based generation */
59
+ BROWSER_GENERATION_LIMITATION = "BROWSER_GENERATION_LIMITATION",
60
+ /** Specified sheet range in Excel/ODS export was not found */
61
+ SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND",
62
+ /** Buffer content type does not match the provided or expected file extension */
63
+ BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH",
64
+ /** Failed to detect file type from buffer due to library error or incompatibility */
65
+ FILE_TYPE_DETECTION_FAILED = "FILE_TYPE_DETECTION_FAILED",
66
+ /** No chunks were generated for the document given the current strategy */
67
+ EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
68
+ /** A node was skipped because it only contained whitespace */
69
+ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
70
+ }
1
71
  /**
2
72
  * Configuration options for OCR.
3
73
  */
@@ -16,16 +86,19 @@ export interface OcrConfig {
16
86
  /**
17
87
  * Path to the Tesseract worker script.
18
88
  * Primarily used for offline/air-gapped environments.
89
+ * Default is ''.
19
90
  */
20
91
  workerPath?: string;
21
92
  /**
22
93
  * Path to the Tesseract core script.
23
94
  * Primarily used for offline/air-gapped environments.
95
+ * Default is ''.
24
96
  */
25
97
  corePath?: string;
26
98
  /**
27
99
  * Path for Tesseract language files (traineddata).
28
100
  * Primarily used for offline/air-gapped environments.
101
+ * Default is ''.
29
102
  */
30
103
  langPath?: string;
31
104
  /**
@@ -40,10 +113,17 @@ export interface OcrConfig {
40
113
  */
41
114
  export interface OfficeParserConfig {
42
115
  /**
116
+ * @deprecated Use `onWarning` instead.
43
117
  * Flag to show all the logs to console in case of an error irrespective of your own handling.
44
118
  * Default is false.
45
119
  */
46
120
  outputErrorToConsole?: boolean;
121
+ /**
122
+ * Callback for warnings or non-fatal errors encountered during parsing.
123
+ * Allows you to capture issues like OCR failures or attachment extraction errors
124
+ * without stopping the parsing process.
125
+ */
126
+ onWarning?: (issue: OfficeIssue) => void;
47
127
  /**
48
128
  * The delimiter used for every new line in places that allow multiline text like word.
49
129
  * Default is \n.
@@ -112,7 +192,7 @@ export interface OfficeParserConfig {
112
192
  * The URL/path to the PDF.js worker script.
113
193
  *
114
194
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
115
- * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
195
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
116
196
  * You can override this with your own local path or a different CDN link.
117
197
  */
118
198
  pdfWorkerSrc?: string;
@@ -123,19 +203,655 @@ export interface OfficeParserConfig {
123
203
  * Default is false
124
204
  */
125
205
  includeBreakNodes?: boolean;
206
+ /**
207
+ * Flag to ignore all internal (anchor) links during parsing.
208
+ * When true, all bookmarks, cross-references, and internal document jumps are stripped
209
+ * from the AST. Only external URLs will be preserved.
210
+ *
211
+ * Use this if you want a "flat" document without any internal interactivity.
212
+ *
213
+ * Default is false.
214
+ */
215
+ ignoreInternalLinks?: boolean;
216
+ /**
217
+ * Optional hint for the file format.
218
+ * When a Buffer or ArrayBuffer is passed, the parser relies on magic bytes to detect the file type.
219
+ * Text-based formats like 'md', 'html', and 'csv' lack reliable magic bytes.
220
+ * If you are parsing these formats from a Buffer, you must provide this fileType hint.
221
+ *
222
+ * This is authoritative and is used to determine the file type, so it should be accurate.
223
+ * If provided, this bypasses the magic bytes detection and the file extension-based detection either way.
224
+ *
225
+ * Default is null.
226
+ */
227
+ fileType?: SupportedFileType | null;
228
+ /**
229
+ * Custom delimiter for CSV files.
230
+ * Defaults to ',' but can be overridden (e.g., ';', '\t').
231
+ */
232
+ csvDelimiter?: string;
233
+ }
234
+ /**
235
+ * A fully-populated parser configuration containing all options.
236
+ * Used internally for merging and resolution.
237
+ */
238
+ export type FullOfficeParserConfig = DeepRequired<OfficeParserConfig>;
239
+ /**
240
+ * Represents a single issue (warning, error, or info) generated during document processing.
241
+ */
242
+ export interface OfficeIssue {
243
+ /** The severity of the issue. */
244
+ type: 'warning' | 'info' | 'error';
245
+ /** Human-readable message text. */
246
+ message: string;
247
+ /** The specific AST node that triggered this issue, if applicable. */
248
+ node?: OfficeContentNode;
249
+ /** A unique error code for programmatic handling. */
250
+ code: OfficeWarningType | OfficeErrorType;
251
+ /** Optional additional context or original error object. */
252
+ details?: any;
253
+ }
254
+ /**
255
+ * The result of a document conversion operation.
256
+ */
257
+ export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
258
+ /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
259
+ value: D extends 'pdf' ? Uint8Array : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
260
+ /** A collection of issues (warnings/infos) generated during the process. */
261
+ messages: OfficeIssue[];
262
+ }
263
+ /**
264
+ * Universal formats supported by all source types for generation.
265
+ */
266
+ export type UniversalGeneratorFormat = 'text' | 'md' | 'html' | 'pdf' | 'csv' | 'rtf' | 'chunks';
267
+ /**
268
+ * Allowed destination formats for a given source type.
269
+ * Currently, all generators are universal across all source formats.
270
+ */
271
+ export type SupportedDestination<_T extends SupportedFileType = SupportedFileType> = UniversalGeneratorFormat;
272
+ /**
273
+ * Configuration options for the OfficeGenerator.
274
+ */
275
+ /**
276
+ * Common configuration options for all generators.
277
+ */
278
+ export interface CommonGeneratorConfig {
279
+ /**
280
+ * Callback called for every node during generation.
281
+ * Allows users to modify nodes before processing, completely override rendering, or filter them out.
282
+ *
283
+ * #### Callback Capabilities:
284
+ * 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
285
+ * 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
286
+ * 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
287
+ * 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
288
+ */
289
+ onNode?: (node: OfficeContentNode) => string | false | Promise<string | false | void> | void;
290
+ /**
291
+ * Callback for warnings, non-fatal errors, or issues encountered during generation.
292
+ * Allows the process to continue while reporting skipping or approximation of content.
293
+ */
294
+ onWarning?: (issue: OfficeIssue) => void;
295
+ /**
296
+ * Map document styles (e.g., 'Heading 1', 'Intense Quote') to specific semantic elements.
297
+ *
298
+ * DESIGN PHILOSOPHY:
299
+ * This is the primary way to customize how the library interprets the visual
300
+ * structure of your source documents.
301
+ *
302
+ * To disable all semantic translation and use raw AST types only,
303
+ * set `ignoreDefaultStyleMap: true` and leave `styleMap` empty.
304
+ *
305
+ * It supports two formats:
306
+ *
307
+ * 1. LEGACY STRING DSL:
308
+ * Simple "selector => output" syntax. Highly compatible with mammoth.js style maps.
309
+ * @example ["p[style-name='Heading 1'] => h1"]
310
+ * @example ["p[style='Quote'] => blockquote"]
311
+ *
312
+ * 2. STRUCTURED OBJECTS (Recommended):
313
+ * More powerful and strictly typed. Ideal for complex logic or when you
314
+ * need to apply specific classes/attributes for the HTML generator.
315
+ * @example
316
+ * [
317
+ * {
318
+ * selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
319
+ * output: { tag: 'h1', classes: ['main-title'], attributes: { id: 'top' } }
320
+ * }
321
+ * ]
322
+ *
323
+ * Note: This property works in conjunction with `ignoreDefaultStyleMap`.
324
+ * Defaults to a robust built-in map that covers common standard Office styles.
325
+ */
326
+ styleMap?: string[] | StructuredStyleMapping[];
327
+ /**
328
+ * Whether to include visual formatting like font size, font family, and colors in the output.
329
+ * Set to false for clean, semantic output.
330
+ * Defaults to true.
331
+ */
332
+ includeFormatting?: boolean;
333
+ /**
334
+ * Whether to automatically generate unique slug-based IDs for headings.
335
+ * Useful for table-of-contents and anchor links.
336
+ * Defaults to true.
337
+ */
338
+ generateIds?: boolean;
339
+ /**
340
+ * Whether to render document metadata (title, author, etc.) as visible content
341
+ * in the generated output (e.g., a header block in HTML or plain text).
342
+ * Structural metadata (HTML <meta> tags, Markdown YAML frontmatter) is always included.
343
+ * Defaults to false.
344
+ */
345
+ renderMetadata?: boolean;
346
+ /**
347
+ * Whether to ignore the built-in default style mappings (e.g. "Heading 1" -> h1).
348
+ * Set to true if you want full control over style mapping.
349
+ * Defaults to false.
350
+ */
351
+ ignoreDefaultStyleMap?: boolean;
352
+ /**
353
+ * Whether to include images in the generated output.
354
+ * Defaults to true.
355
+ */
356
+ includeImages?: boolean;
357
+ /**
358
+ * Whether to include interactive charts in the generated output (HTML only).
359
+ * Defaults to true.
360
+ */
361
+ includeCharts?: boolean;
362
+ /**
363
+ * Whether to ignore all internal (anchor) links and anchor IDs during generation.
364
+ * When true, all bookmarks, cross-references, and internal document jumps are stripped.
365
+ * Specifically for Markdown, this removes the {#id} block from headings.
366
+ * Defaults to false.
367
+ */
368
+ ignoreInternalLinks?: boolean;
369
+ }
370
+ /**
371
+ * Destination-aware generator configuration.
372
+ * Restricts format-specific configurations to their respective destinations.
373
+ */
374
+ /**
375
+ * Mapping of destination formats to their specific configuration interfaces.
376
+ */
377
+ export interface GeneratorSubConfigMap {
378
+ html: HtmlGeneratorConfig;
379
+ md: MdGeneratorConfig;
380
+ pdf: PdfGeneratorConfig;
381
+ csv: CsvGeneratorConfig;
382
+ text: TextGeneratorConfig;
383
+ rtf: RtfGeneratorConfig;
384
+ chunks: ChunkingConfig;
385
+ }
386
+ /**
387
+ * Configuration options for document generators.
388
+ *
389
+ * This interface is designed to be format-aware. When you specify a destination format
390
+ * (e.g., `OfficeGenerator.generate(ast, 'html', config)`), the generic parameter `D`
391
+ * ensures that only the relevant sub-configuration (e.g., `htmlConfig`) is available
392
+ * for type checking.
393
+ *
394
+ * @template D The destination format string. Defaults to `string` for a general configuration.
395
+ */
396
+ export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
397
+ [K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
398
+ };
399
+ /**
400
+ * Configuration options for the OfficeConverter.
401
+ * Combines relevant parser and generator settings for a seamless one-step conversion.
402
+ *
403
+ * @template D The destination format string.
404
+ */
405
+ /**
406
+ * Configuration options for the OfficeConverter.
407
+ * Combines general generator settings with a specific subset of parser settings.
408
+ *
409
+ * @template D The destination format string.
410
+ * @template T The source file type.
411
+ */
412
+ export type OfficeConverterConfig<D extends string = string, T extends SupportedFileType = SupportedFileType> = {
413
+ /**
414
+ * Specific configuration for the source parsing phase.
415
+ */
416
+ parseConfig?: OfficeParserConfig & {
417
+ fileType?: T;
418
+ };
419
+ /**
420
+ * Specific configuration for the destination generation phase.
421
+ */
422
+ generatorConfig?: GeneratorConfig<D>;
423
+ /**
424
+ * Callback for warnings or non-fatal errors encountered during the entire conversion process.
425
+ * This is passed to both the parser and the generator.
426
+ * If provided, this takes precedence over callbacks inside parseConfig or generatorConfig.
427
+ */
428
+ onWarning?: (issue: OfficeIssue) => void;
429
+ };
430
+ /**
431
+ * Deeply required type helper.
432
+ */
433
+ export type DeepRequired<T> = T extends Function | Date | Buffer | RegExp ? T : T extends Array<infer U> ? Array<DeepRequired<U>> : T extends object ? {
434
+ [P in keyof T]-?: DeepRequired<T[P]>;
435
+ } : T;
436
+ /**
437
+ * A fully-populated generator configuration containing all sub-configs.
438
+ * Used internally for merging and resolution.
439
+ * `chunksConfig` is typed as `ChunkingConfig` directly (not DeepRequired) because
440
+ * it is a discriminated union whose members cannot be uniformly deep-required.
441
+ */
442
+ export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
443
+ [K in keyof Omit<GeneratorSubConfigMap, 'chunks'> as `${K}Config`]: GeneratorSubConfigMap[K];
444
+ }> & {
445
+ chunksConfig: ChunkingConfig;
446
+ };
447
+ /**
448
+ * Configuration options for HTML generation.
449
+ */
450
+ export interface HtmlGeneratorConfig {
451
+ /**
452
+ * Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
453
+ * Defaults to true.
454
+ */
455
+ standalone?: boolean;
456
+ /**
457
+ * URL for the Chart.js library to use when 'includeCharts' is true.
458
+ * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
459
+ */
460
+ chartJsSrc?: string;
461
+ }
462
+ /**
463
+ * Configuration options for PDF generation.
464
+ * Maps closely to Puppeteer's PDF options.
465
+ */
466
+ export interface PdfGeneratorConfig {
467
+ /** Paper format. Defaults to 'A4'. */
468
+ format?: 'letter' | 'legal' | 'tabloid' | 'ledger' | 'a0' | 'a1' | 'a2' | 'a3' | 'a4' | 'a5' | 'a6' | 'Letter' | 'Legal' | 'Tabloid' | 'Ledger' | 'A0' | 'A1' | 'A2' | 'A3' | 'A4' | 'A5' | 'A6';
469
+ /** Paper width, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
470
+ width?: string | number;
471
+ /** Paper height, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
472
+ height?: string | number;
473
+ /** Whether to print in landscape orientation. Defaults to false. */
474
+ landscape?: boolean;
475
+ /** Whether to print background graphics. Defaults to true. */
476
+ printBackground?: boolean;
477
+ /** Scale of the webpage rendering. Defaults to 1. */
478
+ scale?: number;
479
+ /** Paper margins. */
480
+ margin?: {
481
+ top?: string | number;
482
+ right?: string | number;
483
+ bottom?: string | number;
484
+ left?: string | number;
485
+ };
486
+ /** Whether to display header and footer. Defaults to false. */
487
+ displayHeaderFooter?: boolean;
488
+ /** HTML template for the print header. */
489
+ headerTemplate?: string;
490
+ /** HTML template for the print footer. */
491
+ footerTemplate?: string;
492
+ /**
493
+ * Optional Puppeteer launch options for Node.js environment.
494
+ * Useful for setting custom executable paths or args in CI/CD.
495
+ */
496
+ launchOptions?: any;
497
+ }
498
+ /**
499
+ * Structured style mapping definition for the StyleMapper.
500
+ *
501
+ * DESIGN PHILOSOPHY: "Semantic Translation"
502
+ * -----------------------------------------
503
+ * Office documents (Word, RTF, PPTX) often use custom or localized style names
504
+ * (e.g., "Heading 1" in English vs "Titre 1" in French, or "MyCompany-Quote").
505
+ *
506
+ * This interface allows you to create a "semantic bridge" between these arbitrary
507
+ * source styles and a universal vocabulary of document elements.
508
+ *
509
+ * WHY USE HTML TAGS FOR NON-HTML OUTPUT?
510
+ * --------------------------------------
511
+ * We use HTML tags (`h1`, `blockquote`, `code`, `pre`) as a "Universal Intermediate
512
+ * Language". By mapping a custom Word style to `blockquote`, you are defining its
513
+ * SEMANTIC MEANING rather than its physical appearance.
514
+ *
515
+ * Each generator then interprets this meaning natively:
516
+ * - HTML Generator: Directly renders the `<blockquote>` tag with your classes.
517
+ * - Markdown Generator: Sees 'blockquote' and renders the standard `> ` prefix.
518
+ * - Text Generator: Sees 'blockquote' and applies appropriate structural indentation.
519
+ */
520
+ export interface StructuredStyleMapping {
521
+ /**
522
+ * The criteria used to identify which AST nodes should be transformed.
523
+ * Think of this as the "Source Filter".
524
+ */
525
+ selector: {
526
+ /**
527
+ * The structural type of the node (e.g., 'paragraph', 'heading', 'text').
528
+ * Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
529
+ */
530
+ nodeType?: string;
531
+ /**
532
+ * A dictionary of attributes to match on the node.
533
+ *
534
+ * The most common use case is matching the 'style' attribute from
535
+ * Word documents (e.g., { style: 'Intense Quote' }).
536
+ *
537
+ * Matchers:
538
+ * - Literal: `style: 'Heading 1'` matches exactly.
539
+ * - Operator: `{ value: 'Title', operator: '~=' }` matches if the word 'Title'
540
+ * is found within the style name.
541
+ */
542
+ attributes?: Record<string, string | number | boolean | {
543
+ value: string | number | boolean;
544
+ operator: '=' | '~=';
545
+ }>;
546
+ };
547
+ /**
548
+ * The target representation for the matched node.
549
+ * Think of this as the "Semantic Meaning" you want to assign to the match.
550
+ */
551
+ output: {
552
+ /**
553
+ * The universal semantic tag (e.g., 'h1', 'h2', 'blockquote', 'code', 'pre', 'u').
554
+ * All generators use this tag to decide their native output syntax.
555
+ */
556
+ tag: string;
557
+ /**
558
+ * CSS classes to apply to the output.
559
+ * This is utilized by the HTML generator to allow for downstream CSS styling.
560
+ */
561
+ classes?: string[];
562
+ /**
563
+ * Key-value pair of HTML attributes (like 'id', 'data-*', or 'style') to apply.
564
+ * Primarily used by the HTML generator for high-fidelity conversion.
565
+ */
566
+ attributes?: Record<string, string>;
567
+ /**
568
+ * If true, prevents the generator from collapsing this element into
569
+ * adjacent elements of the same type.
570
+ *
571
+ * For example, multiple paragraphs mapped to 'blockquote' normally merge into
572
+ * one big blockquote. Setting `fresh: true` forces them to be separate blocks.
573
+ */
574
+ fresh?: boolean;
575
+ };
576
+ }
577
+ /**
578
+ * Configuration options for RTF generation.
579
+ */
580
+ export interface RtfGeneratorConfig {
581
+ }
582
+ /**
583
+ * Configuration options for CSV generation.
584
+ */
585
+ export interface CsvGeneratorConfig {
586
+ /**
587
+ * Range of sheets to export.
588
+ * Supports formats like "1", "1-3", "1,2", "1,3-5,7".
589
+ * 1-based indexing.
590
+ * Default is '' (all sheets).
591
+ */
592
+ sheets?: string;
593
+ /**
594
+ * Whether to merge all selected sheets into a single CSV.
595
+ * If false, returns a ZIP archive containing individual CSV files.
596
+ * Defaults to false.
597
+ */
598
+ mergeSheets?: boolean;
599
+ /**
600
+ * Custom delimiter for CSV files.
601
+ * Defaults to ','.
602
+ */
603
+ columnDelimiter?: string;
604
+ }
605
+ /**
606
+ * Configuration options for Markdown generation.
607
+ */
608
+ export interface MdGeneratorConfig {
609
+ /**
610
+ * Whether to fallback to HTML tags for features not supported by standard Markdown.
611
+ *
612
+ * Markdown has limited support for complex document structures. This flag controls how
613
+ * the generator handles features that cannot be represented in pure Markdown:
614
+ *
615
+ * 1. If a feature is NOT supported natively by Markdown (e.g., nested tables, text alignment,
616
+ * underline, subscript/superscript):
617
+ * - If true: The generator will use HTML tags (<u>, <sub>, <div>, <table>, etc.) to
618
+ * maintain high fidelity.
619
+ * - If false: The generator will skip or simplify the feature (e.g., ignoring alignment,
620
+ * skipping underline, or hoisting nested tables out of their cells).
621
+ *
622
+ * 2. If a feature IS supported by Markdown but a higher quality version is possible
623
+ * via HTML (e.g., tables with merged cells):
624
+ * - If true: Use HTML for better fidelity.
625
+ * - If false: Use native Markdown syntax (e.g., a standard GFM table grid).
626
+ *
627
+ * Defaults to true.
628
+ */
629
+ fallbackToHtml?: boolean;
630
+ }
631
+ /**
632
+ * Configuration options for plain text generation.
633
+ */
634
+ export interface TextGeneratorConfig {
635
+ /**
636
+ * The delimiter used for every new line.
637
+ * Defaults to '\n'.
638
+ */
639
+ newlineDelimiter?: string;
640
+ /**
641
+ * Whether to attempt to preserve the original document layout.
642
+ * If true, tables will be rendered with separators and aligned columns.
643
+ * If false, output will be a flat stream of text nodes.
644
+ * Defaults to false.
645
+ */
646
+ preserveLayout?: boolean;
647
+ }
648
+ /**
649
+ * The strategy used for chunking a document for RAG pipelines.
650
+ * - 'fixed-size': Traditional character/token count based splitting.
651
+ * - 'document-structure': Leverages the AST to split at natural document boundaries.
652
+ * - 'semantic': Uses embedding similarity to find natural topic breakpoints.
653
+ */
654
+ export type ChunkingStrategy = 'fixed-size' | 'document-structure' | 'semantic';
655
+ /**
656
+ * Base configuration applicable to all chunking strategies.
657
+ */
658
+ export interface BaseChunkingConfig {
659
+ /**
660
+ * The strategy used for chunking.
661
+ * Default is 'document-structure'.
662
+ */
663
+ strategy?: ChunkingStrategy;
664
+ /**
665
+ * A function that measures the size of a text string.
666
+ * Defaults to character count: `(text) => text.length`.
667
+ * Override with a token counter (e.g., `tiktoken`) for strict LLM context window adherence.
668
+ */
669
+ lengthFunction?: (text: string) => number;
670
+ /**
671
+ * Whether to strip leading/trailing whitespace from each chunk.
672
+ * Default is true.
673
+ */
674
+ stripWhitespace?: boolean;
675
+ /**
676
+ * Whether to include rich AST metadata (page number, slide number, heading, etc.)
677
+ * in the generated chunk objects.
678
+ * Default is true.
679
+ */
680
+ includeMetadata?: boolean;
681
+ /**
682
+ * Whether to include the starting character index of each chunk
683
+ * relative to the whole document. Useful for UI text highlighting.
684
+ * Default is false.
685
+ */
686
+ addStartIndex?: boolean;
687
+ /**
688
+ * Optional custom regex (as string or RegExp object) to identify sentence boundaries.
689
+ * Use this for languages or specific document types that require custom splitting logic.
690
+ * If provided, it overrides or augments the default segmenter.
691
+ * @example /[。?!]/
692
+ */
693
+ sentenceBoundaryRegex?: string | RegExp;
694
+ /**
695
+ * Optional list of abbreviations to ignore when splitting text into sentences.
696
+ * These words, if followed by a period, will not be treated as sentence boundaries.
697
+ * Use this to handle language-specific or domain-specific abbreviations.
698
+ * @example ["Inc", "Ltd", "approx"]
699
+ */
700
+ abbreviations?: string[];
701
+ }
702
+ /**
703
+ * Configuration for Fixed-Size Chunking.
704
+ * Cuts text based on a maximum size limit with an optional overlap.
705
+ * This is equivalent to LangChain's `RecursiveCharacterTextSplitter`.
706
+ */
707
+ export interface FixedSizeChunkingConfig extends BaseChunkingConfig {
708
+ strategy: 'fixed-size';
709
+ /**
710
+ * Maximum size of the chunk, measured by `lengthFunction`.
711
+ * Default is 1000 characters.
712
+ */
713
+ chunkSize?: number;
714
+ /**
715
+ * Number of characters/tokens to overlap between consecutive chunks
716
+ * to avoid losing context at boundaries.
717
+ * Rule of thumb: ~10–20% of `chunkSize`.
718
+ * Default is 200.
719
+ */
720
+ chunkOverlap?: number;
721
+ /**
722
+ * Ordered list of separators to try when splitting.
723
+ * The chunker tries each in order; if a split would exceed `chunkSize`,
724
+ * it tries the next separator.
725
+ * Default is ['\n\n', '\n', ' ', ''].
726
+ */
727
+ separators?: string[];
728
+ }
729
+ /**
730
+ * Configuration for Document-Structure Chunking.
731
+ * Uses the officeParser AST to split at natural document boundaries like
732
+ * headings, paragraphs, slides, or pages. This is the recommended strategy
733
+ * as it preserves semantic context from the document's own structure.
734
+ */
735
+ export interface DocumentStructureChunkingConfig extends BaseChunkingConfig {
736
+ strategy: 'document-structure';
737
+ /**
738
+ * The primary structural element at which to force a chunk boundary.
739
+ * - 'paragraph': Never cross a paragraph boundary (finest-grained, most precise).
740
+ * - 'heading': Split at every heading change.
741
+ * - 'page': Chunks never span multiple pages (PDF only).
742
+ * - 'slide': Chunks never span multiple slides (PPTX/ODP only).
743
+ * - 'sheet': Chunks never span multiple sheets (XLSX/ODS only).
744
+ * Default is 'paragraph'.
745
+ */
746
+ splitBy?: 'page' | 'slide' | 'sheet' | 'heading' | 'paragraph';
747
+ /**
748
+ * Maximum size of a chunk (measured by `lengthFunction`).
749
+ * If a single structural unit (e.g., one paragraph) exceeds this limit,
750
+ * it will be further split using a recursive character splitter.
751
+ * Default is 1000 characters.
752
+ */
753
+ maxChunkSize?: number;
754
+ /**
755
+ * How to handle table nodes when splitting.
756
+ * - 'row': Split by rows, REPEATING the header row in every chunk so the LLM
757
+ * always understands what the columns mean. (Highly recommended for RAG)
758
+ * - 'flatten': Convert the table to plain text and split like a regular block.
759
+ * Default is 'row'.
760
+ */
761
+ tableSplitStrategy?: 'row' | 'flatten';
762
+ }
763
+ /**
764
+ * Configuration for Semantic Chunking.
765
+ * Uses an embedding model to detect topic shifts and create boundaries
766
+ * where content meaning naturally changes. Computationally expensive but
767
+ * produces the highest quality chunks.
768
+ */
769
+ export interface SemanticChunkingConfig extends BaseChunkingConfig {
770
+ strategy: 'semantic';
771
+ /**
772
+ * A user-provided async function to generate vector embeddings for a text string.
773
+ * Required. Example: a wrapper around OpenAI's `text-embedding-3-small`.
774
+ * @example async (text) => await openai.embeddings.create({ input: text, model: 'text-embedding-3-small' }).then(r => r.data[0].embedding)
775
+ */
776
+ embeddingFunction: (text: string) => Promise<number[]>;
777
+ /**
778
+ * The cosine similarity threshold below which a chunk boundary is created.
779
+ * When the similarity between two adjacent sentences drops below this value,
780
+ * a new chunk starts. Higher = more splits, smaller chunks.
781
+ * Default is 0.8.
782
+ */
783
+ similarityThreshold?: number;
784
+ /**
785
+ * Maximum size of a chunk even if semantic similarity remains high.
786
+ * Prevents runaway chunks when an entire document is on one topic.
787
+ * Default is 2000 characters.
788
+ */
789
+ maxChunkSize?: number;
790
+ /**
791
+ * Number of surrounding sentences to include when computing similarity
792
+ * for a sentence. A larger window reduces noise from single odd sentences.
793
+ * Default is 1.
794
+ */
795
+ bufferSize?: number;
796
+ /**
797
+ * Number of sentences to process in a single batch when calling the embedding function.
798
+ * Higher values are faster but may trigger API rate limits.
799
+ * Default is 50.
800
+ */
801
+ embeddingBatchSize?: number;
802
+ }
803
+ /**
804
+ * Discriminated union of all chunking strategy configurations.
805
+ */
806
+ export type ChunkingConfig = FixedSizeChunkingConfig | DocumentStructureChunkingConfig | SemanticChunkingConfig;
807
+ /**
808
+ * Represents a single document chunk ready for a RAG (Retrieval-Augmented Generation) pipeline.
809
+ *
810
+ * Chunks are the result of splitting a document into smaller, semantically coherent
811
+ * pieces that fit within the context window of an LLM. Each chunk includes the
812
+ * extracted text and rich AST-derived metadata for citations and filtered retrieval.
813
+ */
814
+ export interface OfficeChunk {
815
+ /** The text content of this chunk. This is what gets embedded. */
816
+ text: string;
817
+ /**
818
+ * Rich contextual metadata extracted from the AST.
819
+ * Use this to populate vector DB metadata fields for filtered retrieval
820
+ * and for LLM citations.
821
+ */
822
+ metadata: {
823
+ /** The source file format (e.g., 'docx', 'pptx', 'pdf'). */
824
+ sourceType: SupportedFileType;
825
+ /** Page number (1-based), if available (PDF). */
826
+ pageNumber?: number;
827
+ /** Slide number (1-based), if available (PPTX/ODP). */
828
+ slideNumber?: number;
829
+ /** Sheet name, if available (XLSX/ODS). */
830
+ sheetName?: string;
831
+ /** The text of the nearest heading above this chunk in the document. */
832
+ closestHeading?: string;
833
+ /** True if this chunk is part of a table split. */
834
+ isTableChunk?: boolean;
835
+ /** Extensible for user-defined metadata. */
836
+ [key: string]: any;
837
+ };
838
+ /** The start character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
839
+ startIndex?: number;
840
+ /** The end character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
841
+ endIndex?: number;
126
842
  }
127
843
  /**
128
844
  * Supported file types for parsing.
129
845
  */
130
- export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf';
846
+ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf' | 'md' | 'html' | 'csv';
131
847
  /**
132
848
  * Types of content nodes in the AST.
133
849
  */
134
- export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break';
850
+ export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment';
135
851
  /**
136
852
  * Supported MIME types for attachments.
137
853
  */
138
- export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation';
854
+ export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet' | 'application/vnd.openxmlformats-officedocument.presentationml.presentation' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation' | 'application/rtf' | 'text/csv' | 'text/markdown' | 'text/html';
139
855
  /**
140
856
  * Text formatting options available for text content.
141
857
  * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
@@ -224,6 +940,8 @@ export interface SlideMetadata {
224
940
  noteId?: string;
225
941
  /** The style of the slide. */
226
942
  style?: string;
943
+ /** Unique anchor IDs for internal linking. */
944
+ anchorIds?: string[];
227
945
  }
228
946
  /**
229
947
  * Metadata for a sheet in Excel.
@@ -233,6 +951,8 @@ export interface SheetMetadata {
233
951
  sheetName: string;
234
952
  /** The style of the sheet. */
235
953
  style?: string;
954
+ /** Unique anchor IDs for internal linking. */
955
+ anchorIds?: string[];
236
956
  }
237
957
  /**
238
958
  * Detailed indentation information for paragraphs and headings.
@@ -260,6 +980,8 @@ export interface HeadingMetadata {
260
980
  style?: string;
261
981
  /** Detailed indentation information. */
262
982
  paragraphIndentation?: IndentationMetadata;
983
+ /** Unique anchor IDs for internal linking. */
984
+ anchorIds?: string[];
263
985
  }
264
986
  /**
265
987
  * Metadata for a paragraph.
@@ -271,6 +993,8 @@ export interface ParagraphMetadata {
271
993
  style?: string;
272
994
  /** Detailed indentation information. */
273
995
  paragraphIndentation?: IndentationMetadata;
996
+ /** Unique anchor IDs for internal linking. */
997
+ anchorIds?: string[];
274
998
  }
275
999
  /**
276
1000
  * Metadata for a list item.
@@ -310,6 +1034,8 @@ export interface ListMetadata {
310
1034
  * @example "ListParagraph"
311
1035
  */
312
1036
  style?: string;
1037
+ /** Unique anchor IDs for internal linking. */
1038
+ anchorIds?: string[];
313
1039
  }
314
1040
  /**
315
1041
  * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
@@ -338,6 +1064,8 @@ export interface CellMetadata {
338
1064
  colSpan?: number;
339
1065
  /** The style of the cell. */
340
1066
  style?: string;
1067
+ /** Unique anchor IDs for internal linking. */
1068
+ anchorIds?: string[];
341
1069
  }
342
1070
  /**
343
1071
  * Metadata for a chart node in the document.
@@ -350,6 +1078,8 @@ export interface ChartMetadata {
350
1078
  * @example "chart1.xml"
351
1079
  */
352
1080
  attachmentName: string;
1081
+ /** Unique anchor IDs for internal linking. */
1082
+ anchorIds?: string[];
353
1083
  }
354
1084
  /**
355
1085
  * Metadata for an image node in the document.
@@ -368,6 +1098,14 @@ export interface ImageMetadata {
368
1098
  * @example "Company logo"
369
1099
  */
370
1100
  altText?: string;
1101
+ /**
1102
+ * URL of the image if it is an external link.
1103
+ * Typical for HTML or Markdown images that point to remote servers.
1104
+ * @example "https://example.com/image.png"
1105
+ */
1106
+ url?: string;
1107
+ /** Unique anchor IDs for internal linking. */
1108
+ anchorIds?: string[];
371
1109
  }
372
1110
  /**
373
1111
  * Metadata for PDF page nodes.
@@ -413,6 +1151,8 @@ export interface NoteMetadata {
413
1151
  * @example "1", "2"
414
1152
  */
415
1153
  noteId?: string;
1154
+ /** Unique anchor IDs for internal linking. */
1155
+ anchorIds?: string[];
416
1156
  }
417
1157
  /**
418
1158
  * Metadata for break nodes.
@@ -439,10 +1179,19 @@ export interface BreakMetadata {
439
1179
  */
440
1180
  clear?: 'all' | 'left' | 'none' | 'right';
441
1181
  }
1182
+ /**
1183
+ * Metadata for a code block.
1184
+ */
1185
+ export interface CodeMetadata {
1186
+ /** The programming language of the code block (e.g., 'typescript', 'python') */
1187
+ language?: string;
1188
+ /** Unique anchor IDs for internal linking. */
1189
+ anchorIds?: string[];
1190
+ }
442
1191
  /**
443
1192
  * Union type for content metadata.
444
1193
  */
445
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
1194
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
446
1195
  /**
447
1196
  * Represents a node in the document content tree.
448
1197
  * This is the core building block of the parsed document structure.
@@ -680,9 +1429,19 @@ export interface OfficeMetadata {
680
1429
  * console.log(ast.metadata.author); // 'John Doe'
681
1430
  * console.log(ast.content.length); // Number of top-level content nodes
682
1431
  * console.log(ast.toText()); // Plain text representation
1432
+ * console.log((await ast.to('md')).value); // Markdown representation
1433
+ * console.log((await ast.to('html')).value); // HTML representation
1434
+ * console.log((await ast.to('rtf')).value); // RTF representation
1435
+ * console.log((await ast.to('csv')).value); // CSV representation
1436
+ * console.log((await ast.to('chunks')).value); // Chunks representation
683
1437
  * ```
684
1438
  */
685
1439
  export interface OfficeParserAST {
1440
+ /**
1441
+ * The original configuration used to parse this document.
1442
+ * This includes options like OCR settings, delimiter choices, and filtering flags.
1443
+ */
1444
+ config: OfficeParserConfig;
686
1445
  /**
687
1446
  * The type of the parsed file.
688
1447
  * Indicates which parser was used and what format the input was in.
@@ -717,7 +1476,12 @@ export interface OfficeParserAST {
717
1476
  * @example [{ type: 'image', mimeType: 'image/png', data: 'base64...', name: 'image1.png' }]
718
1477
  */
719
1478
  attachments: OfficeAttachment[];
1479
+ /** Any warnings or non-fatal issues encountered during parsing. */
1480
+ warnings: OfficeIssue[];
720
1481
  /**
1482
+ * @deprecated Use `.to('text')` instead.
1483
+ * Note: This method is synchronous, while the new `.to()` method is asynchronous.
1484
+ *
721
1485
  * Converts the entire AST to plain text.
722
1486
  * This method flattens the document structure and returns just the text content,
723
1487
  * stripping out all formatting, metadata, and structure.
@@ -732,4 +1496,18 @@ export interface OfficeParserAST {
732
1496
  * ```
733
1497
  */
734
1498
  toText(): string;
1499
+ /**
1500
+ * Converts this AST to the specified destination format.
1501
+ * This is the recommended way to convert the AST to different formats.
1502
+ *
1503
+ * @param destination The target format (e.g., 'text', 'md', 'html', 'pdf').
1504
+ * @param config Optional configuration for the generator.
1505
+ * @returns A promise resolving to the generated content (string or Buffer).
1506
+ * @example
1507
+ * ```typescript
1508
+ * const html = await ast.to('html', { includeFormatting: false });
1509
+ * const md = await ast.to('md');
1510
+ * ```
1511
+ */
1512
+ to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
735
1513
  }