officeparser 6.1.1 → 7.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. package/README.md +301 -26
  2. package/dist/OfficeConverter.d.ts +46 -0
  3. package/dist/OfficeConverter.js +72 -0
  4. package/dist/OfficeGenerator.d.ts +19 -0
  5. package/dist/OfficeGenerator.js +48 -0
  6. package/dist/OfficeParser.d.ts +6 -0
  7. package/dist/OfficeParser.js +74 -31
  8. package/dist/cli.d.ts +3 -1
  9. package/dist/cli.js +106 -22
  10. package/dist/defaults.d.ts +41 -0
  11. package/dist/defaults.js +172 -0
  12. package/dist/generators/BaseGenerator.d.ts +58 -0
  13. package/dist/generators/BaseGenerator.js +107 -0
  14. package/dist/generators/ChunkingGenerator.d.ts +81 -0
  15. package/dist/generators/ChunkingGenerator.js +683 -0
  16. package/dist/generators/CsvGenerator.d.ts +30 -0
  17. package/dist/generators/CsvGenerator.js +233 -0
  18. package/dist/generators/HtmlGenerator.d.ts +37 -0
  19. package/dist/generators/HtmlGenerator.js +1013 -0
  20. package/dist/generators/MarkdownGenerator.d.ts +59 -0
  21. package/dist/generators/MarkdownGenerator.js +481 -0
  22. package/dist/generators/PdfGenerator.d.ts +22 -0
  23. package/dist/generators/PdfGenerator.js +118 -0
  24. package/dist/generators/RtfGenerator.d.ts +15 -0
  25. package/dist/generators/RtfGenerator.js +208 -0
  26. package/dist/generators/TextGenerator.d.ts +13 -0
  27. package/dist/generators/TextGenerator.js +108 -0
  28. package/dist/index.d.ts +11 -3
  29. package/dist/index.js +17 -2
  30. package/dist/index.mjs +2 -2
  31. package/dist/officeparser.browser.d.ts +828 -5
  32. package/dist/officeparser.browser.iife.js +703 -52
  33. package/dist/officeparser.browser.mjs +703 -52
  34. package/dist/parsers/CsvParser.d.ts +9 -0
  35. package/dist/parsers/CsvParser.js +110 -0
  36. package/dist/parsers/ExcelParser.d.ts +2 -2
  37. package/dist/parsers/ExcelParser.js +145 -114
  38. package/dist/parsers/HtmlParser.d.ts +2 -0
  39. package/dist/parsers/HtmlParser.js +539 -0
  40. package/dist/parsers/MarkdownParser.d.ts +2 -0
  41. package/dist/parsers/MarkdownParser.js +360 -0
  42. package/dist/parsers/OpenOfficeParser.d.ts +2 -2
  43. package/dist/parsers/OpenOfficeParser.js +140 -79
  44. package/dist/parsers/PdfParser.d.ts +2 -2
  45. package/dist/parsers/PdfParser.js +52 -49
  46. package/dist/parsers/PowerPointParser.d.ts +2 -2
  47. package/dist/parsers/PowerPointParser.js +20 -23
  48. package/dist/parsers/RtfParser.d.ts +2 -2
  49. package/dist/parsers/RtfParser.js +1291 -1240
  50. package/dist/parsers/WordParser.d.ts +2 -2
  51. package/dist/parsers/WordParser.js +232 -97
  52. package/dist/sbom.cdx.json +99 -99
  53. package/dist/types.d.ts +783 -5
  54. package/dist/types.js +73 -0
  55. package/dist/utils/astUtils.d.ts +16 -0
  56. package/dist/utils/astUtils.js +32 -0
  57. package/dist/utils/configUtils.d.ts +26 -0
  58. package/dist/utils/configUtils.js +140 -0
  59. package/dist/utils/envUtils.d.ts +8 -3
  60. package/dist/utils/envUtils.js +117 -34
  61. package/dist/utils/errorUtils.d.ts +17 -29
  62. package/dist/utils/errorUtils.js +110 -52
  63. package/dist/utils/moduleLoader.js +19 -11
  64. package/dist/utils/ocrUtils.js +2 -1
  65. package/dist/utils/sheetUtils.d.ts +7 -0
  66. package/dist/utils/sheetUtils.js +35 -0
  67. package/dist/utils/styleMapper.d.ts +36 -0
  68. package/dist/utils/styleMapper.js +224 -0
  69. package/dist/utils/xmlUtils.d.ts +0 -8
  70. package/dist/utils/xmlUtils.js +2 -1
  71. package/package.json +26 -7
@@ -1,5 +1,75 @@
1
1
  // Generated by dts-bundle-generator v9.5.1
2
2
 
3
+ /**
4
+ * Standard error types for OfficeParser.
5
+ * Use these to identify the kind of error being reported.
6
+ */
7
+ export declare enum OfficeErrorType {
8
+ /** Unsupported file extension */
9
+ EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
10
+ /** File appears to be corrupted or malformed */
11
+ FILE_CORRUPTED = "FILE_CORRUPTED",
12
+ /** File could not be found at the specified path */
13
+ FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST",
14
+ /** Specified location/directory is not reachable or is a directory */
15
+ LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND",
16
+ /** Arguments passed to the function are missing or invalid */
17
+ IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
18
+ /** Error occurred while reading or processing file buffers */
19
+ IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
20
+ /** Input type is not a supported type (string, Buffer, ArrayBuffer) */
21
+ INVALID_INPUT = "INVALID_INPUT",
22
+ /** PDF worker source is missing (required in browser) */
23
+ PDF_WORKER_MISSING = "PDF_WORKER_MISSING",
24
+ /** Attempted to use Node.js-only features in a browser environment */
25
+ FEATURE_NOT_SUPPORTED_IN_BROWSER = "FEATURE_NOT_SUPPORTED_IN_BROWSER",
26
+ /** Style mapping string is malformed */
27
+ INVALID_STYLE_MAPPING = "INVALID_STYLE_MAPPING",
28
+ /** Selector in style mapping is invalid */
29
+ INVALID_SELECTOR = "INVALID_SELECTOR",
30
+ /** Output mapping in style mapping is invalid */
31
+ INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
32
+ /** Semantic chunking strategy is selected but no embedding function is provided */
33
+ MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
34
+ }
35
+ /**
36
+ * Standard warning types for OfficeParser.
37
+ * Use these for reporting non-fatal issues or performance tips.
38
+ */
39
+ export declare enum OfficeWarningType {
40
+ /** Performance advice (e.g., Rosetta translation on Mac) */
41
+ PERFORMANCE_TIP = "PERFORMANCE_TIP",
42
+ /** OCR processing failed for an attachment */
43
+ OCR_FAILED = "OCR_FAILED",
44
+ /** Extraction of structured chart data failed */
45
+ CHART_DATA_EXTRACTION_FAILED = "CHART_DATA_EXTRACTION_FAILED",
46
+ /** Automatic worker path failed, falling back to CDN */
47
+ PDF_WORKER_FALLBACK = "PDF_WORKER_FALLBACK",
48
+ /** General attachment extraction failure */
49
+ ATTACHMENT_EXTRACTION_FAILED = "ATTACHMENT_EXTRACTION_FAILED",
50
+ /** Failed to load a specific page in a multi-page document */
51
+ PAGE_LOAD_FAILED = "PAGE_LOAD_FAILED",
52
+ /** Failed to load a required dynamic dependency */
53
+ DEPENDENCY_LOAD_FAILED = "DEPENDENCY_LOAD_FAILED",
54
+ /** Failed to extract images from a source */
55
+ IMAGE_EXTRACTION_FAILED = "IMAGE_EXTRACTION_FAILED",
56
+ /** Failed to extract annotations from a document */
57
+ ANNOTATION_EXTRACTION_FAILED = "ANNOTATION_EXTRACTION_FAILED",
58
+ /** Failed to process an extracted image bitmap */
59
+ IMAGE_PROCESSING_FAILED = "IMAGE_PROCESSING_FAILED",
60
+ /** Warning about limitations of browser-based generation */
61
+ BROWSER_GENERATION_LIMITATION = "BROWSER_GENERATION_LIMITATION",
62
+ /** Specified sheet range in Excel/ODS export was not found */
63
+ SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND",
64
+ /** Buffer content type does not match the provided or expected file extension */
65
+ BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH",
66
+ /** Failed to detect file type from buffer due to library error or incompatibility */
67
+ FILE_TYPE_DETECTION_FAILED = "FILE_TYPE_DETECTION_FAILED",
68
+ /** No chunks were generated for the document given the current strategy */
69
+ EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
70
+ /** A node was skipped because it only contained whitespace */
71
+ WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
72
+ }
3
73
  /**
4
74
  * Configuration options for OCR.
5
75
  */
@@ -18,16 +88,19 @@ export interface OcrConfig {
18
88
  /**
19
89
  * Path to the Tesseract worker script.
20
90
  * Primarily used for offline/air-gapped environments.
91
+ * Default is ''.
21
92
  */
22
93
  workerPath?: string;
23
94
  /**
24
95
  * Path to the Tesseract core script.
25
96
  * Primarily used for offline/air-gapped environments.
97
+ * Default is ''.
26
98
  */
27
99
  corePath?: string;
28
100
  /**
29
101
  * Path for Tesseract language files (traineddata).
30
102
  * Primarily used for offline/air-gapped environments.
103
+ * Default is ''.
31
104
  */
32
105
  langPath?: string;
33
106
  /**
@@ -42,10 +115,17 @@ export interface OcrConfig {
42
115
  */
43
116
  export interface OfficeParserConfig {
44
117
  /**
118
+ * @deprecated Use `onWarning` instead.
45
119
  * Flag to show all the logs to console in case of an error irrespective of your own handling.
46
120
  * Default is false.
47
121
  */
48
122
  outputErrorToConsole?: boolean;
123
+ /**
124
+ * Callback for warnings or non-fatal errors encountered during parsing.
125
+ * Allows you to capture issues like OCR failures or attachment extraction errors
126
+ * without stopping the parsing process.
127
+ */
128
+ onWarning?: (issue: OfficeIssue) => void;
49
129
  /**
50
130
  * The delimiter used for every new line in places that allow multiline text like word.
51
131
  * Default is \n.
@@ -114,7 +194,7 @@ export interface OfficeParserConfig {
114
194
  * The URL/path to the PDF.js worker script.
115
195
  *
116
196
  * **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
117
- * If not provided, it defaults to `https://unpkg.com/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
197
+ * If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
118
198
  * You can override this with your own local path or a different CDN link.
119
199
  */
120
200
  pdfWorkerSrc?: string;
@@ -125,19 +205,633 @@ export interface OfficeParserConfig {
125
205
  * Default is false
126
206
  */
127
207
  includeBreakNodes?: boolean;
208
+ /**
209
+ * Flag to ignore all internal (anchor) links during parsing.
210
+ * When true, all bookmarks, cross-references, and internal document jumps are stripped
211
+ * from the AST. Only external URLs will be preserved.
212
+ *
213
+ * Use this if you want a "flat" document without any internal interactivity.
214
+ *
215
+ * Default is false.
216
+ */
217
+ ignoreInternalLinks?: boolean;
218
+ /**
219
+ * Optional hint for the file format.
220
+ * When a Buffer or ArrayBuffer is passed, the parser relies on magic bytes to detect the file type.
221
+ * Text-based formats like 'md', 'html', and 'csv' lack reliable magic bytes.
222
+ * If you are parsing these formats from a Buffer, you must provide this fileType hint.
223
+ *
224
+ * This is authoritative and is used to determine the file type, so it should be accurate.
225
+ * If provided, this bypasses the magic bytes detection and the file extension-based detection either way.
226
+ *
227
+ * Default is null.
228
+ */
229
+ fileType?: SupportedFileType | null;
230
+ /**
231
+ * Custom delimiter for CSV files.
232
+ * Defaults to ',' but can be overridden (e.g., ';', '\t').
233
+ */
234
+ csvDelimiter?: string;
235
+ }
236
+ /**
237
+ * Represents a single issue (warning, error, or info) generated during document processing.
238
+ */
239
+ export interface OfficeIssue {
240
+ /** The severity of the issue. */
241
+ type: "warning" | "info" | "error";
242
+ /** Human-readable message text. */
243
+ message: string;
244
+ /** The specific AST node that triggered this issue, if applicable. */
245
+ node?: OfficeContentNode;
246
+ /** A unique error code for programmatic handling. */
247
+ code: OfficeWarningType | OfficeErrorType;
248
+ /** Optional additional context or original error object. */
249
+ details?: any;
250
+ }
251
+ /**
252
+ * The result of a document conversion operation.
253
+ */
254
+ export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
255
+ /** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
256
+ value: D extends "pdf" ? Uint8Array : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
257
+ /** A collection of issues (warnings/infos) generated during the process. */
258
+ messages: OfficeIssue[];
259
+ }
260
+ /**
261
+ * Universal formats supported by all source types for generation.
262
+ */
263
+ export type UniversalGeneratorFormat = "text" | "md" | "html" | "pdf" | "csv" | "rtf" | "chunks";
264
+ /**
265
+ * Allowed destination formats for a given source type.
266
+ * Currently, all generators are universal across all source formats.
267
+ */
268
+ export type SupportedDestination<_T extends SupportedFileType = SupportedFileType> = UniversalGeneratorFormat;
269
+ /**
270
+ * Configuration options for the OfficeGenerator.
271
+ */
272
+ /**
273
+ * Common configuration options for all generators.
274
+ */
275
+ export interface CommonGeneratorConfig {
276
+ /**
277
+ * Callback called for every node during generation.
278
+ * Allows users to modify nodes before processing, completely override rendering, or filter them out.
279
+ *
280
+ * #### Callback Capabilities:
281
+ * 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
282
+ * 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
283
+ * 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
284
+ * 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
285
+ */
286
+ onNode?: (node: OfficeContentNode) => string | false | Promise<string | false | void> | void;
287
+ /**
288
+ * Callback for warnings, non-fatal errors, or issues encountered during generation.
289
+ * Allows the process to continue while reporting skipping or approximation of content.
290
+ */
291
+ onWarning?: (issue: OfficeIssue) => void;
292
+ /**
293
+ * Map document styles (e.g., 'Heading 1', 'Intense Quote') to specific semantic elements.
294
+ *
295
+ * DESIGN PHILOSOPHY:
296
+ * This is the primary way to customize how the library interprets the visual
297
+ * structure of your source documents.
298
+ *
299
+ * To disable all semantic translation and use raw AST types only,
300
+ * set `ignoreDefaultStyleMap: true` and leave `styleMap` empty.
301
+ *
302
+ * It supports two formats:
303
+ *
304
+ * 1. LEGACY STRING DSL:
305
+ * Simple "selector => output" syntax. Highly compatible with mammoth.js style maps.
306
+ * @example ["p[style-name='Heading 1'] => h1"]
307
+ * @example ["p[style='Quote'] => blockquote"]
308
+ *
309
+ * 2. STRUCTURED OBJECTS (Recommended):
310
+ * More powerful and strictly typed. Ideal for complex logic or when you
311
+ * need to apply specific classes/attributes for the HTML generator.
312
+ * @example
313
+ * [
314
+ * {
315
+ * selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
316
+ * output: { tag: 'h1', classes: ['main-title'], attributes: { id: 'top' } }
317
+ * }
318
+ * ]
319
+ *
320
+ * Note: This property works in conjunction with `ignoreDefaultStyleMap`.
321
+ * Defaults to a robust built-in map that covers common standard Office styles.
322
+ */
323
+ styleMap?: string[] | StructuredStyleMapping[];
324
+ /**
325
+ * Whether to include visual formatting like font size, font family, and colors in the output.
326
+ * Set to false for clean, semantic output.
327
+ * Defaults to true.
328
+ */
329
+ includeFormatting?: boolean;
330
+ /**
331
+ * Whether to automatically generate unique slug-based IDs for headings.
332
+ * Useful for table-of-contents and anchor links.
333
+ * Defaults to true.
334
+ */
335
+ generateIds?: boolean;
336
+ /**
337
+ * Whether to render document metadata (title, author, etc.) as visible content
338
+ * in the generated output (e.g., a header block in HTML or plain text).
339
+ * Structural metadata (HTML <meta> tags, Markdown YAML frontmatter) is always included.
340
+ * Defaults to false.
341
+ */
342
+ renderMetadata?: boolean;
343
+ /**
344
+ * Whether to ignore the built-in default style mappings (e.g. "Heading 1" -> h1).
345
+ * Set to true if you want full control over style mapping.
346
+ * Defaults to false.
347
+ */
348
+ ignoreDefaultStyleMap?: boolean;
349
+ /**
350
+ * Whether to include images in the generated output.
351
+ * Defaults to true.
352
+ */
353
+ includeImages?: boolean;
354
+ /**
355
+ * Whether to include interactive charts in the generated output (HTML only).
356
+ * Defaults to true.
357
+ */
358
+ includeCharts?: boolean;
359
+ /**
360
+ * Whether to ignore all internal (anchor) links and anchor IDs during generation.
361
+ * When true, all bookmarks, cross-references, and internal document jumps are stripped.
362
+ * Specifically for Markdown, this removes the {#id} block from headings.
363
+ * Defaults to false.
364
+ */
365
+ ignoreInternalLinks?: boolean;
366
+ }
367
+ /**
368
+ * Destination-aware generator configuration.
369
+ * Restricts format-specific configurations to their respective destinations.
370
+ */
371
+ /**
372
+ * Mapping of destination formats to their specific configuration interfaces.
373
+ */
374
+ export interface GeneratorSubConfigMap {
375
+ html: HtmlGeneratorConfig;
376
+ md: MdGeneratorConfig;
377
+ pdf: PdfGeneratorConfig;
378
+ csv: CsvGeneratorConfig;
379
+ text: TextGeneratorConfig;
380
+ rtf: RtfGeneratorConfig;
381
+ chunks: ChunkingConfig;
382
+ }
383
+ /**
384
+ * Configuration options for document generators.
385
+ *
386
+ * This interface is designed to be format-aware. When you specify a destination format
387
+ * (e.g., `OfficeGenerator.generate(ast, 'html', config)`), the generic parameter `D`
388
+ * ensures that only the relevant sub-configuration (e.g., `htmlConfig`) is available
389
+ * for type checking.
390
+ *
391
+ * @template D The destination format string. Defaults to `string` for a general configuration.
392
+ */
393
+ export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
394
+ [K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
395
+ };
396
+ /**
397
+ * Configuration options for the OfficeConverter.
398
+ * Combines relevant parser and generator settings for a seamless one-step conversion.
399
+ *
400
+ * @template D The destination format string.
401
+ */
402
+ /**
403
+ * Configuration options for the OfficeConverter.
404
+ * Combines general generator settings with a specific subset of parser settings.
405
+ *
406
+ * @template D The destination format string.
407
+ * @template T The source file type.
408
+ */
409
+ export type OfficeConverterConfig<D extends string = string, T extends SupportedFileType = SupportedFileType> = {
410
+ /**
411
+ * Specific configuration for the source parsing phase.
412
+ */
413
+ parseConfig?: OfficeParserConfig & {
414
+ fileType?: T;
415
+ };
416
+ /**
417
+ * Specific configuration for the destination generation phase.
418
+ */
419
+ generatorConfig?: GeneratorConfig<D>;
420
+ /**
421
+ * Callback for warnings or non-fatal errors encountered during the entire conversion process.
422
+ * This is passed to both the parser and the generator.
423
+ * If provided, this takes precedence over callbacks inside parseConfig or generatorConfig.
424
+ */
425
+ onWarning?: (issue: OfficeIssue) => void;
426
+ };
427
+ /**
428
+ * Configuration options for HTML generation.
429
+ */
430
+ export interface HtmlGeneratorConfig {
431
+ /**
432
+ * Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
433
+ * Defaults to true.
434
+ */
435
+ standalone?: boolean;
436
+ /**
437
+ * URL for the Chart.js library to use when 'includeCharts' is true.
438
+ * Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
439
+ */
440
+ chartJsSrc?: string;
441
+ }
442
+ /**
443
+ * Configuration options for PDF generation.
444
+ * Maps closely to Puppeteer's PDF options.
445
+ */
446
+ export interface PdfGeneratorConfig {
447
+ /** Paper format. Defaults to 'A4'. */
448
+ format?: "letter" | "legal" | "tabloid" | "ledger" | "a0" | "a1" | "a2" | "a3" | "a4" | "a5" | "a6" | "Letter" | "Legal" | "Tabloid" | "Ledger" | "A0" | "A1" | "A2" | "A3" | "A4" | "A5" | "A6";
449
+ /** Paper width, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
450
+ width?: string | number;
451
+ /** Paper height, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
452
+ height?: string | number;
453
+ /** Whether to print in landscape orientation. Defaults to false. */
454
+ landscape?: boolean;
455
+ /** Whether to print background graphics. Defaults to true. */
456
+ printBackground?: boolean;
457
+ /** Scale of the webpage rendering. Defaults to 1. */
458
+ scale?: number;
459
+ /** Paper margins. */
460
+ margin?: {
461
+ top?: string | number;
462
+ right?: string | number;
463
+ bottom?: string | number;
464
+ left?: string | number;
465
+ };
466
+ /** Whether to display header and footer. Defaults to false. */
467
+ displayHeaderFooter?: boolean;
468
+ /** HTML template for the print header. */
469
+ headerTemplate?: string;
470
+ /** HTML template for the print footer. */
471
+ footerTemplate?: string;
472
+ /**
473
+ * Optional Puppeteer launch options for Node.js environment.
474
+ * Useful for setting custom executable paths or args in CI/CD.
475
+ */
476
+ launchOptions?: any;
477
+ }
478
+ /**
479
+ * Structured style mapping definition for the StyleMapper.
480
+ *
481
+ * DESIGN PHILOSOPHY: "Semantic Translation"
482
+ * -----------------------------------------
483
+ * Office documents (Word, RTF, PPTX) often use custom or localized style names
484
+ * (e.g., "Heading 1" in English vs "Titre 1" in French, or "MyCompany-Quote").
485
+ *
486
+ * This interface allows you to create a "semantic bridge" between these arbitrary
487
+ * source styles and a universal vocabulary of document elements.
488
+ *
489
+ * WHY USE HTML TAGS FOR NON-HTML OUTPUT?
490
+ * --------------------------------------
491
+ * We use HTML tags (`h1`, `blockquote`, `code`, `pre`) as a "Universal Intermediate
492
+ * Language". By mapping a custom Word style to `blockquote`, you are defining its
493
+ * SEMANTIC MEANING rather than its physical appearance.
494
+ *
495
+ * Each generator then interprets this meaning natively:
496
+ * - HTML Generator: Directly renders the `<blockquote>` tag with your classes.
497
+ * - Markdown Generator: Sees 'blockquote' and renders the standard `> ` prefix.
498
+ * - Text Generator: Sees 'blockquote' and applies appropriate structural indentation.
499
+ */
500
+ export interface StructuredStyleMapping {
501
+ /**
502
+ * The criteria used to identify which AST nodes should be transformed.
503
+ * Think of this as the "Source Filter".
504
+ */
505
+ selector: {
506
+ /**
507
+ * The structural type of the node (e.g., 'paragraph', 'heading', 'text').
508
+ * Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
509
+ */
510
+ nodeType?: string;
511
+ /**
512
+ * A dictionary of attributes to match on the node.
513
+ *
514
+ * The most common use case is matching the 'style' attribute from
515
+ * Word documents (e.g., { style: 'Intense Quote' }).
516
+ *
517
+ * Matchers:
518
+ * - Literal: `style: 'Heading 1'` matches exactly.
519
+ * - Operator: `{ value: 'Title', operator: '~=' }` matches if the word 'Title'
520
+ * is found within the style name.
521
+ */
522
+ attributes?: Record<string, string | number | boolean | {
523
+ value: string | number | boolean;
524
+ operator: "=" | "~=";
525
+ }>;
526
+ };
527
+ /**
528
+ * The target representation for the matched node.
529
+ * Think of this as the "Semantic Meaning" you want to assign to the match.
530
+ */
531
+ output: {
532
+ /**
533
+ * The universal semantic tag (e.g., 'h1', 'h2', 'blockquote', 'code', 'pre', 'u').
534
+ * All generators use this tag to decide their native output syntax.
535
+ */
536
+ tag: string;
537
+ /**
538
+ * CSS classes to apply to the output.
539
+ * This is utilized by the HTML generator to allow for downstream CSS styling.
540
+ */
541
+ classes?: string[];
542
+ /**
543
+ * Key-value pair of HTML attributes (like 'id', 'data-*', or 'style') to apply.
544
+ * Primarily used by the HTML generator for high-fidelity conversion.
545
+ */
546
+ attributes?: Record<string, string>;
547
+ /**
548
+ * If true, prevents the generator from collapsing this element into
549
+ * adjacent elements of the same type.
550
+ *
551
+ * For example, multiple paragraphs mapped to 'blockquote' normally merge into
552
+ * one big blockquote. Setting `fresh: true` forces them to be separate blocks.
553
+ */
554
+ fresh?: boolean;
555
+ };
556
+ }
557
+ /**
558
+ * Configuration options for RTF generation.
559
+ */
560
+ export interface RtfGeneratorConfig {
561
+ }
562
+ /**
563
+ * Configuration options for CSV generation.
564
+ */
565
+ export interface CsvGeneratorConfig {
566
+ /**
567
+ * Range of sheets to export.
568
+ * Supports formats like "1", "1-3", "1,2", "1,3-5,7".
569
+ * 1-based indexing.
570
+ * Default is '' (all sheets).
571
+ */
572
+ sheets?: string;
573
+ /**
574
+ * Whether to merge all selected sheets into a single CSV.
575
+ * If false, returns a ZIP archive containing individual CSV files.
576
+ * Defaults to false.
577
+ */
578
+ mergeSheets?: boolean;
579
+ /**
580
+ * Custom delimiter for CSV files.
581
+ * Defaults to ','.
582
+ */
583
+ columnDelimiter?: string;
584
+ }
585
+ /**
586
+ * Configuration options for Markdown generation.
587
+ */
588
+ export interface MdGeneratorConfig {
589
+ /**
590
+ * Whether to fallback to HTML tags for features not supported by standard Markdown.
591
+ *
592
+ * Markdown has limited support for complex document structures. This flag controls how
593
+ * the generator handles features that cannot be represented in pure Markdown:
594
+ *
595
+ * 1. If a feature is NOT supported natively by Markdown (e.g., nested tables, text alignment,
596
+ * underline, subscript/superscript):
597
+ * - If true: The generator will use HTML tags (<u>, <sub>, <div>, <table>, etc.) to
598
+ * maintain high fidelity.
599
+ * - If false: The generator will skip or simplify the feature (e.g., ignoring alignment,
600
+ * skipping underline, or hoisting nested tables out of their cells).
601
+ *
602
+ * 2. If a feature IS supported by Markdown but a higher quality version is possible
603
+ * via HTML (e.g., tables with merged cells):
604
+ * - If true: Use HTML for better fidelity.
605
+ * - If false: Use native Markdown syntax (e.g., a standard GFM table grid).
606
+ *
607
+ * Defaults to true.
608
+ */
609
+ fallbackToHtml?: boolean;
610
+ }
611
+ /**
612
+ * Configuration options for plain text generation.
613
+ */
614
+ export interface TextGeneratorConfig {
615
+ /**
616
+ * The delimiter used for every new line.
617
+ * Defaults to '\n'.
618
+ */
619
+ newlineDelimiter?: string;
620
+ /**
621
+ * Whether to attempt to preserve the original document layout.
622
+ * If true, tables will be rendered with separators and aligned columns.
623
+ * If false, output will be a flat stream of text nodes.
624
+ * Defaults to false.
625
+ */
626
+ preserveLayout?: boolean;
627
+ }
628
+ /**
629
+ * The strategy used for chunking a document for RAG pipelines.
630
+ * - 'fixed-size': Traditional character/token count based splitting.
631
+ * - 'document-structure': Leverages the AST to split at natural document boundaries.
632
+ * - 'semantic': Uses embedding similarity to find natural topic breakpoints.
633
+ */
634
+ export type ChunkingStrategy = "fixed-size" | "document-structure" | "semantic";
635
+ /**
636
+ * Base configuration applicable to all chunking strategies.
637
+ */
638
+ export interface BaseChunkingConfig {
639
+ /**
640
+ * The strategy used for chunking.
641
+ * Default is 'document-structure'.
642
+ */
643
+ strategy?: ChunkingStrategy;
644
+ /**
645
+ * A function that measures the size of a text string.
646
+ * Defaults to character count: `(text) => text.length`.
647
+ * Override with a token counter (e.g., `tiktoken`) for strict LLM context window adherence.
648
+ */
649
+ lengthFunction?: (text: string) => number;
650
+ /**
651
+ * Whether to strip leading/trailing whitespace from each chunk.
652
+ * Default is true.
653
+ */
654
+ stripWhitespace?: boolean;
655
+ /**
656
+ * Whether to include rich AST metadata (page number, slide number, heading, etc.)
657
+ * in the generated chunk objects.
658
+ * Default is true.
659
+ */
660
+ includeMetadata?: boolean;
661
+ /**
662
+ * Whether to include the starting character index of each chunk
663
+ * relative to the whole document. Useful for UI text highlighting.
664
+ * Default is false.
665
+ */
666
+ addStartIndex?: boolean;
667
+ /**
668
+ * Optional custom regex (as string or RegExp object) to identify sentence boundaries.
669
+ * Use this for languages or specific document types that require custom splitting logic.
670
+ * If provided, it overrides or augments the default segmenter.
671
+ * @example /[。?!]/
672
+ */
673
+ sentenceBoundaryRegex?: string | RegExp;
674
+ /**
675
+ * Optional list of abbreviations to ignore when splitting text into sentences.
676
+ * These words, if followed by a period, will not be treated as sentence boundaries.
677
+ * Use this to handle language-specific or domain-specific abbreviations.
678
+ * @example ["Inc", "Ltd", "approx"]
679
+ */
680
+ abbreviations?: string[];
681
+ }
682
+ /**
683
+ * Configuration for Fixed-Size Chunking.
684
+ * Cuts text based on a maximum size limit with an optional overlap.
685
+ * This is equivalent to LangChain's `RecursiveCharacterTextSplitter`.
686
+ */
687
+ export interface FixedSizeChunkingConfig extends BaseChunkingConfig {
688
+ strategy: "fixed-size";
689
+ /**
690
+ * Maximum size of the chunk, measured by `lengthFunction`.
691
+ * Default is 1000 characters.
692
+ */
693
+ chunkSize?: number;
694
+ /**
695
+ * Number of characters/tokens to overlap between consecutive chunks
696
+ * to avoid losing context at boundaries.
697
+ * Rule of thumb: ~10–20% of `chunkSize`.
698
+ * Default is 200.
699
+ */
700
+ chunkOverlap?: number;
701
+ /**
702
+ * Ordered list of separators to try when splitting.
703
+ * The chunker tries each in order; if a split would exceed `chunkSize`,
704
+ * it tries the next separator.
705
+ * Default is ['\n\n', '\n', ' ', ''].
706
+ */
707
+ separators?: string[];
708
+ }
709
+ /**
710
+ * Configuration for Document-Structure Chunking.
711
+ * Uses the officeParser AST to split at natural document boundaries like
712
+ * headings, paragraphs, slides, or pages. This is the recommended strategy
713
+ * as it preserves semantic context from the document's own structure.
714
+ */
715
+ export interface DocumentStructureChunkingConfig extends BaseChunkingConfig {
716
+ strategy: "document-structure";
717
+ /**
718
+ * The primary structural element at which to force a chunk boundary.
719
+ * - 'paragraph': Never cross a paragraph boundary (finest-grained, most precise).
720
+ * - 'heading': Split at every heading change.
721
+ * - 'page': Chunks never span multiple pages (PDF only).
722
+ * - 'slide': Chunks never span multiple slides (PPTX/ODP only).
723
+ * - 'sheet': Chunks never span multiple sheets (XLSX/ODS only).
724
+ * Default is 'paragraph'.
725
+ */
726
+ splitBy?: "page" | "slide" | "sheet" | "heading" | "paragraph";
727
+ /**
728
+ * Maximum size of a chunk (measured by `lengthFunction`).
729
+ * If a single structural unit (e.g., one paragraph) exceeds this limit,
730
+ * it will be further split using a recursive character splitter.
731
+ * Default is 1000 characters.
732
+ */
733
+ maxChunkSize?: number;
734
+ /**
735
+ * How to handle table nodes when splitting.
736
+ * - 'row': Split by rows, REPEATING the header row in every chunk so the LLM
737
+ * always understands what the columns mean. (Highly recommended for RAG)
738
+ * - 'flatten': Convert the table to plain text and split like a regular block.
739
+ * Default is 'row'.
740
+ */
741
+ tableSplitStrategy?: "row" | "flatten";
742
+ }
743
+ /**
744
+ * Configuration for Semantic Chunking.
745
+ * Uses an embedding model to detect topic shifts and create boundaries
746
+ * where content meaning naturally changes. Computationally expensive but
747
+ * produces the highest quality chunks.
748
+ */
749
+ export interface SemanticChunkingConfig extends BaseChunkingConfig {
750
+ strategy: "semantic";
751
+ /**
752
+ * A user-provided async function to generate vector embeddings for a text string.
753
+ * Required. Example: a wrapper around OpenAI's `text-embedding-3-small`.
754
+ * @example async (text) => await openai.embeddings.create({ input: text, model: 'text-embedding-3-small' }).then(r => r.data[0].embedding)
755
+ */
756
+ embeddingFunction: (text: string) => Promise<number[]>;
757
+ /**
758
+ * The cosine similarity threshold below which a chunk boundary is created.
759
+ * When the similarity between two adjacent sentences drops below this value,
760
+ * a new chunk starts. Higher = more splits, smaller chunks.
761
+ * Default is 0.8.
762
+ */
763
+ similarityThreshold?: number;
764
+ /**
765
+ * Maximum size of a chunk even if semantic similarity remains high.
766
+ * Prevents runaway chunks when an entire document is on one topic.
767
+ * Default is 2000 characters.
768
+ */
769
+ maxChunkSize?: number;
770
+ /**
771
+ * Number of surrounding sentences to include when computing similarity
772
+ * for a sentence. A larger window reduces noise from single odd sentences.
773
+ * Default is 1.
774
+ */
775
+ bufferSize?: number;
776
+ /**
777
+ * Number of sentences to process in a single batch when calling the embedding function.
778
+ * Higher values are faster but may trigger API rate limits.
779
+ * Default is 50.
780
+ */
781
+ embeddingBatchSize?: number;
782
+ }
783
+ /**
784
+ * Discriminated union of all chunking strategy configurations.
785
+ */
786
+ export type ChunkingConfig = FixedSizeChunkingConfig | DocumentStructureChunkingConfig | SemanticChunkingConfig;
787
+ /**
788
+ * Represents a single document chunk ready for a RAG (Retrieval-Augmented Generation) pipeline.
789
+ *
790
+ * Chunks are the result of splitting a document into smaller, semantically coherent
791
+ * pieces that fit within the context window of an LLM. Each chunk includes the
792
+ * extracted text and rich AST-derived metadata for citations and filtered retrieval.
793
+ */
794
+ export interface OfficeChunk {
795
+ /** The text content of this chunk. This is what gets embedded. */
796
+ text: string;
797
+ /**
798
+ * Rich contextual metadata extracted from the AST.
799
+ * Use this to populate vector DB metadata fields for filtered retrieval
800
+ * and for LLM citations.
801
+ */
802
+ metadata: {
803
+ /** The source file format (e.g., 'docx', 'pptx', 'pdf'). */
804
+ sourceType: SupportedFileType;
805
+ /** Page number (1-based), if available (PDF). */
806
+ pageNumber?: number;
807
+ /** Slide number (1-based), if available (PPTX/ODP). */
808
+ slideNumber?: number;
809
+ /** Sheet name, if available (XLSX/ODS). */
810
+ sheetName?: string;
811
+ /** The text of the nearest heading above this chunk in the document. */
812
+ closestHeading?: string;
813
+ /** True if this chunk is part of a table split. */
814
+ isTableChunk?: boolean;
815
+ /** Extensible for user-defined metadata. */
816
+ [key: string]: any;
817
+ };
818
+ /** The start character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
819
+ startIndex?: number;
820
+ /** The end character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
821
+ endIndex?: number;
128
822
  }
129
823
  /**
130
824
  * Supported file types for parsing.
131
825
  */
132
- export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf";
826
+ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv";
133
827
  /**
134
828
  * Types of content nodes in the AST.
135
829
  */
136
- export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break";
830
+ export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment";
137
831
  /**
138
832
  * Supported MIME types for attachments.
139
833
  */
140
- export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation";
834
+ export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" | "application/vnd.openxmlformats-officedocument.presentationml.presentation" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation" | "application/rtf" | "text/csv" | "text/markdown" | "text/html";
141
835
  /**
142
836
  * Text formatting options available for text content.
143
837
  * Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
@@ -226,6 +920,8 @@ export interface SlideMetadata {
226
920
  noteId?: string;
227
921
  /** The style of the slide. */
228
922
  style?: string;
923
+ /** Unique anchor IDs for internal linking. */
924
+ anchorIds?: string[];
229
925
  }
230
926
  /**
231
927
  * Metadata for a sheet in Excel.
@@ -235,6 +931,8 @@ export interface SheetMetadata {
235
931
  sheetName: string;
236
932
  /** The style of the sheet. */
237
933
  style?: string;
934
+ /** Unique anchor IDs for internal linking. */
935
+ anchorIds?: string[];
238
936
  }
239
937
  /**
240
938
  * Detailed indentation information for paragraphs and headings.
@@ -262,6 +960,8 @@ export interface HeadingMetadata {
262
960
  style?: string;
263
961
  /** Detailed indentation information. */
264
962
  paragraphIndentation?: IndentationMetadata;
963
+ /** Unique anchor IDs for internal linking. */
964
+ anchorIds?: string[];
265
965
  }
266
966
  /**
267
967
  * Metadata for a paragraph.
@@ -273,6 +973,8 @@ export interface ParagraphMetadata {
273
973
  style?: string;
274
974
  /** Detailed indentation information. */
275
975
  paragraphIndentation?: IndentationMetadata;
976
+ /** Unique anchor IDs for internal linking. */
977
+ anchorIds?: string[];
276
978
  }
277
979
  /**
278
980
  * Metadata for a list item.
@@ -312,6 +1014,8 @@ export interface ListMetadata {
312
1014
  * @example "ListParagraph"
313
1015
  */
314
1016
  style?: string;
1017
+ /** Unique anchor IDs for internal linking. */
1018
+ anchorIds?: string[];
315
1019
  }
316
1020
  /**
317
1021
  * Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
@@ -340,6 +1044,8 @@ export interface CellMetadata {
340
1044
  colSpan?: number;
341
1045
  /** The style of the cell. */
342
1046
  style?: string;
1047
+ /** Unique anchor IDs for internal linking. */
1048
+ anchorIds?: string[];
343
1049
  }
344
1050
  /**
345
1051
  * Metadata for a chart node in the document.
@@ -352,6 +1058,8 @@ export interface ChartMetadata {
352
1058
  * @example "chart1.xml"
353
1059
  */
354
1060
  attachmentName: string;
1061
+ /** Unique anchor IDs for internal linking. */
1062
+ anchorIds?: string[];
355
1063
  }
356
1064
  /**
357
1065
  * Metadata for an image node in the document.
@@ -370,6 +1078,14 @@ export interface ImageMetadata {
370
1078
  * @example "Company logo"
371
1079
  */
372
1080
  altText?: string;
1081
+ /**
1082
+ * URL of the image if it is an external link.
1083
+ * Typical for HTML or Markdown images that point to remote servers.
1084
+ * @example "https://example.com/image.png"
1085
+ */
1086
+ url?: string;
1087
+ /** Unique anchor IDs for internal linking. */
1088
+ anchorIds?: string[];
373
1089
  }
374
1090
  /**
375
1091
  * Metadata for PDF page nodes.
@@ -415,6 +1131,8 @@ export interface NoteMetadata {
415
1131
  * @example "1", "2"
416
1132
  */
417
1133
  noteId?: string;
1134
+ /** Unique anchor IDs for internal linking. */
1135
+ anchorIds?: string[];
418
1136
  }
419
1137
  /**
420
1138
  * Metadata for break nodes.
@@ -441,10 +1159,19 @@ export interface BreakMetadata {
441
1159
  */
442
1160
  clear?: "all" | "left" | "none" | "right";
443
1161
  }
1162
+ /**
1163
+ * Metadata for a code block.
1164
+ */
1165
+ export interface CodeMetadata {
1166
+ /** The programming language of the code block (e.g., 'typescript', 'python') */
1167
+ language?: string;
1168
+ /** Unique anchor IDs for internal linking. */
1169
+ anchorIds?: string[];
1170
+ }
444
1171
  /**
445
1172
  * Union type for content metadata.
446
1173
  */
447
- export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
1174
+ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
448
1175
  /**
449
1176
  * Represents a node in the document content tree.
450
1177
  * This is the core building block of the parsed document structure.
@@ -682,9 +1409,19 @@ export interface OfficeMetadata {
682
1409
  * console.log(ast.metadata.author); // 'John Doe'
683
1410
  * console.log(ast.content.length); // Number of top-level content nodes
684
1411
  * console.log(ast.toText()); // Plain text representation
1412
+ * console.log((await ast.to('md')).value); // Markdown representation
1413
+ * console.log((await ast.to('html')).value); // HTML representation
1414
+ * console.log((await ast.to('rtf')).value); // RTF representation
1415
+ * console.log((await ast.to('csv')).value); // CSV representation
1416
+ * console.log((await ast.to('chunks')).value); // Chunks representation
685
1417
  * ```
686
1418
  */
687
1419
  export interface OfficeParserAST {
1420
+ /**
1421
+ * The original configuration used to parse this document.
1422
+ * This includes options like OCR settings, delimiter choices, and filtering flags.
1423
+ */
1424
+ config: OfficeParserConfig;
688
1425
  /**
689
1426
  * The type of the parsed file.
690
1427
  * Indicates which parser was used and what format the input was in.
@@ -719,7 +1456,12 @@ export interface OfficeParserAST {
719
1456
  * @example [{ type: 'image', mimeType: 'image/png', data: 'base64...', name: 'image1.png' }]
720
1457
  */
721
1458
  attachments: OfficeAttachment[];
1459
+ /** Any warnings or non-fatal issues encountered during parsing. */
1460
+ warnings: OfficeIssue[];
722
1461
  /**
1462
+ * @deprecated Use `.to('text')` instead.
1463
+ * Note: This method is synchronous, while the new `.to()` method is asynchronous.
1464
+ *
723
1465
  * Converts the entire AST to plain text.
724
1466
  * This method flattens the document structure and returns just the text content,
725
1467
  * stripping out all formatting, metadata, and structure.
@@ -734,6 +1476,20 @@ export interface OfficeParserAST {
734
1476
  * ```
735
1477
  */
736
1478
  toText(): string;
1479
+ /**
1480
+ * Converts this AST to the specified destination format.
1481
+ * This is the recommended way to convert the AST to different formats.
1482
+ *
1483
+ * @param destination The target format (e.g., 'text', 'md', 'html', 'pdf').
1484
+ * @param config Optional configuration for the generator.
1485
+ * @returns A promise resolving to the generated content (string or Buffer).
1486
+ * @example
1487
+ * ```typescript
1488
+ * const html = await ast.to('html', { includeFormatting: false });
1489
+ * const md = await ast.to('md');
1490
+ * ```
1491
+ */
1492
+ to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
737
1493
  }
738
1494
  /**
739
1495
  * Main parser class providing office document parsing functionality.
@@ -762,6 +1518,9 @@ export declare class OfficeParser {
762
1518
  * - `.odt`, `.odp`, `.ods` → OpenOfficeParser (ODF)
763
1519
  * - `.pdf` → PdfParser (PDF.js)
764
1520
  * - `.rtf` → RtfParser (custom RTF parser)
1521
+ * - `.csv` → CsvParser
1522
+ * - `.md` → MarkdownParser
1523
+ * - `.html` → HtmlParser
765
1524
  *
766
1525
  * @param file - File path (string), Buffer, or ArrayBuffer containing the document
767
1526
  * @param config - Optional configuration object (defaults applied for all omitted options)
@@ -798,8 +1557,72 @@ export declare class OfficeParser {
798
1557
  */
799
1558
  static terminateOcr(): Promise<void>;
800
1559
  }
1560
+ /**
1561
+ * Main generator class providing document conversion functionality.
1562
+ */
1563
+ export declare class OfficeGenerator {
1564
+ /**
1565
+ * Generates a file of the specified type from an AST.
1566
+ * This is the single source of truth for generation logic.
1567
+ *
1568
+ * @param ast - The OfficeParserAST to generate from
1569
+ * @param destination - The target format (e.g., 'text', 'md', 'html', 'pdf')
1570
+ * @param config - Optional configuration for the generator
1571
+ * @returns A promise resolving to the ConversionResult containing the value and messages
1572
+ * @throws {Error} If the destination format is unsupported
1573
+ */
1574
+ static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
1575
+ type: T;
1576
+ }, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
1577
+ }
1578
+ /**
1579
+ * Utility type to infer the file type from a file path string literal.
1580
+ */
1581
+ export type InferFileTypeFromPath<T> = T extends `${string}.${infer E}` ? (Lowercase<E> extends SupportedFileType ? Lowercase<E> : SupportedFileType) : SupportedFileType;
1582
+ /**
1583
+ * Main converter class providing a streamlined one-step API for document conversion.
1584
+ *
1585
+ * This class coordinates the `OfficeParser` and `OfficeGenerator` to transform
1586
+ * documents from one format to another (e.g., DOCX to Markdown, PDF to HTML).
1587
+ */
1588
+ export declare class OfficeConverter {
1589
+ /**
1590
+ * Converts an office document from its source format to a specified destination format.
1591
+ *
1592
+ * This method:
1593
+ * 1. Detects the source file type and parses it into a unified AST using `OfficeParser`.
1594
+ * 2. Automatically configures the parser based on the generator requirements (e.g., enabling
1595
+ * attachment extraction if images are requested in the output).
1596
+ * 3. Generates the destination document from the AST using `OfficeGenerator`.
1597
+ *
1598
+ * @template F The inferred type of the input file (path string or buffer).
1599
+ * @template T The authoritative source file type (inferred from path or config).
1600
+ *
1601
+ * @param file - File path (string), Buffer, or ArrayBuffer containing the source document.
1602
+ * @param destination - The target format (e.g., 'md', 'html', 'pdf', 'text', 'chunks').
1603
+ * @param config - Optional unified configuration for both the parser and generator phases.
1604
+ *
1605
+ * @returns A promise resolving to the ConversionResult containing the value and messages.
1606
+ * @throws {Error} If the source format is unsupported or parsing/generation fails.
1607
+ *
1608
+ * @example
1609
+ * ```typescript
1610
+ * // Convert Word to Markdown with a single call
1611
+ * const { value: markdown } = await OfficeConverter.convert('report.docx', 'md');
1612
+ *
1613
+ * // Convert PDF to HTML with OCR enabled for images
1614
+ * const { value: html } = await OfficeConverter.convert(buffer, 'html', {
1615
+ * ocr: true,
1616
+ * includeImages: true
1617
+ * });
1618
+ * ```
1619
+ */
1620
+ static convert<F extends string | Buffer | ArrayBuffer, T extends SupportedFileType = InferFileTypeFromPath<F>>(file: F, destination: SupportedDestination<T>, config?: OfficeConverterConfig<SupportedDestination<T>, T>): Promise<ConversionResult<SupportedDestination<T>>>;
1621
+ }
801
1622
  export declare const parseOffice: typeof OfficeParser.parseOffice;
802
1623
  export declare const terminateOcr: typeof OfficeParser.terminateOcr;
1624
+ export declare const convert: typeof OfficeConverter.convert;
1625
+ export declare const generate: typeof OfficeGenerator.generate;
803
1626
 
804
1627
  export {
805
1628
  OfficeParser as default,