officeparser 6.1.1 → 7.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +219 -26
- package/dist/OfficeConverter.d.ts +46 -0
- package/dist/OfficeConverter.js +72 -0
- package/dist/OfficeGenerator.d.ts +19 -0
- package/dist/OfficeGenerator.js +48 -0
- package/dist/OfficeParser.d.ts +6 -0
- package/dist/OfficeParser.js +55 -29
- package/dist/cli.d.ts +3 -1
- package/dist/cli.js +106 -22
- package/dist/defaults.d.ts +41 -0
- package/dist/defaults.js +172 -0
- package/dist/generators/BaseGenerator.d.ts +58 -0
- package/dist/generators/BaseGenerator.js +107 -0
- package/dist/generators/ChunkingGenerator.d.ts +81 -0
- package/dist/generators/ChunkingGenerator.js +683 -0
- package/dist/generators/CsvGenerator.d.ts +30 -0
- package/dist/generators/CsvGenerator.js +233 -0
- package/dist/generators/HtmlGenerator.d.ts +37 -0
- package/dist/generators/HtmlGenerator.js +1013 -0
- package/dist/generators/MarkdownGenerator.d.ts +59 -0
- package/dist/generators/MarkdownGenerator.js +481 -0
- package/dist/generators/PdfGenerator.d.ts +22 -0
- package/dist/generators/PdfGenerator.js +118 -0
- package/dist/generators/RtfGenerator.d.ts +15 -0
- package/dist/generators/RtfGenerator.js +208 -0
- package/dist/generators/TextGenerator.d.ts +13 -0
- package/dist/generators/TextGenerator.js +108 -0
- package/dist/index.d.ts +11 -3
- package/dist/index.js +17 -2
- package/dist/index.mjs +2 -2
- package/dist/officeparser.browser.d.ts +826 -5
- package/dist/officeparser.browser.iife.js +703 -52
- package/dist/officeparser.browser.mjs +703 -52
- package/dist/parsers/CsvParser.d.ts +9 -0
- package/dist/parsers/CsvParser.js +110 -0
- package/dist/parsers/ExcelParser.d.ts +2 -2
- package/dist/parsers/ExcelParser.js +145 -114
- package/dist/parsers/HtmlParser.d.ts +2 -0
- package/dist/parsers/HtmlParser.js +539 -0
- package/dist/parsers/MarkdownParser.d.ts +2 -0
- package/dist/parsers/MarkdownParser.js +360 -0
- package/dist/parsers/OpenOfficeParser.d.ts +2 -2
- package/dist/parsers/OpenOfficeParser.js +140 -79
- package/dist/parsers/PdfParser.d.ts +2 -2
- package/dist/parsers/PdfParser.js +52 -49
- package/dist/parsers/PowerPointParser.d.ts +2 -2
- package/dist/parsers/PowerPointParser.js +20 -23
- package/dist/parsers/RtfParser.d.ts +2 -2
- package/dist/parsers/RtfParser.js +1291 -1240
- package/dist/parsers/WordParser.d.ts +2 -2
- package/dist/parsers/WordParser.js +232 -97
- package/dist/sbom.cdx.json +99 -99
- package/dist/types.d.ts +781 -5
- package/dist/types.js +71 -0
- package/dist/utils/astUtils.d.ts +16 -0
- package/dist/utils/astUtils.js +32 -0
- package/dist/utils/configUtils.d.ts +26 -0
- package/dist/utils/configUtils.js +140 -0
- package/dist/utils/envUtils.js +56 -2
- package/dist/utils/errorUtils.d.ts +17 -29
- package/dist/utils/errorUtils.js +109 -52
- package/dist/utils/moduleLoader.js +15 -9
- package/dist/utils/ocrUtils.js +2 -1
- package/dist/utils/sheetUtils.d.ts +7 -0
- package/dist/utils/sheetUtils.js +35 -0
- package/dist/utils/styleMapper.d.ts +36 -0
- package/dist/utils/styleMapper.js +224 -0
- package/dist/utils/xmlUtils.d.ts +0 -8
- package/dist/utils/xmlUtils.js +2 -1
- package/package.json +27 -8
package/dist/types.d.ts
CHANGED
|
@@ -1,3 +1,71 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Standard error types for OfficeParser.
|
|
3
|
+
* Use these to identify the kind of error being reported.
|
|
4
|
+
*/
|
|
5
|
+
export declare enum OfficeErrorType {
|
|
6
|
+
/** Unsupported file extension */
|
|
7
|
+
EXTENSION_UNSUPPORTED = "EXTENSION_UNSUPPORTED",
|
|
8
|
+
/** File appears to be corrupted or malformed */
|
|
9
|
+
FILE_CORRUPTED = "FILE_CORRUPTED",
|
|
10
|
+
/** File could not be found at the specified path */
|
|
11
|
+
FILE_DOES_NOT_EXIST = "FILE_DOES_NOT_EXIST",
|
|
12
|
+
/** Specified location/directory is not reachable or is a directory */
|
|
13
|
+
LOCATION_NOT_FOUND = "LOCATION_NOT_FOUND",
|
|
14
|
+
/** Arguments passed to the function are missing or invalid */
|
|
15
|
+
IMPROPER_ARGUMENTS = "IMPROPER_ARGUMENTS",
|
|
16
|
+
/** Error occurred while reading or processing file buffers */
|
|
17
|
+
IMPROPER_BUFFERS = "IMPROPER_BUFFERS",
|
|
18
|
+
/** Input type is not a supported type (string, Buffer, ArrayBuffer) */
|
|
19
|
+
INVALID_INPUT = "INVALID_INPUT",
|
|
20
|
+
/** PDF worker source is missing (required in browser) */
|
|
21
|
+
PDF_WORKER_MISSING = "PDF_WORKER_MISSING",
|
|
22
|
+
/** Attempted to use Node.js-only features in a browser environment */
|
|
23
|
+
FEATURE_NOT_SUPPORTED_IN_BROWSER = "FEATURE_NOT_SUPPORTED_IN_BROWSER",
|
|
24
|
+
/** Style mapping string is malformed */
|
|
25
|
+
INVALID_STYLE_MAPPING = "INVALID_STYLE_MAPPING",
|
|
26
|
+
/** Selector in style mapping is invalid */
|
|
27
|
+
INVALID_SELECTOR = "INVALID_SELECTOR",
|
|
28
|
+
/** Output mapping in style mapping is invalid */
|
|
29
|
+
INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
|
|
30
|
+
/** Semantic chunking strategy is selected but no embedding function is provided */
|
|
31
|
+
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Standard warning types for OfficeParser.
|
|
35
|
+
* Use these for reporting non-fatal issues or performance tips.
|
|
36
|
+
*/
|
|
37
|
+
export declare enum OfficeWarningType {
|
|
38
|
+
/** Performance advice (e.g., Rosetta translation on Mac) */
|
|
39
|
+
PERFORMANCE_TIP = "PERFORMANCE_TIP",
|
|
40
|
+
/** OCR processing failed for an attachment */
|
|
41
|
+
OCR_FAILED = "OCR_FAILED",
|
|
42
|
+
/** Extraction of structured chart data failed */
|
|
43
|
+
CHART_DATA_EXTRACTION_FAILED = "CHART_DATA_EXTRACTION_FAILED",
|
|
44
|
+
/** Automatic worker path failed, falling back to CDN */
|
|
45
|
+
PDF_WORKER_FALLBACK = "PDF_WORKER_FALLBACK",
|
|
46
|
+
/** General attachment extraction failure */
|
|
47
|
+
ATTACHMENT_EXTRACTION_FAILED = "ATTACHMENT_EXTRACTION_FAILED",
|
|
48
|
+
/** Failed to load a specific page in a multi-page document */
|
|
49
|
+
PAGE_LOAD_FAILED = "PAGE_LOAD_FAILED",
|
|
50
|
+
/** Failed to load a required dynamic dependency */
|
|
51
|
+
DEPENDENCY_LOAD_FAILED = "DEPENDENCY_LOAD_FAILED",
|
|
52
|
+
/** Failed to extract images from a source */
|
|
53
|
+
IMAGE_EXTRACTION_FAILED = "IMAGE_EXTRACTION_FAILED",
|
|
54
|
+
/** Failed to extract annotations from a document */
|
|
55
|
+
ANNOTATION_EXTRACTION_FAILED = "ANNOTATION_EXTRACTION_FAILED",
|
|
56
|
+
/** Failed to process an extracted image bitmap */
|
|
57
|
+
IMAGE_PROCESSING_FAILED = "IMAGE_PROCESSING_FAILED",
|
|
58
|
+
/** Warning about limitations of browser-based generation */
|
|
59
|
+
BROWSER_GENERATION_LIMITATION = "BROWSER_GENERATION_LIMITATION",
|
|
60
|
+
/** Specified sheet range in Excel/ODS export was not found */
|
|
61
|
+
SHEET_RANGE_NOT_FOUND = "SHEET_RANGE_NOT_FOUND",
|
|
62
|
+
/** Buffer content type does not match the provided or expected file extension */
|
|
63
|
+
BUFFER_TYPE_MISMATCH = "BUFFER_TYPE_MISMATCH",
|
|
64
|
+
/** No chunks were generated for the document given the current strategy */
|
|
65
|
+
EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
|
|
66
|
+
/** A node was skipped because it only contained whitespace */
|
|
67
|
+
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
|
|
68
|
+
}
|
|
1
69
|
/**
|
|
2
70
|
* Configuration options for OCR.
|
|
3
71
|
*/
|
|
@@ -16,16 +84,19 @@ export interface OcrConfig {
|
|
|
16
84
|
/**
|
|
17
85
|
* Path to the Tesseract worker script.
|
|
18
86
|
* Primarily used for offline/air-gapped environments.
|
|
87
|
+
* Default is ''.
|
|
19
88
|
*/
|
|
20
89
|
workerPath?: string;
|
|
21
90
|
/**
|
|
22
91
|
* Path to the Tesseract core script.
|
|
23
92
|
* Primarily used for offline/air-gapped environments.
|
|
93
|
+
* Default is ''.
|
|
24
94
|
*/
|
|
25
95
|
corePath?: string;
|
|
26
96
|
/**
|
|
27
97
|
* Path for Tesseract language files (traineddata).
|
|
28
98
|
* Primarily used for offline/air-gapped environments.
|
|
99
|
+
* Default is ''.
|
|
29
100
|
*/
|
|
30
101
|
langPath?: string;
|
|
31
102
|
/**
|
|
@@ -40,10 +111,17 @@ export interface OcrConfig {
|
|
|
40
111
|
*/
|
|
41
112
|
export interface OfficeParserConfig {
|
|
42
113
|
/**
|
|
114
|
+
* @deprecated Use `onWarning` instead.
|
|
43
115
|
* Flag to show all the logs to console in case of an error irrespective of your own handling.
|
|
44
116
|
* Default is false.
|
|
45
117
|
*/
|
|
46
118
|
outputErrorToConsole?: boolean;
|
|
119
|
+
/**
|
|
120
|
+
* Callback for warnings or non-fatal errors encountered during parsing.
|
|
121
|
+
* Allows you to capture issues like OCR failures or attachment extraction errors
|
|
122
|
+
* without stopping the parsing process.
|
|
123
|
+
*/
|
|
124
|
+
onWarning?: (issue: OfficeIssue) => void;
|
|
47
125
|
/**
|
|
48
126
|
* The delimiter used for every new line in places that allow multiline text like word.
|
|
49
127
|
* Default is \n.
|
|
@@ -112,7 +190,7 @@ export interface OfficeParserConfig {
|
|
|
112
190
|
* The URL/path to the PDF.js worker script.
|
|
113
191
|
*
|
|
114
192
|
* **Mandatory** when using PDF parsing in browser environments to avoid worker configuration errors.
|
|
115
|
-
* If not provided, it defaults to `https://
|
|
193
|
+
* If not provided, it defaults to `https://cdn.jsdelivr.net/npm/pdfjs-dist@5.6.205/build/pdf.worker.min.mjs`.
|
|
116
194
|
* You can override this with your own local path or a different CDN link.
|
|
117
195
|
*/
|
|
118
196
|
pdfWorkerSrc?: string;
|
|
@@ -123,19 +201,655 @@ export interface OfficeParserConfig {
|
|
|
123
201
|
* Default is false
|
|
124
202
|
*/
|
|
125
203
|
includeBreakNodes?: boolean;
|
|
204
|
+
/**
|
|
205
|
+
* Flag to ignore all internal (anchor) links during parsing.
|
|
206
|
+
* When true, all bookmarks, cross-references, and internal document jumps are stripped
|
|
207
|
+
* from the AST. Only external URLs will be preserved.
|
|
208
|
+
*
|
|
209
|
+
* Use this if you want a "flat" document without any internal interactivity.
|
|
210
|
+
*
|
|
211
|
+
* Default is false.
|
|
212
|
+
*/
|
|
213
|
+
ignoreInternalLinks?: boolean;
|
|
214
|
+
/**
|
|
215
|
+
* Optional hint for the file format.
|
|
216
|
+
* When a Buffer or ArrayBuffer is passed, the parser relies on magic bytes to detect the file type.
|
|
217
|
+
* Text-based formats like 'md', 'html', and 'csv' lack reliable magic bytes.
|
|
218
|
+
* If you are parsing these formats from a Buffer, you must provide this fileType hint.
|
|
219
|
+
*
|
|
220
|
+
* This is authoritative and is used to determine the file type, so it should be accurate.
|
|
221
|
+
* If provided, this bypasses the magic bytes detection and the file extension-based detection either way.
|
|
222
|
+
*
|
|
223
|
+
* Default is null.
|
|
224
|
+
*/
|
|
225
|
+
fileType?: SupportedFileType | null;
|
|
226
|
+
/**
|
|
227
|
+
* Custom delimiter for CSV files.
|
|
228
|
+
* Defaults to ',' but can be overridden (e.g., ';', '\t').
|
|
229
|
+
*/
|
|
230
|
+
csvDelimiter?: string;
|
|
231
|
+
}
|
|
232
|
+
/**
|
|
233
|
+
* A fully-populated parser configuration containing all options.
|
|
234
|
+
* Used internally for merging and resolution.
|
|
235
|
+
*/
|
|
236
|
+
export type FullOfficeParserConfig = DeepRequired<OfficeParserConfig>;
|
|
237
|
+
/**
|
|
238
|
+
* Represents a single issue (warning, error, or info) generated during document processing.
|
|
239
|
+
*/
|
|
240
|
+
export interface OfficeIssue {
|
|
241
|
+
/** The severity of the issue. */
|
|
242
|
+
type: 'warning' | 'info' | 'error';
|
|
243
|
+
/** Human-readable message text. */
|
|
244
|
+
message: string;
|
|
245
|
+
/** The specific AST node that triggered this issue, if applicable. */
|
|
246
|
+
node?: OfficeContentNode;
|
|
247
|
+
/** A unique error code for programmatic handling. */
|
|
248
|
+
code: OfficeWarningType | OfficeErrorType;
|
|
249
|
+
/** Optional additional context or original error object. */
|
|
250
|
+
details?: any;
|
|
251
|
+
}
|
|
252
|
+
/**
|
|
253
|
+
* The result of a document conversion operation.
|
|
254
|
+
*/
|
|
255
|
+
export interface ConversionResult<D extends string = UniversalGeneratorFormat> {
|
|
256
|
+
/** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
|
|
257
|
+
value: D extends 'pdf' ? Uint8Array : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : D extends UniversalGeneratorFormat ? string : never;
|
|
258
|
+
/** A collection of issues (warnings/infos) generated during the process. */
|
|
259
|
+
messages: OfficeIssue[];
|
|
260
|
+
}
|
|
261
|
+
/**
|
|
262
|
+
* Universal formats supported by all source types for generation.
|
|
263
|
+
*/
|
|
264
|
+
export type UniversalGeneratorFormat = 'text' | 'md' | 'html' | 'pdf' | 'csv' | 'rtf' | 'chunks';
|
|
265
|
+
/**
|
|
266
|
+
* Allowed destination formats for a given source type.
|
|
267
|
+
* Currently, all generators are universal across all source formats.
|
|
268
|
+
*/
|
|
269
|
+
export type SupportedDestination<_T extends SupportedFileType = SupportedFileType> = UniversalGeneratorFormat;
|
|
270
|
+
/**
|
|
271
|
+
* Configuration options for the OfficeGenerator.
|
|
272
|
+
*/
|
|
273
|
+
/**
|
|
274
|
+
* Common configuration options for all generators.
|
|
275
|
+
*/
|
|
276
|
+
export interface CommonGeneratorConfig {
|
|
277
|
+
/**
|
|
278
|
+
* Callback called for every node during generation.
|
|
279
|
+
* Allows users to modify nodes before processing, completely override rendering, or filter them out.
|
|
280
|
+
*
|
|
281
|
+
* #### Callback Capabilities:
|
|
282
|
+
* 1. **Filter/Remove Nodes**: Return `false` to skip a node and all its children.
|
|
283
|
+
* 2. **Override Rendering**: Return a `string` to use that exact text as the output, bypassing default logic and recursion.
|
|
284
|
+
* 3. **Mutate Nodes**: Modify the `node` object directly (e.g., changing `node.text`) and return `void` to let the generator proceed with your changes.
|
|
285
|
+
* 4. **Async Support**: The callback can be `async`, allowing you to fetch external data or perform complex logic during generation.
|
|
286
|
+
*/
|
|
287
|
+
onNode?: (node: OfficeContentNode) => string | false | Promise<string | false | void> | void;
|
|
288
|
+
/**
|
|
289
|
+
* Callback for warnings, non-fatal errors, or issues encountered during generation.
|
|
290
|
+
* Allows the process to continue while reporting skipping or approximation of content.
|
|
291
|
+
*/
|
|
292
|
+
onWarning?: (issue: OfficeIssue) => void;
|
|
293
|
+
/**
|
|
294
|
+
* Map document styles (e.g., 'Heading 1', 'Intense Quote') to specific semantic elements.
|
|
295
|
+
*
|
|
296
|
+
* DESIGN PHILOSOPHY:
|
|
297
|
+
* This is the primary way to customize how the library interprets the visual
|
|
298
|
+
* structure of your source documents.
|
|
299
|
+
*
|
|
300
|
+
* To disable all semantic translation and use raw AST types only,
|
|
301
|
+
* set `ignoreDefaultStyleMap: true` and leave `styleMap` empty.
|
|
302
|
+
*
|
|
303
|
+
* It supports two formats:
|
|
304
|
+
*
|
|
305
|
+
* 1. LEGACY STRING DSL:
|
|
306
|
+
* Simple "selector => output" syntax. Highly compatible with mammoth.js style maps.
|
|
307
|
+
* @example ["p[style-name='Heading 1'] => h1"]
|
|
308
|
+
* @example ["p[style='Quote'] => blockquote"]
|
|
309
|
+
*
|
|
310
|
+
* 2. STRUCTURED OBJECTS (Recommended):
|
|
311
|
+
* More powerful and strictly typed. Ideal for complex logic or when you
|
|
312
|
+
* need to apply specific classes/attributes for the HTML generator.
|
|
313
|
+
* @example
|
|
314
|
+
* [
|
|
315
|
+
* {
|
|
316
|
+
* selector: { nodeType: 'paragraph', attributes: { style: 'Heading 1' } },
|
|
317
|
+
* output: { tag: 'h1', classes: ['main-title'], attributes: { id: 'top' } }
|
|
318
|
+
* }
|
|
319
|
+
* ]
|
|
320
|
+
*
|
|
321
|
+
* Note: This property works in conjunction with `ignoreDefaultStyleMap`.
|
|
322
|
+
* Defaults to a robust built-in map that covers common standard Office styles.
|
|
323
|
+
*/
|
|
324
|
+
styleMap?: string[] | StructuredStyleMapping[];
|
|
325
|
+
/**
|
|
326
|
+
* Whether to include visual formatting like font size, font family, and colors in the output.
|
|
327
|
+
* Set to false for clean, semantic output.
|
|
328
|
+
* Defaults to true.
|
|
329
|
+
*/
|
|
330
|
+
includeFormatting?: boolean;
|
|
331
|
+
/**
|
|
332
|
+
* Whether to automatically generate unique slug-based IDs for headings.
|
|
333
|
+
* Useful for table-of-contents and anchor links.
|
|
334
|
+
* Defaults to true.
|
|
335
|
+
*/
|
|
336
|
+
generateIds?: boolean;
|
|
337
|
+
/**
|
|
338
|
+
* Whether to render document metadata (title, author, etc.) as visible content
|
|
339
|
+
* in the generated output (e.g., a header block in HTML or plain text).
|
|
340
|
+
* Structural metadata (HTML <meta> tags, Markdown YAML frontmatter) is always included.
|
|
341
|
+
* Defaults to false.
|
|
342
|
+
*/
|
|
343
|
+
renderMetadata?: boolean;
|
|
344
|
+
/**
|
|
345
|
+
* Whether to ignore the built-in default style mappings (e.g. "Heading 1" -> h1).
|
|
346
|
+
* Set to true if you want full control over style mapping.
|
|
347
|
+
* Defaults to false.
|
|
348
|
+
*/
|
|
349
|
+
ignoreDefaultStyleMap?: boolean;
|
|
350
|
+
/**
|
|
351
|
+
* Whether to include images in the generated output.
|
|
352
|
+
* Defaults to true.
|
|
353
|
+
*/
|
|
354
|
+
includeImages?: boolean;
|
|
355
|
+
/**
|
|
356
|
+
* Whether to include interactive charts in the generated output (HTML only).
|
|
357
|
+
* Defaults to true.
|
|
358
|
+
*/
|
|
359
|
+
includeCharts?: boolean;
|
|
360
|
+
/**
|
|
361
|
+
* Whether to ignore all internal (anchor) links and anchor IDs during generation.
|
|
362
|
+
* When true, all bookmarks, cross-references, and internal document jumps are stripped.
|
|
363
|
+
* Specifically for Markdown, this removes the {#id} block from headings.
|
|
364
|
+
* Defaults to false.
|
|
365
|
+
*/
|
|
366
|
+
ignoreInternalLinks?: boolean;
|
|
367
|
+
}
|
|
368
|
+
/**
|
|
369
|
+
* Destination-aware generator configuration.
|
|
370
|
+
* Restricts format-specific configurations to their respective destinations.
|
|
371
|
+
*/
|
|
372
|
+
/**
|
|
373
|
+
* Mapping of destination formats to their specific configuration interfaces.
|
|
374
|
+
*/
|
|
375
|
+
export interface GeneratorSubConfigMap {
|
|
376
|
+
html: HtmlGeneratorConfig;
|
|
377
|
+
md: MdGeneratorConfig;
|
|
378
|
+
pdf: PdfGeneratorConfig;
|
|
379
|
+
csv: CsvGeneratorConfig;
|
|
380
|
+
text: TextGeneratorConfig;
|
|
381
|
+
rtf: RtfGeneratorConfig;
|
|
382
|
+
chunks: ChunkingConfig;
|
|
383
|
+
}
|
|
384
|
+
/**
|
|
385
|
+
* Configuration options for document generators.
|
|
386
|
+
*
|
|
387
|
+
* This interface is designed to be format-aware. When you specify a destination format
|
|
388
|
+
* (e.g., `OfficeGenerator.generate(ast, 'html', config)`), the generic parameter `D`
|
|
389
|
+
* ensures that only the relevant sub-configuration (e.g., `htmlConfig`) is available
|
|
390
|
+
* for type checking.
|
|
391
|
+
*
|
|
392
|
+
* @template D The destination format string. Defaults to `string` for a general configuration.
|
|
393
|
+
*/
|
|
394
|
+
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & {
|
|
395
|
+
[K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
|
|
396
|
+
};
|
|
397
|
+
/**
|
|
398
|
+
* Configuration options for the OfficeConverter.
|
|
399
|
+
* Combines relevant parser and generator settings for a seamless one-step conversion.
|
|
400
|
+
*
|
|
401
|
+
* @template D The destination format string.
|
|
402
|
+
*/
|
|
403
|
+
/**
|
|
404
|
+
* Configuration options for the OfficeConverter.
|
|
405
|
+
* Combines general generator settings with a specific subset of parser settings.
|
|
406
|
+
*
|
|
407
|
+
* @template D The destination format string.
|
|
408
|
+
* @template T The source file type.
|
|
409
|
+
*/
|
|
410
|
+
export type OfficeConverterConfig<D extends string = string, T extends SupportedFileType = SupportedFileType> = {
|
|
411
|
+
/**
|
|
412
|
+
* Specific configuration for the source parsing phase.
|
|
413
|
+
*/
|
|
414
|
+
parseConfig?: OfficeParserConfig & {
|
|
415
|
+
fileType?: T;
|
|
416
|
+
};
|
|
417
|
+
/**
|
|
418
|
+
* Specific configuration for the destination generation phase.
|
|
419
|
+
*/
|
|
420
|
+
generatorConfig?: GeneratorConfig<D>;
|
|
421
|
+
/**
|
|
422
|
+
* Callback for warnings or non-fatal errors encountered during the entire conversion process.
|
|
423
|
+
* This is passed to both the parser and the generator.
|
|
424
|
+
* If provided, this takes precedence over callbacks inside parseConfig or generatorConfig.
|
|
425
|
+
*/
|
|
426
|
+
onWarning?: (issue: OfficeIssue) => void;
|
|
427
|
+
};
|
|
428
|
+
/**
|
|
429
|
+
* Deeply required type helper.
|
|
430
|
+
*/
|
|
431
|
+
export type DeepRequired<T> = T extends Function | Date | Buffer | RegExp ? T : T extends Array<infer U> ? Array<DeepRequired<U>> : T extends object ? {
|
|
432
|
+
[P in keyof T]-?: DeepRequired<T[P]>;
|
|
433
|
+
} : T;
|
|
434
|
+
/**
|
|
435
|
+
* A fully-populated generator configuration containing all sub-configs.
|
|
436
|
+
* Used internally for merging and resolution.
|
|
437
|
+
* `chunksConfig` is typed as `ChunkingConfig` directly (not DeepRequired) because
|
|
438
|
+
* it is a discriminated union whose members cannot be uniformly deep-required.
|
|
439
|
+
*/
|
|
440
|
+
export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
|
|
441
|
+
[K in keyof Omit<GeneratorSubConfigMap, 'chunks'> as `${K}Config`]: GeneratorSubConfigMap[K];
|
|
442
|
+
}> & {
|
|
443
|
+
chunksConfig: ChunkingConfig;
|
|
444
|
+
};
|
|
445
|
+
/**
|
|
446
|
+
* Configuration options for HTML generation.
|
|
447
|
+
*/
|
|
448
|
+
export interface HtmlGeneratorConfig {
|
|
449
|
+
/**
|
|
450
|
+
* Whether to wrap the output in a full HTML document structure (e.g., <html>, <head>, etc.).
|
|
451
|
+
* Defaults to true.
|
|
452
|
+
*/
|
|
453
|
+
standalone?: boolean;
|
|
454
|
+
/**
|
|
455
|
+
* URL for the Chart.js library to use when 'includeCharts' is true.
|
|
456
|
+
* Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
|
|
457
|
+
*/
|
|
458
|
+
chartJsSrc?: string;
|
|
459
|
+
}
|
|
460
|
+
/**
|
|
461
|
+
* Configuration options for PDF generation.
|
|
462
|
+
* Maps closely to Puppeteer's PDF options.
|
|
463
|
+
*/
|
|
464
|
+
export interface PdfGeneratorConfig {
|
|
465
|
+
/** Paper format. Defaults to 'A4'. */
|
|
466
|
+
format?: 'letter' | 'legal' | 'tabloid' | 'ledger' | 'a0' | 'a1' | 'a2' | 'a3' | 'a4' | 'a5' | 'a6' | 'Letter' | 'Legal' | 'Tabloid' | 'Ledger' | 'A0' | 'A1' | 'A2' | 'A3' | 'A4' | 'A5' | 'A6';
|
|
467
|
+
/** Paper width, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
|
|
468
|
+
width?: string | number;
|
|
469
|
+
/** Paper height, accepts values labeled with units (e.g., '5in', '3cm') or numbers (in pixels). */
|
|
470
|
+
height?: string | number;
|
|
471
|
+
/** Whether to print in landscape orientation. Defaults to false. */
|
|
472
|
+
landscape?: boolean;
|
|
473
|
+
/** Whether to print background graphics. Defaults to true. */
|
|
474
|
+
printBackground?: boolean;
|
|
475
|
+
/** Scale of the webpage rendering. Defaults to 1. */
|
|
476
|
+
scale?: number;
|
|
477
|
+
/** Paper margins. */
|
|
478
|
+
margin?: {
|
|
479
|
+
top?: string | number;
|
|
480
|
+
right?: string | number;
|
|
481
|
+
bottom?: string | number;
|
|
482
|
+
left?: string | number;
|
|
483
|
+
};
|
|
484
|
+
/** Whether to display header and footer. Defaults to false. */
|
|
485
|
+
displayHeaderFooter?: boolean;
|
|
486
|
+
/** HTML template for the print header. */
|
|
487
|
+
headerTemplate?: string;
|
|
488
|
+
/** HTML template for the print footer. */
|
|
489
|
+
footerTemplate?: string;
|
|
490
|
+
/**
|
|
491
|
+
* Optional Puppeteer launch options for Node.js environment.
|
|
492
|
+
* Useful for setting custom executable paths or args in CI/CD.
|
|
493
|
+
*/
|
|
494
|
+
launchOptions?: any;
|
|
495
|
+
}
|
|
496
|
+
/**
|
|
497
|
+
* Structured style mapping definition for the StyleMapper.
|
|
498
|
+
*
|
|
499
|
+
* DESIGN PHILOSOPHY: "Semantic Translation"
|
|
500
|
+
* -----------------------------------------
|
|
501
|
+
* Office documents (Word, RTF, PPTX) often use custom or localized style names
|
|
502
|
+
* (e.g., "Heading 1" in English vs "Titre 1" in French, or "MyCompany-Quote").
|
|
503
|
+
*
|
|
504
|
+
* This interface allows you to create a "semantic bridge" between these arbitrary
|
|
505
|
+
* source styles and a universal vocabulary of document elements.
|
|
506
|
+
*
|
|
507
|
+
* WHY USE HTML TAGS FOR NON-HTML OUTPUT?
|
|
508
|
+
* --------------------------------------
|
|
509
|
+
* We use HTML tags (`h1`, `blockquote`, `code`, `pre`) as a "Universal Intermediate
|
|
510
|
+
* Language". By mapping a custom Word style to `blockquote`, you are defining its
|
|
511
|
+
* SEMANTIC MEANING rather than its physical appearance.
|
|
512
|
+
*
|
|
513
|
+
* Each generator then interprets this meaning natively:
|
|
514
|
+
* - HTML Generator: Directly renders the `<blockquote>` tag with your classes.
|
|
515
|
+
* - Markdown Generator: Sees 'blockquote' and renders the standard `> ` prefix.
|
|
516
|
+
* - Text Generator: Sees 'blockquote' and applies appropriate structural indentation.
|
|
517
|
+
*/
|
|
518
|
+
export interface StructuredStyleMapping {
|
|
519
|
+
/**
|
|
520
|
+
* The criteria used to identify which AST nodes should be transformed.
|
|
521
|
+
* Think of this as the "Source Filter".
|
|
522
|
+
*/
|
|
523
|
+
selector: {
|
|
524
|
+
/**
|
|
525
|
+
* The structural type of the node (e.g., 'paragraph', 'heading', 'text').
|
|
526
|
+
* Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
|
|
527
|
+
*/
|
|
528
|
+
nodeType?: string;
|
|
529
|
+
/**
|
|
530
|
+
* A dictionary of attributes to match on the node.
|
|
531
|
+
*
|
|
532
|
+
* The most common use case is matching the 'style' attribute from
|
|
533
|
+
* Word documents (e.g., { style: 'Intense Quote' }).
|
|
534
|
+
*
|
|
535
|
+
* Matchers:
|
|
536
|
+
* - Literal: `style: 'Heading 1'` matches exactly.
|
|
537
|
+
* - Operator: `{ value: 'Title', operator: '~=' }` matches if the word 'Title'
|
|
538
|
+
* is found within the style name.
|
|
539
|
+
*/
|
|
540
|
+
attributes?: Record<string, string | number | boolean | {
|
|
541
|
+
value: string | number | boolean;
|
|
542
|
+
operator: '=' | '~=';
|
|
543
|
+
}>;
|
|
544
|
+
};
|
|
545
|
+
/**
|
|
546
|
+
* The target representation for the matched node.
|
|
547
|
+
* Think of this as the "Semantic Meaning" you want to assign to the match.
|
|
548
|
+
*/
|
|
549
|
+
output: {
|
|
550
|
+
/**
|
|
551
|
+
* The universal semantic tag (e.g., 'h1', 'h2', 'blockquote', 'code', 'pre', 'u').
|
|
552
|
+
* All generators use this tag to decide their native output syntax.
|
|
553
|
+
*/
|
|
554
|
+
tag: string;
|
|
555
|
+
/**
|
|
556
|
+
* CSS classes to apply to the output.
|
|
557
|
+
* This is utilized by the HTML generator to allow for downstream CSS styling.
|
|
558
|
+
*/
|
|
559
|
+
classes?: string[];
|
|
560
|
+
/**
|
|
561
|
+
* Key-value pair of HTML attributes (like 'id', 'data-*', or 'style') to apply.
|
|
562
|
+
* Primarily used by the HTML generator for high-fidelity conversion.
|
|
563
|
+
*/
|
|
564
|
+
attributes?: Record<string, string>;
|
|
565
|
+
/**
|
|
566
|
+
* If true, prevents the generator from collapsing this element into
|
|
567
|
+
* adjacent elements of the same type.
|
|
568
|
+
*
|
|
569
|
+
* For example, multiple paragraphs mapped to 'blockquote' normally merge into
|
|
570
|
+
* one big blockquote. Setting `fresh: true` forces them to be separate blocks.
|
|
571
|
+
*/
|
|
572
|
+
fresh?: boolean;
|
|
573
|
+
};
|
|
574
|
+
}
|
|
575
|
+
/**
|
|
576
|
+
* Configuration options for RTF generation.
|
|
577
|
+
*/
|
|
578
|
+
export interface RtfGeneratorConfig {
|
|
579
|
+
}
|
|
580
|
+
/**
|
|
581
|
+
* Configuration options for CSV generation.
|
|
582
|
+
*/
|
|
583
|
+
export interface CsvGeneratorConfig {
|
|
584
|
+
/**
|
|
585
|
+
* Range of sheets to export.
|
|
586
|
+
* Supports formats like "1", "1-3", "1,2", "1,3-5,7".
|
|
587
|
+
* 1-based indexing.
|
|
588
|
+
* Default is '' (all sheets).
|
|
589
|
+
*/
|
|
590
|
+
sheets?: string;
|
|
591
|
+
/**
|
|
592
|
+
* Whether to merge all selected sheets into a single CSV.
|
|
593
|
+
* If false, returns a ZIP archive containing individual CSV files.
|
|
594
|
+
* Defaults to false.
|
|
595
|
+
*/
|
|
596
|
+
mergeSheets?: boolean;
|
|
597
|
+
/**
|
|
598
|
+
* Custom delimiter for CSV files.
|
|
599
|
+
* Defaults to ','.
|
|
600
|
+
*/
|
|
601
|
+
columnDelimiter?: string;
|
|
602
|
+
}
|
|
603
|
+
/**
|
|
604
|
+
* Configuration options for Markdown generation.
|
|
605
|
+
*/
|
|
606
|
+
export interface MdGeneratorConfig {
|
|
607
|
+
/**
|
|
608
|
+
* Whether to fallback to HTML tags for features not supported by standard Markdown.
|
|
609
|
+
*
|
|
610
|
+
* Markdown has limited support for complex document structures. This flag controls how
|
|
611
|
+
* the generator handles features that cannot be represented in pure Markdown:
|
|
612
|
+
*
|
|
613
|
+
* 1. If a feature is NOT supported natively by Markdown (e.g., nested tables, text alignment,
|
|
614
|
+
* underline, subscript/superscript):
|
|
615
|
+
* - If true: The generator will use HTML tags (<u>, <sub>, <div>, <table>, etc.) to
|
|
616
|
+
* maintain high fidelity.
|
|
617
|
+
* - If false: The generator will skip or simplify the feature (e.g., ignoring alignment,
|
|
618
|
+
* skipping underline, or hoisting nested tables out of their cells).
|
|
619
|
+
*
|
|
620
|
+
* 2. If a feature IS supported by Markdown but a higher quality version is possible
|
|
621
|
+
* via HTML (e.g., tables with merged cells):
|
|
622
|
+
* - If true: Use HTML for better fidelity.
|
|
623
|
+
* - If false: Use native Markdown syntax (e.g., a standard GFM table grid).
|
|
624
|
+
*
|
|
625
|
+
* Defaults to true.
|
|
626
|
+
*/
|
|
627
|
+
fallbackToHtml?: boolean;
|
|
628
|
+
}
|
|
629
|
+
/**
|
|
630
|
+
* Configuration options for plain text generation.
|
|
631
|
+
*/
|
|
632
|
+
export interface TextGeneratorConfig {
|
|
633
|
+
/**
|
|
634
|
+
* The delimiter used for every new line.
|
|
635
|
+
* Defaults to '\n'.
|
|
636
|
+
*/
|
|
637
|
+
newlineDelimiter?: string;
|
|
638
|
+
/**
|
|
639
|
+
* Whether to attempt to preserve the original document layout.
|
|
640
|
+
* If true, tables will be rendered with separators and aligned columns.
|
|
641
|
+
* If false, output will be a flat stream of text nodes.
|
|
642
|
+
* Defaults to false.
|
|
643
|
+
*/
|
|
644
|
+
preserveLayout?: boolean;
|
|
645
|
+
}
|
|
646
|
+
/**
|
|
647
|
+
* The strategy used for chunking a document for RAG pipelines.
|
|
648
|
+
* - 'fixed-size': Traditional character/token count based splitting.
|
|
649
|
+
* - 'document-structure': Leverages the AST to split at natural document boundaries.
|
|
650
|
+
* - 'semantic': Uses embedding similarity to find natural topic breakpoints.
|
|
651
|
+
*/
|
|
652
|
+
export type ChunkingStrategy = 'fixed-size' | 'document-structure' | 'semantic';
|
|
653
|
+
/**
|
|
654
|
+
* Base configuration applicable to all chunking strategies.
|
|
655
|
+
*/
|
|
656
|
+
export interface BaseChunkingConfig {
|
|
657
|
+
/**
|
|
658
|
+
* The strategy used for chunking.
|
|
659
|
+
* Default is 'document-structure'.
|
|
660
|
+
*/
|
|
661
|
+
strategy?: ChunkingStrategy;
|
|
662
|
+
/**
|
|
663
|
+
* A function that measures the size of a text string.
|
|
664
|
+
* Defaults to character count: `(text) => text.length`.
|
|
665
|
+
* Override with a token counter (e.g., `tiktoken`) for strict LLM context window adherence.
|
|
666
|
+
*/
|
|
667
|
+
lengthFunction?: (text: string) => number;
|
|
668
|
+
/**
|
|
669
|
+
* Whether to strip leading/trailing whitespace from each chunk.
|
|
670
|
+
* Default is true.
|
|
671
|
+
*/
|
|
672
|
+
stripWhitespace?: boolean;
|
|
673
|
+
/**
|
|
674
|
+
* Whether to include rich AST metadata (page number, slide number, heading, etc.)
|
|
675
|
+
* in the generated chunk objects.
|
|
676
|
+
* Default is true.
|
|
677
|
+
*/
|
|
678
|
+
includeMetadata?: boolean;
|
|
679
|
+
/**
|
|
680
|
+
* Whether to include the starting character index of each chunk
|
|
681
|
+
* relative to the whole document. Useful for UI text highlighting.
|
|
682
|
+
* Default is false.
|
|
683
|
+
*/
|
|
684
|
+
addStartIndex?: boolean;
|
|
685
|
+
/**
|
|
686
|
+
* Optional custom regex (as string or RegExp object) to identify sentence boundaries.
|
|
687
|
+
* Use this for languages or specific document types that require custom splitting logic.
|
|
688
|
+
* If provided, it overrides or augments the default segmenter.
|
|
689
|
+
* @example /[。?!]/
|
|
690
|
+
*/
|
|
691
|
+
sentenceBoundaryRegex?: string | RegExp;
|
|
692
|
+
/**
|
|
693
|
+
* Optional list of abbreviations to ignore when splitting text into sentences.
|
|
694
|
+
* These words, if followed by a period, will not be treated as sentence boundaries.
|
|
695
|
+
* Use this to handle language-specific or domain-specific abbreviations.
|
|
696
|
+
* @example ["Inc", "Ltd", "approx"]
|
|
697
|
+
*/
|
|
698
|
+
abbreviations?: string[];
|
|
699
|
+
}
|
|
700
|
+
/**
|
|
701
|
+
* Configuration for Fixed-Size Chunking.
|
|
702
|
+
* Cuts text based on a maximum size limit with an optional overlap.
|
|
703
|
+
* This is equivalent to LangChain's `RecursiveCharacterTextSplitter`.
|
|
704
|
+
*/
|
|
705
|
+
export interface FixedSizeChunkingConfig extends BaseChunkingConfig {
|
|
706
|
+
strategy: 'fixed-size';
|
|
707
|
+
/**
|
|
708
|
+
* Maximum size of the chunk, measured by `lengthFunction`.
|
|
709
|
+
* Default is 1000 characters.
|
|
710
|
+
*/
|
|
711
|
+
chunkSize?: number;
|
|
712
|
+
/**
|
|
713
|
+
* Number of characters/tokens to overlap between consecutive chunks
|
|
714
|
+
* to avoid losing context at boundaries.
|
|
715
|
+
* Rule of thumb: ~10–20% of `chunkSize`.
|
|
716
|
+
* Default is 200.
|
|
717
|
+
*/
|
|
718
|
+
chunkOverlap?: number;
|
|
719
|
+
/**
|
|
720
|
+
* Ordered list of separators to try when splitting.
|
|
721
|
+
* The chunker tries each in order; if a split would exceed `chunkSize`,
|
|
722
|
+
* it tries the next separator.
|
|
723
|
+
* Default is ['\n\n', '\n', ' ', ''].
|
|
724
|
+
*/
|
|
725
|
+
separators?: string[];
|
|
726
|
+
}
|
|
727
|
+
/**
|
|
728
|
+
* Configuration for Document-Structure Chunking.
|
|
729
|
+
* Uses the officeParser AST to split at natural document boundaries like
|
|
730
|
+
* headings, paragraphs, slides, or pages. This is the recommended strategy
|
|
731
|
+
* as it preserves semantic context from the document's own structure.
|
|
732
|
+
*/
|
|
733
|
+
export interface DocumentStructureChunkingConfig extends BaseChunkingConfig {
|
|
734
|
+
strategy: 'document-structure';
|
|
735
|
+
/**
|
|
736
|
+
* The primary structural element at which to force a chunk boundary.
|
|
737
|
+
* - 'paragraph': Never cross a paragraph boundary (finest-grained, most precise).
|
|
738
|
+
* - 'heading': Split at every heading change.
|
|
739
|
+
* - 'page': Chunks never span multiple pages (PDF only).
|
|
740
|
+
* - 'slide': Chunks never span multiple slides (PPTX/ODP only).
|
|
741
|
+
* - 'sheet': Chunks never span multiple sheets (XLSX/ODS only).
|
|
742
|
+
* Default is 'paragraph'.
|
|
743
|
+
*/
|
|
744
|
+
splitBy?: 'page' | 'slide' | 'sheet' | 'heading' | 'paragraph';
|
|
745
|
+
/**
|
|
746
|
+
* Maximum size of a chunk (measured by `lengthFunction`).
|
|
747
|
+
* If a single structural unit (e.g., one paragraph) exceeds this limit,
|
|
748
|
+
* it will be further split using a recursive character splitter.
|
|
749
|
+
* Default is 1000 characters.
|
|
750
|
+
*/
|
|
751
|
+
maxChunkSize?: number;
|
|
752
|
+
/**
|
|
753
|
+
* How to handle table nodes when splitting.
|
|
754
|
+
* - 'row': Split by rows, REPEATING the header row in every chunk so the LLM
|
|
755
|
+
* always understands what the columns mean. (Highly recommended for RAG)
|
|
756
|
+
* - 'flatten': Convert the table to plain text and split like a regular block.
|
|
757
|
+
* Default is 'row'.
|
|
758
|
+
*/
|
|
759
|
+
tableSplitStrategy?: 'row' | 'flatten';
|
|
760
|
+
}
|
|
761
|
+
/**
|
|
762
|
+
* Configuration for Semantic Chunking.
|
|
763
|
+
* Uses an embedding model to detect topic shifts and create boundaries
|
|
764
|
+
* where content meaning naturally changes. Computationally expensive but
|
|
765
|
+
* produces the highest quality chunks.
|
|
766
|
+
*/
|
|
767
|
+
export interface SemanticChunkingConfig extends BaseChunkingConfig {
|
|
768
|
+
strategy: 'semantic';
|
|
769
|
+
/**
|
|
770
|
+
* A user-provided async function to generate vector embeddings for a text string.
|
|
771
|
+
* Required. Example: a wrapper around OpenAI's `text-embedding-3-small`.
|
|
772
|
+
* @example async (text) => await openai.embeddings.create({ input: text, model: 'text-embedding-3-small' }).then(r => r.data[0].embedding)
|
|
773
|
+
*/
|
|
774
|
+
embeddingFunction: (text: string) => Promise<number[]>;
|
|
775
|
+
/**
|
|
776
|
+
* The cosine similarity threshold below which a chunk boundary is created.
|
|
777
|
+
* When the similarity between two adjacent sentences drops below this value,
|
|
778
|
+
* a new chunk starts. Higher = more splits, smaller chunks.
|
|
779
|
+
* Default is 0.8.
|
|
780
|
+
*/
|
|
781
|
+
similarityThreshold?: number;
|
|
782
|
+
/**
|
|
783
|
+
* Maximum size of a chunk even if semantic similarity remains high.
|
|
784
|
+
* Prevents runaway chunks when an entire document is on one topic.
|
|
785
|
+
* Default is 2000 characters.
|
|
786
|
+
*/
|
|
787
|
+
maxChunkSize?: number;
|
|
788
|
+
/**
|
|
789
|
+
* Number of surrounding sentences to include when computing similarity
|
|
790
|
+
* for a sentence. A larger window reduces noise from single odd sentences.
|
|
791
|
+
* Default is 1.
|
|
792
|
+
*/
|
|
793
|
+
bufferSize?: number;
|
|
794
|
+
/**
|
|
795
|
+
* Number of sentences to process in a single batch when calling the embedding function.
|
|
796
|
+
* Higher values are faster but may trigger API rate limits.
|
|
797
|
+
* Default is 50.
|
|
798
|
+
*/
|
|
799
|
+
embeddingBatchSize?: number;
|
|
800
|
+
}
|
|
801
|
+
/**
|
|
802
|
+
* Discriminated union of all chunking strategy configurations.
|
|
803
|
+
*/
|
|
804
|
+
export type ChunkingConfig = FixedSizeChunkingConfig | DocumentStructureChunkingConfig | SemanticChunkingConfig;
|
|
805
|
+
/**
|
|
806
|
+
* Represents a single document chunk ready for a RAG (Retrieval-Augmented Generation) pipeline.
|
|
807
|
+
*
|
|
808
|
+
* Chunks are the result of splitting a document into smaller, semantically coherent
|
|
809
|
+
* pieces that fit within the context window of an LLM. Each chunk includes the
|
|
810
|
+
* extracted text and rich AST-derived metadata for citations and filtered retrieval.
|
|
811
|
+
*/
|
|
812
|
+
export interface OfficeChunk {
|
|
813
|
+
/** The text content of this chunk. This is what gets embedded. */
|
|
814
|
+
text: string;
|
|
815
|
+
/**
|
|
816
|
+
* Rich contextual metadata extracted from the AST.
|
|
817
|
+
* Use this to populate vector DB metadata fields for filtered retrieval
|
|
818
|
+
* and for LLM citations.
|
|
819
|
+
*/
|
|
820
|
+
metadata: {
|
|
821
|
+
/** The source file format (e.g., 'docx', 'pptx', 'pdf'). */
|
|
822
|
+
sourceType: SupportedFileType;
|
|
823
|
+
/** Page number (1-based), if available (PDF). */
|
|
824
|
+
pageNumber?: number;
|
|
825
|
+
/** Slide number (1-based), if available (PPTX/ODP). */
|
|
826
|
+
slideNumber?: number;
|
|
827
|
+
/** Sheet name, if available (XLSX/ODS). */
|
|
828
|
+
sheetName?: string;
|
|
829
|
+
/** The text of the nearest heading above this chunk in the document. */
|
|
830
|
+
closestHeading?: string;
|
|
831
|
+
/** True if this chunk is part of a table split. */
|
|
832
|
+
isTableChunk?: boolean;
|
|
833
|
+
/** Extensible for user-defined metadata. */
|
|
834
|
+
[key: string]: any;
|
|
835
|
+
};
|
|
836
|
+
/** The start character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
|
|
837
|
+
startIndex?: number;
|
|
838
|
+
/** The end character index of this chunk in the full document text. Only set when `addStartIndex` is true. */
|
|
839
|
+
endIndex?: number;
|
|
126
840
|
}
|
|
127
841
|
/**
|
|
128
842
|
* Supported file types for parsing.
|
|
129
843
|
*/
|
|
130
|
-
export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf';
|
|
844
|
+
export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods' | 'pdf' | 'rtf' | 'md' | 'html' | 'csv';
|
|
131
845
|
/**
|
|
132
846
|
* Types of content nodes in the AST.
|
|
133
847
|
*/
|
|
134
|
-
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break';
|
|
848
|
+
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment';
|
|
135
849
|
/**
|
|
136
850
|
* Supported MIME types for attachments.
|
|
137
851
|
*/
|
|
138
|
-
export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation';
|
|
852
|
+
export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet' | 'application/vnd.openxmlformats-officedocument.presentationml.presentation' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation' | 'application/rtf' | 'text/csv' | 'text/markdown' | 'text/html';
|
|
139
853
|
/**
|
|
140
854
|
* Text formatting options available for text content.
|
|
141
855
|
* Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
|
|
@@ -224,6 +938,8 @@ export interface SlideMetadata {
|
|
|
224
938
|
noteId?: string;
|
|
225
939
|
/** The style of the slide. */
|
|
226
940
|
style?: string;
|
|
941
|
+
/** Unique anchor IDs for internal linking. */
|
|
942
|
+
anchorIds?: string[];
|
|
227
943
|
}
|
|
228
944
|
/**
|
|
229
945
|
* Metadata for a sheet in Excel.
|
|
@@ -233,6 +949,8 @@ export interface SheetMetadata {
|
|
|
233
949
|
sheetName: string;
|
|
234
950
|
/** The style of the sheet. */
|
|
235
951
|
style?: string;
|
|
952
|
+
/** Unique anchor IDs for internal linking. */
|
|
953
|
+
anchorIds?: string[];
|
|
236
954
|
}
|
|
237
955
|
/**
|
|
238
956
|
* Detailed indentation information for paragraphs and headings.
|
|
@@ -260,6 +978,8 @@ export interface HeadingMetadata {
|
|
|
260
978
|
style?: string;
|
|
261
979
|
/** Detailed indentation information. */
|
|
262
980
|
paragraphIndentation?: IndentationMetadata;
|
|
981
|
+
/** Unique anchor IDs for internal linking. */
|
|
982
|
+
anchorIds?: string[];
|
|
263
983
|
}
|
|
264
984
|
/**
|
|
265
985
|
* Metadata for a paragraph.
|
|
@@ -271,6 +991,8 @@ export interface ParagraphMetadata {
|
|
|
271
991
|
style?: string;
|
|
272
992
|
/** Detailed indentation information. */
|
|
273
993
|
paragraphIndentation?: IndentationMetadata;
|
|
994
|
+
/** Unique anchor IDs for internal linking. */
|
|
995
|
+
anchorIds?: string[];
|
|
274
996
|
}
|
|
275
997
|
/**
|
|
276
998
|
* Metadata for a list item.
|
|
@@ -310,6 +1032,8 @@ export interface ListMetadata {
|
|
|
310
1032
|
* @example "ListParagraph"
|
|
311
1033
|
*/
|
|
312
1034
|
style?: string;
|
|
1035
|
+
/** Unique anchor IDs for internal linking. */
|
|
1036
|
+
anchorIds?: string[];
|
|
313
1037
|
}
|
|
314
1038
|
/**
|
|
315
1039
|
* Metadata for a table cell (primarily used in Excel/spreadsheet parsing).
|
|
@@ -338,6 +1062,8 @@ export interface CellMetadata {
|
|
|
338
1062
|
colSpan?: number;
|
|
339
1063
|
/** The style of the cell. */
|
|
340
1064
|
style?: string;
|
|
1065
|
+
/** Unique anchor IDs for internal linking. */
|
|
1066
|
+
anchorIds?: string[];
|
|
341
1067
|
}
|
|
342
1068
|
/**
|
|
343
1069
|
* Metadata for a chart node in the document.
|
|
@@ -350,6 +1076,8 @@ export interface ChartMetadata {
|
|
|
350
1076
|
* @example "chart1.xml"
|
|
351
1077
|
*/
|
|
352
1078
|
attachmentName: string;
|
|
1079
|
+
/** Unique anchor IDs for internal linking. */
|
|
1080
|
+
anchorIds?: string[];
|
|
353
1081
|
}
|
|
354
1082
|
/**
|
|
355
1083
|
* Metadata for an image node in the document.
|
|
@@ -368,6 +1096,14 @@ export interface ImageMetadata {
|
|
|
368
1096
|
* @example "Company logo"
|
|
369
1097
|
*/
|
|
370
1098
|
altText?: string;
|
|
1099
|
+
/**
|
|
1100
|
+
* URL of the image if it is an external link.
|
|
1101
|
+
* Typical for HTML or Markdown images that point to remote servers.
|
|
1102
|
+
* @example "https://example.com/image.png"
|
|
1103
|
+
*/
|
|
1104
|
+
url?: string;
|
|
1105
|
+
/** Unique anchor IDs for internal linking. */
|
|
1106
|
+
anchorIds?: string[];
|
|
371
1107
|
}
|
|
372
1108
|
/**
|
|
373
1109
|
* Metadata for PDF page nodes.
|
|
@@ -413,6 +1149,8 @@ export interface NoteMetadata {
|
|
|
413
1149
|
* @example "1", "2"
|
|
414
1150
|
*/
|
|
415
1151
|
noteId?: string;
|
|
1152
|
+
/** Unique anchor IDs for internal linking. */
|
|
1153
|
+
anchorIds?: string[];
|
|
416
1154
|
}
|
|
417
1155
|
/**
|
|
418
1156
|
* Metadata for break nodes.
|
|
@@ -439,10 +1177,19 @@ export interface BreakMetadata {
|
|
|
439
1177
|
*/
|
|
440
1178
|
clear?: 'all' | 'left' | 'none' | 'right';
|
|
441
1179
|
}
|
|
1180
|
+
/**
|
|
1181
|
+
* Metadata for a code block.
|
|
1182
|
+
*/
|
|
1183
|
+
export interface CodeMetadata {
|
|
1184
|
+
/** The programming language of the code block (e.g., 'typescript', 'python') */
|
|
1185
|
+
language?: string;
|
|
1186
|
+
/** Unique anchor IDs for internal linking. */
|
|
1187
|
+
anchorIds?: string[];
|
|
1188
|
+
}
|
|
442
1189
|
/**
|
|
443
1190
|
* Union type for content metadata.
|
|
444
1191
|
*/
|
|
445
|
-
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | undefined;
|
|
1192
|
+
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
|
|
446
1193
|
/**
|
|
447
1194
|
* Represents a node in the document content tree.
|
|
448
1195
|
* This is the core building block of the parsed document structure.
|
|
@@ -680,9 +1427,19 @@ export interface OfficeMetadata {
|
|
|
680
1427
|
* console.log(ast.metadata.author); // 'John Doe'
|
|
681
1428
|
* console.log(ast.content.length); // Number of top-level content nodes
|
|
682
1429
|
* console.log(ast.toText()); // Plain text representation
|
|
1430
|
+
* console.log((await ast.to('md')).value); // Markdown representation
|
|
1431
|
+
* console.log((await ast.to('html')).value); // HTML representation
|
|
1432
|
+
* console.log((await ast.to('rtf')).value); // RTF representation
|
|
1433
|
+
* console.log((await ast.to('csv')).value); // CSV representation
|
|
1434
|
+
* console.log((await ast.to('chunks')).value); // Chunks representation
|
|
683
1435
|
* ```
|
|
684
1436
|
*/
|
|
685
1437
|
export interface OfficeParserAST {
|
|
1438
|
+
/**
|
|
1439
|
+
* The original configuration used to parse this document.
|
|
1440
|
+
* This includes options like OCR settings, delimiter choices, and filtering flags.
|
|
1441
|
+
*/
|
|
1442
|
+
config: OfficeParserConfig;
|
|
686
1443
|
/**
|
|
687
1444
|
* The type of the parsed file.
|
|
688
1445
|
* Indicates which parser was used and what format the input was in.
|
|
@@ -717,7 +1474,12 @@ export interface OfficeParserAST {
|
|
|
717
1474
|
* @example [{ type: 'image', mimeType: 'image/png', data: 'base64...', name: 'image1.png' }]
|
|
718
1475
|
*/
|
|
719
1476
|
attachments: OfficeAttachment[];
|
|
1477
|
+
/** Any warnings or non-fatal issues encountered during parsing. */
|
|
1478
|
+
warnings: OfficeIssue[];
|
|
720
1479
|
/**
|
|
1480
|
+
* @deprecated Use `.to('text')` instead.
|
|
1481
|
+
* Note: This method is synchronous, while the new `.to()` method is asynchronous.
|
|
1482
|
+
*
|
|
721
1483
|
* Converts the entire AST to plain text.
|
|
722
1484
|
* This method flattens the document structure and returns just the text content,
|
|
723
1485
|
* stripping out all formatting, metadata, and structure.
|
|
@@ -732,4 +1494,18 @@ export interface OfficeParserAST {
|
|
|
732
1494
|
* ```
|
|
733
1495
|
*/
|
|
734
1496
|
toText(): string;
|
|
1497
|
+
/**
|
|
1498
|
+
* Converts this AST to the specified destination format.
|
|
1499
|
+
* This is the recommended way to convert the AST to different formats.
|
|
1500
|
+
*
|
|
1501
|
+
* @param destination The target format (e.g., 'text', 'md', 'html', 'pdf').
|
|
1502
|
+
* @param config Optional configuration for the generator.
|
|
1503
|
+
* @returns A promise resolving to the generated content (string or Buffer).
|
|
1504
|
+
* @example
|
|
1505
|
+
* ```typescript
|
|
1506
|
+
* const html = await ast.to('html', { includeFormatting: false });
|
|
1507
|
+
* const md = await ast.to('md');
|
|
1508
|
+
* ```
|
|
1509
|
+
*/
|
|
1510
|
+
to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult>;
|
|
735
1511
|
}
|