officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
|
@@ -30,7 +30,9 @@ export declare enum OfficeErrorType {
|
|
|
30
30
|
/** Output mapping in style mapping is invalid */
|
|
31
31
|
INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
|
|
32
32
|
/** Semantic chunking strategy is selected but no embedding function is provided */
|
|
33
|
-
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
|
|
33
|
+
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
|
|
34
|
+
/** The operation was aborted */
|
|
35
|
+
OPERATION_ABORTED = "OPERATION_ABORTED"
|
|
34
36
|
}
|
|
35
37
|
/**
|
|
36
38
|
* Standard warning types for OfficeParser.
|
|
@@ -68,7 +70,71 @@ export declare enum OfficeWarningType {
|
|
|
68
70
|
/** No chunks were generated for the document given the current strategy */
|
|
69
71
|
EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
|
|
70
72
|
/** A node was skipped because it only contained whitespace */
|
|
71
|
-
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
|
|
73
|
+
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
|
|
74
|
+
/** The HTML generator containerWidth option is invalid */
|
|
75
|
+
INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Consolidated timeout settings for OCR operations.
|
|
79
|
+
* Preferred over the individual flat timeout properties on {@link OcrConfig},
|
|
80
|
+
* which are now deprecated.
|
|
81
|
+
*
|
|
82
|
+
* If a key is present here, it takes priority over the corresponding deprecated
|
|
83
|
+
* flat property (e.g. `timeout.autoTerminate` wins over `autoTerminateTimeout`).
|
|
84
|
+
* Set any value to `0` to disable that specific timeout.
|
|
85
|
+
*/
|
|
86
|
+
export interface OcrTimeoutConfig {
|
|
87
|
+
/**
|
|
88
|
+
* Timeout in milliseconds of inactivity before the OCR worker pool is
|
|
89
|
+
* automatically terminated and freed.
|
|
90
|
+
*
|
|
91
|
+
* The timer resets every time a new OCR job is enqueued. When the last
|
|
92
|
+
* job completes and this duration passes without a new one, the entire
|
|
93
|
+
* worker pool is torn down so that no background threads keep the Node.js
|
|
94
|
+
* process alive unnecessarily.
|
|
95
|
+
*
|
|
96
|
+
* Set to `0` to keep workers alive indefinitely (useful when you want to
|
|
97
|
+
* call {@link terminateOcr} manually at shutdown time).
|
|
98
|
+
* Default is 10,000 ms (10 seconds).
|
|
99
|
+
*/
|
|
100
|
+
autoTerminate?: number;
|
|
101
|
+
/**
|
|
102
|
+
* Timeout in milliseconds for initializing a Tesseract worker
|
|
103
|
+
* (loading the JS runtime, downloading or loading the `.traineddata`
|
|
104
|
+
* language file) or for re-initializing an existing worker with a
|
|
105
|
+
* different language.
|
|
106
|
+
*
|
|
107
|
+
* Multi-language combinations (e.g. `'por+eng+spa'`) must download a
|
|
108
|
+
* separate `.traineddata` file for each language and are therefore
|
|
109
|
+
* particularly susceptible to slow networks. Tune this value upward if
|
|
110
|
+
* your OCR environment has high network latency or if you are loading
|
|
111
|
+
* languages from disk in a large container image.
|
|
112
|
+
*
|
|
113
|
+
* When the timeout fires, the failed job is rejected with a non-fatal
|
|
114
|
+
* {@link OfficeWarningType.OCR_FAILED} warning and parsing continues
|
|
115
|
+
* without OCR output for that image. The stalled worker is terminated
|
|
116
|
+
* and removed from the pool to prevent thread leaks.
|
|
117
|
+
*
|
|
118
|
+
* Set to `0` to wait indefinitely (not recommended for production; a hung
|
|
119
|
+
* network request will block the entire OCR queue for that language).
|
|
120
|
+
* Default is 60,000 ms (60 seconds).
|
|
121
|
+
*/
|
|
122
|
+
workerLoad?: number;
|
|
123
|
+
/**
|
|
124
|
+
* Timeout in milliseconds for the actual OCR text-recognition call
|
|
125
|
+
* (`worker.recognize(image)`) on an already-initialized Tesseract worker.
|
|
126
|
+
*
|
|
127
|
+
* Recognition time scales with image resolution and the number of active
|
|
128
|
+
* languages. Very high-resolution scans or unusual character sets can
|
|
129
|
+
* exceed the default. If this timeout fires, the job is rejected with a
|
|
130
|
+
* non-fatal {@link OfficeWarningType.OCR_FAILED} warning; the worker is
|
|
131
|
+
* terminated and evicted from the pool because its internal state after a
|
|
132
|
+
* mid-recognition timeout is undefined.
|
|
133
|
+
*
|
|
134
|
+
* Set to `0` to wait indefinitely.
|
|
135
|
+
* Default is 30,000 ms (30 seconds).
|
|
136
|
+
*/
|
|
137
|
+
recognition?: number;
|
|
72
138
|
}
|
|
73
139
|
/**
|
|
74
140
|
* Configuration options for OCR.
|
|
@@ -104,11 +170,33 @@ export interface OcrConfig {
|
|
|
104
170
|
*/
|
|
105
171
|
langPath?: string;
|
|
106
172
|
/**
|
|
173
|
+
* Consolidated timeout settings for all OCR operations.
|
|
174
|
+
*
|
|
175
|
+
* Prefer this over the deprecated flat timeout properties.
|
|
176
|
+
* If `timeout.autoTerminate` is set, it takes priority over the deprecated `autoTerminateTimeout`.
|
|
177
|
+
*/
|
|
178
|
+
timeout?: OcrTimeoutConfig;
|
|
179
|
+
/**
|
|
180
|
+
* @deprecated Use `timeout.autoTerminate` instead.
|
|
181
|
+
*
|
|
107
182
|
* Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
|
|
108
183
|
* Set to 0 to disable auto-termination.
|
|
109
184
|
* Default is 10,000 (10 seconds).
|
|
185
|
+
*
|
|
186
|
+
* If `timeout.autoTerminate` is also set, that value takes priority over this one.
|
|
110
187
|
*/
|
|
111
188
|
autoTerminateTimeout?: number;
|
|
189
|
+
/**
|
|
190
|
+
* An optional AbortSignal propagated from the main parser configuration to abort active OCR jobs.
|
|
191
|
+
* If the signal is aborted:
|
|
192
|
+
* 1. Any pending OCR jobs in the scheduler queue are rejected immediately.
|
|
193
|
+
* 2. Any active OCR job running on a Tesseract worker will reject, the worker will be
|
|
194
|
+
* terminated, and it will be removed from the pool to avoid hanging worker threads.
|
|
195
|
+
*
|
|
196
|
+
* Developers should prefer passing this at the top level of `parseOffice` (as `config.abortSignal`),
|
|
197
|
+
* which automatically propagates here.
|
|
198
|
+
*/
|
|
199
|
+
abortSignal?: AbortSignal | null;
|
|
112
200
|
}
|
|
113
201
|
/**
|
|
114
202
|
* Configuration options for the OfficeParser.
|
|
@@ -137,10 +225,23 @@ export interface OfficeParserConfig {
|
|
|
137
225
|
*/
|
|
138
226
|
ignoreNotes?: boolean;
|
|
139
227
|
/**
|
|
140
|
-
* Flag
|
|
141
|
-
* Default is false.
|
|
142
|
-
|
|
143
|
-
|
|
228
|
+
* Flag to ignore comments from parsing.
|
|
229
|
+
* Default is false.
|
|
230
|
+
*/
|
|
231
|
+
ignoreComments?: boolean;
|
|
232
|
+
/**
|
|
233
|
+
* Flag to ignore headers and footers from parsing.
|
|
234
|
+
* Default is false.
|
|
235
|
+
*/
|
|
236
|
+
ignoreHeadersAndFooters?: boolean;
|
|
237
|
+
/**
|
|
238
|
+
* Flag to ignore slide masters from parsing in PowerPoint.
|
|
239
|
+
* Default is false.
|
|
240
|
+
*/
|
|
241
|
+
ignoreSlideMasters?: boolean;
|
|
242
|
+
/**
|
|
243
|
+
* @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`.
|
|
244
|
+
* This option is now completely ignored by all parsers.
|
|
144
245
|
*/
|
|
145
246
|
putNotesAtLast?: boolean;
|
|
146
247
|
/**
|
|
@@ -175,6 +276,20 @@ export interface OfficeParserConfig {
|
|
|
175
276
|
* If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
|
|
176
277
|
*/
|
|
177
278
|
ocrConfig?: OcrConfig;
|
|
279
|
+
/**
|
|
280
|
+
* An optional AbortSignal to cancel the parsing operation.
|
|
281
|
+
* When aborted, the parser immediately rejects with a standard AbortError (DOMException).
|
|
282
|
+
*
|
|
283
|
+
* ### Format-Specific Abort Behavior:
|
|
284
|
+
* - **PDF**: Checked between page loads and before individual image OCR operations.
|
|
285
|
+
* - **RTF**: Checked before parsing/traversal and before running OCR on image attachments.
|
|
286
|
+
* - **DOCX/XLSX/PPTX/ODF**: Checked during zip decompression before loading and parsing XML files.
|
|
287
|
+
* - **CSV/MD/HTML**: Checked at the start of the parsing phase.
|
|
288
|
+
*
|
|
289
|
+
* Note: If an OCR operation is currently running on a Tesseract worker when aborted,
|
|
290
|
+
* the worker will be terminated and removed from the worker pool automatically to prevent leaks.
|
|
291
|
+
*/
|
|
292
|
+
abortSignal?: AbortSignal | null;
|
|
178
293
|
/**
|
|
179
294
|
* Flag to serialize raw content (XML) as clean, formatted strings.
|
|
180
295
|
* Only relevant when `includeRawContent` is true.
|
|
@@ -251,9 +366,10 @@ export interface OfficeIssue {
|
|
|
251
366
|
/**
|
|
252
367
|
* The result of a document conversion operation.
|
|
253
368
|
*/
|
|
254
|
-
export
|
|
369
|
+
export type ConversionValue<D extends UniversalGeneratorFormat> = D extends "pdf" ? Uint8Array | string : D extends "chunks" ? OfficeChunk[] : D extends "csv" ? string | Uint8Array : string;
|
|
370
|
+
export interface ConversionResult<D extends UniversalGeneratorFormat> {
|
|
255
371
|
/** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
|
|
256
|
-
value: D
|
|
372
|
+
value: ConversionValue<D>;
|
|
257
373
|
/** A collection of issues (warnings/infos) generated during the process. */
|
|
258
374
|
messages: OfficeIssue[];
|
|
259
375
|
}
|
|
@@ -363,23 +479,43 @@ export interface CommonGeneratorConfig {
|
|
|
363
479
|
* Defaults to false.
|
|
364
480
|
*/
|
|
365
481
|
ignoreInternalLinks?: boolean;
|
|
482
|
+
/**
|
|
483
|
+
* An optional AbortSignal to cancel the generation operation.
|
|
484
|
+
* When aborted, the generator immediately rejects with a standard AbortError.
|
|
485
|
+
* Currently supported by PdfGenerator and ChunkingGenerator.
|
|
486
|
+
*/
|
|
487
|
+
abortSignal?: AbortSignal | null;
|
|
366
488
|
}
|
|
367
489
|
/**
|
|
368
490
|
* Destination-aware generator configuration.
|
|
369
491
|
* Restricts format-specific configurations to their respective destinations.
|
|
370
492
|
*/
|
|
371
493
|
/**
|
|
372
|
-
*
|
|
494
|
+
* Maps a destination format string to its corresponding specific configuration object type.
|
|
373
495
|
*/
|
|
374
|
-
export
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
}
|
|
496
|
+
export type GeneratorSpecificConfig<D extends string> = D extends "html" ? {
|
|
497
|
+
htmlConfig?: HtmlGeneratorConfig;
|
|
498
|
+
} : D extends "md" ? {
|
|
499
|
+
mdConfig?: MdGeneratorConfig;
|
|
500
|
+
} : D extends "pdf" ? {
|
|
501
|
+
pdfConfig?: PdfGeneratorConfig;
|
|
502
|
+
} : D extends "csv" ? {
|
|
503
|
+
csvConfig?: CsvGeneratorConfig;
|
|
504
|
+
} : D extends "text" ? {
|
|
505
|
+
textConfig?: TextGeneratorConfig;
|
|
506
|
+
} : D extends "rtf" ? {
|
|
507
|
+
rtfConfig?: RtfGeneratorConfig;
|
|
508
|
+
} : D extends "chunks" ? {
|
|
509
|
+
chunksConfig?: ChunkingConfig;
|
|
510
|
+
} : Partial<{
|
|
511
|
+
htmlConfig: HtmlGeneratorConfig;
|
|
512
|
+
mdConfig: MdGeneratorConfig;
|
|
513
|
+
pdfConfig: PdfGeneratorConfig;
|
|
514
|
+
csvConfig: CsvGeneratorConfig;
|
|
515
|
+
textConfig: TextGeneratorConfig;
|
|
516
|
+
rtfConfig: RtfGeneratorConfig;
|
|
517
|
+
chunksConfig: ChunkingConfig;
|
|
518
|
+
}>;
|
|
383
519
|
/**
|
|
384
520
|
* Configuration options for document generators.
|
|
385
521
|
*
|
|
@@ -390,9 +526,7 @@ export interface GeneratorSubConfigMap {
|
|
|
390
526
|
*
|
|
391
527
|
* @template D The destination format string. Defaults to `string` for a general configuration.
|
|
392
528
|
*/
|
|
393
|
-
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig &
|
|
394
|
-
[K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
|
|
395
|
-
};
|
|
529
|
+
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & GeneratorSpecificConfig<D>;
|
|
396
530
|
/**
|
|
397
531
|
* Configuration options for the OfficeConverter.
|
|
398
532
|
* Combines relevant parser and generator settings for a seamless one-step conversion.
|
|
@@ -424,6 +558,19 @@ export type OfficeConverterConfig<D extends string = string, T extends Supported
|
|
|
424
558
|
*/
|
|
425
559
|
onWarning?: (issue: OfficeIssue) => void;
|
|
426
560
|
};
|
|
561
|
+
/**
|
|
562
|
+
* Configuration options for granular raw HTML injections.
|
|
563
|
+
*/
|
|
564
|
+
export interface HtmlInjectionConfig {
|
|
565
|
+
/** Raw HTML injected immediately after the opening <head> tag */
|
|
566
|
+
headStart?: string;
|
|
567
|
+
/** Raw HTML injected immediately before the closing </head> tag */
|
|
568
|
+
headEnd?: string;
|
|
569
|
+
/** Raw HTML injected immediately after the opening <body> tag */
|
|
570
|
+
bodyStart?: string;
|
|
571
|
+
/** Raw HTML injected immediately before the closing </body> tag */
|
|
572
|
+
bodyEnd?: string;
|
|
573
|
+
}
|
|
427
574
|
/**
|
|
428
575
|
* Configuration options for HTML generation.
|
|
429
576
|
*/
|
|
@@ -438,6 +585,25 @@ export interface HtmlGeneratorConfig {
|
|
|
438
585
|
* Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
|
|
439
586
|
*/
|
|
440
587
|
chartJsSrc?: string;
|
|
588
|
+
/**
|
|
589
|
+
* Custom container width for the generated HTML.
|
|
590
|
+
* Can be a number (pixels) or string (e.g., '900px', '100%').
|
|
591
|
+
* If not specified or set to 'auto', it defaults based on the content type:
|
|
592
|
+
* - Spreadsheet: '100%'
|
|
593
|
+
* - Presentation/Slides: '1100px'
|
|
594
|
+
* - Standard Document (PDF/DOCX/RTF/etc.): '900px'
|
|
595
|
+
*/
|
|
596
|
+
containerWidth?: string | number;
|
|
597
|
+
/**
|
|
598
|
+
* Custom CSS to append to the generated HTML document.
|
|
599
|
+
* This CSS will be included in the `<style>` block and can be used to style
|
|
600
|
+
* custom classes added during AST manipulation or override default styles.
|
|
601
|
+
*/
|
|
602
|
+
customCss?: string;
|
|
603
|
+
/**
|
|
604
|
+
* Granular injection points for custom HTML, scripts, and styles.
|
|
605
|
+
*/
|
|
606
|
+
injections?: HtmlInjectionConfig;
|
|
441
607
|
}
|
|
442
608
|
/**
|
|
443
609
|
* Configuration options for PDF generation.
|
|
@@ -474,6 +640,12 @@ export interface PdfGeneratorConfig {
|
|
|
474
640
|
* Useful for setting custom executable paths or args in CI/CD.
|
|
475
641
|
*/
|
|
476
642
|
launchOptions?: any;
|
|
643
|
+
/**
|
|
644
|
+
* Timeout in milliseconds for PDF generation.
|
|
645
|
+
* Limits the time spent waiting for Puppeteer to launch, load content, and render PDF.
|
|
646
|
+
* Defaults to 30000 ms (30 seconds). Set to 0 to disable.
|
|
647
|
+
*/
|
|
648
|
+
timeout?: number;
|
|
477
649
|
}
|
|
478
650
|
/**
|
|
479
651
|
* Structured style mapping definition for the StyleMapper.
|
|
@@ -507,7 +679,7 @@ export interface StructuredStyleMapping {
|
|
|
507
679
|
* The structural type of the node (e.g., 'paragraph', 'heading', 'text').
|
|
508
680
|
* Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
|
|
509
681
|
*/
|
|
510
|
-
nodeType?:
|
|
682
|
+
nodeType?: OfficeContentNodeType;
|
|
511
683
|
/**
|
|
512
684
|
* A dictionary of attributes to match on the node.
|
|
513
685
|
*
|
|
@@ -573,7 +745,7 @@ export interface CsvGeneratorConfig {
|
|
|
573
745
|
/**
|
|
574
746
|
* Whether to merge all selected sheets into a single CSV.
|
|
575
747
|
* If false, returns a ZIP archive containing individual CSV files.
|
|
576
|
-
* Defaults to
|
|
748
|
+
* Defaults to true.
|
|
577
749
|
*/
|
|
578
750
|
mergeSheets?: boolean;
|
|
579
751
|
/**
|
|
@@ -779,6 +951,11 @@ export interface SemanticChunkingConfig extends BaseChunkingConfig {
|
|
|
779
951
|
* Default is 50.
|
|
780
952
|
*/
|
|
781
953
|
embeddingBatchSize?: number;
|
|
954
|
+
/**
|
|
955
|
+
* Timeout in milliseconds for individual embedding API calls.
|
|
956
|
+
* Defaults to 10000 ms (10 seconds). Set to 0 to disable.
|
|
957
|
+
*/
|
|
958
|
+
timeout?: number;
|
|
782
959
|
}
|
|
783
960
|
/**
|
|
784
961
|
* Discriminated union of all chunking strategy configurations.
|
|
@@ -827,11 +1004,16 @@ export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods"
|
|
|
827
1004
|
/**
|
|
828
1005
|
* Types of content nodes in the AST.
|
|
829
1006
|
*/
|
|
830
|
-
export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment";
|
|
1007
|
+
export type OfficeContentNodeType = "paragraph" | "heading" | "table" | "list" | "text" | "image" | "chart" | "drawing" | "slide" | "note" | "sheet" | "row" | "cell" | "page" | "break" | "code" | "comment" | "header" | "footer" | "slideMaster";
|
|
831
1008
|
/**
|
|
832
1009
|
* Supported MIME types for attachments.
|
|
833
1010
|
*/
|
|
834
1011
|
export type OfficeMimeType = "image/jpeg" | "image/png" | "image/gif" | "image/bmp" | "image/tiff" | "image/svg+xml" | "application/pdf" | "application/vnd.openxmlformats-officedocument.wordprocessingml.document" | "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" | "application/vnd.openxmlformats-officedocument.presentationml.presentation" | "application/vnd.oasis.opendocument.chart" | "application/vnd.oasis.opendocument.spreadsheet" | "application/vnd.oasis.opendocument.text" | "application/vnd.oasis.opendocument.presentation" | "application/rtf" | "text/csv" | "text/markdown" | "text/html";
|
|
1012
|
+
/**
|
|
1013
|
+
* Text alignment options.
|
|
1014
|
+
* Common in spreadsheet cells, paragraph styles, and text elements.
|
|
1015
|
+
*/
|
|
1016
|
+
export type TextAlignment = "left" | "center" | "right" | "justify";
|
|
835
1017
|
/**
|
|
836
1018
|
* Text formatting options available for text content.
|
|
837
1019
|
* Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
|
|
@@ -905,7 +1087,7 @@ export interface TextFormatting {
|
|
|
905
1087
|
* Common in spreadsheet cells or paragraph styles.
|
|
906
1088
|
* @example "center", "right"
|
|
907
1089
|
*/
|
|
908
|
-
alignment?:
|
|
1090
|
+
alignment?: TextAlignment;
|
|
909
1091
|
}
|
|
910
1092
|
/**
|
|
911
1093
|
* Metadata for a slide in PowerPoint.
|
|
@@ -955,7 +1137,7 @@ export interface HeadingMetadata {
|
|
|
955
1137
|
/** The heading level (e.g., 1 for H1). */
|
|
956
1138
|
level: number;
|
|
957
1139
|
/** The alignment of the heading. */
|
|
958
|
-
alignment?:
|
|
1140
|
+
alignment?: TextAlignment;
|
|
959
1141
|
/** The style of the heading. */
|
|
960
1142
|
style?: string;
|
|
961
1143
|
/** Detailed indentation information. */
|
|
@@ -968,7 +1150,7 @@ export interface HeadingMetadata {
|
|
|
968
1150
|
*/
|
|
969
1151
|
export interface ParagraphMetadata {
|
|
970
1152
|
/** The alignment of the paragraph. */
|
|
971
|
-
alignment?:
|
|
1153
|
+
alignment?: TextAlignment;
|
|
972
1154
|
/** The style of the paragraph. */
|
|
973
1155
|
style?: string;
|
|
974
1156
|
/** Detailed indentation information. */
|
|
@@ -996,7 +1178,7 @@ export interface ListMetadata {
|
|
|
996
1178
|
* Text alignment of the list item.
|
|
997
1179
|
* @example 'left', 'center', 'right', 'justify'
|
|
998
1180
|
*/
|
|
999
|
-
alignment:
|
|
1181
|
+
alignment: TextAlignment;
|
|
1000
1182
|
/**
|
|
1001
1183
|
* The list ID from the Word document's numbering definition.
|
|
1002
1184
|
* Used to identify which list definition this item belongs to.
|
|
@@ -1046,6 +1228,15 @@ export interface CellMetadata {
|
|
|
1046
1228
|
style?: string;
|
|
1047
1229
|
/** Unique anchor IDs for internal linking. */
|
|
1048
1230
|
anchorIds?: string[];
|
|
1231
|
+
/** Background color for this cell in hex format (e.g. #FFFFFF). */
|
|
1232
|
+
backgroundColor?: string;
|
|
1233
|
+
}
|
|
1234
|
+
/**
|
|
1235
|
+
* Metadata for a table.
|
|
1236
|
+
*/
|
|
1237
|
+
export interface TableMetadata {
|
|
1238
|
+
/** Unique anchor IDs for internal linking. */
|
|
1239
|
+
anchorIds?: string[];
|
|
1049
1240
|
}
|
|
1050
1241
|
/**
|
|
1051
1242
|
* Metadata for a chart node in the document.
|
|
@@ -1133,6 +1324,8 @@ export interface NoteMetadata {
|
|
|
1133
1324
|
noteId?: string;
|
|
1134
1325
|
/** Unique anchor IDs for internal linking. */
|
|
1135
1326
|
anchorIds?: string[];
|
|
1327
|
+
/** The slide number this note is associated with (used in PowerPoint). */
|
|
1328
|
+
slideNumber?: number;
|
|
1136
1329
|
}
|
|
1137
1330
|
/**
|
|
1138
1331
|
* Metadata for break nodes.
|
|
@@ -1168,10 +1361,25 @@ export interface CodeMetadata {
|
|
|
1168
1361
|
/** Unique anchor IDs for internal linking. */
|
|
1169
1362
|
anchorIds?: string[];
|
|
1170
1363
|
}
|
|
1364
|
+
/**
|
|
1365
|
+
* Metadata for a comment/annotation.
|
|
1366
|
+
*/
|
|
1367
|
+
export interface CommentMetadata {
|
|
1368
|
+
author?: string;
|
|
1369
|
+
initials?: string;
|
|
1370
|
+
date?: string;
|
|
1371
|
+
commentId?: string;
|
|
1372
|
+
}
|
|
1373
|
+
/**
|
|
1374
|
+
* Metadata for a header or footer.
|
|
1375
|
+
*/
|
|
1376
|
+
export interface HeaderFooterMetadata {
|
|
1377
|
+
type: "default" | "first" | "even" | string;
|
|
1378
|
+
}
|
|
1171
1379
|
/**
|
|
1172
1380
|
* Union type for content metadata.
|
|
1173
1381
|
*/
|
|
1174
|
-
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
|
|
1382
|
+
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
|
|
1175
1383
|
/**
|
|
1176
1384
|
* Represents a node in the document content tree.
|
|
1177
1385
|
* This is the core building block of the parsed document structure.
|
|
@@ -1198,13 +1406,10 @@ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata |
|
|
|
1198
1406
|
* children: [...]
|
|
1199
1407
|
* }
|
|
1200
1408
|
*/
|
|
1201
|
-
|
|
1202
|
-
|
|
1203
|
-
|
|
1204
|
-
|
|
1205
|
-
* Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
|
|
1206
|
-
*/
|
|
1207
|
-
type: OfficeContentNodeType;
|
|
1409
|
+
/**
|
|
1410
|
+
* Shared properties available on all document content nodes.
|
|
1411
|
+
*/
|
|
1412
|
+
export interface BaseContentNode {
|
|
1208
1413
|
/**
|
|
1209
1414
|
* The complete text content of the node and all its children combined.
|
|
1210
1415
|
* For container nodes (paragraph, heading), this is the concatenation of all child text.
|
|
@@ -1222,6 +1427,16 @@ export interface OfficeContentNode {
|
|
|
1222
1427
|
* @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
|
|
1223
1428
|
*/
|
|
1224
1429
|
children?: OfficeContentNode[];
|
|
1430
|
+
/**
|
|
1431
|
+
* Comments attached to this specific node.
|
|
1432
|
+
* Keeps annotations completely separate from the actual content flow.
|
|
1433
|
+
*/
|
|
1434
|
+
comments?: OfficeContentNode[];
|
|
1435
|
+
/**
|
|
1436
|
+
* Notes (like footnotes or slide notes) attached to this specific node.
|
|
1437
|
+
* Keeps notes separate from the actual structural children.
|
|
1438
|
+
*/
|
|
1439
|
+
notes?: OfficeContentNode[];
|
|
1225
1440
|
/**
|
|
1226
1441
|
* Text formatting applied to this node.
|
|
1227
1442
|
* Only applicable to text-containing nodes.
|
|
@@ -1229,16 +1444,6 @@ export interface OfficeContentNode {
|
|
|
1229
1444
|
* @example { bold: true, size: "12", font: "Arial" }
|
|
1230
1445
|
*/
|
|
1231
1446
|
formatting?: TextFormatting;
|
|
1232
|
-
/**
|
|
1233
|
-
* Type-specific metadata providing additional context about the node.
|
|
1234
|
-
* The metadata structure depends on the node type:
|
|
1235
|
-
* - Headings: { level: 1 }
|
|
1236
|
-
* - Lists: { listType: 'ordered', indentation: 0 }
|
|
1237
|
-
* - Cells: { row: 0, col: 0 }
|
|
1238
|
-
* - Slides: { slideNumber: 1 }
|
|
1239
|
-
* @example { level: 1 } for a heading
|
|
1240
|
-
*/
|
|
1241
|
-
metadata?: ContentMetadata;
|
|
1242
1447
|
/**
|
|
1243
1448
|
* The raw source content for this node.
|
|
1244
1449
|
* - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
|
|
@@ -1250,6 +1455,93 @@ export interface OfficeContentNode {
|
|
|
1250
1455
|
*/
|
|
1251
1456
|
rawContent?: string;
|
|
1252
1457
|
}
|
|
1458
|
+
/**
|
|
1459
|
+
* Represents a node in the document content tree.
|
|
1460
|
+
* This is the core building block of the parsed document structure.
|
|
1461
|
+
* Content nodes can be nested to represent hierarchical document structures
|
|
1462
|
+
* (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
|
|
1463
|
+
*
|
|
1464
|
+
* @example
|
|
1465
|
+
* // A simple paragraph with formatted text
|
|
1466
|
+
* {
|
|
1467
|
+
* type: 'paragraph',
|
|
1468
|
+
* text: 'Hello world',
|
|
1469
|
+
* children: [
|
|
1470
|
+
* { type: 'text', text: 'Hello ', formatting: { bold: true } },
|
|
1471
|
+
* { type: 'text', text: 'world', formatting: { italic: true } }
|
|
1472
|
+
* ]
|
|
1473
|
+
* }
|
|
1474
|
+
*
|
|
1475
|
+
* @example
|
|
1476
|
+
* // A heading with metadata
|
|
1477
|
+
* {
|
|
1478
|
+
* type: 'heading',
|
|
1479
|
+
* text: 'Chapter 1',
|
|
1480
|
+
* metadata: { level: 1 },
|
|
1481
|
+
* children: [...]
|
|
1482
|
+
* }
|
|
1483
|
+
*/
|
|
1484
|
+
export type OfficeContentNode = BaseContentNode & ({
|
|
1485
|
+
type: "slide";
|
|
1486
|
+
metadata?: SlideMetadata;
|
|
1487
|
+
} | {
|
|
1488
|
+
type: "sheet";
|
|
1489
|
+
metadata?: SheetMetadata;
|
|
1490
|
+
} | {
|
|
1491
|
+
type: "heading";
|
|
1492
|
+
metadata?: HeadingMetadata;
|
|
1493
|
+
} | {
|
|
1494
|
+
type: "list";
|
|
1495
|
+
metadata?: ListMetadata;
|
|
1496
|
+
} | {
|
|
1497
|
+
type: "cell";
|
|
1498
|
+
metadata?: CellMetadata;
|
|
1499
|
+
} | {
|
|
1500
|
+
type: "image";
|
|
1501
|
+
metadata?: ImageMetadata;
|
|
1502
|
+
} | {
|
|
1503
|
+
type: "chart";
|
|
1504
|
+
metadata?: ChartMetadata;
|
|
1505
|
+
} | {
|
|
1506
|
+
type: "page";
|
|
1507
|
+
metadata?: PageMetadata;
|
|
1508
|
+
} | {
|
|
1509
|
+
type: "paragraph";
|
|
1510
|
+
metadata?: ParagraphMetadata;
|
|
1511
|
+
} | {
|
|
1512
|
+
type: "text";
|
|
1513
|
+
metadata?: TextMetadata;
|
|
1514
|
+
} | {
|
|
1515
|
+
type: "note";
|
|
1516
|
+
metadata?: NoteMetadata;
|
|
1517
|
+
} | {
|
|
1518
|
+
type: "break";
|
|
1519
|
+
metadata?: BreakMetadata;
|
|
1520
|
+
} | {
|
|
1521
|
+
type: "code";
|
|
1522
|
+
metadata?: CodeMetadata;
|
|
1523
|
+
} | {
|
|
1524
|
+
type: "comment";
|
|
1525
|
+
metadata?: CommentMetadata;
|
|
1526
|
+
} | {
|
|
1527
|
+
type: "header";
|
|
1528
|
+
metadata?: HeaderFooterMetadata;
|
|
1529
|
+
} | {
|
|
1530
|
+
type: "footer";
|
|
1531
|
+
metadata?: HeaderFooterMetadata;
|
|
1532
|
+
} | {
|
|
1533
|
+
type: "table";
|
|
1534
|
+
metadata?: TableMetadata;
|
|
1535
|
+
} | {
|
|
1536
|
+
type: "row";
|
|
1537
|
+
metadata?: undefined;
|
|
1538
|
+
} | {
|
|
1539
|
+
type: "drawing";
|
|
1540
|
+
metadata?: undefined;
|
|
1541
|
+
} | {
|
|
1542
|
+
type: "slideMaster";
|
|
1543
|
+
metadata?: SlideMetadata;
|
|
1544
|
+
});
|
|
1253
1545
|
/**
|
|
1254
1546
|
* Structured information extracted from a chart.
|
|
1255
1547
|
*/
|
|
@@ -1389,14 +1681,40 @@ export interface OfficeMetadata {
|
|
|
1389
1681
|
* Values are typed as string, number, boolean, or Date where the source format provides type information.
|
|
1390
1682
|
*/
|
|
1391
1683
|
customProperties?: Record<string, string | number | boolean | Date>;
|
|
1684
|
+
/** Keywords associated with the document. */
|
|
1685
|
+
keywords?: string;
|
|
1686
|
+
/**
|
|
1687
|
+
* Contains all format-specific metadata fields extracted verbatim.
|
|
1688
|
+
* Consumers can use this to access properties not mapped to the standard OfficeMetadata fields.
|
|
1689
|
+
* Examples: all <meta> tags in HTML, app.xml properties in DOCX, XMP dicts in PDF.
|
|
1690
|
+
*/
|
|
1691
|
+
nativeProperties?: Record<string, any>;
|
|
1692
|
+
}
|
|
1693
|
+
/**
|
|
1694
|
+
* Contains out-of-band layout elements and templates that are not part of the main document flow.
|
|
1695
|
+
*/
|
|
1696
|
+
export interface OfficeAuxiliaryContent {
|
|
1697
|
+
/** Headers extracted from the document. */
|
|
1698
|
+
headers?: OfficeContentNode[];
|
|
1699
|
+
/** Footers extracted from the document. */
|
|
1700
|
+
footers?: OfficeContentNode[];
|
|
1701
|
+
/** Slide Masters extracted from presentations. */
|
|
1702
|
+
slideMasters?: OfficeContentNode[];
|
|
1392
1703
|
}
|
|
1393
1704
|
/**
|
|
1394
|
-
* The Abstract Syntax Tree (AST)
|
|
1395
|
-
* This is the
|
|
1705
|
+
* The Root Abstract Syntax Tree (AST) representing a parsed Office Document.
|
|
1706
|
+
* This is the ultimate output of `OfficeParser.parseOffice()`.
|
|
1396
1707
|
*
|
|
1397
|
-
*
|
|
1398
|
-
*
|
|
1399
|
-
*
|
|
1708
|
+
* DESIGN PHILOSOPHY:
|
|
1709
|
+
* The AST is designed to be a universal, format-agnostic representation of document content.
|
|
1710
|
+
* Whether the input was a PDF, DOCX, XLSX, Markdown, or HTML file, the resulting AST
|
|
1711
|
+
* uses the same consistent structure (`OfficeContentNode` trees).
|
|
1712
|
+
*
|
|
1713
|
+
* ### Key Top-Level Properties:
|
|
1714
|
+
* - `metadata`: Document-level properties (author, title, stats).
|
|
1715
|
+
* - `content`: The main sequential flow of the document (paragraphs, tables, slides, sheets).
|
|
1716
|
+
* - `attachments`: Extracted binary assets (images, embedded files).
|
|
1717
|
+
* - `auxiliary`: Out-of-band layout/template elements (headers, footers, slide masters).
|
|
1400
1718
|
*
|
|
1401
1719
|
* @example
|
|
1402
1720
|
* ```typescript
|
|
@@ -1446,6 +1764,12 @@ export interface OfficeParserAST {
|
|
|
1446
1764
|
* @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
|
|
1447
1765
|
*/
|
|
1448
1766
|
content: OfficeContentNode[];
|
|
1767
|
+
/**
|
|
1768
|
+
* Out-of-band layout and template elements that are not part of the main text flow.
|
|
1769
|
+
* Extracted only if the respective `ignore...` config flags are false.
|
|
1770
|
+
* Contains elements like `headers`, `footers`, and `slideMasters`.
|
|
1771
|
+
*/
|
|
1772
|
+
auxiliary?: OfficeAuxiliaryContent;
|
|
1449
1773
|
/**
|
|
1450
1774
|
* Attachments extracted from the document (images, charts, embedded files).
|
|
1451
1775
|
* Only populated when `config.extractAttachments` is true.
|
|
@@ -1489,7 +1813,7 @@ export interface OfficeParserAST {
|
|
|
1489
1813
|
* const md = await ast.to('md');
|
|
1490
1814
|
* ```
|
|
1491
1815
|
*/
|
|
1492
|
-
to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
1816
|
+
to<T extends this, D extends SupportedDestination<T["type"]>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
1493
1817
|
}
|
|
1494
1818
|
/**
|
|
1495
1819
|
* Main parser class providing office document parsing functionality.
|
|
@@ -1573,7 +1897,7 @@ export declare class OfficeGenerator {
|
|
|
1573
1897
|
*/
|
|
1574
1898
|
static generate<T extends SupportedFileType, D extends SupportedDestination<T>>(ast: OfficeParserAST & {
|
|
1575
1899
|
type: T;
|
|
1576
|
-
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
1900
|
+
}, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
1577
1901
|
}
|
|
1578
1902
|
/**
|
|
1579
1903
|
* Utility type to infer the file type from a file path string literal.
|