officeparser 7.0.3 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +152 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/OfficeParser.js +6 -0
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +27 -1
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +31 -4
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +51 -10
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +377 -53
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +6 -1
- package/dist/parsers/ExcelParser.js +69 -21
- package/dist/parsers/HtmlParser.js +15 -1
- package/dist/parsers/MarkdownParser.js +18 -10
- package/dist/parsers/OpenOfficeParser.js +61 -34
- package/dist/parsers/PdfParser.js +26 -1
- package/dist/parsers/PowerPointParser.js +168 -40
- package/dist/parsers/RtfParser.js +30 -24
- package/dist/parsers/WordParser.js +158 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +383 -53
- package/dist/types.js +4 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +69 -2
- package/dist/utils/errorUtils.d.ts +20 -0
- package/dist/utils/errorUtils.js +39 -3
- package/dist/utils/moduleLoader.js +3 -3
- package/dist/utils/ocrUtils.js +271 -66
- package/dist/utils/xmlUtils.d.ts +17 -0
- package/dist/utils/xmlUtils.js +85 -1
- package/package.json +3 -2
package/dist/types.d.ts
CHANGED
|
@@ -28,7 +28,9 @@ export declare enum OfficeErrorType {
|
|
|
28
28
|
/** Output mapping in style mapping is invalid */
|
|
29
29
|
INVALID_OUTPUT_MAPPING = "INVALID_OUTPUT_MAPPING",
|
|
30
30
|
/** Semantic chunking strategy is selected but no embedding function is provided */
|
|
31
|
-
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION"
|
|
31
|
+
MISSING_EMBEDDING_FUNCTION = "MISSING_EMBEDDING_FUNCTION",
|
|
32
|
+
/** The operation was aborted */
|
|
33
|
+
OPERATION_ABORTED = "OPERATION_ABORTED"
|
|
32
34
|
}
|
|
33
35
|
/**
|
|
34
36
|
* Standard warning types for OfficeParser.
|
|
@@ -66,7 +68,71 @@ export declare enum OfficeWarningType {
|
|
|
66
68
|
/** No chunks were generated for the document given the current strategy */
|
|
67
69
|
EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
|
|
68
70
|
/** A node was skipped because it only contained whitespace */
|
|
69
|
-
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
|
|
71
|
+
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
|
|
72
|
+
/** The HTML generator containerWidth option is invalid */
|
|
73
|
+
INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Consolidated timeout settings for OCR operations.
|
|
77
|
+
* Preferred over the individual flat timeout properties on {@link OcrConfig},
|
|
78
|
+
* which are now deprecated.
|
|
79
|
+
*
|
|
80
|
+
* If a key is present here, it takes priority over the corresponding deprecated
|
|
81
|
+
* flat property (e.g. `timeout.autoTerminate` wins over `autoTerminateTimeout`).
|
|
82
|
+
* Set any value to `0` to disable that specific timeout.
|
|
83
|
+
*/
|
|
84
|
+
export interface OcrTimeoutConfig {
|
|
85
|
+
/**
|
|
86
|
+
* Timeout in milliseconds of inactivity before the OCR worker pool is
|
|
87
|
+
* automatically terminated and freed.
|
|
88
|
+
*
|
|
89
|
+
* The timer resets every time a new OCR job is enqueued. When the last
|
|
90
|
+
* job completes and this duration passes without a new one, the entire
|
|
91
|
+
* worker pool is torn down so that no background threads keep the Node.js
|
|
92
|
+
* process alive unnecessarily.
|
|
93
|
+
*
|
|
94
|
+
* Set to `0` to keep workers alive indefinitely (useful when you want to
|
|
95
|
+
* call {@link terminateOcr} manually at shutdown time).
|
|
96
|
+
* Default is 10,000 ms (10 seconds).
|
|
97
|
+
*/
|
|
98
|
+
autoTerminate?: number;
|
|
99
|
+
/**
|
|
100
|
+
* Timeout in milliseconds for initializing a Tesseract worker
|
|
101
|
+
* (loading the JS runtime, downloading or loading the `.traineddata`
|
|
102
|
+
* language file) or for re-initializing an existing worker with a
|
|
103
|
+
* different language.
|
|
104
|
+
*
|
|
105
|
+
* Multi-language combinations (e.g. `'por+eng+spa'`) must download a
|
|
106
|
+
* separate `.traineddata` file for each language and are therefore
|
|
107
|
+
* particularly susceptible to slow networks. Tune this value upward if
|
|
108
|
+
* your OCR environment has high network latency or if you are loading
|
|
109
|
+
* languages from disk in a large container image.
|
|
110
|
+
*
|
|
111
|
+
* When the timeout fires, the failed job is rejected with a non-fatal
|
|
112
|
+
* {@link OfficeWarningType.OCR_FAILED} warning and parsing continues
|
|
113
|
+
* without OCR output for that image. The stalled worker is terminated
|
|
114
|
+
* and removed from the pool to prevent thread leaks.
|
|
115
|
+
*
|
|
116
|
+
* Set to `0` to wait indefinitely (not recommended for production; a hung
|
|
117
|
+
* network request will block the entire OCR queue for that language).
|
|
118
|
+
* Default is 60,000 ms (60 seconds).
|
|
119
|
+
*/
|
|
120
|
+
workerLoad?: number;
|
|
121
|
+
/**
|
|
122
|
+
* Timeout in milliseconds for the actual OCR text-recognition call
|
|
123
|
+
* (`worker.recognize(image)`) on an already-initialized Tesseract worker.
|
|
124
|
+
*
|
|
125
|
+
* Recognition time scales with image resolution and the number of active
|
|
126
|
+
* languages. Very high-resolution scans or unusual character sets can
|
|
127
|
+
* exceed the default. If this timeout fires, the job is rejected with a
|
|
128
|
+
* non-fatal {@link OfficeWarningType.OCR_FAILED} warning; the worker is
|
|
129
|
+
* terminated and evicted from the pool because its internal state after a
|
|
130
|
+
* mid-recognition timeout is undefined.
|
|
131
|
+
*
|
|
132
|
+
* Set to `0` to wait indefinitely.
|
|
133
|
+
* Default is 30,000 ms (30 seconds).
|
|
134
|
+
*/
|
|
135
|
+
recognition?: number;
|
|
70
136
|
}
|
|
71
137
|
/**
|
|
72
138
|
* Configuration options for OCR.
|
|
@@ -102,11 +168,33 @@ export interface OcrConfig {
|
|
|
102
168
|
*/
|
|
103
169
|
langPath?: string;
|
|
104
170
|
/**
|
|
171
|
+
* Consolidated timeout settings for all OCR operations.
|
|
172
|
+
*
|
|
173
|
+
* Prefer this over the deprecated flat timeout properties.
|
|
174
|
+
* If `timeout.autoTerminate` is set, it takes priority over the deprecated `autoTerminateTimeout`.
|
|
175
|
+
*/
|
|
176
|
+
timeout?: OcrTimeoutConfig;
|
|
177
|
+
/**
|
|
178
|
+
* @deprecated Use `timeout.autoTerminate` instead.
|
|
179
|
+
*
|
|
105
180
|
* Timeout in milliseconds of inactivity before the OCR worker pool is automatically terminated.
|
|
106
181
|
* Set to 0 to disable auto-termination.
|
|
107
182
|
* Default is 10,000 (10 seconds).
|
|
183
|
+
*
|
|
184
|
+
* If `timeout.autoTerminate` is also set, that value takes priority over this one.
|
|
108
185
|
*/
|
|
109
186
|
autoTerminateTimeout?: number;
|
|
187
|
+
/**
|
|
188
|
+
* An optional AbortSignal propagated from the main parser configuration to abort active OCR jobs.
|
|
189
|
+
* If the signal is aborted:
|
|
190
|
+
* 1. Any pending OCR jobs in the scheduler queue are rejected immediately.
|
|
191
|
+
* 2. Any active OCR job running on a Tesseract worker will reject, the worker will be
|
|
192
|
+
* terminated, and it will be removed from the pool to avoid hanging worker threads.
|
|
193
|
+
*
|
|
194
|
+
* Developers should prefer passing this at the top level of `parseOffice` (as `config.abortSignal`),
|
|
195
|
+
* which automatically propagates here.
|
|
196
|
+
*/
|
|
197
|
+
abortSignal?: AbortSignal | null;
|
|
110
198
|
}
|
|
111
199
|
/**
|
|
112
200
|
* Configuration options for the OfficeParser.
|
|
@@ -135,10 +223,23 @@ export interface OfficeParserConfig {
|
|
|
135
223
|
*/
|
|
136
224
|
ignoreNotes?: boolean;
|
|
137
225
|
/**
|
|
138
|
-
* Flag
|
|
139
|
-
* Default is false.
|
|
140
|
-
|
|
141
|
-
|
|
226
|
+
* Flag to ignore comments from parsing.
|
|
227
|
+
* Default is false.
|
|
228
|
+
*/
|
|
229
|
+
ignoreComments?: boolean;
|
|
230
|
+
/**
|
|
231
|
+
* Flag to ignore headers and footers from parsing.
|
|
232
|
+
* Default is false.
|
|
233
|
+
*/
|
|
234
|
+
ignoreHeadersAndFooters?: boolean;
|
|
235
|
+
/**
|
|
236
|
+
* Flag to ignore slide masters from parsing in PowerPoint.
|
|
237
|
+
* Default is false.
|
|
238
|
+
*/
|
|
239
|
+
ignoreSlideMasters?: boolean;
|
|
240
|
+
/**
|
|
241
|
+
* @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`.
|
|
242
|
+
* This option is now completely ignored by all parsers.
|
|
142
243
|
*/
|
|
143
244
|
putNotesAtLast?: boolean;
|
|
144
245
|
/**
|
|
@@ -173,6 +274,20 @@ export interface OfficeParserConfig {
|
|
|
173
274
|
* If provided, `ocrLanguage` will be ignored in favor of `ocrConfig.language`.
|
|
174
275
|
*/
|
|
175
276
|
ocrConfig?: OcrConfig;
|
|
277
|
+
/**
|
|
278
|
+
* An optional AbortSignal to cancel the parsing operation.
|
|
279
|
+
* When aborted, the parser immediately rejects with a standard AbortError (DOMException).
|
|
280
|
+
*
|
|
281
|
+
* ### Format-Specific Abort Behavior:
|
|
282
|
+
* - **PDF**: Checked between page loads and before individual image OCR operations.
|
|
283
|
+
* - **RTF**: Checked before parsing/traversal and before running OCR on image attachments.
|
|
284
|
+
* - **DOCX/XLSX/PPTX/ODF**: Checked during zip decompression before loading and parsing XML files.
|
|
285
|
+
* - **CSV/MD/HTML**: Checked at the start of the parsing phase.
|
|
286
|
+
*
|
|
287
|
+
* Note: If an OCR operation is currently running on a Tesseract worker when aborted,
|
|
288
|
+
* the worker will be terminated and removed from the worker pool automatically to prevent leaks.
|
|
289
|
+
*/
|
|
290
|
+
abortSignal?: AbortSignal | null;
|
|
176
291
|
/**
|
|
177
292
|
* Flag to serialize raw content (XML) as clean, formatted strings.
|
|
178
293
|
* Only relevant when `includeRawContent` is true.
|
|
@@ -254,9 +369,10 @@ export interface OfficeIssue {
|
|
|
254
369
|
/**
|
|
255
370
|
* The result of a document conversion operation.
|
|
256
371
|
*/
|
|
257
|
-
|
|
372
|
+
type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : string;
|
|
373
|
+
export interface ConversionResult<D extends UniversalGeneratorFormat> {
|
|
258
374
|
/** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
|
|
259
|
-
value: D
|
|
375
|
+
value: ConversionValue<D>;
|
|
260
376
|
/** A collection of issues (warnings/infos) generated during the process. */
|
|
261
377
|
messages: OfficeIssue[];
|
|
262
378
|
}
|
|
@@ -366,23 +482,43 @@ export interface CommonGeneratorConfig {
|
|
|
366
482
|
* Defaults to false.
|
|
367
483
|
*/
|
|
368
484
|
ignoreInternalLinks?: boolean;
|
|
485
|
+
/**
|
|
486
|
+
* An optional AbortSignal to cancel the generation operation.
|
|
487
|
+
* When aborted, the generator immediately rejects with a standard AbortError.
|
|
488
|
+
* Currently supported by PdfGenerator and ChunkingGenerator.
|
|
489
|
+
*/
|
|
490
|
+
abortSignal?: AbortSignal | null;
|
|
369
491
|
}
|
|
370
492
|
/**
|
|
371
493
|
* Destination-aware generator configuration.
|
|
372
494
|
* Restricts format-specific configurations to their respective destinations.
|
|
373
495
|
*/
|
|
374
496
|
/**
|
|
375
|
-
*
|
|
497
|
+
* Maps a destination format string to its corresponding specific configuration object type.
|
|
376
498
|
*/
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
}
|
|
499
|
+
type GeneratorSpecificConfig<D extends string> = D extends 'html' ? {
|
|
500
|
+
htmlConfig?: HtmlGeneratorConfig;
|
|
501
|
+
} : D extends 'md' ? {
|
|
502
|
+
mdConfig?: MdGeneratorConfig;
|
|
503
|
+
} : D extends 'pdf' ? {
|
|
504
|
+
pdfConfig?: PdfGeneratorConfig;
|
|
505
|
+
} : D extends 'csv' ? {
|
|
506
|
+
csvConfig?: CsvGeneratorConfig;
|
|
507
|
+
} : D extends 'text' ? {
|
|
508
|
+
textConfig?: TextGeneratorConfig;
|
|
509
|
+
} : D extends 'rtf' ? {
|
|
510
|
+
rtfConfig?: RtfGeneratorConfig;
|
|
511
|
+
} : D extends 'chunks' ? {
|
|
512
|
+
chunksConfig?: ChunkingConfig;
|
|
513
|
+
} : Partial<{
|
|
514
|
+
htmlConfig: HtmlGeneratorConfig;
|
|
515
|
+
mdConfig: MdGeneratorConfig;
|
|
516
|
+
pdfConfig: PdfGeneratorConfig;
|
|
517
|
+
csvConfig: CsvGeneratorConfig;
|
|
518
|
+
textConfig: TextGeneratorConfig;
|
|
519
|
+
rtfConfig: RtfGeneratorConfig;
|
|
520
|
+
chunksConfig: ChunkingConfig;
|
|
521
|
+
}>;
|
|
386
522
|
/**
|
|
387
523
|
* Configuration options for document generators.
|
|
388
524
|
*
|
|
@@ -393,9 +529,7 @@ export interface GeneratorSubConfigMap {
|
|
|
393
529
|
*
|
|
394
530
|
* @template D The destination format string. Defaults to `string` for a general configuration.
|
|
395
531
|
*/
|
|
396
|
-
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig &
|
|
397
|
-
[K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
|
|
398
|
-
};
|
|
532
|
+
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & GeneratorSpecificConfig<D>;
|
|
399
533
|
/**
|
|
400
534
|
* Configuration options for the OfficeConverter.
|
|
401
535
|
* Combines relevant parser and generator settings for a seamless one-step conversion.
|
|
@@ -440,10 +574,28 @@ export type DeepRequired<T> = T extends Function | Date | Buffer | RegExp ? T :
|
|
|
440
574
|
* it is a discriminated union whose members cannot be uniformly deep-required.
|
|
441
575
|
*/
|
|
442
576
|
export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
|
|
443
|
-
|
|
577
|
+
htmlConfig: HtmlGeneratorConfig;
|
|
578
|
+
mdConfig: MdGeneratorConfig;
|
|
579
|
+
pdfConfig: PdfGeneratorConfig;
|
|
580
|
+
csvConfig: CsvGeneratorConfig;
|
|
581
|
+
textConfig: TextGeneratorConfig;
|
|
582
|
+
rtfConfig: RtfGeneratorConfig;
|
|
444
583
|
}> & {
|
|
445
584
|
chunksConfig: ChunkingConfig;
|
|
446
585
|
};
|
|
586
|
+
/**
|
|
587
|
+
* Configuration options for granular raw HTML injections.
|
|
588
|
+
*/
|
|
589
|
+
export interface HtmlInjectionConfig {
|
|
590
|
+
/** Raw HTML injected immediately after the opening <head> tag */
|
|
591
|
+
headStart?: string;
|
|
592
|
+
/** Raw HTML injected immediately before the closing </head> tag */
|
|
593
|
+
headEnd?: string;
|
|
594
|
+
/** Raw HTML injected immediately after the opening <body> tag */
|
|
595
|
+
bodyStart?: string;
|
|
596
|
+
/** Raw HTML injected immediately before the closing </body> tag */
|
|
597
|
+
bodyEnd?: string;
|
|
598
|
+
}
|
|
447
599
|
/**
|
|
448
600
|
* Configuration options for HTML generation.
|
|
449
601
|
*/
|
|
@@ -458,6 +610,25 @@ export interface HtmlGeneratorConfig {
|
|
|
458
610
|
* Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
|
|
459
611
|
*/
|
|
460
612
|
chartJsSrc?: string;
|
|
613
|
+
/**
|
|
614
|
+
* Custom container width for the generated HTML.
|
|
615
|
+
* Can be a number (pixels) or string (e.g., '900px', '100%').
|
|
616
|
+
* If not specified or set to 'auto', it defaults based on the content type:
|
|
617
|
+
* - Spreadsheet: '100%'
|
|
618
|
+
* - Presentation/Slides: '1100px'
|
|
619
|
+
* - Standard Document (PDF/DOCX/RTF/etc.): '900px'
|
|
620
|
+
*/
|
|
621
|
+
containerWidth?: string | number;
|
|
622
|
+
/**
|
|
623
|
+
* Custom CSS to append to the generated HTML document.
|
|
624
|
+
* This CSS will be included in the `<style>` block and can be used to style
|
|
625
|
+
* custom classes added during AST manipulation or override default styles.
|
|
626
|
+
*/
|
|
627
|
+
customCss?: string;
|
|
628
|
+
/**
|
|
629
|
+
* Granular injection points for custom HTML, scripts, and styles.
|
|
630
|
+
*/
|
|
631
|
+
injections?: HtmlInjectionConfig;
|
|
461
632
|
}
|
|
462
633
|
/**
|
|
463
634
|
* Configuration options for PDF generation.
|
|
@@ -494,6 +665,12 @@ export interface PdfGeneratorConfig {
|
|
|
494
665
|
* Useful for setting custom executable paths or args in CI/CD.
|
|
495
666
|
*/
|
|
496
667
|
launchOptions?: any;
|
|
668
|
+
/**
|
|
669
|
+
* Timeout in milliseconds for PDF generation.
|
|
670
|
+
* Limits the time spent waiting for Puppeteer to launch, load content, and render PDF.
|
|
671
|
+
* Defaults to 30000 ms (30 seconds). Set to 0 to disable.
|
|
672
|
+
*/
|
|
673
|
+
timeout?: number;
|
|
497
674
|
}
|
|
498
675
|
/**
|
|
499
676
|
* Structured style mapping definition for the StyleMapper.
|
|
@@ -527,7 +704,7 @@ export interface StructuredStyleMapping {
|
|
|
527
704
|
* The structural type of the node (e.g., 'paragraph', 'heading', 'text').
|
|
528
705
|
* Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
|
|
529
706
|
*/
|
|
530
|
-
nodeType?:
|
|
707
|
+
nodeType?: OfficeContentNodeType;
|
|
531
708
|
/**
|
|
532
709
|
* A dictionary of attributes to match on the node.
|
|
533
710
|
*
|
|
@@ -593,7 +770,7 @@ export interface CsvGeneratorConfig {
|
|
|
593
770
|
/**
|
|
594
771
|
* Whether to merge all selected sheets into a single CSV.
|
|
595
772
|
* If false, returns a ZIP archive containing individual CSV files.
|
|
596
|
-
* Defaults to
|
|
773
|
+
* Defaults to true.
|
|
597
774
|
*/
|
|
598
775
|
mergeSheets?: boolean;
|
|
599
776
|
/**
|
|
@@ -799,6 +976,11 @@ export interface SemanticChunkingConfig extends BaseChunkingConfig {
|
|
|
799
976
|
* Default is 50.
|
|
800
977
|
*/
|
|
801
978
|
embeddingBatchSize?: number;
|
|
979
|
+
/**
|
|
980
|
+
* Timeout in milliseconds for individual embedding API calls.
|
|
981
|
+
* Defaults to 10000 ms (10 seconds). Set to 0 to disable.
|
|
982
|
+
*/
|
|
983
|
+
timeout?: number;
|
|
802
984
|
}
|
|
803
985
|
/**
|
|
804
986
|
* Discriminated union of all chunking strategy configurations.
|
|
@@ -847,11 +1029,16 @@ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods'
|
|
|
847
1029
|
/**
|
|
848
1030
|
* Types of content nodes in the AST.
|
|
849
1031
|
*/
|
|
850
|
-
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment';
|
|
1032
|
+
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster';
|
|
851
1033
|
/**
|
|
852
1034
|
* Supported MIME types for attachments.
|
|
853
1035
|
*/
|
|
854
1036
|
export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet' | 'application/vnd.openxmlformats-officedocument.presentationml.presentation' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation' | 'application/rtf' | 'text/csv' | 'text/markdown' | 'text/html';
|
|
1037
|
+
/**
|
|
1038
|
+
* Text alignment options.
|
|
1039
|
+
* Common in spreadsheet cells, paragraph styles, and text elements.
|
|
1040
|
+
*/
|
|
1041
|
+
export type TextAlignment = 'left' | 'center' | 'right' | 'justify';
|
|
855
1042
|
/**
|
|
856
1043
|
* Text formatting options available for text content.
|
|
857
1044
|
* Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
|
|
@@ -925,7 +1112,7 @@ export interface TextFormatting {
|
|
|
925
1112
|
* Common in spreadsheet cells or paragraph styles.
|
|
926
1113
|
* @example "center", "right"
|
|
927
1114
|
*/
|
|
928
|
-
alignment?:
|
|
1115
|
+
alignment?: TextAlignment;
|
|
929
1116
|
}
|
|
930
1117
|
/**
|
|
931
1118
|
* Metadata for a slide in PowerPoint.
|
|
@@ -975,7 +1162,7 @@ export interface HeadingMetadata {
|
|
|
975
1162
|
/** The heading level (e.g., 1 for H1). */
|
|
976
1163
|
level: number;
|
|
977
1164
|
/** The alignment of the heading. */
|
|
978
|
-
alignment?:
|
|
1165
|
+
alignment?: TextAlignment;
|
|
979
1166
|
/** The style of the heading. */
|
|
980
1167
|
style?: string;
|
|
981
1168
|
/** Detailed indentation information. */
|
|
@@ -988,7 +1175,7 @@ export interface HeadingMetadata {
|
|
|
988
1175
|
*/
|
|
989
1176
|
export interface ParagraphMetadata {
|
|
990
1177
|
/** The alignment of the paragraph. */
|
|
991
|
-
alignment?:
|
|
1178
|
+
alignment?: TextAlignment;
|
|
992
1179
|
/** The style of the paragraph. */
|
|
993
1180
|
style?: string;
|
|
994
1181
|
/** Detailed indentation information. */
|
|
@@ -1016,7 +1203,7 @@ export interface ListMetadata {
|
|
|
1016
1203
|
* Text alignment of the list item.
|
|
1017
1204
|
* @example 'left', 'center', 'right', 'justify'
|
|
1018
1205
|
*/
|
|
1019
|
-
alignment:
|
|
1206
|
+
alignment: TextAlignment;
|
|
1020
1207
|
/**
|
|
1021
1208
|
* The list ID from the Word document's numbering definition.
|
|
1022
1209
|
* Used to identify which list definition this item belongs to.
|
|
@@ -1066,6 +1253,15 @@ export interface CellMetadata {
|
|
|
1066
1253
|
style?: string;
|
|
1067
1254
|
/** Unique anchor IDs for internal linking. */
|
|
1068
1255
|
anchorIds?: string[];
|
|
1256
|
+
/** Background color for this cell in hex format (e.g. #FFFFFF). */
|
|
1257
|
+
backgroundColor?: string;
|
|
1258
|
+
}
|
|
1259
|
+
/**
|
|
1260
|
+
* Metadata for a table.
|
|
1261
|
+
*/
|
|
1262
|
+
export interface TableMetadata {
|
|
1263
|
+
/** Unique anchor IDs for internal linking. */
|
|
1264
|
+
anchorIds?: string[];
|
|
1069
1265
|
}
|
|
1070
1266
|
/**
|
|
1071
1267
|
* Metadata for a chart node in the document.
|
|
@@ -1153,6 +1349,8 @@ export interface NoteMetadata {
|
|
|
1153
1349
|
noteId?: string;
|
|
1154
1350
|
/** Unique anchor IDs for internal linking. */
|
|
1155
1351
|
anchorIds?: string[];
|
|
1352
|
+
/** The slide number this note is associated with (used in PowerPoint). */
|
|
1353
|
+
slideNumber?: number;
|
|
1156
1354
|
}
|
|
1157
1355
|
/**
|
|
1158
1356
|
* Metadata for break nodes.
|
|
@@ -1188,10 +1386,25 @@ export interface CodeMetadata {
|
|
|
1188
1386
|
/** Unique anchor IDs for internal linking. */
|
|
1189
1387
|
anchorIds?: string[];
|
|
1190
1388
|
}
|
|
1389
|
+
/**
|
|
1390
|
+
* Metadata for a comment/annotation.
|
|
1391
|
+
*/
|
|
1392
|
+
export interface CommentMetadata {
|
|
1393
|
+
author?: string;
|
|
1394
|
+
initials?: string;
|
|
1395
|
+
date?: string;
|
|
1396
|
+
commentId?: string;
|
|
1397
|
+
}
|
|
1398
|
+
/**
|
|
1399
|
+
* Metadata for a header or footer.
|
|
1400
|
+
*/
|
|
1401
|
+
export interface HeaderFooterMetadata {
|
|
1402
|
+
type: 'default' | 'first' | 'even' | string;
|
|
1403
|
+
}
|
|
1191
1404
|
/**
|
|
1192
1405
|
* Union type for content metadata.
|
|
1193
1406
|
*/
|
|
1194
|
-
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
|
|
1407
|
+
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
|
|
1195
1408
|
/**
|
|
1196
1409
|
* Represents a node in the document content tree.
|
|
1197
1410
|
* This is the core building block of the parsed document structure.
|
|
@@ -1218,13 +1431,10 @@ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata |
|
|
|
1218
1431
|
* children: [...]
|
|
1219
1432
|
* }
|
|
1220
1433
|
*/
|
|
1221
|
-
|
|
1222
|
-
|
|
1223
|
-
|
|
1224
|
-
|
|
1225
|
-
* Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
|
|
1226
|
-
*/
|
|
1227
|
-
type: OfficeContentNodeType;
|
|
1434
|
+
/**
|
|
1435
|
+
* Shared properties available on all document content nodes.
|
|
1436
|
+
*/
|
|
1437
|
+
export interface BaseContentNode {
|
|
1228
1438
|
/**
|
|
1229
1439
|
* The complete text content of the node and all its children combined.
|
|
1230
1440
|
* For container nodes (paragraph, heading), this is the concatenation of all child text.
|
|
@@ -1242,6 +1452,16 @@ export interface OfficeContentNode {
|
|
|
1242
1452
|
* @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
|
|
1243
1453
|
*/
|
|
1244
1454
|
children?: OfficeContentNode[];
|
|
1455
|
+
/**
|
|
1456
|
+
* Comments attached to this specific node.
|
|
1457
|
+
* Keeps annotations completely separate from the actual content flow.
|
|
1458
|
+
*/
|
|
1459
|
+
comments?: OfficeContentNode[];
|
|
1460
|
+
/**
|
|
1461
|
+
* Notes (like footnotes or slide notes) attached to this specific node.
|
|
1462
|
+
* Keeps notes separate from the actual structural children.
|
|
1463
|
+
*/
|
|
1464
|
+
notes?: OfficeContentNode[];
|
|
1245
1465
|
/**
|
|
1246
1466
|
* Text formatting applied to this node.
|
|
1247
1467
|
* Only applicable to text-containing nodes.
|
|
@@ -1249,16 +1469,6 @@ export interface OfficeContentNode {
|
|
|
1249
1469
|
* @example { bold: true, size: "12", font: "Arial" }
|
|
1250
1470
|
*/
|
|
1251
1471
|
formatting?: TextFormatting;
|
|
1252
|
-
/**
|
|
1253
|
-
* Type-specific metadata providing additional context about the node.
|
|
1254
|
-
* The metadata structure depends on the node type:
|
|
1255
|
-
* - Headings: { level: 1 }
|
|
1256
|
-
* - Lists: { listType: 'ordered', indentation: 0 }
|
|
1257
|
-
* - Cells: { row: 0, col: 0 }
|
|
1258
|
-
* - Slides: { slideNumber: 1 }
|
|
1259
|
-
* @example { level: 1 } for a heading
|
|
1260
|
-
*/
|
|
1261
|
-
metadata?: ContentMetadata;
|
|
1262
1472
|
/**
|
|
1263
1473
|
* The raw source content for this node.
|
|
1264
1474
|
* - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
|
|
@@ -1270,6 +1480,93 @@ export interface OfficeContentNode {
|
|
|
1270
1480
|
*/
|
|
1271
1481
|
rawContent?: string;
|
|
1272
1482
|
}
|
|
1483
|
+
/**
|
|
1484
|
+
* Represents a node in the document content tree.
|
|
1485
|
+
* This is the core building block of the parsed document structure.
|
|
1486
|
+
* Content nodes can be nested to represent hierarchical document structures
|
|
1487
|
+
* (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
|
|
1488
|
+
*
|
|
1489
|
+
* @example
|
|
1490
|
+
* // A simple paragraph with formatted text
|
|
1491
|
+
* {
|
|
1492
|
+
* type: 'paragraph',
|
|
1493
|
+
* text: 'Hello world',
|
|
1494
|
+
* children: [
|
|
1495
|
+
* { type: 'text', text: 'Hello ', formatting: { bold: true } },
|
|
1496
|
+
* { type: 'text', text: 'world', formatting: { italic: true } }
|
|
1497
|
+
* ]
|
|
1498
|
+
* }
|
|
1499
|
+
*
|
|
1500
|
+
* @example
|
|
1501
|
+
* // A heading with metadata
|
|
1502
|
+
* {
|
|
1503
|
+
* type: 'heading',
|
|
1504
|
+
* text: 'Chapter 1',
|
|
1505
|
+
* metadata: { level: 1 },
|
|
1506
|
+
* children: [...]
|
|
1507
|
+
* }
|
|
1508
|
+
*/
|
|
1509
|
+
export type OfficeContentNode = BaseContentNode & ({
|
|
1510
|
+
type: 'slide';
|
|
1511
|
+
metadata?: SlideMetadata;
|
|
1512
|
+
} | {
|
|
1513
|
+
type: 'sheet';
|
|
1514
|
+
metadata?: SheetMetadata;
|
|
1515
|
+
} | {
|
|
1516
|
+
type: 'heading';
|
|
1517
|
+
metadata?: HeadingMetadata;
|
|
1518
|
+
} | {
|
|
1519
|
+
type: 'list';
|
|
1520
|
+
metadata?: ListMetadata;
|
|
1521
|
+
} | {
|
|
1522
|
+
type: 'cell';
|
|
1523
|
+
metadata?: CellMetadata;
|
|
1524
|
+
} | {
|
|
1525
|
+
type: 'image';
|
|
1526
|
+
metadata?: ImageMetadata;
|
|
1527
|
+
} | {
|
|
1528
|
+
type: 'chart';
|
|
1529
|
+
metadata?: ChartMetadata;
|
|
1530
|
+
} | {
|
|
1531
|
+
type: 'page';
|
|
1532
|
+
metadata?: PageMetadata;
|
|
1533
|
+
} | {
|
|
1534
|
+
type: 'paragraph';
|
|
1535
|
+
metadata?: ParagraphMetadata;
|
|
1536
|
+
} | {
|
|
1537
|
+
type: 'text';
|
|
1538
|
+
metadata?: TextMetadata;
|
|
1539
|
+
} | {
|
|
1540
|
+
type: 'note';
|
|
1541
|
+
metadata?: NoteMetadata;
|
|
1542
|
+
} | {
|
|
1543
|
+
type: 'break';
|
|
1544
|
+
metadata?: BreakMetadata;
|
|
1545
|
+
} | {
|
|
1546
|
+
type: 'code';
|
|
1547
|
+
metadata?: CodeMetadata;
|
|
1548
|
+
} | {
|
|
1549
|
+
type: 'comment';
|
|
1550
|
+
metadata?: CommentMetadata;
|
|
1551
|
+
} | {
|
|
1552
|
+
type: 'header';
|
|
1553
|
+
metadata?: HeaderFooterMetadata;
|
|
1554
|
+
} | {
|
|
1555
|
+
type: 'footer';
|
|
1556
|
+
metadata?: HeaderFooterMetadata;
|
|
1557
|
+
} | {
|
|
1558
|
+
type: 'table';
|
|
1559
|
+
metadata?: TableMetadata;
|
|
1560
|
+
} | {
|
|
1561
|
+
type: 'row';
|
|
1562
|
+
metadata?: undefined;
|
|
1563
|
+
} | {
|
|
1564
|
+
type: 'drawing';
|
|
1565
|
+
metadata?: undefined;
|
|
1566
|
+
} | {
|
|
1567
|
+
type: 'slideMaster';
|
|
1568
|
+
metadata?: SlideMetadata;
|
|
1569
|
+
});
|
|
1273
1570
|
/**
|
|
1274
1571
|
* Structured information extracted from a chart.
|
|
1275
1572
|
*/
|
|
@@ -1409,14 +1706,40 @@ export interface OfficeMetadata {
|
|
|
1409
1706
|
* Values are typed as string, number, boolean, or Date where the source format provides type information.
|
|
1410
1707
|
*/
|
|
1411
1708
|
customProperties?: Record<string, string | number | boolean | Date>;
|
|
1709
|
+
/** Keywords associated with the document. */
|
|
1710
|
+
keywords?: string;
|
|
1711
|
+
/**
|
|
1712
|
+
* Contains all format-specific metadata fields extracted verbatim.
|
|
1713
|
+
* Consumers can use this to access properties not mapped to the standard OfficeMetadata fields.
|
|
1714
|
+
* Examples: all <meta> tags in HTML, app.xml properties in DOCX, XMP dicts in PDF.
|
|
1715
|
+
*/
|
|
1716
|
+
nativeProperties?: Record<string, any>;
|
|
1717
|
+
}
|
|
1718
|
+
/**
|
|
1719
|
+
* Contains out-of-band layout elements and templates that are not part of the main document flow.
|
|
1720
|
+
*/
|
|
1721
|
+
export interface OfficeAuxiliaryContent {
|
|
1722
|
+
/** Headers extracted from the document. */
|
|
1723
|
+
headers?: OfficeContentNode[];
|
|
1724
|
+
/** Footers extracted from the document. */
|
|
1725
|
+
footers?: OfficeContentNode[];
|
|
1726
|
+
/** Slide Masters extracted from presentations. */
|
|
1727
|
+
slideMasters?: OfficeContentNode[];
|
|
1412
1728
|
}
|
|
1413
1729
|
/**
|
|
1414
|
-
* The Abstract Syntax Tree (AST)
|
|
1415
|
-
* This is the
|
|
1730
|
+
* The Root Abstract Syntax Tree (AST) representing a parsed Office Document.
|
|
1731
|
+
* This is the ultimate output of `OfficeParser.parseOffice()`.
|
|
1416
1732
|
*
|
|
1417
|
-
*
|
|
1418
|
-
*
|
|
1419
|
-
*
|
|
1733
|
+
* DESIGN PHILOSOPHY:
|
|
1734
|
+
* The AST is designed to be a universal, format-agnostic representation of document content.
|
|
1735
|
+
* Whether the input was a PDF, DOCX, XLSX, Markdown, or HTML file, the resulting AST
|
|
1736
|
+
* uses the same consistent structure (`OfficeContentNode` trees).
|
|
1737
|
+
*
|
|
1738
|
+
* ### Key Top-Level Properties:
|
|
1739
|
+
* - `metadata`: Document-level properties (author, title, stats).
|
|
1740
|
+
* - `content`: The main sequential flow of the document (paragraphs, tables, slides, sheets).
|
|
1741
|
+
* - `attachments`: Extracted binary assets (images, embedded files).
|
|
1742
|
+
* - `auxiliary`: Out-of-band layout/template elements (headers, footers, slide masters).
|
|
1420
1743
|
*
|
|
1421
1744
|
* @example
|
|
1422
1745
|
* ```typescript
|
|
@@ -1466,6 +1789,12 @@ export interface OfficeParserAST {
|
|
|
1466
1789
|
* @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
|
|
1467
1790
|
*/
|
|
1468
1791
|
content: OfficeContentNode[];
|
|
1792
|
+
/**
|
|
1793
|
+
* Out-of-band layout and template elements that are not part of the main text flow.
|
|
1794
|
+
* Extracted only if the respective `ignore...` config flags are false.
|
|
1795
|
+
* Contains elements like `headers`, `footers`, and `slideMasters`.
|
|
1796
|
+
*/
|
|
1797
|
+
auxiliary?: OfficeAuxiliaryContent;
|
|
1469
1798
|
/**
|
|
1470
1799
|
* Attachments extracted from the document (images, charts, embedded files).
|
|
1471
1800
|
* Only populated when `config.extractAttachments` is true.
|
|
@@ -1509,5 +1838,6 @@ export interface OfficeParserAST {
|
|
|
1509
1838
|
* const md = await ast.to('md');
|
|
1510
1839
|
* ```
|
|
1511
1840
|
*/
|
|
1512
|
-
to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
1841
|
+
to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
1513
1842
|
}
|
|
1843
|
+
export {};
|
package/dist/types.js
CHANGED
|
@@ -33,6 +33,8 @@ var OfficeErrorType;
|
|
|
33
33
|
OfficeErrorType["INVALID_OUTPUT_MAPPING"] = "INVALID_OUTPUT_MAPPING";
|
|
34
34
|
/** Semantic chunking strategy is selected but no embedding function is provided */
|
|
35
35
|
OfficeErrorType["MISSING_EMBEDDING_FUNCTION"] = "MISSING_EMBEDDING_FUNCTION";
|
|
36
|
+
/** The operation was aborted */
|
|
37
|
+
OfficeErrorType["OPERATION_ABORTED"] = "OPERATION_ABORTED";
|
|
36
38
|
})(OfficeErrorType || (exports.OfficeErrorType = OfficeErrorType = {}));
|
|
37
39
|
/**
|
|
38
40
|
* Standard warning types for OfficeParser.
|
|
@@ -72,4 +74,6 @@ var OfficeWarningType;
|
|
|
72
74
|
OfficeWarningType["EMPTY_CHUNK_GENERATED"] = "EMPTY_CHUNK_GENERATED";
|
|
73
75
|
/** A node was skipped because it only contained whitespace */
|
|
74
76
|
OfficeWarningType["WHITESPACE_NODE_SKIPPED"] = "WHITESPACE_NODE_SKIPPED";
|
|
77
|
+
/** The HTML generator containerWidth option is invalid */
|
|
78
|
+
OfficeWarningType["INVALID_CONTAINER_WIDTH"] = "INVALID_CONTAINER_WIDTH";
|
|
75
79
|
})(OfficeWarningType || (exports.OfficeWarningType = OfficeWarningType = {}));
|
package/dist/utils/astUtils.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { OfficeAttachment, OfficeAuxiliaryContent, OfficeContentNode, OfficeMetadata, OfficeParserAST, OfficeParserConfig, SupportedFileType } from '../types.js';
|
|
2
2
|
/**
|
|
3
3
|
* Creates a fully-featured OfficeParserAST object with conversion methods.
|
|
4
4
|
*
|
|
@@ -13,4 +13,4 @@ import { OfficeParserAST, OfficeContentNode, OfficeMetadata, OfficeAttachment, S
|
|
|
13
13
|
* @param toTextSync - Synchronous text extraction logic (for backward compatibility)
|
|
14
14
|
* @returns An object conforming to OfficeParserAST
|
|
15
15
|
*/
|
|
16
|
-
export declare function createAST(type: SupportedFileType, metadata: OfficeMetadata, content: OfficeContentNode[], attachments: OfficeAttachment[], config: OfficeParserConfig, toTextSync: () => string): OfficeParserAST;
|
|
16
|
+
export declare function createAST(type: SupportedFileType, metadata: OfficeMetadata, content: OfficeContentNode[], attachments: OfficeAttachment[], config: OfficeParserConfig, auxiliary: OfficeAuxiliaryContent | undefined, toTextSync: () => string): OfficeParserAST;
|