officeparser 7.1.0 → 7.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -18
- package/dist/OfficeGenerator.d.ts +1 -1
- package/dist/OfficeGenerator.js +16 -7
- package/dist/cli.d.ts +4 -0
- package/dist/cli.js +12 -3
- package/dist/defaults.js +11 -0
- package/dist/generators/BaseGenerator.d.ts +3 -3
- package/dist/generators/ChunkingGenerator.js +8 -1
- package/dist/generators/CsvGenerator.d.ts +1 -1
- package/dist/generators/HtmlGenerator.d.ts +2 -1
- package/dist/generators/HtmlGenerator.js +462 -40
- package/dist/generators/MarkdownGenerator.d.ts +1 -1
- package/dist/generators/MarkdownGenerator.js +3 -1
- package/dist/generators/PdfGenerator.d.ts +1 -1
- package/dist/generators/PdfGenerator.js +0 -6
- package/dist/generators/RtfGenerator.d.ts +2 -1
- package/dist/generators/RtfGenerator.js +43 -6
- package/dist/generators/TextGenerator.d.ts +1 -1
- package/dist/officeparser.browser.d.ts +259 -52
- package/dist/officeparser.browser.iife.js +380 -93
- package/dist/officeparser.browser.mjs +380 -93
- package/dist/parsers/CsvParser.js +1 -1
- package/dist/parsers/ExcelParser.js +63 -19
- package/dist/parsers/HtmlParser.js +10 -1
- package/dist/parsers/MarkdownParser.js +13 -10
- package/dist/parsers/OpenOfficeParser.js +57 -34
- package/dist/parsers/PdfParser.js +23 -1
- package/dist/parsers/PowerPointParser.js +164 -40
- package/dist/parsers/RtfParser.js +28 -24
- package/dist/parsers/WordParser.js +154 -11
- package/dist/sbom.cdx.json +100 -100
- package/dist/types.d.ts +265 -52
- package/dist/types.js +2 -0
- package/dist/utils/astUtils.d.ts +2 -2
- package/dist/utils/astUtils.js +2 -1
- package/dist/utils/configUtils.d.ts +5 -0
- package/dist/utils/configUtils.js +55 -1
- package/dist/utils/errorUtils.js +2 -1
- package/dist/utils/xmlUtils.d.ts +9 -0
- package/dist/utils/xmlUtils.js +53 -1
- package/package.json +1 -1
package/dist/types.d.ts
CHANGED
|
@@ -68,7 +68,9 @@ export declare enum OfficeWarningType {
|
|
|
68
68
|
/** No chunks were generated for the document given the current strategy */
|
|
69
69
|
EMPTY_CHUNK_GENERATED = "EMPTY_CHUNK_GENERATED",
|
|
70
70
|
/** A node was skipped because it only contained whitespace */
|
|
71
|
-
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED"
|
|
71
|
+
WHITESPACE_NODE_SKIPPED = "WHITESPACE_NODE_SKIPPED",
|
|
72
|
+
/** The HTML generator containerWidth option is invalid */
|
|
73
|
+
INVALID_CONTAINER_WIDTH = "INVALID_CONTAINER_WIDTH"
|
|
72
74
|
}
|
|
73
75
|
/**
|
|
74
76
|
* Consolidated timeout settings for OCR operations.
|
|
@@ -221,10 +223,23 @@ export interface OfficeParserConfig {
|
|
|
221
223
|
*/
|
|
222
224
|
ignoreNotes?: boolean;
|
|
223
225
|
/**
|
|
224
|
-
* Flag
|
|
225
|
-
* Default is false.
|
|
226
|
-
|
|
227
|
-
|
|
226
|
+
* Flag to ignore comments from parsing.
|
|
227
|
+
* Default is false.
|
|
228
|
+
*/
|
|
229
|
+
ignoreComments?: boolean;
|
|
230
|
+
/**
|
|
231
|
+
* Flag to ignore headers and footers from parsing.
|
|
232
|
+
* Default is false.
|
|
233
|
+
*/
|
|
234
|
+
ignoreHeadersAndFooters?: boolean;
|
|
235
|
+
/**
|
|
236
|
+
* Flag to ignore slide masters from parsing in PowerPoint.
|
|
237
|
+
* Default is false.
|
|
238
|
+
*/
|
|
239
|
+
ignoreSlideMasters?: boolean;
|
|
240
|
+
/**
|
|
241
|
+
* @deprecated Notes are now structurally attached to the specific nodes they belong to via `node.notes`.
|
|
242
|
+
* This option is now completely ignored by all parsers.
|
|
228
243
|
*/
|
|
229
244
|
putNotesAtLast?: boolean;
|
|
230
245
|
/**
|
|
@@ -354,9 +369,10 @@ export interface OfficeIssue {
|
|
|
354
369
|
/**
|
|
355
370
|
* The result of a document conversion operation.
|
|
356
371
|
*/
|
|
357
|
-
|
|
372
|
+
type ConversionValue<D extends UniversalGeneratorFormat> = D extends 'pdf' ? Uint8Array | string : D extends 'chunks' ? OfficeChunk[] : D extends 'csv' ? string | Uint8Array : string;
|
|
373
|
+
export interface ConversionResult<D extends UniversalGeneratorFormat> {
|
|
358
374
|
/** The actual generated content (HTML, Markdown, Text, OfficeChunk[], etc.). */
|
|
359
|
-
value: D
|
|
375
|
+
value: ConversionValue<D>;
|
|
360
376
|
/** A collection of issues (warnings/infos) generated during the process. */
|
|
361
377
|
messages: OfficeIssue[];
|
|
362
378
|
}
|
|
@@ -478,17 +494,31 @@ export interface CommonGeneratorConfig {
|
|
|
478
494
|
* Restricts format-specific configurations to their respective destinations.
|
|
479
495
|
*/
|
|
480
496
|
/**
|
|
481
|
-
*
|
|
497
|
+
* Maps a destination format string to its corresponding specific configuration object type.
|
|
482
498
|
*/
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
|
|
489
|
-
|
|
490
|
-
|
|
491
|
-
}
|
|
499
|
+
type GeneratorSpecificConfig<D extends string> = D extends 'html' ? {
|
|
500
|
+
htmlConfig?: HtmlGeneratorConfig;
|
|
501
|
+
} : D extends 'md' ? {
|
|
502
|
+
mdConfig?: MdGeneratorConfig;
|
|
503
|
+
} : D extends 'pdf' ? {
|
|
504
|
+
pdfConfig?: PdfGeneratorConfig;
|
|
505
|
+
} : D extends 'csv' ? {
|
|
506
|
+
csvConfig?: CsvGeneratorConfig;
|
|
507
|
+
} : D extends 'text' ? {
|
|
508
|
+
textConfig?: TextGeneratorConfig;
|
|
509
|
+
} : D extends 'rtf' ? {
|
|
510
|
+
rtfConfig?: RtfGeneratorConfig;
|
|
511
|
+
} : D extends 'chunks' ? {
|
|
512
|
+
chunksConfig?: ChunkingConfig;
|
|
513
|
+
} : Partial<{
|
|
514
|
+
htmlConfig: HtmlGeneratorConfig;
|
|
515
|
+
mdConfig: MdGeneratorConfig;
|
|
516
|
+
pdfConfig: PdfGeneratorConfig;
|
|
517
|
+
csvConfig: CsvGeneratorConfig;
|
|
518
|
+
textConfig: TextGeneratorConfig;
|
|
519
|
+
rtfConfig: RtfGeneratorConfig;
|
|
520
|
+
chunksConfig: ChunkingConfig;
|
|
521
|
+
}>;
|
|
492
522
|
/**
|
|
493
523
|
* Configuration options for document generators.
|
|
494
524
|
*
|
|
@@ -499,9 +529,7 @@ export interface GeneratorSubConfigMap {
|
|
|
499
529
|
*
|
|
500
530
|
* @template D The destination format string. Defaults to `string` for a general configuration.
|
|
501
531
|
*/
|
|
502
|
-
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig &
|
|
503
|
-
[K in keyof GeneratorSubConfigMap as `${K & string}Config`]?: string extends D ? GeneratorSubConfigMap[K] : (D extends K ? GeneratorSubConfigMap[K] : never);
|
|
504
|
-
};
|
|
532
|
+
export type GeneratorConfig<D extends string = string> = CommonGeneratorConfig & GeneratorSpecificConfig<D>;
|
|
505
533
|
/**
|
|
506
534
|
* Configuration options for the OfficeConverter.
|
|
507
535
|
* Combines relevant parser and generator settings for a seamless one-step conversion.
|
|
@@ -546,10 +574,28 @@ export type DeepRequired<T> = T extends Function | Date | Buffer | RegExp ? T :
|
|
|
546
574
|
* it is a discriminated union whose members cannot be uniformly deep-required.
|
|
547
575
|
*/
|
|
548
576
|
export type FullGeneratorConfig = DeepRequired<CommonGeneratorConfig & {
|
|
549
|
-
|
|
577
|
+
htmlConfig: HtmlGeneratorConfig;
|
|
578
|
+
mdConfig: MdGeneratorConfig;
|
|
579
|
+
pdfConfig: PdfGeneratorConfig;
|
|
580
|
+
csvConfig: CsvGeneratorConfig;
|
|
581
|
+
textConfig: TextGeneratorConfig;
|
|
582
|
+
rtfConfig: RtfGeneratorConfig;
|
|
550
583
|
}> & {
|
|
551
584
|
chunksConfig: ChunkingConfig;
|
|
552
585
|
};
|
|
586
|
+
/**
|
|
587
|
+
* Configuration options for granular raw HTML injections.
|
|
588
|
+
*/
|
|
589
|
+
export interface HtmlInjectionConfig {
|
|
590
|
+
/** Raw HTML injected immediately after the opening <head> tag */
|
|
591
|
+
headStart?: string;
|
|
592
|
+
/** Raw HTML injected immediately before the closing </head> tag */
|
|
593
|
+
headEnd?: string;
|
|
594
|
+
/** Raw HTML injected immediately after the opening <body> tag */
|
|
595
|
+
bodyStart?: string;
|
|
596
|
+
/** Raw HTML injected immediately before the closing </body> tag */
|
|
597
|
+
bodyEnd?: string;
|
|
598
|
+
}
|
|
553
599
|
/**
|
|
554
600
|
* Configuration options for HTML generation.
|
|
555
601
|
*/
|
|
@@ -564,6 +610,25 @@ export interface HtmlGeneratorConfig {
|
|
|
564
610
|
* Defaults to 'https://cdn.jsdelivr.net/npm/chart.js'.
|
|
565
611
|
*/
|
|
566
612
|
chartJsSrc?: string;
|
|
613
|
+
/**
|
|
614
|
+
* Custom container width for the generated HTML.
|
|
615
|
+
* Can be a number (pixels) or string (e.g., '900px', '100%').
|
|
616
|
+
* If not specified or set to 'auto', it defaults based on the content type:
|
|
617
|
+
* - Spreadsheet: '100%'
|
|
618
|
+
* - Presentation/Slides: '1100px'
|
|
619
|
+
* - Standard Document (PDF/DOCX/RTF/etc.): '900px'
|
|
620
|
+
*/
|
|
621
|
+
containerWidth?: string | number;
|
|
622
|
+
/**
|
|
623
|
+
* Custom CSS to append to the generated HTML document.
|
|
624
|
+
* This CSS will be included in the `<style>` block and can be used to style
|
|
625
|
+
* custom classes added during AST manipulation or override default styles.
|
|
626
|
+
*/
|
|
627
|
+
customCss?: string;
|
|
628
|
+
/**
|
|
629
|
+
* Granular injection points for custom HTML, scripts, and styles.
|
|
630
|
+
*/
|
|
631
|
+
injections?: HtmlInjectionConfig;
|
|
567
632
|
}
|
|
568
633
|
/**
|
|
569
634
|
* Configuration options for PDF generation.
|
|
@@ -639,7 +704,7 @@ export interface StructuredStyleMapping {
|
|
|
639
704
|
* The structural type of the node (e.g., 'paragraph', 'heading', 'text').
|
|
640
705
|
* Most style mappings target 'paragraph' nodes to convert them into headers or blocks.
|
|
641
706
|
*/
|
|
642
|
-
nodeType?:
|
|
707
|
+
nodeType?: OfficeContentNodeType;
|
|
643
708
|
/**
|
|
644
709
|
* A dictionary of attributes to match on the node.
|
|
645
710
|
*
|
|
@@ -705,7 +770,7 @@ export interface CsvGeneratorConfig {
|
|
|
705
770
|
/**
|
|
706
771
|
* Whether to merge all selected sheets into a single CSV.
|
|
707
772
|
* If false, returns a ZIP archive containing individual CSV files.
|
|
708
|
-
* Defaults to
|
|
773
|
+
* Defaults to true.
|
|
709
774
|
*/
|
|
710
775
|
mergeSheets?: boolean;
|
|
711
776
|
/**
|
|
@@ -964,11 +1029,16 @@ export type SupportedFileType = 'docx' | 'pptx' | 'xlsx' | 'odt' | 'odp' | 'ods'
|
|
|
964
1029
|
/**
|
|
965
1030
|
* Types of content nodes in the AST.
|
|
966
1031
|
*/
|
|
967
|
-
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment';
|
|
1032
|
+
export type OfficeContentNodeType = 'paragraph' | 'heading' | 'table' | 'list' | 'text' | 'image' | 'chart' | 'drawing' | 'slide' | 'note' | 'sheet' | 'row' | 'cell' | 'page' | 'break' | 'code' | 'comment' | 'header' | 'footer' | 'slideMaster';
|
|
968
1033
|
/**
|
|
969
1034
|
* Supported MIME types for attachments.
|
|
970
1035
|
*/
|
|
971
1036
|
export type OfficeMimeType = 'image/jpeg' | 'image/png' | 'image/gif' | 'image/bmp' | 'image/tiff' | 'image/svg+xml' | 'application/pdf' | 'application/vnd.openxmlformats-officedocument.wordprocessingml.document' | 'application/vnd.openxmlformats-officedocument.spreadsheetml.sheet' | 'application/vnd.openxmlformats-officedocument.presentationml.presentation' | 'application/vnd.oasis.opendocument.chart' | 'application/vnd.oasis.opendocument.spreadsheet' | 'application/vnd.oasis.opendocument.text' | 'application/vnd.oasis.opendocument.presentation' | 'application/rtf' | 'text/csv' | 'text/markdown' | 'text/html';
|
|
1037
|
+
/**
|
|
1038
|
+
* Text alignment options.
|
|
1039
|
+
* Common in spreadsheet cells, paragraph styles, and text elements.
|
|
1040
|
+
*/
|
|
1041
|
+
export type TextAlignment = 'left' | 'center' | 'right' | 'justify';
|
|
972
1042
|
/**
|
|
973
1043
|
* Text formatting options available for text content.
|
|
974
1044
|
* Represents common formatting attributes found in office documents (DOCX, RTF, PPTX, etc.).
|
|
@@ -1042,7 +1112,7 @@ export interface TextFormatting {
|
|
|
1042
1112
|
* Common in spreadsheet cells or paragraph styles.
|
|
1043
1113
|
* @example "center", "right"
|
|
1044
1114
|
*/
|
|
1045
|
-
alignment?:
|
|
1115
|
+
alignment?: TextAlignment;
|
|
1046
1116
|
}
|
|
1047
1117
|
/**
|
|
1048
1118
|
* Metadata for a slide in PowerPoint.
|
|
@@ -1092,7 +1162,7 @@ export interface HeadingMetadata {
|
|
|
1092
1162
|
/** The heading level (e.g., 1 for H1). */
|
|
1093
1163
|
level: number;
|
|
1094
1164
|
/** The alignment of the heading. */
|
|
1095
|
-
alignment?:
|
|
1165
|
+
alignment?: TextAlignment;
|
|
1096
1166
|
/** The style of the heading. */
|
|
1097
1167
|
style?: string;
|
|
1098
1168
|
/** Detailed indentation information. */
|
|
@@ -1105,7 +1175,7 @@ export interface HeadingMetadata {
|
|
|
1105
1175
|
*/
|
|
1106
1176
|
export interface ParagraphMetadata {
|
|
1107
1177
|
/** The alignment of the paragraph. */
|
|
1108
|
-
alignment?:
|
|
1178
|
+
alignment?: TextAlignment;
|
|
1109
1179
|
/** The style of the paragraph. */
|
|
1110
1180
|
style?: string;
|
|
1111
1181
|
/** Detailed indentation information. */
|
|
@@ -1133,7 +1203,7 @@ export interface ListMetadata {
|
|
|
1133
1203
|
* Text alignment of the list item.
|
|
1134
1204
|
* @example 'left', 'center', 'right', 'justify'
|
|
1135
1205
|
*/
|
|
1136
|
-
alignment:
|
|
1206
|
+
alignment: TextAlignment;
|
|
1137
1207
|
/**
|
|
1138
1208
|
* The list ID from the Word document's numbering definition.
|
|
1139
1209
|
* Used to identify which list definition this item belongs to.
|
|
@@ -1183,6 +1253,15 @@ export interface CellMetadata {
|
|
|
1183
1253
|
style?: string;
|
|
1184
1254
|
/** Unique anchor IDs for internal linking. */
|
|
1185
1255
|
anchorIds?: string[];
|
|
1256
|
+
/** Background color for this cell in hex format (e.g. #FFFFFF). */
|
|
1257
|
+
backgroundColor?: string;
|
|
1258
|
+
}
|
|
1259
|
+
/**
|
|
1260
|
+
* Metadata for a table.
|
|
1261
|
+
*/
|
|
1262
|
+
export interface TableMetadata {
|
|
1263
|
+
/** Unique anchor IDs for internal linking. */
|
|
1264
|
+
anchorIds?: string[];
|
|
1186
1265
|
}
|
|
1187
1266
|
/**
|
|
1188
1267
|
* Metadata for a chart node in the document.
|
|
@@ -1270,6 +1349,8 @@ export interface NoteMetadata {
|
|
|
1270
1349
|
noteId?: string;
|
|
1271
1350
|
/** Unique anchor IDs for internal linking. */
|
|
1272
1351
|
anchorIds?: string[];
|
|
1352
|
+
/** The slide number this note is associated with (used in PowerPoint). */
|
|
1353
|
+
slideNumber?: number;
|
|
1273
1354
|
}
|
|
1274
1355
|
/**
|
|
1275
1356
|
* Metadata for break nodes.
|
|
@@ -1305,10 +1386,25 @@ export interface CodeMetadata {
|
|
|
1305
1386
|
/** Unique anchor IDs for internal linking. */
|
|
1306
1387
|
anchorIds?: string[];
|
|
1307
1388
|
}
|
|
1389
|
+
/**
|
|
1390
|
+
* Metadata for a comment/annotation.
|
|
1391
|
+
*/
|
|
1392
|
+
export interface CommentMetadata {
|
|
1393
|
+
author?: string;
|
|
1394
|
+
initials?: string;
|
|
1395
|
+
date?: string;
|
|
1396
|
+
commentId?: string;
|
|
1397
|
+
}
|
|
1398
|
+
/**
|
|
1399
|
+
* Metadata for a header or footer.
|
|
1400
|
+
*/
|
|
1401
|
+
export interface HeaderFooterMetadata {
|
|
1402
|
+
type: 'default' | 'first' | 'even' | string;
|
|
1403
|
+
}
|
|
1308
1404
|
/**
|
|
1309
1405
|
* Union type for content metadata.
|
|
1310
1406
|
*/
|
|
1311
|
-
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | undefined;
|
|
1407
|
+
export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata | ListMetadata | CellMetadata | ImageMetadata | ChartMetadata | PageMetadata | ParagraphMetadata | TextMetadata | NoteMetadata | BreakMetadata | CodeMetadata | CommentMetadata | HeaderFooterMetadata | TableMetadata | undefined;
|
|
1312
1408
|
/**
|
|
1313
1409
|
* Represents a node in the document content tree.
|
|
1314
1410
|
* This is the core building block of the parsed document structure.
|
|
@@ -1335,13 +1431,10 @@ export type ContentMetadata = SlideMetadata | SheetMetadata | HeadingMetadata |
|
|
|
1335
1431
|
* children: [...]
|
|
1336
1432
|
* }
|
|
1337
1433
|
*/
|
|
1338
|
-
|
|
1339
|
-
|
|
1340
|
-
|
|
1341
|
-
|
|
1342
|
-
* Common types: 'paragraph', 'heading', 'table', 'list', 'text', 'image', etc.
|
|
1343
|
-
*/
|
|
1344
|
-
type: OfficeContentNodeType;
|
|
1434
|
+
/**
|
|
1435
|
+
* Shared properties available on all document content nodes.
|
|
1436
|
+
*/
|
|
1437
|
+
export interface BaseContentNode {
|
|
1345
1438
|
/**
|
|
1346
1439
|
* The complete text content of the node and all its children combined.
|
|
1347
1440
|
* For container nodes (paragraph, heading), this is the concatenation of all child text.
|
|
@@ -1359,6 +1452,16 @@ export interface OfficeContentNode {
|
|
|
1359
1452
|
* @example [{ type: 'text', text: 'Hello', formatting: { bold: true } }]
|
|
1360
1453
|
*/
|
|
1361
1454
|
children?: OfficeContentNode[];
|
|
1455
|
+
/**
|
|
1456
|
+
* Comments attached to this specific node.
|
|
1457
|
+
* Keeps annotations completely separate from the actual content flow.
|
|
1458
|
+
*/
|
|
1459
|
+
comments?: OfficeContentNode[];
|
|
1460
|
+
/**
|
|
1461
|
+
* Notes (like footnotes or slide notes) attached to this specific node.
|
|
1462
|
+
* Keeps notes separate from the actual structural children.
|
|
1463
|
+
*/
|
|
1464
|
+
notes?: OfficeContentNode[];
|
|
1362
1465
|
/**
|
|
1363
1466
|
* Text formatting applied to this node.
|
|
1364
1467
|
* Only applicable to text-containing nodes.
|
|
@@ -1366,16 +1469,6 @@ export interface OfficeContentNode {
|
|
|
1366
1469
|
* @example { bold: true, size: "12", font: "Arial" }
|
|
1367
1470
|
*/
|
|
1368
1471
|
formatting?: TextFormatting;
|
|
1369
|
-
/**
|
|
1370
|
-
* Type-specific metadata providing additional context about the node.
|
|
1371
|
-
* The metadata structure depends on the node type:
|
|
1372
|
-
* - Headings: { level: 1 }
|
|
1373
|
-
* - Lists: { listType: 'ordered', indentation: 0 }
|
|
1374
|
-
* - Cells: { row: 0, col: 0 }
|
|
1375
|
-
* - Slides: { slideNumber: 1 }
|
|
1376
|
-
* @example { level: 1 } for a heading
|
|
1377
|
-
*/
|
|
1378
|
-
metadata?: ContentMetadata;
|
|
1379
1472
|
/**
|
|
1380
1473
|
* The raw source content for this node.
|
|
1381
1474
|
* - For XML-based formats (DOCX, XLSX, PPTX): contains the raw XML
|
|
@@ -1387,6 +1480,93 @@ export interface OfficeContentNode {
|
|
|
1387
1480
|
*/
|
|
1388
1481
|
rawContent?: string;
|
|
1389
1482
|
}
|
|
1483
|
+
/**
|
|
1484
|
+
* Represents a node in the document content tree.
|
|
1485
|
+
* This is the core building block of the parsed document structure.
|
|
1486
|
+
* Content nodes can be nested to represent hierarchical document structures
|
|
1487
|
+
* (e.g., paragraphs containing text runs, tables containing rows, rows containing cells).
|
|
1488
|
+
*
|
|
1489
|
+
* @example
|
|
1490
|
+
* // A simple paragraph with formatted text
|
|
1491
|
+
* {
|
|
1492
|
+
* type: 'paragraph',
|
|
1493
|
+
* text: 'Hello world',
|
|
1494
|
+
* children: [
|
|
1495
|
+
* { type: 'text', text: 'Hello ', formatting: { bold: true } },
|
|
1496
|
+
* { type: 'text', text: 'world', formatting: { italic: true } }
|
|
1497
|
+
* ]
|
|
1498
|
+
* }
|
|
1499
|
+
*
|
|
1500
|
+
* @example
|
|
1501
|
+
* // A heading with metadata
|
|
1502
|
+
* {
|
|
1503
|
+
* type: 'heading',
|
|
1504
|
+
* text: 'Chapter 1',
|
|
1505
|
+
* metadata: { level: 1 },
|
|
1506
|
+
* children: [...]
|
|
1507
|
+
* }
|
|
1508
|
+
*/
|
|
1509
|
+
export type OfficeContentNode = BaseContentNode & ({
|
|
1510
|
+
type: 'slide';
|
|
1511
|
+
metadata?: SlideMetadata;
|
|
1512
|
+
} | {
|
|
1513
|
+
type: 'sheet';
|
|
1514
|
+
metadata?: SheetMetadata;
|
|
1515
|
+
} | {
|
|
1516
|
+
type: 'heading';
|
|
1517
|
+
metadata?: HeadingMetadata;
|
|
1518
|
+
} | {
|
|
1519
|
+
type: 'list';
|
|
1520
|
+
metadata?: ListMetadata;
|
|
1521
|
+
} | {
|
|
1522
|
+
type: 'cell';
|
|
1523
|
+
metadata?: CellMetadata;
|
|
1524
|
+
} | {
|
|
1525
|
+
type: 'image';
|
|
1526
|
+
metadata?: ImageMetadata;
|
|
1527
|
+
} | {
|
|
1528
|
+
type: 'chart';
|
|
1529
|
+
metadata?: ChartMetadata;
|
|
1530
|
+
} | {
|
|
1531
|
+
type: 'page';
|
|
1532
|
+
metadata?: PageMetadata;
|
|
1533
|
+
} | {
|
|
1534
|
+
type: 'paragraph';
|
|
1535
|
+
metadata?: ParagraphMetadata;
|
|
1536
|
+
} | {
|
|
1537
|
+
type: 'text';
|
|
1538
|
+
metadata?: TextMetadata;
|
|
1539
|
+
} | {
|
|
1540
|
+
type: 'note';
|
|
1541
|
+
metadata?: NoteMetadata;
|
|
1542
|
+
} | {
|
|
1543
|
+
type: 'break';
|
|
1544
|
+
metadata?: BreakMetadata;
|
|
1545
|
+
} | {
|
|
1546
|
+
type: 'code';
|
|
1547
|
+
metadata?: CodeMetadata;
|
|
1548
|
+
} | {
|
|
1549
|
+
type: 'comment';
|
|
1550
|
+
metadata?: CommentMetadata;
|
|
1551
|
+
} | {
|
|
1552
|
+
type: 'header';
|
|
1553
|
+
metadata?: HeaderFooterMetadata;
|
|
1554
|
+
} | {
|
|
1555
|
+
type: 'footer';
|
|
1556
|
+
metadata?: HeaderFooterMetadata;
|
|
1557
|
+
} | {
|
|
1558
|
+
type: 'table';
|
|
1559
|
+
metadata?: TableMetadata;
|
|
1560
|
+
} | {
|
|
1561
|
+
type: 'row';
|
|
1562
|
+
metadata?: undefined;
|
|
1563
|
+
} | {
|
|
1564
|
+
type: 'drawing';
|
|
1565
|
+
metadata?: undefined;
|
|
1566
|
+
} | {
|
|
1567
|
+
type: 'slideMaster';
|
|
1568
|
+
metadata?: SlideMetadata;
|
|
1569
|
+
});
|
|
1390
1570
|
/**
|
|
1391
1571
|
* Structured information extracted from a chart.
|
|
1392
1572
|
*/
|
|
@@ -1526,14 +1706,40 @@ export interface OfficeMetadata {
|
|
|
1526
1706
|
* Values are typed as string, number, boolean, or Date where the source format provides type information.
|
|
1527
1707
|
*/
|
|
1528
1708
|
customProperties?: Record<string, string | number | boolean | Date>;
|
|
1709
|
+
/** Keywords associated with the document. */
|
|
1710
|
+
keywords?: string;
|
|
1711
|
+
/**
|
|
1712
|
+
* Contains all format-specific metadata fields extracted verbatim.
|
|
1713
|
+
* Consumers can use this to access properties not mapped to the standard OfficeMetadata fields.
|
|
1714
|
+
* Examples: all <meta> tags in HTML, app.xml properties in DOCX, XMP dicts in PDF.
|
|
1715
|
+
*/
|
|
1716
|
+
nativeProperties?: Record<string, any>;
|
|
1717
|
+
}
|
|
1718
|
+
/**
|
|
1719
|
+
* Contains out-of-band layout elements and templates that are not part of the main document flow.
|
|
1720
|
+
*/
|
|
1721
|
+
export interface OfficeAuxiliaryContent {
|
|
1722
|
+
/** Headers extracted from the document. */
|
|
1723
|
+
headers?: OfficeContentNode[];
|
|
1724
|
+
/** Footers extracted from the document. */
|
|
1725
|
+
footers?: OfficeContentNode[];
|
|
1726
|
+
/** Slide Masters extracted from presentations. */
|
|
1727
|
+
slideMasters?: OfficeContentNode[];
|
|
1529
1728
|
}
|
|
1530
1729
|
/**
|
|
1531
|
-
* The Abstract Syntax Tree (AST)
|
|
1532
|
-
* This is the
|
|
1730
|
+
* The Root Abstract Syntax Tree (AST) representing a parsed Office Document.
|
|
1731
|
+
* This is the ultimate output of `OfficeParser.parseOffice()`.
|
|
1732
|
+
*
|
|
1733
|
+
* DESIGN PHILOSOPHY:
|
|
1734
|
+
* The AST is designed to be a universal, format-agnostic representation of document content.
|
|
1735
|
+
* Whether the input was a PDF, DOCX, XLSX, Markdown, or HTML file, the resulting AST
|
|
1736
|
+
* uses the same consistent structure (`OfficeContentNode` trees).
|
|
1533
1737
|
*
|
|
1534
|
-
*
|
|
1535
|
-
*
|
|
1536
|
-
*
|
|
1738
|
+
* ### Key Top-Level Properties:
|
|
1739
|
+
* - `metadata`: Document-level properties (author, title, stats).
|
|
1740
|
+
* - `content`: The main sequential flow of the document (paragraphs, tables, slides, sheets).
|
|
1741
|
+
* - `attachments`: Extracted binary assets (images, embedded files).
|
|
1742
|
+
* - `auxiliary`: Out-of-band layout/template elements (headers, footers, slide masters).
|
|
1537
1743
|
*
|
|
1538
1744
|
* @example
|
|
1539
1745
|
* ```typescript
|
|
@@ -1583,6 +1789,12 @@ export interface OfficeParserAST {
|
|
|
1583
1789
|
* @example [{ type: 'paragraph', text: 'Hello' }, { type: 'heading', text: 'Chapter 1' }]
|
|
1584
1790
|
*/
|
|
1585
1791
|
content: OfficeContentNode[];
|
|
1792
|
+
/**
|
|
1793
|
+
* Out-of-band layout and template elements that are not part of the main text flow.
|
|
1794
|
+
* Extracted only if the respective `ignore...` config flags are false.
|
|
1795
|
+
* Contains elements like `headers`, `footers`, and `slideMasters`.
|
|
1796
|
+
*/
|
|
1797
|
+
auxiliary?: OfficeAuxiliaryContent;
|
|
1586
1798
|
/**
|
|
1587
1799
|
* Attachments extracted from the document (images, charts, embedded files).
|
|
1588
1800
|
* Only populated when `config.extractAttachments` is true.
|
|
@@ -1626,5 +1838,6 @@ export interface OfficeParserAST {
|
|
|
1626
1838
|
* const md = await ast.to('md');
|
|
1627
1839
|
* ```
|
|
1628
1840
|
*/
|
|
1629
|
-
to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult
|
|
1841
|
+
to<T extends this, D extends SupportedDestination<T['type']>>(this: T, destination: D, config?: GeneratorConfig<D>): Promise<ConversionResult<D>>;
|
|
1630
1842
|
}
|
|
1843
|
+
export {};
|
package/dist/types.js
CHANGED
|
@@ -74,4 +74,6 @@ var OfficeWarningType;
|
|
|
74
74
|
OfficeWarningType["EMPTY_CHUNK_GENERATED"] = "EMPTY_CHUNK_GENERATED";
|
|
75
75
|
/** A node was skipped because it only contained whitespace */
|
|
76
76
|
OfficeWarningType["WHITESPACE_NODE_SKIPPED"] = "WHITESPACE_NODE_SKIPPED";
|
|
77
|
+
/** The HTML generator containerWidth option is invalid */
|
|
78
|
+
OfficeWarningType["INVALID_CONTAINER_WIDTH"] = "INVALID_CONTAINER_WIDTH";
|
|
77
79
|
})(OfficeWarningType || (exports.OfficeWarningType = OfficeWarningType = {}));
|
package/dist/utils/astUtils.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import { OfficeAttachment, OfficeAuxiliaryContent, OfficeContentNode, OfficeMetadata, OfficeParserAST, OfficeParserConfig, SupportedFileType } from '../types.js';
|
|
2
2
|
/**
|
|
3
3
|
* Creates a fully-featured OfficeParserAST object with conversion methods.
|
|
4
4
|
*
|
|
@@ -13,4 +13,4 @@ import { OfficeParserAST, OfficeContentNode, OfficeMetadata, OfficeAttachment, S
|
|
|
13
13
|
* @param toTextSync - Synchronous text extraction logic (for backward compatibility)
|
|
14
14
|
* @returns An object conforming to OfficeParserAST
|
|
15
15
|
*/
|
|
16
|
-
export declare function createAST(type: SupportedFileType, metadata: OfficeMetadata, content: OfficeContentNode[], attachments: OfficeAttachment[], config: OfficeParserConfig, toTextSync: () => string): OfficeParserAST;
|
|
16
|
+
export declare function createAST(type: SupportedFileType, metadata: OfficeMetadata, content: OfficeContentNode[], attachments: OfficeAttachment[], config: OfficeParserConfig, auxiliary: OfficeAuxiliaryContent | undefined, toTextSync: () => string): OfficeParserAST;
|
package/dist/utils/astUtils.js
CHANGED
|
@@ -16,13 +16,14 @@ const OfficeGenerator_js_1 = require("../OfficeGenerator.js");
|
|
|
16
16
|
* @param toTextSync - Synchronous text extraction logic (for backward compatibility)
|
|
17
17
|
* @returns An object conforming to OfficeParserAST
|
|
18
18
|
*/
|
|
19
|
-
function createAST(type, metadata, content, attachments, config, toTextSync) {
|
|
19
|
+
function createAST(type, metadata, content, attachments, config, auxiliary, toTextSync) {
|
|
20
20
|
return {
|
|
21
21
|
config,
|
|
22
22
|
type,
|
|
23
23
|
metadata,
|
|
24
24
|
content,
|
|
25
25
|
attachments,
|
|
26
|
+
auxiliary,
|
|
26
27
|
warnings: [],
|
|
27
28
|
toText: toTextSync,
|
|
28
29
|
async to(destination, genConfig) {
|
|
@@ -24,3 +24,8 @@ export declare function resolveParserConfig(userConfig?: OfficeParserConfig | Fu
|
|
|
24
24
|
* @returns A fully populated configuration object
|
|
25
25
|
*/
|
|
26
26
|
export declare function resolveGeneratorConfig<D extends string>(destination: D, astConfig?: OfficeParserConfig, userConfig?: GeneratorConfig<D> | FullGeneratorConfig): FullGeneratorConfig;
|
|
27
|
+
/**
|
|
28
|
+
* Validates the containerWidth option for HTML generation.
|
|
29
|
+
* Can be 'auto', a positive number, or a positive CSS length/percentage string.
|
|
30
|
+
*/
|
|
31
|
+
export declare function isValidContainerWidth(width: any): boolean;
|
|
@@ -4,7 +4,10 @@ exports.isFullGeneratorConfig = isFullGeneratorConfig;
|
|
|
4
4
|
exports.isFullParserConfig = isFullParserConfig;
|
|
5
5
|
exports.resolveParserConfig = resolveParserConfig;
|
|
6
6
|
exports.resolveGeneratorConfig = resolveGeneratorConfig;
|
|
7
|
+
exports.isValidContainerWidth = isValidContainerWidth;
|
|
7
8
|
const defaults_js_1 = require("../defaults.js");
|
|
9
|
+
const types_js_1 = require("../types.js");
|
|
10
|
+
const errorUtils_js_1 = require("./errorUtils.js");
|
|
8
11
|
/**
|
|
9
12
|
* Deep clones an object, specifically handling arrays and plain objects.
|
|
10
13
|
*/
|
|
@@ -100,6 +103,7 @@ function resolveGeneratorConfig(destination, astConfig, userConfig) {
|
|
|
100
103
|
// If it's already a full config and we don't need to merge AST config, return it as is.
|
|
101
104
|
// We assume FullGeneratorConfig is already "safe" (references resolved).
|
|
102
105
|
if (isFullGeneratorConfig(userConfig) && !astConfig) {
|
|
106
|
+
validateHtmlConfigWidth(userConfig.htmlConfig, userConfig);
|
|
103
107
|
return userConfig;
|
|
104
108
|
}
|
|
105
109
|
// 1. Start with full defaults (deep cloned to avoid reference sharing)
|
|
@@ -115,7 +119,22 @@ function resolveGeneratorConfig(destination, astConfig, userConfig) {
|
|
|
115
119
|
return;
|
|
116
120
|
for (const key in source) {
|
|
117
121
|
if (source[key] !== undefined) {
|
|
118
|
-
|
|
122
|
+
// Deep merge plain objects (like injections or margin)
|
|
123
|
+
if (typeof source[key] === 'object' &&
|
|
124
|
+
source[key] !== null &&
|
|
125
|
+
!Array.isArray(source[key]) &&
|
|
126
|
+
!(source[key] instanceof Function) &&
|
|
127
|
+
!(source[key] instanceof Date) &&
|
|
128
|
+
!(source[key] instanceof RegExp) &&
|
|
129
|
+
!(source[key] instanceof Buffer)) {
|
|
130
|
+
if (!target[key] || typeof target[key] !== 'object') {
|
|
131
|
+
target[key] = {};
|
|
132
|
+
}
|
|
133
|
+
mergeSubConfig(target[key], source[key]);
|
|
134
|
+
}
|
|
135
|
+
else {
|
|
136
|
+
target[key] = source[key];
|
|
137
|
+
}
|
|
119
138
|
}
|
|
120
139
|
}
|
|
121
140
|
};
|
|
@@ -149,5 +168,40 @@ function resolveGeneratorConfig(destination, astConfig, userConfig) {
|
|
|
149
168
|
// Since FullGeneratorConfig doesn't have an 'mdConfig', we rely on the generator implementation.
|
|
150
169
|
}
|
|
151
170
|
}
|
|
171
|
+
validateHtmlConfigWidth(config.htmlConfig, config);
|
|
152
172
|
return config;
|
|
153
173
|
}
|
|
174
|
+
/**
|
|
175
|
+
* Validates the containerWidth option for HTML generation.
|
|
176
|
+
* Can be 'auto', a positive number, or a positive CSS length/percentage string.
|
|
177
|
+
*/
|
|
178
|
+
function isValidContainerWidth(width) {
|
|
179
|
+
if (width === 'auto')
|
|
180
|
+
return true;
|
|
181
|
+
if (typeof width === 'number') {
|
|
182
|
+
return Number.isFinite(width) && width > 0;
|
|
183
|
+
}
|
|
184
|
+
if (typeof width === 'string') {
|
|
185
|
+
const val = width.trim().toLowerCase();
|
|
186
|
+
if (val === 'auto')
|
|
187
|
+
return true;
|
|
188
|
+
const match = val.match(/^((?:\d*\.)?\d+)(px|%|em|rem|vw|vh|vmin|vmax|ch|in|cm|mm|pt|pc)?$/);
|
|
189
|
+
if (!match)
|
|
190
|
+
return false;
|
|
191
|
+
const numericValue = parseFloat(match[1]);
|
|
192
|
+
return numericValue > 0;
|
|
193
|
+
}
|
|
194
|
+
return false;
|
|
195
|
+
}
|
|
196
|
+
/**
|
|
197
|
+
* Emits a warning and falls back to 'auto' if the HTML containerWidth is invalid.
|
|
198
|
+
*/
|
|
199
|
+
function validateHtmlConfigWidth(htmlConfig, config) {
|
|
200
|
+
if (htmlConfig?.containerWidth !== undefined) {
|
|
201
|
+
const width = htmlConfig.containerWidth;
|
|
202
|
+
if (!isValidContainerWidth(width)) {
|
|
203
|
+
(0, errorUtils_js_1.logWarning)(types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH, config, width);
|
|
204
|
+
htmlConfig.containerWidth = 'auto';
|
|
205
|
+
}
|
|
206
|
+
}
|
|
207
|
+
}
|
package/dist/utils/errorUtils.js
CHANGED
|
@@ -50,7 +50,8 @@ const WARNING_MESSAGES = {
|
|
|
50
50
|
[types_js_1.OfficeWarningType.BUFFER_TYPE_MISMATCH]: (info) => `File content type mismatch: Detected '${info.detected}' but expected/provided '${info.expected}'. Parsing will proceed with '${info.expected}' as requested.`,
|
|
51
51
|
[types_js_1.OfficeWarningType.FILE_TYPE_DETECTION_FAILED]: `Auto-detection of file type failed. This can happen on older Node.js versions with modern file-type versions. Please provide the 'fileType' hint in the configuration if parsing fails.`,
|
|
52
52
|
[types_js_1.OfficeWarningType.EMPTY_CHUNK_GENERATED]: (strategy) => `No chunks generated for document. Check if the document content is compatible with the '${strategy}' strategy.`,
|
|
53
|
-
[types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED]: (nodeType) => `Skipped whitespace-only node of type: ${nodeType}
|
|
53
|
+
[types_js_1.OfficeWarningType.WHITESPACE_NODE_SKIPPED]: (nodeType) => `Skipped whitespace-only node of type: ${nodeType}`,
|
|
54
|
+
[types_js_1.OfficeWarningType.INVALID_CONTAINER_WIDTH]: (val) => `Invalid HTML containerWidth: ${JSON.stringify(val)}. Falling back to "auto". Width must be a positive number, a valid CSS length string (e.g., "900px", "100%", "50vw"), or "auto".`
|
|
54
55
|
};
|
|
55
56
|
/**
|
|
56
57
|
* Creates a formatted warning message for a specific warning type.
|
package/dist/utils/xmlUtils.d.ts
CHANGED
|
@@ -144,6 +144,15 @@ export declare const parseOfficeMetadata: (xmlContent: string) => OfficeMetadata
|
|
|
144
144
|
* ```
|
|
145
145
|
*/
|
|
146
146
|
export declare const parseOOXMLCustomProperties: (xmlContent: string) => Record<string, string | number | boolean | Date>;
|
|
147
|
+
/**
|
|
148
|
+
* Parses OOXML application properties from `docProps/app.xml`.
|
|
149
|
+
*
|
|
150
|
+
* Application properties contain document statistics and application settings.
|
|
151
|
+
*
|
|
152
|
+
* @param xmlContent - Raw XML string from `docProps/app.xml`
|
|
153
|
+
* @returns A record of property name -> typed value
|
|
154
|
+
*/
|
|
155
|
+
export declare const parseOOXMLAppProperties: (xmlContent: string) => Record<string, string | number | boolean>;
|
|
147
156
|
/**
|
|
148
157
|
* Decodes XML entities (standard named entities, decimal, and hexadecimal entities) in a string.
|
|
149
158
|
* Useful when parsing content with regular expressions instead of a full DOM parser.
|