officeparser 7.5.1 → 7.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +19 -6
- package/dist/OfficeConverter.d.ts +2 -2
- package/dist/OfficeParser.d.ts +2 -2
- package/dist/OfficeParser.js +10 -0
- package/dist/cli.js +3 -0
- package/dist/defaults.js +3 -1
- package/dist/generators/ChunkingGenerator.d.ts +2 -1
- package/dist/generators/ChunkingGenerator.js +73 -6
- package/dist/generators/EpubGenerator.js +4 -1
- package/dist/generators/HtmlGenerator.js +117 -29
- package/dist/generators/MarkdownGenerator.js +120 -29
- package/dist/generators/PdfGenerator.js +4 -1
- package/dist/index.d.ts +2 -2
- package/dist/officeparser.browser.d.ts +70 -12
- package/dist/officeparser.browser.iife.js +211 -202
- package/dist/officeparser.browser.mjs +211 -202
- package/dist/officeparser.browser.slim.d.ts +70 -12
- package/dist/officeparser.browser.slim.iife.js +213 -204
- package/dist/officeparser.browser.slim.mjs +213 -204
- package/dist/parsers/HtmlParser.js +206 -38
- package/dist/parsers/MarkdownParser.js +135 -18
- package/dist/parsers/WordParser.js +7 -17
- package/dist/sbom.cdx.json +95 -95
- package/dist/types.d.ts +68 -10
- package/dist/utils/configUtils.js +3 -1
- package/dist/utils/sanitize.d.ts +9 -0
- package/dist/utils/sanitize.js +26 -0
- package/package.json +5 -5
|
@@ -403,6 +403,18 @@ export interface HtmlParserConfig {
|
|
|
403
403
|
* Defaults to false.
|
|
404
404
|
*/
|
|
405
405
|
preserveAttributes?: boolean;
|
|
406
|
+
/**
|
|
407
|
+
* Preserve `<iframe>` embeds that aren't recognized as a known provider (YouTube is always
|
|
408
|
+
* recognized). By default every non-YouTube iframe is dropped, which is a deliberate security
|
|
409
|
+
* posture other consumers rely on; set this to opt back in. `true` preserves any iframe; an
|
|
410
|
+
* array is a hostname allowlist (an entry matches the src's host exactly or as a `.`-suffix,
|
|
411
|
+
* so `"vimeo.com"` also matches `player.vimeo.com`). Preserved iframes become `embed` nodes
|
|
412
|
+
* with `embedType: 'iframe'`; on generation the `src` is still scheme-checked (only http/https
|
|
413
|
+
* survive). This also governs a raw `<iframe>` block encountered in Markdown input.
|
|
414
|
+
*
|
|
415
|
+
* Defaults to false.
|
|
416
|
+
*/
|
|
417
|
+
preserveIframes?: boolean | string[];
|
|
406
418
|
}
|
|
407
419
|
/**
|
|
408
420
|
* Maps an input format string to its corresponding format-specific parser configuration, mirroring
|
|
@@ -859,6 +871,18 @@ export interface HtmlGeneratorConfig {
|
|
|
859
871
|
* Granular injection points for custom HTML, scripts, and styles.
|
|
860
872
|
*/
|
|
861
873
|
injections?: HtmlInjectionConfig;
|
|
874
|
+
/**
|
|
875
|
+
* Carry each rich node's raw source in a `data-*` attribute, with undelimited text content,
|
|
876
|
+
* so attribute-driven structured consumers (rich-text editors, custom viewers) can rehydrate
|
|
877
|
+
* the node from the markup rather than re-parsing the display text. Affects wikilinks
|
|
878
|
+
* (adds `data-wikilink`/`data-target`/`data-alias`), citations (a `<span class="citation">`
|
|
879
|
+
* carrying `data-key` instead of `<cite>`), math (the LaTeX in `data-math`, undelimited) and
|
|
880
|
+
* mermaid (a `<div class="mermaid" data-mermaid>` instead of `<pre><code>`).
|
|
881
|
+
*
|
|
882
|
+
* Off by default; the default output is byte-identical to previous releases. The widened
|
|
883
|
+
* `HtmlParser` reads every shape this emits, so output stays self-round-trippable.
|
|
884
|
+
*/
|
|
885
|
+
sourceAttributes?: boolean;
|
|
862
886
|
}
|
|
863
887
|
/**
|
|
864
888
|
* Configuration options for PDF generation.
|
|
@@ -1074,6 +1098,14 @@ export interface FallbackToHtmlConfig {
|
|
|
1074
1098
|
embeds?: boolean;
|
|
1075
1099
|
/** Multi-line table cell content joined with `<br>` instead of a space. */
|
|
1076
1100
|
cellLineBreaks?: boolean;
|
|
1101
|
+
/**
|
|
1102
|
+
* Inline text color, highlight, and font size via a `<span style="color:...;background-color:...;
|
|
1103
|
+
* font-size:...">` run, which the Markdown parser reads back. These have no Markdown syntax and
|
|
1104
|
+
* are silently lost otherwise. Unlike the other fields this is **off by default even when
|
|
1105
|
+
* `fallbackToHtml` is `true`**, because it changes default output; enable it explicitly with
|
|
1106
|
+
* `fallbackToHtml: { inlineFormatting: true }`.
|
|
1107
|
+
*/
|
|
1108
|
+
inlineFormatting?: boolean;
|
|
1077
1109
|
}
|
|
1078
1110
|
/**
|
|
1079
1111
|
* Configuration options for Markdown generation.
|
|
@@ -1345,6 +1377,16 @@ export interface OfficeChunk {
|
|
|
1345
1377
|
* Supported file types for parsing.
|
|
1346
1378
|
*/
|
|
1347
1379
|
export type SupportedFileType = "docx" | "pptx" | "xlsx" | "odt" | "odp" | "ods" | "pdf" | "rtf" | "md" | "html" | "csv" | "epub";
|
|
1380
|
+
/**
|
|
1381
|
+
* A structural stand-in for the web `Blob`/`File` so `parseOffice`/`convert` accept them in the
|
|
1382
|
+
* browser without pulling the DOM lib into this package's types. Any object with an
|
|
1383
|
+
* `arrayBuffer()` method qualifies. When `name` is present (as on a `File`) it is used only for
|
|
1384
|
+
* extension-based type detection, never as a filesystem path.
|
|
1385
|
+
*/
|
|
1386
|
+
export interface BlobLike {
|
|
1387
|
+
arrayBuffer(): Promise<ArrayBuffer>;
|
|
1388
|
+
name?: string;
|
|
1389
|
+
}
|
|
1348
1390
|
/**
|
|
1349
1391
|
* Types of content nodes in the AST.
|
|
1350
1392
|
*/
|
|
@@ -1586,7 +1628,7 @@ export interface TableMetadata {
|
|
|
1586
1628
|
/** Unique anchor IDs for internal linking. */
|
|
1587
1629
|
anchorIds?: string[];
|
|
1588
1630
|
/**
|
|
1589
|
-
* Layout alignment of the table on the page (e.g.
|
|
1631
|
+
* Layout alignment of the table on the page (e.g. an editor's custom table node).
|
|
1590
1632
|
* @example 'center'
|
|
1591
1633
|
*/
|
|
1592
1634
|
align?: "left" | "center" | "right";
|
|
@@ -1631,12 +1673,12 @@ export interface ImageMetadata {
|
|
|
1631
1673
|
/** Unique anchor IDs for internal linking. */
|
|
1632
1674
|
anchorIds?: string[];
|
|
1633
1675
|
/**
|
|
1634
|
-
* Display width of the image (e.g.
|
|
1676
|
+
* Display width of the image (e.g. an editor's custom image node), as a CSS length or percentage.
|
|
1635
1677
|
* @example "50%"
|
|
1636
1678
|
*/
|
|
1637
1679
|
width?: string;
|
|
1638
1680
|
/**
|
|
1639
|
-
* Layout alignment of the image (e.g.
|
|
1681
|
+
* Layout alignment of the image (e.g. an editor's custom image node).
|
|
1640
1682
|
* @example 'center'
|
|
1641
1683
|
*/
|
|
1642
1684
|
align?: "left" | "center" | "right";
|
|
@@ -1646,14 +1688,19 @@ export interface ImageMetadata {
|
|
|
1646
1688
|
* Markdown has no native syntax for this - see `MarkdownGenerator`'s `embed` case.
|
|
1647
1689
|
*/
|
|
1648
1690
|
export interface EmbedMetadata {
|
|
1649
|
-
/**
|
|
1650
|
-
|
|
1651
|
-
|
|
1652
|
-
|
|
1653
|
-
|
|
1691
|
+
/**
|
|
1692
|
+
* The kind of embed. 'youtube' is recognized from a `data-youtube-video` wrapper or a YouTube
|
|
1693
|
+
* iframe; 'iframe' is a generic preserved iframe (opt-in via `HtmlParserConfig.preserveIframes`).
|
|
1694
|
+
*/
|
|
1695
|
+
embedType: "youtube" | "iframe";
|
|
1696
|
+
/** The provider-specific video ID (e.g. the 11-character YouTube video ID). Absent for generic iframes. */
|
|
1697
|
+
videoId?: string;
|
|
1698
|
+
/** The original/canonical URL of the embedded media, if known. For a generic iframe, its `src`. */
|
|
1654
1699
|
url?: string;
|
|
1655
1700
|
/** Display width, as a CSS length or percentage. */
|
|
1656
1701
|
width?: string;
|
|
1702
|
+
/** Display height, as a CSS length or percentage (generic iframes). */
|
|
1703
|
+
height?: string;
|
|
1657
1704
|
/** Layout alignment of the embed. */
|
|
1658
1705
|
align?: "left" | "center" | "right";
|
|
1659
1706
|
}
|
|
@@ -1737,6 +1784,14 @@ export interface NoteMetadata {
|
|
|
1737
1784
|
anchorIds?: string[];
|
|
1738
1785
|
/** The slide number this note is associated with (used in PowerPoint). */
|
|
1739
1786
|
slideNumber?: number;
|
|
1787
|
+
/**
|
|
1788
|
+
* True for a footnote/endnote definition that no reference points at (an "orphan").
|
|
1789
|
+
* The Markdown parser sets this when it recovers a `[^id]: ...` definition with no matching
|
|
1790
|
+
* `[^id]` reference so the definition is preserved rather than dropped. Generators route such
|
|
1791
|
+
* notes into their footnotes section without a citation marker, and the HTML generator omits
|
|
1792
|
+
* the (otherwise dangling) back-link.
|
|
1793
|
+
*/
|
|
1794
|
+
unreferenced?: boolean;
|
|
1740
1795
|
}
|
|
1741
1796
|
/**
|
|
1742
1797
|
* Metadata for break nodes.
|
|
@@ -1751,8 +1806,11 @@ export interface BreakMetadata {
|
|
|
1751
1806
|
* - 'lastRenderedPage': The editing application has inserted a soft break on the last save.
|
|
1752
1807
|
* - 'textWrapping' (default, assumed when not specified): The next text will be placed on the next line.
|
|
1753
1808
|
* - 'carriageReturn': An explicit carriage return (w:cr) equivalent to a hard line break.
|
|
1809
|
+
* - 'thematic': A thematic break (Markdown `---`/`***`/`___`, HTML `<hr>`) - a horizontal
|
|
1810
|
+
* rule separating sections, distinct from a page break. Emitted as `---` in Markdown and
|
|
1811
|
+
* `<hr>` in HTML.
|
|
1754
1812
|
*/
|
|
1755
|
-
breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn";
|
|
1813
|
+
breakType: "column" | "page" | "lastRenderedPage" | "textWrapping" | "carriageReturn" | "thematic";
|
|
1756
1814
|
/**
|
|
1757
1815
|
* Specifies the location which shall be used as the next available line when breakType
|
|
1758
1816
|
* has a value of 'textWrapping'. Should be ignored for other break types.
|
|
@@ -1774,7 +1832,7 @@ export interface CodeMetadata {
|
|
|
1774
1832
|
/**
|
|
1775
1833
|
* When set, this node is a LaTeX math expression rather than a code block. `node.text`
|
|
1776
1834
|
* holds the bare LaTeX (delimiters excluded); 'inline' round-trips as `$...$`,
|
|
1777
|
-
* 'block' as `$$...$$`. Matches
|
|
1835
|
+
* 'block' as `$$...$$`. Matches attribute-driven editors' math nodes.
|
|
1778
1836
|
*/
|
|
1779
1837
|
math?: "inline" | "block";
|
|
1780
1838
|
}
|
|
@@ -2342,7 +2400,7 @@ export declare class OfficeParser {
|
|
|
2342
2400
|
* const text = ast.toText();
|
|
2343
2401
|
* ```
|
|
2344
2402
|
*/
|
|
2345
|
-
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
2403
|
+
static parseOffice(file: string | Buffer | ArrayBuffer | Uint8Array | BlobLike, configOrCallback?: OfficeParserConfig | ((ast: OfficeParserAST, err?: any) => void), config?: OfficeParserConfig): Promise<OfficeParserAST>;
|
|
2346
2404
|
/**
|
|
2347
2405
|
* Terminates all active OCR workers and cleans up resources.
|
|
2348
2406
|
*
|
|
@@ -2418,7 +2476,7 @@ export declare class OfficeConverter {
|
|
|
2418
2476
|
* });
|
|
2419
2477
|
* ```
|
|
2420
2478
|
*/
|
|
2421
|
-
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
2479
|
+
static convert<F extends string | Buffer | ArrayBuffer | Uint8Array | BlobLike, T extends SupportedFileType = InferFileTypeFromPath<F>, D extends SupportedDestination<T> = SupportedDestination<T>>(file: F, destination: D, config?: OfficeConverterConfig<D, T>): Promise<ConversionResult<D>>;
|
|
2422
2480
|
}
|
|
2423
2481
|
export declare const parseOffice: typeof OfficeParser.parseOffice;
|
|
2424
2482
|
export declare const terminateOcr: typeof OfficeParser.terminateOcr;
|