@docx-editor.dev/docx-to-markdown 0.0.1 → 2.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,251 @@
1
+ /// <reference lib="dom" />
2
+ import { SemanticDrawingVisit, SemanticCommentArtifactRecord, SemanticTrackedChangeArtifactRecord, SemanticReviewArtifactRecord, RevisionDisplayMode } from '@docx-editor.dev/core/layout';
3
+ import { OpenDocumentForExportOptions, FontOrigin, ExportFontResolutionReport } from '@docx-editor.dev/core/export';
4
+
5
+ /** One physical occurrence of an extracted image. @public */
6
+ interface MarkdownImageOccurrence {
7
+ readonly pageNumber: number;
8
+ readonly story: SemanticDrawingVisit['story'];
9
+ readonly rootStory: SemanticDrawingVisit['rootStory'];
10
+ readonly partName: string;
11
+ readonly drawingNodeId: string;
12
+ readonly paragraphId: string;
13
+ /** Source offset within the paragraph, in UTF-16 code units. */
14
+ readonly start: number;
15
+ /** Word's extent width in CSS pixels (96 px per inch), before crop or rotation. */
16
+ readonly displayWidthPx: number;
17
+ /** Word's extent height in CSS pixels (96 px per inch), before crop or rotation. */
18
+ readonly displayHeightPx: number;
19
+ /** Anchored images render at their paragraph position; Markdown does not reproduce text wrapping. */
20
+ readonly kind: 'inline' | 'anchored';
21
+ readonly decorative: boolean;
22
+ readonly alt: string;
23
+ }
24
+ /** Image bytes and provenance supplied to a URL resolver. @public */
25
+ interface MarkdownImageData {
26
+ /** Core's SHA-256 content identifier for the exported bytes. */
27
+ readonly id: string;
28
+ /** Portable relative filename, independent of the resolved URL. */
29
+ readonly path: string;
30
+ readonly mimeType: string;
31
+ /** Owned bytes. Treat as read-only to preserve the content identifier. */
32
+ readonly bytes: Uint8Array;
33
+ readonly byteLength: number;
34
+ /** Intrinsic pixel width of the exported bytes; use occurrence dimensions for display. */
35
+ readonly pixelWidth: number;
36
+ /** Intrinsic pixel height of the exported bytes; use occurrence dimensions for display. */
37
+ readonly pixelHeight: number;
38
+ readonly occurrences: readonly MarkdownImageOccurrence[];
39
+ }
40
+ /** One unique image returned by a Markdown export. @public */
41
+ interface MarkdownImageAsset extends MarkdownImageData {
42
+ /** Raw URL; Markdown output escapes destination syntax separately. */
43
+ readonly url: string;
44
+ }
45
+ /** Portable image extraction and optional application-owned storage. @public */
46
+ interface MarkdownImageOptions {
47
+ /**
48
+ * Default: `markdown` emits standard image links without dimensions.
49
+ * `html` emits an escaped `<img>` with each occurrence's display width and height,
50
+ * rounded to whole CSS pixels. Your renderer must allow HTML and these attributes.
51
+ * Neither syntax reproduces cropping, rotation, or floating text wrapping.
52
+ */
53
+ readonly syntax?: 'markdown' | 'html';
54
+ /**
55
+ * Called sequentially once per unique image, after all extraction succeeds. Receives a
56
+ * separate byte copy; mutations cannot change result bytes. Return a relative path or HTTP(S)
57
+ * URL. The application owns completed uploads, their cleanup, and signed-URL expiry.
58
+ */
59
+ readonly resolveUrl?: (image: MarkdownImageData, context: {
60
+ readonly signal?: AbortSignal;
61
+ }) => string | Promise<string>;
62
+ /** Unique extracted bytes; default 64 MiB. Not a total parsing/layout memory limit. */
63
+ readonly maxTotalBytes?: number;
64
+ }
65
+ /** Projection controls for an already-open export session. @public */
66
+ interface MarkdownProjectionOptions {
67
+ /** `true` or `{}` extracts images with portable relative URLs. Default: false. */
68
+ readonly images?: boolean | MarkdownImageOptions;
69
+ /** Cancels export waits; cannot interrupt synchronous parsing or layout. */
70
+ readonly signal?: AbortSignal;
71
+ }
72
+
73
+ /** Page/story/source provenance for one exported review artifact. @public */
74
+ type MarkdownReviewOccurrence = SemanticReviewArtifactRecord['occurrences'][number];
75
+ /**
76
+ * Normalized DOCX comment, independent of editor UI state.
77
+ *
78
+ * Its ID is opaque and stable only within one {@link MarkdownExportResult}. Never persist the ID
79
+ * across exports; pair citations with the caller's own document version or content hash.
80
+ * @public
81
+ */
82
+ type MarkdownComment = SemanticCommentArtifactRecord;
83
+ /**
84
+ * Normalized DOCX tracked change, independent of editor UI state.
85
+ *
86
+ * Its ID is opaque and stable only within one {@link MarkdownExportResult}. Never persist the ID
87
+ * across exports; pair citations with the caller's own document version or content hash.
88
+ * @public
89
+ */
90
+ type MarkdownTrackedChange = SemanticTrackedChangeArtifactRecord;
91
+ /**
92
+ * Comment or tracked change returned by Markdown export. IDs are opaque and result-local.
93
+ * @public
94
+ */
95
+ type MarkdownReviewArtifact = SemanticReviewArtifactRecord;
96
+ /** A UTF-16 range suitable for slicing the named Markdown projection. @public */
97
+ interface MarkdownReviewRange {
98
+ /** Inclusive start offset in the selected Markdown string. */
99
+ readonly start: number;
100
+ /** Exclusive end offset in the selected Markdown string. */
101
+ readonly end: number;
102
+ /** Offsets use JavaScript string indexing and can be passed directly to `String.slice()`. */
103
+ readonly unit: 'utf16-code-unit';
104
+ /** Whether this is an exact text mapping or the smallest generated construct containing it. */
105
+ readonly precision: MarkdownReviewRangePrecision;
106
+ }
107
+ /** Precision of a source-to-Markdown review range. @public */
108
+ type MarkdownReviewRangePrecision = 'exact' | 'containing-construct';
109
+ /** How much of one Core source occurrence is represented by the returned ranges. @public */
110
+ type MarkdownReviewCoverage = 'complete' | 'partial' | 'none';
111
+ /** Markdown string containing one review binding. @public */
112
+ type MarkdownReviewProjection = {
113
+ /** Selects `MarkdownExportResult.markdown`. */
114
+ readonly kind: 'document';
115
+ } | {
116
+ /** Selects one field on one entry in `MarkdownExportResult.pages`. */
117
+ readonly kind: 'page';
118
+ /** Zero-based index into `MarkdownExportResult.pages`. */
119
+ readonly pageIndex: number;
120
+ /** One-based page number, included for citation and validation. */
121
+ readonly pageNumber: number;
122
+ /** Markdown page field containing the bound ranges. */
123
+ readonly field: 'markdown' | 'headerMarkdown' | 'footerMarkdown';
124
+ };
125
+ /** Why a source occurrence has no range in a particular Markdown projection. @public */
126
+ type MarkdownReviewUnmappedReason = 'not-represented-in-markdown' | 'non-linear-structural-change' | 'omitted-story-content';
127
+ /** Honest snapshot-local mapping from a Core review occurrence to generated Markdown. @public */
128
+ interface MarkdownReviewBinding {
129
+ /** ID of the corresponding artifact in this export result; do not join it across snapshots. */
130
+ readonly artifactId: string;
131
+ /** Discriminant of the corresponding review artifact. */
132
+ readonly artifactKind: MarkdownReviewArtifact['kind'];
133
+ /** Index into this snapshot's immutable artifact `occurrences` array. */
134
+ readonly occurrenceIndex: number;
135
+ /** Generated Markdown string containing this binding. */
136
+ readonly projection: MarkdownReviewProjection;
137
+ /** Ordered, non-overlapping ranges in the selected projection. */
138
+ readonly ranges: readonly MarkdownReviewRange[];
139
+ /** Whether the ranges represent all, some, or none of the source occurrence. */
140
+ readonly coverage: MarkdownReviewCoverage;
141
+ /** Present when the source occurrence has no honest linear range in this projection. */
142
+ readonly unmappedReason?: MarkdownReviewUnmappedReason;
143
+ }
144
+
145
+ /** Markdown emitted for one physical layout page. @public */
146
+ interface MarkdownPage {
147
+ /** Identifier for this page within this export result. */
148
+ readonly id: string;
149
+ /** One-based physical page number. */
150
+ readonly number: number;
151
+ /** Body projection, plus local note definitions or labelled continuation blocks. */
152
+ readonly markdown: string;
153
+ /** Header story for this page, kept separate from logical document content. */
154
+ readonly headerMarkdown: string;
155
+ /** Footer story for this page, kept separate from logical document content. */
156
+ readonly footerMarkdown: string;
157
+ /**
158
+ * Membership view of comments occurring on this page. Each entry is the complete document-wide
159
+ * artifact and can contain occurrences from other pages. For page-local provenance, filter with
160
+ * `occurrence.pageIndex === page.number - 1`.
161
+ */
162
+ readonly comments: readonly MarkdownComment[];
163
+ /**
164
+ * Membership view of tracked changes occurring on this page. Each entry is the complete
165
+ * document-wide artifact and can contain occurrences from other pages. For page-local
166
+ * provenance, filter with `occurrence.pageIndex === page.number - 1`.
167
+ */
168
+ readonly trackedChanges: readonly MarkdownTrackedChange[];
169
+ }
170
+ /** Machine-readable scope of the page numbers returned by this export. @public */
171
+ interface MarkdownPaginationInfo {
172
+ /** Pages come from the docx-editor semantic layout engine, not stale DOCX page-break hints. */
173
+ readonly source: 'layout-engine';
174
+ /** Page numbers describe this exact export result. */
175
+ readonly scope: 'export-snapshot';
176
+ /** Core store revision from which this layout snapshot was produced. */
177
+ readonly layoutRevision: number;
178
+ /** Tracked-change display mode used to paginate and translate this snapshot. */
179
+ readonly displayMode: RevisionDisplayMode;
180
+ }
181
+ /** Content or resources that could not be represented fully. @public */
182
+ interface MarkdownWarning {
183
+ /** Stable code for filtering warnings without parsing messages. */
184
+ readonly code: 'omitted-drawing' | 'omitted-textbox' | 'font-origin-failed' | 'incomplete-font' | 'content-scan-limit' | 'image-placement-fallback';
185
+ /** Human-readable description of the limitation. */
186
+ readonly message: string;
187
+ /** One-based page number, when the warning concerns a drawing occurrence. */
188
+ readonly pageNumber?: number;
189
+ /** Package part when the warning comes from source inspection. */
190
+ readonly partName?: string;
191
+ }
192
+ /** Full logical document plus page-scoped projections. @public */
193
+ interface MarkdownExportResult {
194
+ /** Unique extracted images; empty unless images are enabled. */
195
+ readonly media: readonly MarkdownImageAsset[];
196
+ /** Omitted content and font problems that may affect completeness or pagination. */
197
+ readonly warnings: readonly MarkdownWarning[];
198
+ /** Primary physical page projections, preserving Word layout boundaries and furniture. */
199
+ readonly pages: readonly MarkdownPage[];
200
+ /**
201
+ * Every normalized comment and tracked change, including artifacts without a page occurrence.
202
+ * Artifact IDs are opaque and stable only within this result.
203
+ */
204
+ readonly reviewArtifacts: readonly MarkdownReviewArtifact[];
205
+ /** Markdown offsets valid only within this immutable result, with explicit mapping fidelity. */
206
+ readonly reviewBindings: readonly MarkdownReviewBinding[];
207
+ /** Structured font-resolution evidence, or null when the layout's font origin is unavailable. */
208
+ readonly fontResolution: ExportFontResolutionReport | null;
209
+ /** How page numbers and tracked changes were projected for this result. */
210
+ readonly pagination: MarkdownPaginationInfo;
211
+ /** Convenience logical Markdown with split records joined and repeated furniture excluded. */
212
+ readonly markdown: string;
213
+ }
214
+ /** One caller-controlled font origin used for headless pagination. @public */
215
+ type MarkdownFontOrigin = FontOrigin;
216
+ /** One font origin, or an ordered first-wins list of origins. @public */
217
+ type MarkdownFontsSource = MarkdownFontOrigin | readonly MarkdownFontOrigin[];
218
+ /** Layout controls for a reusable Markdown export session. @public */
219
+ interface OpenMarkdownDocumentForExportOptions extends OpenDocumentForExportOptions {
220
+ /**
221
+ * Retains incremental state for a live view or caller-measured session. Document-aware byte
222
+ * sessions are immutable and reject `true` instead of silently ignoring it.
223
+ */
224
+ readonly reuseAcrossRevisions?: boolean;
225
+ /**
226
+ * Caller-supplied font bytes or resolvers, in first-wins order. These take precedence over the
227
+ * package's bundled metric-compatible Word substitutes. A resolver is invoked after the DOCX is
228
+ * parsed with the bounded family list layout can render. Custom origins require immutable DOCX
229
+ * bytes; for a live view, supply a revision-stable `measurer` instead. An explicit measurer takes
230
+ * precedence and font origins are not invoked.
231
+ */
232
+ readonly fonts?: MarkdownFontsSource;
233
+ /**
234
+ * Opt-in origins consulted only after caller fonts and bundled substitutes. Put
235
+ * `googleFonts()` here to fetch catalogued families the local origins cannot paint. This has the
236
+ * same immutable-bytes restriction and explicit-measurer precedence as {@link fonts}.
237
+ */
238
+ readonly fallbackFonts?: MarkdownFontsSource;
239
+ /** `strict` refuses failed origins or any requested family missing one of four static faces. */
240
+ readonly fontPolicy?: 'best-effort' | 'strict';
241
+ /**
242
+ * Fire-and-forget evidence for the exact direct/substituted faces behind page breaks. Returned
243
+ * promises are observed for rejection but do not delay export.
244
+ */
245
+ readonly onFontResolution?: (report: ExportFontResolutionReport) => void;
246
+ }
247
+ /** Layout and resource controls for one-shot Markdown export. @public */
248
+ interface MarkdownExportOptions extends OpenMarkdownDocumentForExportOptions, MarkdownProjectionOptions {
249
+ }
250
+
251
+ export type { MarkdownImageAsset as M, OpenMarkdownDocumentForExportOptions as O, MarkdownExportResult as a, MarkdownExportOptions as b, MarkdownProjectionOptions as c, MarkdownComment as d, MarkdownFontOrigin as e, MarkdownFontsSource as f, MarkdownImageData as g, MarkdownImageOccurrence as h, MarkdownImageOptions as i, MarkdownPage as j, MarkdownPaginationInfo as k, MarkdownReviewArtifact as l, MarkdownReviewBinding as m, MarkdownReviewCoverage as n, MarkdownReviewOccurrence as o, MarkdownReviewProjection as p, MarkdownReviewRange as q, MarkdownReviewRangePrecision as r, MarkdownReviewUnmappedReason as s, MarkdownTrackedChange as t, MarkdownWarning as u };
package/dist/node.cjs ADDED
@@ -0,0 +1 @@
1
+ 'use strict';var chunkNP5KYCWK_cjs=require('./chunk-NP5KYCWK.cjs'),promises=require('fs/promises'),path=require('path'),fs=require('fs');function C(e){return typeof e=="object"&&e!==null&&"code"in e&&e.code==="ENOENT"}async function f(e){let i=path.parse(e).root,n=i;for(let a of e.slice(i.length).split(path.sep).filter(Boolean)){n=path.join(n,a);let s=await promises.lstat(n);if(s.isSymbolicLink()||!s.isDirectory())throw new chunkNP5KYCWK_cjs.b("write-failed","Output directories must be real directories, not symbolic links.",{path:n})}}async function j(e,i){let n=chunkNP5KYCWK_cjs.d(e);if(typeof i?.directory!="string"||!i.directory.trim())throw new TypeError("directory must be a nonempty path");let a=path.resolve(i.directory),s=[],w=[];try{let r=await promises.realpath(path.dirname(a)),t=path.join(r,path.basename(a));await f(r);try{await promises.mkdir(t),w.push(t);}catch(o){if(!(typeof o=="object"&&o&&"code"in o&&o.code==="EEXIST"))throw o}if(await f(t),(await promises.readdir(t)).length)throw new chunkNP5KYCWK_cjs.b("output-not-empty","Choose a new or empty directory; existing files are never overwritten.",{path:t});if(e.media.length){let o=path.join(t,"media");await promises.mkdir(o),w.push(o);}for(let[o,k]of n){let p=path.join(t,o);await f(path.dirname(p));let m=await promises.open(p,fs.constants.O_WRONLY|fs.constants.O_CREAT|fs.constants.O_EXCL|fs.constants.O_NOFOLLOW,384);s.push(p);try{await m.writeFile(k);}finally{await m.close();}}}catch(r){for(let t of s.reverse())try{await promises.unlink(t);}catch{}for(let t of w.reverse())try{await promises.rmdir(t);}catch{}throw r instanceof chunkNP5KYCWK_cjs.b?r:new chunkNP5KYCWK_cjs.b("write-failed",C(r)?"The parent output directory must exist.":"Could not write the Markdown bundle.",{path:a,cause:r})}}exports.writeMarkdownBundle=j;
@@ -0,0 +1,14 @@
1
+ /// <reference lib="dom" />
2
+ import { a as MarkdownExportResult } from './markdown-types-Cn-MizR0.cjs';
3
+ import '@docx-editor.dev/core/layout';
4
+ import '@docx-editor.dev/core/export';
5
+
6
+ /** Output folder for a portable bundle. @public */
7
+ interface WriteMarkdownBundleOptions {
8
+ /** New or existing empty directory. Files are never overwritten. */
9
+ readonly directory: string;
10
+ }
11
+ /** Write document.md, document.json, and media/ into a new or empty directory. @public */
12
+ declare function writeMarkdownBundle(result: MarkdownExportResult, options: WriteMarkdownBundleOptions): Promise<void>;
13
+
14
+ export { type WriteMarkdownBundleOptions, writeMarkdownBundle };
package/dist/node.d.ts ADDED
@@ -0,0 +1,14 @@
1
+ /// <reference lib="dom" />
2
+ import { a as MarkdownExportResult } from './markdown-types-Cn-MizR0.js';
3
+ import '@docx-editor.dev/core/layout';
4
+ import '@docx-editor.dev/core/export';
5
+
6
+ /** Output folder for a portable bundle. @public */
7
+ interface WriteMarkdownBundleOptions {
8
+ /** New or existing empty directory. Files are never overwritten. */
9
+ readonly directory: string;
10
+ }
11
+ /** Write document.md, document.json, and media/ into a new or empty directory. @public */
12
+ declare function writeMarkdownBundle(result: MarkdownExportResult, options: WriteMarkdownBundleOptions): Promise<void>;
13
+
14
+ export { type WriteMarkdownBundleOptions, writeMarkdownBundle };
package/dist/node.js ADDED
@@ -0,0 +1 @@
1
+ import {d,b}from'./chunk-CYF5JG4K.js';import {realpath,mkdir,readdir,open,unlink,rmdir,lstat}from'fs/promises';import {resolve,dirname,join,basename,parse,sep}from'path';import {constants}from'fs';function C(e){return typeof e=="object"&&e!==null&&"code"in e&&e.code==="ENOENT"}async function f(e){let i=parse(e).root,n=i;for(let a of e.slice(i.length).split(sep).filter(Boolean)){n=join(n,a);let s=await lstat(n);if(s.isSymbolicLink()||!s.isDirectory())throw new b("write-failed","Output directories must be real directories, not symbolic links.",{path:n})}}async function j(e,i){let n=d(e);if(typeof i?.directory!="string"||!i.directory.trim())throw new TypeError("directory must be a nonempty path");let a=resolve(i.directory),s=[],w=[];try{let r=await realpath(dirname(a)),t=join(r,basename(a));await f(r);try{await mkdir(t),w.push(t);}catch(o){if(!(typeof o=="object"&&o&&"code"in o&&o.code==="EEXIST"))throw o}if(await f(t),(await readdir(t)).length)throw new b("output-not-empty","Choose a new or empty directory; existing files are never overwritten.",{path:t});if(e.media.length){let o=join(t,"media");await mkdir(o),w.push(o);}for(let[o,k]of n){let p=join(t,o);await f(dirname(p));let m=await open(p,constants.O_WRONLY|constants.O_CREAT|constants.O_EXCL|constants.O_NOFOLLOW,384);s.push(p);try{await m.writeFile(k);}finally{await m.close();}}}catch(r){for(let t of s.reverse())try{await unlink(t);}catch{}for(let t of w.reverse())try{await rmdir(t);}catch{}throw r instanceof b?r:new b("write-failed",C(r)?"The parent output directory must exist.":"Could not write the Markdown bundle.",{path:a,cause:r})}}export{j as writeMarkdownBundle};
package/docs/api.md ADDED
@@ -0,0 +1,330 @@
1
+ # DOCX to Markdown API reference
2
+
3
+ Use this reference to choose an export function, configure conversion, and inspect the result.
4
+ For installation and a first export, see the [package quickstart](../README.md).
5
+
6
+ ## Public interface
7
+
8
+ ```ts
9
+ exportMarkdown(
10
+ source: Uint8Array | HeadlessDocumentView,
11
+ options?: MarkdownExportOptions
12
+ ): Promise<MarkdownExportResult>;
13
+
14
+ openDocumentForExport(
15
+ source: Uint8Array | HeadlessDocumentView,
16
+ options?: OpenMarkdownDocumentForExportOptions
17
+ ): Promise<OpenMarkdownDocumentForExportResult>;
18
+
19
+ exportMarkdownFrom(
20
+ session: ExportSession,
21
+ options?: MarkdownProjectionOptions
22
+ ): Promise<MarkdownExportResult>;
23
+
24
+ exportMarkdownLayout(layout: ExportSemanticLayout): MarkdownExportResult;
25
+ ```
26
+
27
+ Use `exportMarkdown` for a single export. To reuse or inspect a layout, use `openDocumentForExport` and `exportMarkdownFrom`. To convert a layout after disposing its session, use `exportMarkdownLayout`.
28
+
29
+ ### Result shape
30
+
31
+ ```ts
32
+ interface MarkdownExportResult {
33
+ /** Unique extracted assets, or [] when images are disabled. */
34
+ readonly media: readonly MarkdownImageAsset[];
35
+ /** Omitted content and font problems. */
36
+ readonly warnings: readonly MarkdownWarning[];
37
+ /** Primary output: physical page projections with page furniture and provenance. */
38
+ readonly pages: readonly MarkdownPage[];
39
+ /** All comments and tracked changes, including artifacts without a page occurrence. */
40
+ readonly reviewArtifacts: readonly MarkdownReviewArtifact[];
41
+ /** Offsets and artifact IDs valid only within this immutable result. */
42
+ readonly reviewBindings: readonly MarkdownReviewBinding[];
43
+ /** Font families and faces used for pagination, or null when the layout's font origin is unavailable. */
44
+ readonly fontResolution: ExportFontResolutionReport | null;
45
+ /** How this result's pages and revision content were produced. */
46
+ readonly pagination: {
47
+ readonly source: 'layout-engine';
48
+ readonly scope: 'export-snapshot';
49
+ readonly layoutRevision: number;
50
+ readonly displayMode: 'all-markup' | 'proposed' | 'original';
51
+ };
52
+ /** Convenience logical, full-document Markdown. */
53
+ readonly markdown: string;
54
+ }
55
+
56
+ interface MarkdownPage {
57
+ /** Identifier for this page within this export result. */
58
+ readonly id: string;
59
+ /** One-based physical page number. */
60
+ readonly number: number;
61
+ /** Body content and page-local note definitions or continuations. */
62
+ readonly markdown: string;
63
+ /** Header and footer are separate from logical document content. */
64
+ readonly headerMarkdown: string;
65
+ readonly footerMarkdown: string;
66
+ /** Membership views: complete artifacts with at least one occurrence on this page. */
67
+ readonly comments: readonly MarkdownComment[];
68
+ readonly trackedChanges: readonly MarkdownTrackedChange[];
69
+ }
70
+
71
+ interface MarkdownReviewBinding {
72
+ readonly artifactId: string;
73
+ readonly artifactKind: 'comment' | 'tracked-change';
74
+ readonly occurrenceIndex: number;
75
+ readonly coverage: 'complete' | 'partial' | 'none';
76
+ readonly projection:
77
+ | { readonly kind: 'document' }
78
+ | {
79
+ readonly kind: 'page';
80
+ readonly pageIndex: number;
81
+ readonly pageNumber: number;
82
+ readonly field: 'markdown' | 'headerMarkdown' | 'footerMarkdown';
83
+ };
84
+ readonly ranges: readonly {
85
+ readonly start: number;
86
+ readonly end: number;
87
+ readonly unit: 'utf16-code-unit';
88
+ readonly precision: 'exact' | 'containing-construct';
89
+ }[];
90
+ readonly unmappedReason?:
91
+ | 'not-represented-in-markdown'
92
+ | 'non-linear-structural-change'
93
+ | 'omitted-story-content';
94
+ }
95
+ ```
96
+
97
+ `fontResolution` lists requested families, resolved and substituted faces, coverage (`complete`, `partial`, or `none`), and nonfatal `originFailures`. Document-aware byte sessions return this report. Reusable sessions also expose it as `session.fontResolution`. It is `null` for detached layouts, custom measurers, ordinary Core sessions, and live views using shared shaping. Fatal failures throw `DocumentOpenError` or `ExportResourceError`.
98
+
99
+ `pages` contains the page output from the editor's layout engine. Page numbers and IDs apply only to this export. Different fonts or Microsoft Word versions can produce different page breaks.
100
+
101
+ `pagination` records the layout source, export scope, Core revision, and tracked-change display mode.
102
+
103
+ For citations that must survive storage or document updates, retain your own document version or content hash alongside the page number. For example:
104
+
105
+ ```ts
106
+ const citation = {
107
+ documentVersion: contract.sha256,
108
+ engineVersion: applicationBuild.docxEditorVersion,
109
+ pageNumber: result.pages[11]!.number,
110
+ pageId: result.pages[11]!.id,
111
+ pagination: result.pagination,
112
+ };
113
+ ```
114
+
115
+ `result.markdown` contains the whole document. It joins content split across pages and excludes repeated headers and footers. Use `pages` when you need page citations.
116
+
117
+ ### Comments and tracked changes
118
+
119
+ The exporter returns comments and revision metadata separately from Markdown. Review IDs are opaque and valid only within one export. For stored citations, include your own document version or content hash.
120
+
121
+ `page.comments` and `page.trackedChanges` contain complete artifacts with occurrences on that page. Their `occurrences` can include other pages. Filter them to avoid double counting:
122
+
123
+ ```ts
124
+ const localComments = page.comments.flatMap((artifact) =>
125
+ artifact.occurrences
126
+ .filter(({ physicalPageNumber }) => physicalPageNumber === page.number)
127
+ .map((occurrence) => ({ artifact, occurrence }))
128
+ );
129
+ ```
130
+
131
+ Page artifacts can occur in the body, headers, footers, footnotes, endnotes, or note separators. `result.reviewArtifacts` contains all artifacts, including those without a page occurrence.
132
+
133
+ `result.reviewBindings` connects each occurrence to offsets in `result.markdown`, `page.markdown`, `page.headerMarkdown`, or `page.footerMarkdown`. Offsets use JavaScript UTF-16 string indexing and can be passed directly to `slice()`:
134
+
135
+ ```ts
136
+ for (const binding of result.reviewBindings) {
137
+ const output =
138
+ binding.projection.kind === 'document'
139
+ ? result.markdown
140
+ : result.pages[binding.projection.pageIndex]![binding.projection.field];
141
+
142
+ for (const range of binding.ranges) {
143
+ console.log(output.slice(range.start, range.end));
144
+ }
145
+ }
146
+ ```
147
+
148
+ For source-aligned edits, require `coverage === 'complete'` and `precision === 'exact'`. For citations or display, you can use partial or `containing-construct` bindings if you retain that precision information. Existing bindings with no mapped ranges include an `unmappedReason`.
149
+
150
+ Artifacts without occurrences have no bindings or `unmappedReason`. These include orphan comments and comments entirely hidden by the selected revision mode. Inspect `result.reviewArtifacts` as well as `result.reviewBindings` to retain them.
151
+
152
+ Artifact IDs, occurrence indexes, page IDs, and offsets are valid only within this export.
153
+
154
+ Tracked changes also participate in layout through `displayMode`: `all-markup` (default) keeps inserted and deleted text visible, `proposed` includes pending insertions and hides pending deletions, and `original` hides pending insertions and shows pending deletions. Revision mode applies to the whole document.
155
+
156
+ ## Images and portable delivery
157
+
158
+ Enable `images: true` for relative image links and `result.media` bytes. Use `{ images: { resolveUrl, maxTotalBytes } }` for custom delivery. The default extracted-byte limit is 64 MiB; image extraction is opt-in. `exportMarkdownFrom(session, options)` accepts the same image options and an abort signal.
159
+
160
+ Set `images: { syntax: 'html' }` to emit `<img>` tags with each occurrence's displayed width and height in whole CSS pixels.
161
+ The default `syntax: 'markdown'` emits standard image links without size attributes.
162
+ Your renderer must support sanitized HTML and retain `width` and `height`.
163
+ Occurrences expose exact `displayWidthPx`, `displayHeightPx`, and `kind` (`inline` or `anchored`).
164
+ Asset `pixelWidth` and `pixelHeight` describe the image bytes, not their displayed size.
165
+ Crop, rotation, and floating text wrapping are not reproduced.
166
+
167
+ `createMarkdownZip(result)` and `toMarkdownJSON(result)` are exported from the main package. `writeMarkdownBundle(result, { directory })` comes from `@docx-editor.dev/docx-to-markdown/node`. See [image APIs, errors, ownership, and runnable workflows](images.md).
168
+
169
+ ## Options
170
+
171
+ `MarkdownExportOptions` contains layout and resource controls for the export snapshot.
172
+
173
+ | Option | Meaning |
174
+ | ----------------------- | --------------------------------------------------------------------------------------------- |
175
+ | `displayMode` | Tracked-change projection: `all-markup` (default), `proposed`, or `original`. |
176
+ | `signal` | Aborts resource waits and later layout work. |
177
+ | `resourceTimeoutMs` | Deadline applied separately to initial font provisioning and each layout resource wait. |
178
+ | `reuseAcrossRevisions` | Retains state for live/caller-measured sessions; document-aware byte sessions reject `true`. |
179
+ | `fonts` | Font configuration or resolver. Earlier entries win. Requires immutable DOCX bytes. |
180
+ | `fallbackFonts` | Fallback after bundled fonts. Requires DOCX bytes. Accepts `googleFonts()`. |
181
+ | `fontPolicy` | `best-effort` (default), or `strict` to require all four static faces and no origin failures. |
182
+ | `onFontResolution` | Receives the font-resolution report. |
183
+ | `measurer` | Custom text measurer. Overrides font resolution. |
184
+ | `producer` | Stable identity for a host-owned measurer and its cache entries. |
185
+ | `imageDecodePort` | Custom image metadata decoder. Defaults to the Node.js decoder. |
186
+ | `convertPreservedImage` | Converts preserved EMF, WMF, or TIFF bytes to a supported raster format. |
187
+
188
+ `fontPolicy` and `onFontResolution` require immutable DOCX bytes with the default document-aware font resolution. Both options throw `TypeError` when used with a live `HeadlessDocumentView` or combined with a custom `measurer`, including an explicit `fontPolicy: 'best-effort'`.
189
+
190
+ ### Tracked changes
191
+
192
+ To show the proposed text, set `displayMode` to `proposed`. This changes the export projection; it does not accept changes in the DOCX. Review artifacts remain available:
193
+
194
+ ```ts
195
+ const proposed = await exportMarkdown(docxBytes, {
196
+ displayMode: 'proposed',
197
+ });
198
+ ```
199
+
200
+ ## Reuse an export session
201
+
202
+ ```ts
203
+ import { readFile } from 'node:fs/promises';
204
+ import { exportMarkdownFrom, openDocumentForExport } from '@docx-editor.dev/docx-to-markdown';
205
+
206
+ const docxBytes = await readFile('document.docx');
207
+ const opened = await openDocumentForExport(docxBytes, {
208
+ displayMode: 'all-markup',
209
+ });
210
+
211
+ if (!opened.ok) {
212
+ throw new Error(
213
+ `DOCX was refused: ${opened.reason}${opened.detail ? ` (${opened.detail})` : ''}`
214
+ );
215
+ }
216
+
217
+ try {
218
+ // Resource settlement and layout are cached by the session.
219
+ const layout = await opened.session.layout();
220
+ console.log(`Pages: ${layout.pages.length}`);
221
+
222
+ const first = await exportMarkdownFrom(opened.session);
223
+ const second = await exportMarkdownFrom(opened.session);
224
+ console.log(first.pages.length === second.pages.length);
225
+ } finally {
226
+ opened.session.dispose();
227
+ }
228
+ ```
229
+
230
+ Call `dispose()` on reusable sessions to release caches and pending resource work. Repeated calls are safe. Live views can retain state across revisions; byte sources use one-shot caching by default.
231
+
232
+ For a single export, use `exportMarkdown(docxBytes)`. To keep a layout without retaining session resources, obtain it before disposal, then pass it to `exportMarkdownLayout`. Layouts remain valid after disposal.
233
+
234
+ ## Images
235
+
236
+ Images affect page boundaries and are omitted from Markdown unless you enable `images`. If omitting an inline image would join words, the exporter inserts a space.
237
+
238
+ ## Errors and cancellation
239
+
240
+ For malformed or unsupported DOCX input, `exportMarkdown` throws `DocumentOpenError`. `openDocumentForExport` returns `{ ok: false, reason, detail }` instead.
241
+
242
+ Both workflows throw `TypeError` for unsupported combinations of [export options](#options), such as a custom `measurer` combined with `fontPolicy` or `onFontResolution`.
243
+
244
+ Both workflows can throw `ExportResourceError` with one of these codes:
245
+
246
+ - `aborted`, `timedOut`, `nonConvergent`, or `disposed`.
247
+ - `layoutInvariant`: document geometry exceeds the engine's page model.
248
+ - `layoutFailed`: another layout or host integration failure.
249
+
250
+ The layout errors retain the original diagnostic as `cause`. Aborting a reusable session releases its resources and prevents reuse. Calling `dispose()` afterward is safe.
251
+
252
+ ```ts
253
+ import {
254
+ DocumentOpenError,
255
+ ExportResourceError,
256
+ exportMarkdown,
257
+ } from '@docx-editor.dev/docx-to-markdown';
258
+
259
+ const controller = new AbortController();
260
+
261
+ try {
262
+ const result = await exportMarkdown(docxBytes, {
263
+ signal: controller.signal,
264
+ resourceTimeoutMs: 30_000,
265
+ });
266
+ console.log(result.markdown);
267
+ } catch (error) {
268
+ if (error instanceof DocumentOpenError) {
269
+ console.error(error.reason, error.detail);
270
+ } else if (error instanceof ExportResourceError) {
271
+ console.error(error.code, error.message);
272
+ } else {
273
+ throw error;
274
+ }
275
+ }
276
+ ```
277
+
278
+ ## Layout and fonts
279
+
280
+ Core resolves fonts, images, document geometry, and `displayMode` before pagination. Markdown uses that immutable layout. `signal` and `resourceTimeoutMs` cover resource processing.
281
+
282
+ See [Configure fonts for Markdown layout](fonts.md) for resolution order, bundled substitutes, separate Google Fonts and custom-font examples, strict mode, and troubleshooting.
283
+
284
+ Custom `fonts` and `fallbackFonts` require immutable DOCX bytes. For a live `HeadlessDocumentView`, use a host-owned, revision-stable `measurer` with a stable `producer`. A custom measurer takes precedence and bypasses both font options. Omit `fontPolicy` and `onFontResolution` when using a custom measurer or live view; these combinations throw `TypeError`.
285
+
286
+ Use `onFontResolution` to observe resolved and substituted faces. The exporter does not await callback promises. Callback errors are logged without failing the export. To wait for report storage, save `result.fontResolution` after export and await that operation.
287
+
288
+ Use `googleFonts({ onFailure })` to log fallback failures. Failed or aborted Google Fonts requests are not cached.
289
+
290
+ ### Font limits
291
+
292
+ The package exports these limits:
293
+
294
+ - `HARD_MAX_FONT_BYTES`: 64 MiB per face.
295
+ - `HARD_MAX_FONT_SOURCES`: 256 sources per composition.
296
+ - `HARD_MAX_AGGREGATE_FONT_BYTES`: 128 MiB per composition and across active document font leases.
297
+
298
+ Invalid or oversized origins are reported and skipped. Exceeding the process-wide lease budget causes `layoutFailed`. Return only requested faces from resolvers, limit concurrent exports, and dispose reusable sessions promptly. Use the exported constants in your code.
299
+
300
+ Document-embedded fonts are admitted after caller fonts, bundled substitutes, and optional fallback origins, using the same mapper as the browser editor. Regular, bold, italic, and bold-italic embedded faces must pass the shared font limits.
301
+
302
+ ## Warnings
303
+
304
+ `result.warnings` contains `{ code, message, pageNumber?, partName? }` objects. Codes are stable; messages are for display. `pageNumber` is one-based when present.
305
+
306
+ | Code | Meaning |
307
+ | -------------------- | -------------------------------------------------------------------- |
308
+ | `omitted-drawing` | Images or shapes are omitted. |
309
+ | `omitted-textbox` | Text box content is omitted. |
310
+ | `font-origin-failed` | A font source failed to load. |
311
+ | `content-scan-limit` | Source inspection reached its bound; additional omissions may exist. |
312
+ | `incomplete-font` | A requested family lacks one or more faces; pagination may differ. |
313
+
314
+ Drawing warnings are grouped by category and page, including headers, footers, and notes. Unsupported legacy images, shapes, and text boxes are reported from the source package with `partName`, without a page number. Each warning identifies omitted content.
315
+
316
+ ## Markdown limitations
317
+
318
+ - Use `pages` for page output and `markdown` for the whole document.
319
+ - Page headers and footers are returned separately per page.
320
+ - Merged table cells are flattened.
321
+ - Nested tables use inline HTML (`<table>`, `<tr>`, `<td>`, and `<th>`). Cells retain inline Markdown.
322
+ - Images affect layout. They are omitted by default; `images: true` includes supported images and returns their bytes.
323
+ - Anchored text-box text is omitted because it has no unambiguous linear position; comments and tracked changes inside it remain available as page artifacts with exact text-box provenance.
324
+ - Office Math uses the core semantic equation fallback.
325
+ - A note continued without its reference is emitted as a labeled continuation block in page Markdown.
326
+
327
+ ## See also
328
+
329
+ - [Image extraction and delivery](images.md).
330
+ - [Application integrations](integrations.md).