@docx-editor.dev/docx-to-markdown 0.0.1 → 2.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/docs/fonts.md ADDED
@@ -0,0 +1,172 @@
1
+ # Configure fonts for Markdown layout
2
+
3
+ Use fonts that match the source document when you need accurate page references. The converter measures text to calculate line wrapping, table row heights, and page boundaries before generating Markdown. Markdown does not preserve the DOCX font family; your renderer controls the displayed font.
4
+
5
+ Before you begin, [install the converter](../README.md#install-the-package). The examples use Node.js and DOCX bytes. Save each complete example as `convert.mjs`, place `document.docx` beside it, and run `node convert.mjs`.
6
+
7
+ ## Choose a font source
8
+
9
+ | Your task | Configuration |
10
+ | -------------------------------------------------- | ------------------------------------------------------------------------------------------ |
11
+ | Convert documents that use common Word fonts | Start with the bundled defaults; no font option is required. |
12
+ | Load missing faces from Google Fonts | Set `fallbackFonts: googleFonts()`. Network access is required. |
13
+ | Use your own fonts or replace a bundled substitute | Supply font bytes through `fonts`. |
14
+ | Convert without network access | Use bundled fonts, document-embedded fonts, and local custom fonts. Omit remote resolvers. |
15
+
16
+ Fonts resolve in this order:
17
+
18
+ 1. Your `fonts` configuration or resolvers, with earlier entries taking priority.
19
+ 2. Bundled substitutes.
20
+ 3. Optional `fallbackFonts`.
21
+ 4. Document-embedded fonts for faces that earlier sources did not resolve.
22
+
23
+ A fallback supplies missing faces. It does not replace a face that an earlier source has already resolved. To override a bundled substitute, use `fonts`.
24
+
25
+ ## Start with bundled fonts
26
+
27
+ The converter includes substitutes for common Word fonts:
28
+
29
+ | Word font | Bundled substitute |
30
+ | --------------- | ------------------ |
31
+ | Calibri | Carlito |
32
+ | Cambria | Caladea |
33
+ | Times New Roman | Liberation Serif |
34
+ | Arial | Liberation Sans |
35
+ | Courier New | Liberation Mono |
36
+
37
+ These substitutes target matching character widths for the glyphs they cover. Differences in glyphs and kerning can still change line and page breaks. For matching page references, compare representative exports with Word using the same revision visibility.
38
+
39
+ If a font remains unresolved, the default `fontPolicy: 'best-effort'` allows approximate measurement. Check the resolution report before relying on page citations.
40
+
41
+ ## Add Google Fonts fallback
42
+
43
+ Install the fonts package as a direct dependency of your application:
44
+
45
+ ```sh
46
+ npm install @docx-editor.dev/fonts
47
+ ```
48
+
49
+ This example keeps the bundled substitutes and requests missing faces from the Google Fonts catalog:
50
+
51
+ ```js
52
+ import { readFile } from 'node:fs/promises';
53
+ import { exportMarkdown } from '@docx-editor.dev/docx-to-markdown';
54
+ import { googleFonts } from '@docx-editor.dev/fonts/google';
55
+
56
+ const docxBytes = await readFile('document.docx');
57
+ const result = await exportMarkdown(docxBytes, {
58
+ fallbackFonts: googleFonts(),
59
+ });
60
+
61
+ console.log(result.markdown);
62
+ console.dir(result.fontResolution, { depth: null });
63
+ console.log(result.warnings);
64
+ ```
65
+
66
+ The resolver fetches matching faces from a pinned catalog with verified content hashes. Requests disclose requested font families to the CDN. It cannot supply fonts outside the catalog or guarantee the same pagination as Word.
67
+
68
+ For offline conversion, omit `fallbackFonts` and ensure that any custom resolvers use local data.
69
+
70
+ ## Supply your own fonts
71
+
72
+ Use `fonts` when you have licensed font files that match the document, or when you need to override a bundled substitute. Register each face under the family name requested by the DOCX.
73
+
74
+ This example expects a document that uses Aptos and four licensed files in a `fonts/` directory beside `convert.mjs`: `Aptos.ttf`, `Aptos-Bold.ttf`, `Aptos-Italic.ttf`, and `Aptos-BoldItalic.ttf`. Replace the family name and filenames with your document's fonts. The converter does not include these files.
75
+
76
+ ```js
77
+ import { readFile } from 'node:fs/promises';
78
+ import { createFontSource, exportMarkdown } from '@docx-editor.dev/docx-to-markdown';
79
+
80
+ const faces = [
81
+ { file: 'Aptos.ttf', weight: 400, style: 'normal' },
82
+ { file: 'Aptos-Bold.ttf', weight: 700, style: 'normal' },
83
+ { file: 'Aptos-Italic.ttf', weight: 400, style: 'italic' },
84
+ { file: 'Aptos-BoldItalic.ttf', weight: 700, style: 'italic' },
85
+ ];
86
+
87
+ const sources = [];
88
+ for (const { file, weight, style } of faces) {
89
+ const bytes = new Uint8Array(await readFile(`fonts/${file}`));
90
+ const admitted = createFontSource(bytes, { family: 'Aptos', weight, style });
91
+ if ('failure' in admitted) {
92
+ throw new Error(admitted.failure.diagnostic ?? admitted.failure.reason);
93
+ }
94
+ sources.push(admitted.source);
95
+ }
96
+
97
+ const docxBytes = await readFile('document.docx');
98
+ const result = await exportMarkdown(docxBytes, { fonts: { sources } });
99
+
100
+ console.log(result.markdown);
101
+ console.dir(result.fontResolution, { depth: null });
102
+ ```
103
+
104
+ Repeat registration for other document families. You can combine `fonts` with `fallbackFonts` to resolve faces your local sources do not provide.
105
+
106
+ ## Read the font report
107
+
108
+ `result.fontResolution.families` reports the faces available for each checked family. For example, a family with only its regular face can produce an entry like this; optional identity fields are omitted:
109
+
110
+ ```json
111
+ {
112
+ "family": "Acme Sans",
113
+ "coverage": "partial",
114
+ "faces": [
115
+ {
116
+ "weight": 400,
117
+ "style": "normal",
118
+ "sourceFamily": "Acme Sans",
119
+ "via": "direct"
120
+ }
121
+ ]
122
+ }
123
+ ```
124
+
125
+ This entry means that the bold, italic, and bold-italic faces are missing. Supply those files through `fonts`, or use a fallback that provides them.
126
+
127
+ - `coverage: 'complete'` means all four static faces resolved. It does not certify glyph coverage for every script or that the original Word fonts were used.
128
+ - `coverage: 'partial'` means some faces resolved. `coverage: 'none'` means no faces resolved.
129
+ - `via: 'substitution'` identifies a substituted face. Inspect `sourceFamily` and the face's `substitution` details to understand the choice.
130
+ - `originFailures` records font-source failures, even when another source supplies the missing faces.
131
+
132
+ For a compact view, add this code after either conversion example:
133
+
134
+ ```js
135
+ console.table(
136
+ result.fontResolution?.families.map(({ family, coverage, faces }) => ({
137
+ family,
138
+ coverage,
139
+ faces: faces.map(({ weight, style, via }) => `${weight} ${style} (${via})`).join(', '),
140
+ })) ?? []
141
+ );
142
+ ```
143
+
144
+ Document-aware byte exports return a report. Detached layouts and custom measurers can return `fontResolution: null`; that means evidence is unavailable, not that no fonts were needed.
145
+
146
+ ## Require complete font resolution
147
+
148
+ Add `fontPolicy: 'strict'` to the options in either conversion example to reject font-source failures or missing regular, bold, italic, or bold-italic faces among the checked families. Strict mode can reject an export even if the document only uses regular text or another source recovered from a font-source failure.
149
+
150
+ Strict mode checks font resolution. It does not guarantee the same page breaks as Word, and complete coverage can include substitutes. The resolver checks at most 64 candidate families, prioritizing the body, then headers and footers, then notes. Additional families do not cause strict mode to fail.
151
+
152
+ Save `result.fontResolution`, your package versions, font configuration, document version, and revision mode with exports that require reproducible page citations. Use `toMarkdownJSON(result)` when storing the result as JSON so font failure causes become diagnostic strings.
153
+
154
+ ## Troubleshoot fonts and pagination
155
+
156
+ | Symptom | What to check | What to do |
157
+ | ---------------------------------------------- | ----------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |
158
+ | Page breaks differ from Word | Substituted or incomplete families, and `displayMode` | Supply the source document's fonts through `fonts`, match revision visibility, and compare representative documents. Other layout differences can remain. |
159
+ | `incomplete-font` warning | The family's `coverage` and `faces` | Supply the missing regular, bold, italic, or bold-italic files. Use Google Fonts fallback only if it contains that family and those faces. |
160
+ | `font-origin-failed` warning | `fontResolution.originFailures` | Correct the failing resolver or file path. For remote sources, check network access and the reported failure. |
161
+ | Google Fonts does not change the selected font | Whether an earlier source already resolved the face | Put your replacement font in `fonts`; `fallbackFonts` does not override bundled substitutes. |
162
+ | Strict export fails with `layoutFailed` | The error message and font-source diagnostics | Supply missing faces or fix the failing source. Use best effort only if approximate measurement is acceptable. |
163
+ | Font policy throws `TypeError` | Input type and custom `measurer` | Pass immutable DOCX bytes with the default measurer, or omit `fontPolicy` and `onFontResolution` for a live view or custom measurer. |
164
+ | Deployed export cannot load bundled fonts | Whether deployment retained package assets | Follow the Node.js and Next.js deployment configuration in the integration guide. |
165
+
166
+ For strict failures, use `onFontResolution: (report) => console.dir(report, { depth: null })` to inspect the report before rejection. The exporter does not await callback promises. See the [font resource limits](api.md#font-limits) for file size and memory constraints.
167
+
168
+ ## Next steps
169
+
170
+ - [Deploy the converter](integrations.md#nextjs).
171
+ - [Read page and warning fields](api.md).
172
+ - [Include and deliver images](images.md).
package/docs/images.md ADDED
@@ -0,0 +1,223 @@
1
+ # Include images in Markdown exports
2
+
3
+ Before you begin, [install the converter](../README.md#install-the-package).
4
+ Set `images: true` to include image links and extracted bytes. For this Node.js example, save the code as `convert.mjs`, place `document.docx` beside it, and run `node convert.mjs`:
5
+
6
+ ```ts
7
+ import { readFile } from 'node:fs/promises';
8
+ import { exportMarkdown } from '@docx-editor.dev/docx-to-markdown';
9
+
10
+ const docxBytes = await readFile('document.docx');
11
+ const result = await exportMarkdown(docxBytes, { images: true });
12
+ console.log(result.markdown); // ![Description](media/<digest>.png)
13
+ console.log(result.media); // Unique bytes, paths, URLs, dimensions, and occurrences.
14
+ ```
15
+
16
+ `images: true` and `images: {}` use relative URLs. If you omit `images`, the result contains no image links and returns `media: []`.
17
+
18
+ Each image contains its ID, path, URL, MIME type, bytes, intrinsic dimensions, and occurrences.
19
+ The ID is a hexadecimal SHA-256 digest without the `sha256:` prefix. MIME types and extensions describe the exported bytes, including converted images.
20
+
21
+ Each occurrence records its page, story, source position, displayed dimensions, drawing kind, and alternative text.
22
+ Repeated uses of identical bytes share one asset with separate occurrences. Source offsets use UTF-16 code units.
23
+ Keep the document version with stored citations: occurrence and page identifiers belong to that export snapshot.
24
+ See the [image types](https://github.com/eigenpal/docx-editor/blob/main/packages/docx-to-markdown/src/media-types.ts) for all returned fields.
25
+
26
+ ## Preserve displayed image sizes
27
+
28
+ An image file's `pixelWidth` and `pixelHeight` describe its intrinsic pixels. Word can display the same file at different sizes.
29
+ Use each occurrence's `displayWidthPx` and `displayHeightPx` for the document's displayed size, in CSS pixels at 96 pixels per inch.
30
+ These values retain fractional pixels. `kind` distinguishes inline and anchored drawings.
31
+
32
+ Standard Markdown image syntax has no width or height attributes. To carry each occurrence's size into your Markdown renderer, replace the conversion call in the first example with:
33
+
34
+ ```ts
35
+ const result = await exportMarkdown(docxBytes, {
36
+ images: { syntax: 'html' },
37
+ });
38
+ // <img src="media/<digest>.png" alt="Banner" width="300" height="80">
39
+ ```
40
+
41
+ The converter escapes HTML attributes and rounds the display dimensions to whole CSS pixels. Extents smaller than half a pixel round to zero.
42
+ This option works with local folders, ZIP downloads, and `resolveUrl` for server storage.
43
+ The default `syntax: 'markdown'` keeps standard image links without dimensions.
44
+ Both options return full occurrence metadata.
45
+
46
+ Configure your renderer to parse HTML, sanitize it, and retain `img` attributes `src`, `alt`, `width`, and `height`.
47
+ For example, [`react-markdown`](https://github.com/remarkjs/react-markdown#appendix-a-html-in-markdown) supports `rehype-raw` followed by [`rehype-sanitize`](https://github.com/rehypejs/rehype-sanitize); the default sanitizer retains these attributes.
48
+ If your renderer disables HTML or removes size attributes, use the metadata in a custom preview.
49
+
50
+ For a custom React preview, pass the selected occurrence and its asset's trusted preview URL:
51
+
52
+ ```tsx
53
+ import type { MarkdownImageOccurrence } from '@docx-editor.dev/docx-to-markdown';
54
+
55
+ function ImagePreview({
56
+ imageUrl,
57
+ occurrence,
58
+ }: {
59
+ imageUrl: string;
60
+ occurrence: MarkdownImageOccurrence;
61
+ }) {
62
+ const { displayWidthPx: width, displayHeightPx: height, alt } = occurrence;
63
+ return (
64
+ <img
65
+ src={imageUrl}
66
+ alt={alt}
67
+ width={Math.round(width)}
68
+ height={Math.round(height)}
69
+ style={{
70
+ display: 'inline',
71
+ width,
72
+ maxWidth: '100%',
73
+ height: width === 0 || height === 0 ? height : 'auto',
74
+ ...(width > 0 && height > 0 ? { aspectRatio: `${width} / ${height}` } : {}),
75
+ }}
76
+ />
77
+ );
78
+ }
79
+ ```
80
+
81
+ The explicit aspect ratio preserves Word's displayed proportions when the image shrinks, even if they differ from its intrinsic proportions.
82
+ Use the same styles in a custom Markdown image component, with `width` and `height` from the generated HTML.
83
+ Resolve `src` through your known asset URLs, as shown in the browser example.
84
+
85
+ Do not use an asset's first occurrence to size every image with the same URL.
86
+ Occurrences describe physical layout and include repeated headers; their order is not a Markdown image index.
87
+ Use HTML syntax to attach the correct dimensions directly to each rendered image.
88
+ For a preview built from occurrence metadata, identify the occurrence by its page, story, part, and drawing node.
89
+
90
+ Displayed dimensions describe the drawing's extent before crop and rotation. They do not reproduce cropping, rotation, effects, alignment, or floating text wrapping.
91
+ Anchored images appear at their source paragraph positions in Markdown.
92
+
93
+ ## Save a local folder
94
+
95
+ Use `writeMarkdownBundle()` to save Markdown, JSON metadata, and image files:
96
+
97
+ ```ts
98
+ import { readFile } from 'node:fs/promises';
99
+ import { exportMarkdown } from '@docx-editor.dev/docx-to-markdown';
100
+ import { writeMarkdownBundle } from '@docx-editor.dev/docx-to-markdown/node';
101
+
102
+ const result = await exportMarkdown(await readFile('input.docx'), { images: true });
103
+ await writeMarkdownBundle(result, { directory: './output' });
104
+ ```
105
+
106
+ The helper writes `document.md`, `document.json`, and `media/`.
107
+ The parent directory must exist, and the output directory must be new or empty. Existing files are not overwritten.
108
+ Files use `0o600` permissions where supported. JSON includes page, review, and image metadata without image bytes.
109
+
110
+ ## Save a ZIP file
111
+
112
+ This complete Node.js example writes a portable archive:
113
+
114
+ ```js
115
+ import { readFile, writeFile } from 'node:fs/promises';
116
+ import { createMarkdownZip, exportMarkdown } from '@docx-editor.dev/docx-to-markdown';
117
+
118
+ const result = await exportMarkdown(await readFile('document.docx'), { images: true });
119
+ await writeFile('document.zip', await createMarkdownZip(result));
120
+ ```
121
+
122
+ The ZIP includes `document.md`, `document.json`, and `media/`. Keep generated relative image URLs for ZIP and folder output; exports with hosted URLs cannot be saved as portable bundles.
123
+
124
+ ## Convert and download in the browser
125
+
126
+ Configure your bundler to serve the package's font and WebAssembly assets.
127
+ For a working configuration, see the [browser demo](https://github.com/eigenpal/docx-editor/tree/main/examples/docx-to-markdown).
128
+ Image extraction runs in the browser without a server or storage service.
129
+
130
+ ```ts
131
+ import { exportMarkdown, createMarkdownZip } from '@docx-editor.dev/docx-to-markdown';
132
+
133
+ async function download(file: File) {
134
+ const bytes = new Uint8Array(await file.arrayBuffer());
135
+ const result = await exportMarkdown(bytes, { images: true });
136
+ const zip = await createMarkdownZip(result);
137
+ const url = URL.createObjectURL(new Blob([zip.slice().buffer], { type: 'application/zip' }));
138
+ const link = document.createElement('a');
139
+ link.href = url;
140
+ link.download = file.name.replace(/\.docx$/i, '') + '.zip';
141
+ link.click();
142
+ setTimeout(() => URL.revokeObjectURL(url), 1000);
143
+ }
144
+ ```
145
+
146
+ ZIP and folder output contain the same files. ZIP entry ordering and timestamps are deterministic. ZIP compression yields between bounded chunks; it needs no workers or worker CSP permissions. Both helpers require portable URLs (`image.url === image.path`). For a local bundle, export with `images: true` without `resolveUrl`; hosted-URL results produce `MarkdownBundleError` with `non-portable-image-url`.
147
+
148
+ For an on-screen preview, use your completed browser export as `result` and map paths to object URLs without rewriting Markdown:
149
+
150
+ ```ts
151
+ const imageUrls = new Map(
152
+ result.media.map((image) => [
153
+ image.path,
154
+ URL.createObjectURL(new Blob([image.bytes.slice().buffer], { type: image.mimeType })),
155
+ ])
156
+ );
157
+
158
+ // Give your Markdown renderer a custom image component that looks up only known paths.
159
+ // Continue sanitizing Markdown and HTML. Render SVG through <img>, never inline SVG.
160
+ // On result replacement, unmount, or a discarded async result:
161
+ for (const url of imageUrls.values()) URL.revokeObjectURL(url);
162
+ ```
163
+
164
+ Create URLs outside rendering. Keep them alive while the corresponding preview remains visible. Do not persist or transmit object URLs; they belong to the current browser context. See [MDN object URL lifecycle](https://developer.mozilla.org/en-US/docs/Web/API/URL/createObjectURL_static).
165
+
166
+ ## Return server URLs to a client
167
+
168
+ The application supplies storage and returns its image URLs. The converter does not create a backend or upload files automatically.
169
+
170
+ ```ts
171
+ import { exportMarkdown, toMarkdownJSON } from '@docx-editor.dev/docx-to-markdown';
172
+
173
+ // storage and documentId belong to your application.
174
+ const result = await exportMarkdown(docxBytes, {
175
+ signal: request.signal,
176
+ images: {
177
+ resolveUrl: (image, { signal }) =>
178
+ storage.upload(`documents/${documentId}/${image.path}`, image.bytes, {
179
+ contentType: image.mimeType,
180
+ signal,
181
+ }),
182
+ },
183
+ });
184
+ return Response.json(toMarkdownJSON(result));
185
+ ```
186
+
187
+ `toMarkdownJSON` omits image bytes and converts arbitrary font-origin failure causes to diagnostic strings. It preserves the remaining export fields. Send binary image responses separately. Do not JSON-serialize `Uint8Array` values directly. Metadata stores raw URLs; Markdown destinations are escaped separately. Resolvers may return relative paths or absolute HTTP(S) URLs. Relative paths require application routing or accompanying files.
188
+
189
+ The resolver runs once per unique image, sequentially, after all extraction and budget checks succeed. This makes callback ordering predictable; many remote uploads incur cumulative latency. There are no automatic retries. URLs are finalized before translation, preserving correct review offsets. Applications own completed uploads, cleanup after failure, access control, and signed-URL expiry.
190
+
191
+ Serve untrusted media from an isolated origin without application cookies, or through a controlled response route. Set the actual `Content-Type` and `X-Content-Type-Options: nosniff`. For SVG, use a restrictive response CSP such as `sandbox; default-src 'none'`. Core's image validation is not SVG sanitization. Direct navigation to SVG has different restrictions from SVG displayed in `<img>`; never inline extracted SVG in trusted application markup. See [MDN SVG image restrictions](https://developer.mozilla.org/en-US/docs/Web/SVG/Guides/SVG_as_an_image).
192
+
193
+ ## Sessions, limits, and failure handling
194
+
195
+ For an existing export session, pass image options to `exportMarkdownFrom()`:
196
+
197
+ ```ts
198
+ const result = await exportMarkdownFrom(session, {
199
+ images: { maxTotalBytes: 128 * 1024 * 1024 },
200
+ signal,
201
+ });
202
+ ```
203
+
204
+ `exportMarkdownFrom` leaves the caller's session open. Returned bytes survive disposal. `exportMarkdownLayout` remains synchronous and text-only because detached layouts do not own image bytes.
205
+
206
+ The default limit is 64 MiB of unique extracted bytes, including assets from omitted text-box stories. It is not a total parsing, layout, or transient-memory limit. `maxTotalBytes` must be a positive safe integer. A limit failure rejects the export before any resolver runs. It does not return partial Markdown. Raise the limit or use `images: false`.
207
+
208
+ `MarkdownMediaError.code` is `media-limit`, `url-resolution-failed`, `invalid-image-url`, or `image-bytes-unavailable`. Limit errors include `limitBytes` and `actualBytes`, the total observed when the limit was exceeded. Resolver failures retain `cause` and the asset ID. Cancellation uses the existing `ExportResourceError` with code `aborted`; a resolver receives the signal and pending callback waits end promptly. Cancellation cannot interrupt synchronous parsing/layout or undo completed uploads.
209
+
210
+ The resolver gets a separate byte copy. Mutations cannot corrupt returned assets. Result byte arrays are caller-owned; treat them as read-only to preserve their IDs. Metadata objects and collections are frozen.
211
+
212
+ `MarkdownBundleError.code` is `invalid-media-path`, `duplicate-output-path`, `non-portable-image-url`, `output-not-empty`, `write-failed`, or `archive-failed`. Filesystem failures preserve the cause and relevant path. Helpers remove only files they created during a failed write.
213
+
214
+ ## Output limits
215
+
216
+ Extraction covers validated ready images published by layout, including the body, headers, footers, tables, notes, separators, and nested text boxes. Hidden and revision-suppressed images are not published and are not extracted. Unsupported images, missing resources, external links, and shapes remain omitted with warnings; extraction never fetches document-linked external images. Existing conversion hooks can supply raster replacements for preserved TIFF, EMF, and WMF files.
217
+
218
+ Inline and anchored images appear at their source paragraph positions, including table cells. Anchors within projected field results follow the field; display text cannot provide an exact source position. Decorative images have empty alternative text. Text-box content and note separators remain outside Markdown, although their image bytes and occurrences remain available. Unplaced anchors use an explicit `image-placement-fallback` warning. The logical document and affected page emit separate fallback warnings; page warnings include `pageNumber`. Cropping, rotation, and drawing effects are not reproduced.
219
+
220
+ ## Next steps
221
+
222
+ - [Review export options and warnings](api.md).
223
+ - [Configure your application runtime](integrations.md).
@@ -0,0 +1,112 @@
1
+ # Integrate DOCX to Markdown
2
+
3
+ Before you begin, [install the converter and check the runtime requirements](../README.md#before-you-begin).
4
+ Use `result.markdown` for text and `result.pages` for page citations. See [image delivery](images.md) for browser ZIP downloads, local folders, and server URLs with JSON metadata.
5
+
6
+ ## LangChain
7
+
8
+ Install the packages, then create one LangChain document for each exported page:
9
+
10
+ ```sh
11
+ npm install @docx-editor.dev/docx-to-markdown @docx-editor.dev/core @langchain/core
12
+ ```
13
+
14
+ ```ts
15
+ import { readFile } from 'node:fs/promises';
16
+ import { Document } from '@langchain/core/documents';
17
+ import { exportMarkdown } from '@docx-editor.dev/docx-to-markdown';
18
+
19
+ const filename = 'contract.docx';
20
+ const docxBytes = await readFile(filename);
21
+ const result = await exportMarkdown(docxBytes, { displayMode: 'proposed' });
22
+ const documents = result.pages.map(
23
+ (page) =>
24
+ new Document({
25
+ pageContent: page.markdown,
26
+ metadata: { source: filename, page: page.number },
27
+ })
28
+ );
29
+ ```
30
+
31
+ Pass `documents` to your splitter or vector store. Preserve page metadata when splitting, and add a document version or content hash before storing citations. Review `result.warnings` for omitted content or incomplete fonts.
32
+
33
+ ## Other ingestion pipelines
34
+
35
+ Use this converter for DOCX files in a MarkItDown, Docling, or Unstructured pipeline. Map each page to a text record with source and page metadata.
36
+
37
+ ```ts
38
+ const records = result.pages.map((page) => ({
39
+ text: page.markdown,
40
+ metadata: { source: filename, page: page.number },
41
+ }));
42
+ ```
43
+
44
+ Map these fields to your pipeline's document schema. For Python applications, expose the Node.js converter through a service that returns JSON.
45
+
46
+ ## Next.js
47
+
48
+ Keep the packages external so Node.js can load their bundled font and WebAssembly files. This configuration works with both Webpack and Turbopack.
49
+
50
+ ```js
51
+ // next.config.mjs
52
+ export default {
53
+ serverExternalPackages: [
54
+ '@docx-editor.dev/docx-to-markdown',
55
+ '@docx-editor.dev/core',
56
+ '@docx-editor.dev/fonts',
57
+ ],
58
+ };
59
+ ```
60
+
61
+ ```ts
62
+ // app/api/convert/route.ts
63
+ import {
64
+ DocumentOpenError,
65
+ exportMarkdown,
66
+ toMarkdownJSON,
67
+ } from '@docx-editor.dev/docx-to-markdown';
68
+
69
+ export const runtime = 'nodejs';
70
+
71
+ export async function POST(request: Request) {
72
+ try {
73
+ const result = await exportMarkdown(new Uint8Array(await request.arrayBuffer()), {
74
+ displayMode: 'proposed',
75
+ signal: request.signal,
76
+ resourceTimeoutMs: 15_000,
77
+ });
78
+ return Response.json(toMarkdownJSON(result));
79
+ } catch (error) {
80
+ if (error instanceof DocumentOpenError) {
81
+ return Response.json({ error: 'Document could not be opened' }, { status: 422 });
82
+ }
83
+ throw error;
84
+ }
85
+ }
86
+ ```
87
+
88
+ The example buffers the request. Apply authentication and upload limits before reading the body, and handle resource failures through your application's error handling. `toMarkdownJSON` excludes binary image bytes; if you enable images, deliver their assets separately or use a ZIP.
89
+
90
+ With your development server running, send a local DOCX file and save the response:
91
+
92
+ ```sh
93
+ curl --fail-with-body http://localhost:3000/api/convert \
94
+ -H 'Content-Type: application/vnd.openxmlformats-officedocument.wordprocessingml.document' \
95
+ --data-binary @contract.docx \
96
+ --output contract.json
97
+ ```
98
+
99
+ For self-hosted standalone output, add `output: 'standalone'` to your Next.js configuration and retain the traced package assets. See [Next.js serverExternalPackages](https://nextjs.org/docs/app/api-reference/config/next-config-js/serverExternalPackages).
100
+
101
+ ## Serverless functions and worker threads
102
+
103
+ Use a Node.js runtime that allows WebAssembly and includes the packages' font and WASM assets. The converter uses bundled fonts by default. See [font setup](fonts.md) to configure remote fallback or local custom fonts.
104
+
105
+ `resourceTimeoutMs` limits resource waits, including font provisioning. It is not a deadline for the whole conversion. Parsing and layout run synchronously; `AbortSignal` cannot interrupt JavaScript that is already running. For a hard deadline, run conversion in a worker thread and terminate the worker on timeout. Limit upload size and concurrent conversions to fit your deployment's memory budget.
106
+
107
+ Next.js Edge is unsupported. Validate the production bundle, font coverage, and page counts on your target platform before deployment.
108
+
109
+ ## Next steps
110
+
111
+ - [Deliver image files and JSON metadata](images.md).
112
+ - [Configure export options and handle errors](api.md).
package/package.json CHANGED
@@ -1,24 +1,74 @@
1
1
  {
2
2
  "name": "@docx-editor.dev/docx-to-markdown",
3
- "version": "0.0.1",
4
- "description": "Convert DOCX to Markdown with layout-derived pages, headers, footers, comments, and tracked changes.",
5
- "license": "Apache-2.0",
3
+ "version": "2.19.0",
4
+ "private": false,
6
5
  "publishConfig": {
7
6
  "access": "public"
8
7
  },
8
+ "description": "Convert DOCX to Markdown with layout-derived pages, headers, footers, comments, and tracked changes.",
9
+ "type": "module",
10
+ "sideEffects": false,
11
+ "engines": {
12
+ "node": "^20.16.0 || >=22.3.0"
13
+ },
14
+ "exports": {
15
+ ".": {
16
+ "require": {
17
+ "types": "./types/index.d.cts",
18
+ "default": "./dist/index.cjs"
19
+ },
20
+ "types": "./dist/index.d.ts",
21
+ "import": "./dist/index.js"
22
+ },
23
+ "./package.json": "./package.json",
24
+ "./node": {
25
+ "require": {
26
+ "types": "./types/node.d.cts",
27
+ "default": "./dist/node.cjs"
28
+ },
29
+ "types": "./dist/node.d.ts",
30
+ "import": "./dist/node.js"
31
+ }
32
+ },
9
33
  "files": [
10
- "README.md"
34
+ "dist",
35
+ "types",
36
+ "!dist/metafile-*.json",
37
+ "README.md",
38
+ "docs",
39
+ "THIRD_PARTY_NOTICES.md"
11
40
  ],
41
+ "scripts": {
42
+ "build": "tsup",
43
+ "typecheck": "tsc --noEmit",
44
+ "api:extract": "node ../../scripts/api-extractor.mjs --package @docx-editor.dev/docx-to-markdown --local",
45
+ "api:check": "node ../../scripts/api-extractor.mjs --package @docx-editor.dev/docx-to-markdown",
46
+ "check:consumer": "node ../../scripts/check-markdown-media-consumer.mjs"
47
+ },
48
+ "peerDependencies": {
49
+ "@docx-editor.dev/core": "~2.19.0"
50
+ },
51
+ "dependencies": {
52
+ "@docx-editor.dev/fonts": "~2.19.0",
53
+ "fflate": "^0.8.2"
54
+ },
55
+ "devDependencies": {
56
+ "@happy-dom/global-registrator": "^20.8.3",
57
+ "@docx-editor.dev/core": "workspace:*",
58
+ "micromark": "^4.0.2",
59
+ "micromark-extension-gfm": "^3.0.0",
60
+ "typescript": "^5.3.3"
61
+ },
12
62
  "keywords": [
13
63
  "docx",
14
64
  "markdown",
15
65
  "converter",
16
66
  "headless"
17
67
  ],
68
+ "license": "Apache-2.0",
18
69
  "repository": {
19
70
  "type": "git",
20
71
  "url": "https://github.com/eigenpal/docx-editor.git",
21
72
  "directory": "packages/docx-to-markdown"
22
- },
23
- "homepage": "https://www.docx-editor.dev/docs/2.x/export/markdown"
73
+ }
24
74
  }
@@ -0,0 +1,26 @@
1
+ // CommonJS uses the same contracts as ESM. Explicit type-only import resolution
2
+ // supports TypeScript's Node16 module mode with Core's shared ESM declarations.
3
+ import type * as API from '../dist/index.js' with { 'resolution-mode': 'import' };
4
+ export type * from '../dist/index.js' with { 'resolution-mode': 'import' };
5
+
6
+ export declare const exportMarkdown: typeof API.exportMarkdown;
7
+ export declare const exportMarkdownFrom: typeof API.exportMarkdownFrom;
8
+ export declare const exportMarkdownLayout: typeof API.exportMarkdownLayout;
9
+ export declare const openDocumentForExport: typeof API.openDocumentForExport;
10
+ export declare const createFontSource: typeof API.createFontSource;
11
+ export declare const defineFontResolver: typeof API.defineFontResolver;
12
+ export declare const forEachSemanticDrawing: typeof API.forEachSemanticDrawing;
13
+ export declare const HARD_MAX_AGGREGATE_FONT_BYTES: typeof API.HARD_MAX_AGGREGATE_FONT_BYTES;
14
+ export declare const HARD_MAX_FONT_BYTES: typeof API.HARD_MAX_FONT_BYTES;
15
+ export declare const HARD_MAX_FONT_SOURCES: typeof API.HARD_MAX_FONT_SOURCES;
16
+ export declare const DocumentOpenError: typeof API.DocumentOpenError;
17
+ export type DocumentOpenError = API.DocumentOpenError;
18
+ export declare const ExportResourceError: typeof API.ExportResourceError;
19
+ export type ExportResourceError = API.ExportResourceError;
20
+
21
+ export declare const createMarkdownZip: typeof API.createMarkdownZip;
22
+ export declare const toMarkdownJSON: typeof API.toMarkdownJSON;
23
+ export declare const MarkdownMediaError: typeof API.MarkdownMediaError;
24
+ export type MarkdownMediaError = API.MarkdownMediaError;
25
+ export declare const MarkdownBundleError: typeof API.MarkdownBundleError;
26
+ export type MarkdownBundleError = API.MarkdownBundleError;
@@ -0,0 +1,3 @@
1
+ import type * as API from '../dist/node.js' with { 'resolution-mode': 'import' };
2
+ export type * from '../dist/node.js' with { 'resolution-mode': 'import' };
3
+ export declare const writeMarkdownBundle: typeof API.writeMarkdownBundle;