word-to-markdown 0.2.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +123 -18
- package/build/cli.js +59 -18
- package/build/cli.js.map +1 -1
- package/build/main.d.ts +88 -0
- package/build/main.js +558 -365
- package/build/main.js.map +1 -1
- package/package.json +38 -29
package/README.md
CHANGED
|
@@ -1,6 +1,11 @@
|
|
|
1
1
|
# Word to Markdown
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
[](https://www.npmjs.com/package/word-to-markdown)
|
|
4
|
+
[](https://www.npmjs.com/package/word-to-markdown)
|
|
5
|
+
[](https://github.com/benbalter/word-to-markdown-js/actions/workflows/ci.yml)
|
|
6
|
+
[](https://github.com/benbalter/word-to-markdown-js/blob/main/LICENSE)
|
|
7
|
+
|
|
8
|
+
Convert Word documents to beautiful Markdown. Via command line, as a Node library, or in your browser. An even better version of the original [`word-to-markdown`](https://github.com/benbalter/word-to-markdown).
|
|
4
9
|
|
|
5
10
|
Try it in your browser at [word2md.com](https://word2md.com), or use it from the command line — no clone required:
|
|
6
11
|
|
|
@@ -8,22 +13,46 @@ Try it in your browser at [word2md.com](https://word2md.com), or use it from the
|
|
|
8
13
|
npx word-to-markdown input.docx > output.md
|
|
9
14
|
```
|
|
10
15
|
|
|
11
|
-
##
|
|
16
|
+
## What it converts
|
|
12
17
|
|
|
13
|
-
- Paragraphs
|
|
14
|
-
- Numbered lists
|
|
15
|
-
- Bullet lists
|
|
16
|
-
- Nested Lists
|
|
18
|
+
- Paragraphs and line breaks
|
|
17
19
|
- Headings
|
|
18
|
-
-
|
|
20
|
+
- Bold, italic, and strikethrough
|
|
21
|
+
- Superscript and subscript — preserved as inline `<sup>`/`<sub>` tags
|
|
22
|
+
- Bullet lists, and numbered lists (kept as `1./2./3.` by default)
|
|
23
|
+
- Nested lists
|
|
19
24
|
- Tables
|
|
20
|
-
- Footnotes and endnotes
|
|
21
|
-
- Images
|
|
22
|
-
- Bold, italics, underlines, strikethrough, superscript and subscript.
|
|
23
25
|
- Links
|
|
24
|
-
-
|
|
25
|
-
-
|
|
26
|
-
|
|
26
|
+
- Footnotes and endnotes — converted to GFM/Pandoc `[^1]` footnotes
|
|
27
|
+
- Images — embedded inline as base64 data URIs, or extracted to files
|
|
28
|
+
|
|
29
|
+
### Notes and limitations
|
|
30
|
+
|
|
31
|
+
- **Numbered lists** are kept as `1./2./3.` ordered lists. Pass
|
|
32
|
+
`{ numberedLists: 'bullets' }` (library) or `--bullet-lists` (CLI) to convert
|
|
33
|
+
them to bullet lists instead (matching the original word-to-markdown).
|
|
34
|
+
- **Images** are inlined as base64 data URIs by default. To extract them to
|
|
35
|
+
files with relative links instead, use `{ images: 'extract' }` (library — the
|
|
36
|
+
bytes come back on `ConvertResult.images`) or `--image-dir <dir>` (CLI). Drop
|
|
37
|
+
them entirely with `{ images: 'strip' }` / `--strip-images`. On the web,
|
|
38
|
+
documents with images offer a **Download .zip** (Markdown + an `images/`
|
|
39
|
+
folder). For full control, pass a custom [Mammoth image handler](https://github.com/mwilliamson/mammoth.js/#images)
|
|
40
|
+
via `options.mammoth`.
|
|
41
|
+
- **Underline** is dropped by default (Mammoth's default, since underlines are
|
|
42
|
+
easily confused with links). Pass `{ underline: 'preserve' }` (library) or
|
|
43
|
+
`--underline` (CLI) to keep it as an inline `<u>` tag.
|
|
44
|
+
- **Footnotes and endnotes** become standard GFM/Pandoc footnotes — a `[^1]`
|
|
45
|
+
reference in the body and a `[^1]: …` definition at the end — which render on
|
|
46
|
+
GitHub and in the web preview. Pass `{ footnotes: 'preserve' }` (library) or
|
|
47
|
+
`--preserve-footnotes` (CLI) to instead keep Mammoth's raw `<sup>` links and
|
|
48
|
+
numbered note list (useful for CommonMark targets that lack footnote support).
|
|
49
|
+
A note whose body spans multiple paragraphs (a rare Word construct) is left in
|
|
50
|
+
Mammoth's raw form rather than converted, so its reference and body stay
|
|
51
|
+
linked.
|
|
52
|
+
- **Comments, text boxes, and equations are not converted** — Mammoth drops
|
|
53
|
+
them during the `.docx` → HTML step. When content is dropped this way,
|
|
54
|
+
`convertWithWarnings` surfaces a warning.
|
|
55
|
+
- Heading levels come from Word's paragraph styles, not from font size.
|
|
27
56
|
|
|
28
57
|
## How is this different from the original?
|
|
29
58
|
|
|
@@ -56,8 +85,28 @@ npm install -g word-to-markdown
|
|
|
56
85
|
w2m path/to/your/file.docx > output.md
|
|
57
86
|
```
|
|
58
87
|
|
|
88
|
+
The converted Markdown is written to **stdout** and any document warnings (encryption, sensitivity labels, and the like) to **stderr**, so a redirect captures only the Markdown. The command exits with a non-zero status and a friendly message if the file is missing, unreadable, or not a valid `.docx`. Only `.docx` is supported — re-save older `.doc` files as `.docx` first.
|
|
89
|
+
|
|
90
|
+
Options:
|
|
91
|
+
|
|
92
|
+
- `-o, --output <file>` — write the Markdown to `<file>` instead of stdout
|
|
93
|
+
(warnings still go to stderr).
|
|
94
|
+
- `--bullet-lists` — convert numbered lists to bullets instead of keeping `1./2./3.`.
|
|
95
|
+
- `--underline` — preserve underlined text as inline `<u>` tags (dropped by default).
|
|
96
|
+
- `--strip-images` — remove images instead of embedding them as base64 data URIs.
|
|
97
|
+
- `--image-dir <dir>` — extract images to `<dir>` and link them relatively, instead
|
|
98
|
+
of embedding base64. Links resolve relative to where you save the Markdown, e.g.
|
|
99
|
+
`w2m --image-dir images report.docx > report.md`, or with `-o`,
|
|
100
|
+
`w2m -o out/report.md --image-dir images report.docx` (images land in
|
|
101
|
+
`out/images/`).
|
|
102
|
+
- `--preserve-footnotes` — keep Word footnotes as raw `<sup>` links and a
|
|
103
|
+
numbered note list instead of GFM `[^1]` footnotes.
|
|
104
|
+
- `-V, --version` — print the version.
|
|
105
|
+
|
|
59
106
|
## Use as a library
|
|
60
107
|
|
|
108
|
+
Published to npm as [`word-to-markdown`](https://www.npmjs.com/package/word-to-markdown). It ships as an ES module with TypeScript declarations and requires Node 22.13 or later.
|
|
109
|
+
|
|
61
110
|
```console
|
|
62
111
|
npm install word-to-markdown
|
|
63
112
|
```
|
|
@@ -74,7 +123,63 @@ const { markdown, warnings } = await convertWithWarnings(
|
|
|
74
123
|
);
|
|
75
124
|
```
|
|
76
125
|
|
|
77
|
-
|
|
126
|
+
Both functions accept either a file-path string (Node) or an `ArrayBuffer` (Node or the browser):
|
|
127
|
+
|
|
128
|
+
```js
|
|
129
|
+
const { markdown } = await convertWithWarnings(arrayBuffer);
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
### API
|
|
133
|
+
|
|
134
|
+
- **`convert(input, options?): Promise<string>`** — resolves to the Markdown.
|
|
135
|
+
- **`convertWithWarnings(input, options?): Promise<{ markdown: string; warnings: string[]; images? }>`** — also returns human-readable warnings for encrypted, protected, or sensitivity-labeled documents, and (in `extract` mode) the extracted images.
|
|
136
|
+
|
|
137
|
+
`input` is a file-path `string` (Node) or an `ArrayBuffer`. `options` (type `ConvertOptions`) is optional:
|
|
138
|
+
|
|
139
|
+
- **`images`** — `'inline'` (default) embeds images as base64 data URIs; `'strip'` removes them; `'extract'` replaces each with a relative `` link and returns the bytes on `ConvertResult.images` (use `convertWithWarnings` to retrieve them).
|
|
140
|
+
- **`imageDir`** — link/path prefix for extracted images (default `'images'`); only applies with `images: 'extract'`.
|
|
141
|
+
- **`numberedLists`** — `'ordered'` (default) keeps `1./2./3.`; `'bullets'` converts numbered lists to bullets.
|
|
142
|
+
- **`underline`** — `'ignore'` (default) drops underlines; `'preserve'` keeps them as inline `<u>` tags.
|
|
143
|
+
- **`footnotes`** — `'gfm'` (default) converts footnotes and endnotes to `[^1]` references and definitions; `'preserve'` keeps Mammoth's raw `<sup>` links and numbered note list.
|
|
144
|
+
- **`mammoth`** / **`turndown`** — escape hatches forwarded to [Mammoth](https://github.com/mwilliamson/mammoth.js/) and [Turndown](https://github.com/mixmark-io/turndown) respectively.
|
|
145
|
+
|
|
146
|
+
```js
|
|
147
|
+
// Extract images to files and write them out yourself:
|
|
148
|
+
const { markdown, images } = await convertWithWarnings('file.docx', {
|
|
149
|
+
images: 'extract',
|
|
150
|
+
});
|
|
151
|
+
// markdown → ; images → [{ path, contentType, bytes }]
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
### Error handling
|
|
155
|
+
|
|
156
|
+
Conversion throws typed errors so you can respond to each failure precisely. All of them extend `WordToMarkdownError`, so `error instanceof WordToMarkdownError` catches any of them.
|
|
157
|
+
|
|
158
|
+
```js
|
|
159
|
+
import convert, {
|
|
160
|
+
UnsupportedFileError,
|
|
161
|
+
FileNotFoundError,
|
|
162
|
+
InvalidFileError,
|
|
163
|
+
FilePermissionError,
|
|
164
|
+
ConversionError,
|
|
165
|
+
} from 'word-to-markdown';
|
|
166
|
+
|
|
167
|
+
try {
|
|
168
|
+
const markdown = await convert('path/to/your/file.docx');
|
|
169
|
+
} catch (error) {
|
|
170
|
+
if (error instanceof UnsupportedFileError) {
|
|
171
|
+
// a .doc or password-protected file — only unprotected .docx is supported
|
|
172
|
+
} else if (error instanceof FileNotFoundError) {
|
|
173
|
+
// the path doesn't exist (or runs through something that isn't a directory)
|
|
174
|
+
} else if (error instanceof InvalidFileError) {
|
|
175
|
+
// not a valid or parseable .docx (or the path is a directory)
|
|
176
|
+
} else if (error instanceof FilePermissionError) {
|
|
177
|
+
// the file couldn't be read, or is in a blocked system directory
|
|
178
|
+
} else if (error instanceof ConversionError) {
|
|
179
|
+
// something failed mid-conversion — see error.cause
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
```
|
|
78
183
|
|
|
79
184
|
## Running Locally
|
|
80
185
|
|
|
@@ -98,7 +203,7 @@ To self-host the static site using Docker Compose:
|
|
|
98
203
|
|
|
99
204
|
1. Clone the repository
|
|
100
205
|
2. Run `npm install && npm run build`
|
|
101
|
-
3. Run `docker
|
|
206
|
+
3. Run `docker compose up -d`
|
|
102
207
|
4. Access at http://localhost:3000
|
|
103
208
|
|
|
104
209
|
## More context
|
|
@@ -111,12 +216,12 @@ See the README of [the original Word to Markdown](https://github.com/benbalter/w
|
|
|
111
216
|
|
|
112
217
|
1. Use [LibreOffice](https://www.libreoffice.org/) to convert the Word document to HTML.
|
|
113
218
|
2. Use a bunch of RegEx to clean up the HTML
|
|
114
|
-
3.
|
|
219
|
+
3. Use [Premailer](https://github.com/premailer/premailer) to inline the CSS
|
|
115
220
|
4. Use [Nokogiri](https://nokogiri.org) to manipulate the HTML further
|
|
116
221
|
5. Use [Reverse Markdown](https://github.com/xijo/reverse_markdown) to convert the HTML to Markdown
|
|
117
222
|
6. Use a bunch of RegEx to clean up the Markdown
|
|
118
223
|
|
|
119
|
-
Not only did this process require installing and shelling out to a huge binary (LibreOffice), but it was very fragile, and key projects like Reverse Markdown are no longer maintained. I tried experimenting with Pandoc, but it had many of the same
|
|
224
|
+
Not only did this process require installing and shelling out to a huge binary (LibreOffice), but it was very fragile, and key projects like Reverse Markdown are no longer maintained. I tried experimenting with Pandoc, but it had many of the same limitations.
|
|
120
225
|
|
|
121
226
|
### The new way
|
|
122
227
|
|
|
@@ -126,6 +231,6 @@ Not only did this process require installing and shelling out to a huge binary (
|
|
|
126
231
|
|
|
127
232
|
All three of these projects are actively maintained and heavily used, and allows us to convert the document faster, and entirely in JavaScript. Heck, I think theoretically, this could run in the browser for added privacy.
|
|
128
233
|
|
|
129
|
-
It's still
|
|
234
|
+
It's still young, but so far, I've found the output to be better, with much less manual cleanup required. Notice something is off? Please [open an issue](https://github.com/benbalter/word-to-markdown-js/issues/new).
|
|
130
235
|
|
|
131
236
|
One note: This project does not yet attempt to guess heading levels based on font size. It could, but it's not yet implemented.
|
package/build/cli.js
CHANGED
|
@@ -1,16 +1,59 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import { __awaiter } from "tslib";
|
|
3
2
|
import { Command } from 'commander';
|
|
4
|
-
import {
|
|
3
|
+
import { createRequire } from 'module';
|
|
4
|
+
import { mkdir, writeFile } from 'fs/promises';
|
|
5
|
+
import path from 'path';
|
|
6
|
+
import { convertWithWarnings } from './main.js';
|
|
7
|
+
// Read our own version from package.json. createRequire resolves relative to
|
|
8
|
+
// this module, so `../package.json` points at the package root both from the
|
|
9
|
+
// TypeScript source (src/) and the compiled CLI (build/) after publish. This
|
|
10
|
+
// avoids JSON import attributes, which node16 module resolution would require.
|
|
11
|
+
const require = createRequire(import.meta.url);
|
|
12
|
+
const { version } = require('../package.json');
|
|
5
13
|
const program = new Command();
|
|
6
14
|
program.name('w2m');
|
|
7
15
|
program.description('Convert Word documents to beautiful Markdown');
|
|
16
|
+
program.version(version);
|
|
8
17
|
program
|
|
9
18
|
.command('convert', { isDefault: true })
|
|
10
19
|
.argument('<file>', 'The Word document to convert')
|
|
11
|
-
.
|
|
20
|
+
.option('-o, --output <file>', 'Write the Markdown to <file> instead of stdout. Warnings still print to ' +
|
|
21
|
+
'stderr.')
|
|
22
|
+
.option('--strip-images', 'Remove images instead of embedding them as base64 data URIs')
|
|
23
|
+
.option('--image-dir <dir>', 'Extract images to <dir> and link them relatively, instead of embedding ' +
|
|
24
|
+
'them as base64. Links resolve relative to where you save the Markdown.')
|
|
25
|
+
.option('--bullet-lists', 'Convert numbered lists to bullets rather than keeping them as 1./2./3.')
|
|
26
|
+
.option('--underline', 'Preserve underlined text as inline <u> tags (dropped by default)')
|
|
27
|
+
.option('--preserve-footnotes', 'Keep Word footnotes as raw <sup> links and a numbered note list instead ' +
|
|
28
|
+
'of converting them to GFM [^1] footnotes')
|
|
29
|
+
.action(async (file, options) => {
|
|
12
30
|
try {
|
|
13
|
-
|
|
31
|
+
// --image-dir (extract) takes precedence over --strip-images.
|
|
32
|
+
if (options.imageDir && options.stripImages) {
|
|
33
|
+
console.error('Ignoring --strip-images because --image-dir is set.');
|
|
34
|
+
}
|
|
35
|
+
const images = options.imageDir
|
|
36
|
+
? 'extract'
|
|
37
|
+
: options.stripImages
|
|
38
|
+
? 'strip'
|
|
39
|
+
: 'inline';
|
|
40
|
+
const result = await convertWithWarnings(file, {
|
|
41
|
+
images,
|
|
42
|
+
imageDir: options.imageDir,
|
|
43
|
+
numberedLists: options.bulletLists ? 'bullets' : 'ordered',
|
|
44
|
+
underline: options.underline ? 'preserve' : 'ignore',
|
|
45
|
+
footnotes: options.preserveFootnotes ? 'preserve' : 'gfm',
|
|
46
|
+
});
|
|
47
|
+
// Write extracted images to disk before emitting the Markdown that links
|
|
48
|
+
// them. The links are relative to the Markdown file, so resolve them
|
|
49
|
+
// against its directory (the working directory when writing to stdout).
|
|
50
|
+
if (result.images && result.images.length > 0) {
|
|
51
|
+
const markdownDir = options.output ? path.dirname(options.output) : '.';
|
|
52
|
+
const imageDir = path.resolve(markdownDir, options.imageDir);
|
|
53
|
+
await mkdir(imageDir, { recursive: true });
|
|
54
|
+
await Promise.all(result.images.map((image) => writeFile(path.resolve(markdownDir, image.path), image.bytes)));
|
|
55
|
+
console.error(`Wrote ${result.images.length} image(s) to ${path.relative('.', imageDir) || '.'}/`);
|
|
56
|
+
}
|
|
14
57
|
// Display warnings to stderr if any
|
|
15
58
|
if (result.warnings.length > 0) {
|
|
16
59
|
result.warnings.forEach((warning) => {
|
|
@@ -18,23 +61,21 @@ program
|
|
|
18
61
|
});
|
|
19
62
|
console.error(''); // Empty line for separation
|
|
20
63
|
}
|
|
21
|
-
//
|
|
22
|
-
|
|
64
|
+
// Write the Markdown to the requested file, or stdout by default.
|
|
65
|
+
if (options.output) {
|
|
66
|
+
await mkdir(path.dirname(options.output), { recursive: true });
|
|
67
|
+
await writeFile(options.output, result.markdown);
|
|
68
|
+
console.error(`Wrote Markdown to ${options.output}`);
|
|
69
|
+
}
|
|
70
|
+
else {
|
|
71
|
+
console.log(result.markdown);
|
|
72
|
+
}
|
|
23
73
|
}
|
|
24
74
|
catch (error) {
|
|
25
|
-
//
|
|
26
|
-
if (error instanceof UnsupportedFileError ||
|
|
27
|
-
error instanceof FileNotFoundError ||
|
|
28
|
-
error instanceof InvalidFileError ||
|
|
29
|
-
error instanceof FilePermissionError ||
|
|
30
|
-
error instanceof ConversionError) {
|
|
31
|
-
console.error(`Error: ${error.message}`);
|
|
32
|
-
process.exit(1);
|
|
33
|
-
}
|
|
34
|
-
// Handle unexpected errors (including non-Error objects)
|
|
75
|
+
// Converter errors carry user-friendly messages; print anything else as-is
|
|
35
76
|
console.error('Error:', error instanceof Error ? error.message : String(error));
|
|
36
77
|
process.exit(1);
|
|
37
78
|
}
|
|
38
|
-
})
|
|
39
|
-
program.
|
|
79
|
+
});
|
|
80
|
+
await program.parseAsync();
|
|
40
81
|
//# sourceMappingURL=cli.js.map
|
package/build/cli.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"cli.js","sourceRoot":"","sources":["../src/cli.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"cli.js","sourceRoot":"","sources":["../src/cli.ts"],"names":[],"mappings":";AAEA,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AACpC,OAAO,EAAE,aAAa,EAAE,MAAM,QAAQ,CAAC;AACvC,OAAO,EAAE,KAAK,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAC/C,OAAO,IAAI,MAAM,MAAM,CAAC;AACxB,OAAO,EAAE,mBAAmB,EAAE,MAAM,WAAW,CAAC;AAEhD,6EAA6E;AAC7E,6EAA6E;AAC7E,6EAA6E;AAC7E,+EAA+E;AAC/E,MAAM,OAAO,GAAG,aAAa,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAC/C,MAAM,EAAE,OAAO,EAAE,GAAG,OAAO,CAAC,iBAAiB,CAAwB,CAAC;AAEtE,MAAM,OAAO,GAAG,IAAI,OAAO,EAAE,CAAC;AAC9B,OAAO,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;AACpB,OAAO,CAAC,WAAW,CAAC,8CAA8C,CAAC,CAAC;AACpE,OAAO,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC;AACzB,OAAO;KACJ,OAAO,CAAC,SAAS,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC;KACvC,QAAQ,CAAC,QAAQ,EAAE,8BAA8B,CAAC;KAClD,MAAM,CACL,qBAAqB,EACrB,0EAA0E;IACxE,SAAS,CACZ;KACA,MAAM,CACL,gBAAgB,EAChB,6DAA6D,CAC9D;KACA,MAAM,CACL,mBAAmB,EACnB,yEAAyE;IACvE,wEAAwE,CAC3E;KACA,MAAM,CACL,gBAAgB,EAChB,wEAAwE,CACzE;KACA,MAAM,CACL,aAAa,EACb,kEAAkE,CACnE;KACA,MAAM,CACL,sBAAsB,EACtB,0EAA0E;IACxE,0CAA0C,CAC7C;KACA,MAAM,CAAC,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,EAAE;IAC9B,IAAI,CAAC;QACH,8DAA8D;QAC9D,IAAI,OAAO,CAAC,QAAQ,IAAI,OAAO,CAAC,WAAW,EAAE,CAAC;YAC5C,OAAO,CAAC,KAAK,CAAC,qDAAqD,CAAC,CAAC;QACvE,CAAC;QACD,MAAM,MAAM,GAAG,OAAO,CAAC,QAAQ;YAC7B,CAAC,CAAC,SAAS;YACX,CAAC,CAAC,OAAO,CAAC,WAAW;gBACnB,CAAC,CAAC,OAAO;gBACT,CAAC,CAAC,QAAQ,CAAC;QACf,MAAM,MAAM,GAAG,MAAM,mBAAmB,CAAC,IAAI,EAAE;YAC7C,MAAM;YACN,QAAQ,EAAE,OAAO,CAAC,QAAQ;YAC1B,aAAa,EAAE,OAAO,CAAC,WAAW,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,SAAS;YAC1D,SAAS,EAAE,OAAO,CAAC,SAAS,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,QAAQ;YACpD,SAAS,EAAE,OAAO,CAAC,iBAAiB,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,KAAK;SAC1D,CAAC,CAAC;QAEH,yEAAyE;QACzE,qEAAqE;QACrE,wEAAwE;QACxE,IAAI,MAAM,CAAC,MAAM,IAAI,MAAM,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC9C,MAAM,WAAW,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC;YACxE,MAAM,QAAQ,GAAG,IAAI,CAAC,OAAO,CAAC,WAAW,EAAE,OAAO,CAAC,QAAQ,CAAC,CAAC;YAC7D,MAAM,KAAK,CAAC,QAAQ,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC3C,MAAM,OAAO,CAAC,GAAG,CACf,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAC1B,SAAS,CAAC,IAAI,CAAC,OAAO,CAAC,WAAW,EAAE,KAAK,CAAC,IAAI,CAAC,EAAE,KAAK,CAAC,KAAK,CAAC,CAC9D,CACF,CAAC;YACF,OAAO,CAAC,KAAK,CACX,SAAS,MAAM,CAAC,MAAM,CAAC,MAAM,gBAAgB,IAAI,CAAC,QAAQ,CAAC,GAAG,EAAE,QAAQ,CAAC,IAAI,GAAG,GAAG,CACpF,CAAC;QACJ,CAAC;QAED,oCAAoC;QACpC,IAAI,MAAM,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC/B,MAAM,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC,OAAO,EAAE,EAAE;gBAClC,OAAO,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC;YACzB,CAAC,CAAC,CAAC;YACH,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC,CAAC,4BAA4B;QACjD,CAAC;QAED,kEAAkE;QAClE,IAAI,OAAO,CAAC,MAAM,EAAE,CAAC;YACnB,MAAM,KAAK,CAAC,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,MAAM,CAAC,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC/D,MAAM,SAAS,CAAC,OAAO,CAAC,MAAM,EAAE,MAAM,CAAC,QAAQ,CAAC,CAAC;YACjD,OAAO,CAAC,KAAK,CAAC,qBAAqB,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;QACvD,CAAC;aAAM,CAAC;YACN,OAAO,CAAC,GAAG,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC;QAC/B,CAAC;IACH,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,2EAA2E;QAC3E,OAAO,CAAC,KAAK,CACX,QAAQ,EACR,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CACvD,CAAC;QACF,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IAClB,CAAC;AACH,CAAC,CAAC,CAAC;AAEL,MAAM,OAAO,CAAC,UAAU,EAAE,CAAC","sourcesContent":["#!/usr/bin/env node\n\nimport { Command } from 'commander';\nimport { createRequire } from 'module';\nimport { mkdir, writeFile } from 'fs/promises';\nimport path from 'path';\nimport { convertWithWarnings } from './main.js';\n\n// Read our own version from package.json. createRequire resolves relative to\n// this module, so `../package.json` points at the package root both from the\n// TypeScript source (src/) and the compiled CLI (build/) after publish. This\n// avoids JSON import attributes, which node16 module resolution would require.\nconst require = createRequire(import.meta.url);\nconst { version } = require('../package.json') as { version: string };\n\nconst program = new Command();\nprogram.name('w2m');\nprogram.description('Convert Word documents to beautiful Markdown');\nprogram.version(version);\nprogram\n .command('convert', { isDefault: true })\n .argument('<file>', 'The Word document to convert')\n .option(\n '-o, --output <file>',\n 'Write the Markdown to <file> instead of stdout. Warnings still print to ' +\n 'stderr.',\n )\n .option(\n '--strip-images',\n 'Remove images instead of embedding them as base64 data URIs',\n )\n .option(\n '--image-dir <dir>',\n 'Extract images to <dir> and link them relatively, instead of embedding ' +\n 'them as base64. Links resolve relative to where you save the Markdown.',\n )\n .option(\n '--bullet-lists',\n 'Convert numbered lists to bullets rather than keeping them as 1./2./3.',\n )\n .option(\n '--underline',\n 'Preserve underlined text as inline <u> tags (dropped by default)',\n )\n .option(\n '--preserve-footnotes',\n 'Keep Word footnotes as raw <sup> links and a numbered note list instead ' +\n 'of converting them to GFM [^1] footnotes',\n )\n .action(async (file, options) => {\n try {\n // --image-dir (extract) takes precedence over --strip-images.\n if (options.imageDir && options.stripImages) {\n console.error('Ignoring --strip-images because --image-dir is set.');\n }\n const images = options.imageDir\n ? 'extract'\n : options.stripImages\n ? 'strip'\n : 'inline';\n const result = await convertWithWarnings(file, {\n images,\n imageDir: options.imageDir,\n numberedLists: options.bulletLists ? 'bullets' : 'ordered',\n underline: options.underline ? 'preserve' : 'ignore',\n footnotes: options.preserveFootnotes ? 'preserve' : 'gfm',\n });\n\n // Write extracted images to disk before emitting the Markdown that links\n // them. The links are relative to the Markdown file, so resolve them\n // against its directory (the working directory when writing to stdout).\n if (result.images && result.images.length > 0) {\n const markdownDir = options.output ? path.dirname(options.output) : '.';\n const imageDir = path.resolve(markdownDir, options.imageDir);\n await mkdir(imageDir, { recursive: true });\n await Promise.all(\n result.images.map((image) =>\n writeFile(path.resolve(markdownDir, image.path), image.bytes),\n ),\n );\n console.error(\n `Wrote ${result.images.length} image(s) to ${path.relative('.', imageDir) || '.'}/`,\n );\n }\n\n // Display warnings to stderr if any\n if (result.warnings.length > 0) {\n result.warnings.forEach((warning) => {\n console.error(warning);\n });\n console.error(''); // Empty line for separation\n }\n\n // Write the Markdown to the requested file, or stdout by default.\n if (options.output) {\n await mkdir(path.dirname(options.output), { recursive: true });\n await writeFile(options.output, result.markdown);\n console.error(`Wrote Markdown to ${options.output}`);\n } else {\n console.log(result.markdown);\n }\n } catch (error) {\n // Converter errors carry user-friendly messages; print anything else as-is\n console.error(\n 'Error:',\n error instanceof Error ? error.message : String(error),\n );\n process.exit(1);\n }\n });\n\nawait program.parseAsync();\n"]}
|
package/build/main.d.ts
ADDED
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
export interface ConvertOptions {
|
|
2
|
+
mammoth?: object;
|
|
3
|
+
turndown?: object;
|
|
4
|
+
/**
|
|
5
|
+
* How to handle images. `'inline'` (default) keeps Mammoth's base64 data
|
|
6
|
+
* URIs; `'strip'` removes images entirely (useful to avoid multi-MB output
|
|
7
|
+
* from image-heavy documents); `'extract'` replaces each image with a
|
|
8
|
+
* relative `` link and returns the image bytes on
|
|
9
|
+
* `ConvertResult.images` (use `convertWithWarnings` to retrieve them).
|
|
10
|
+
*/
|
|
11
|
+
images?: 'inline' | 'strip' | 'extract';
|
|
12
|
+
/**
|
|
13
|
+
* Directory prefix used for extracted image links and paths (default
|
|
14
|
+
* `'images'`). Only applies when `images` is `'extract'`. The same value is
|
|
15
|
+
* used for the Markdown link (``) and the returned
|
|
16
|
+
* `ExtractedImage.path`, so links and files always agree.
|
|
17
|
+
*/
|
|
18
|
+
imageDir?: string;
|
|
19
|
+
/**
|
|
20
|
+
* How to render Word's numbered lists. `'ordered'` (default) keeps them as
|
|
21
|
+
* `1.`/`2.`/… ordered lists; `'bullets'` converts them to bullet lists
|
|
22
|
+
* (matching the classic word-to-markdown behavior).
|
|
23
|
+
*/
|
|
24
|
+
numberedLists?: 'bullets' | 'ordered';
|
|
25
|
+
/**
|
|
26
|
+
* How to handle underlined text. `'ignore'` (default) drops the underline —
|
|
27
|
+
* Mammoth's default, since underlines are easily confused with links in HTML.
|
|
28
|
+
* `'preserve'` keeps it as an inline `<u>…</u>` tag (rendered by GitHub-flavored
|
|
29
|
+
* Markdown). Note that superscript and subscript are always preserved as
|
|
30
|
+
* `<sup>`/`<sub>` and need no option.
|
|
31
|
+
*/
|
|
32
|
+
underline?: 'ignore' | 'preserve';
|
|
33
|
+
/**
|
|
34
|
+
* How to render Word's footnotes and endnotes. `'gfm'` (default) rewrites
|
|
35
|
+
* Mammoth's superscript reference links plus trailing note list into standard
|
|
36
|
+
* GitHub-flavored/Pandoc footnote syntax (`[^1]` references and `[^1]:`
|
|
37
|
+
* definitions). `'preserve'` keeps Mammoth's raw `<sup>` links and numbered
|
|
38
|
+
* note list (useful for CommonMark targets that don't support `[^1]`).
|
|
39
|
+
*/
|
|
40
|
+
footnotes?: 'gfm' | 'preserve';
|
|
41
|
+
}
|
|
42
|
+
export interface ExtractedImage {
|
|
43
|
+
path: string;
|
|
44
|
+
contentType: string;
|
|
45
|
+
bytes: Uint8Array;
|
|
46
|
+
}
|
|
47
|
+
export interface ConvertResult {
|
|
48
|
+
markdown: string;
|
|
49
|
+
warnings: string[];
|
|
50
|
+
/** Present (possibly empty) when converting with `images: 'extract'`. */
|
|
51
|
+
images?: ExtractedImage[];
|
|
52
|
+
}
|
|
53
|
+
export interface DocumentProperties {
|
|
54
|
+
sensitivity?: string;
|
|
55
|
+
confidentiality?: string;
|
|
56
|
+
encryption?: boolean;
|
|
57
|
+
protection?: boolean;
|
|
58
|
+
}
|
|
59
|
+
export declare class WordToMarkdownError extends Error {
|
|
60
|
+
}
|
|
61
|
+
export declare class UnsupportedFileError extends WordToMarkdownError {
|
|
62
|
+
constructor(message: string);
|
|
63
|
+
}
|
|
64
|
+
export declare class FileNotFoundError extends WordToMarkdownError {
|
|
65
|
+
constructor(filePath?: string);
|
|
66
|
+
}
|
|
67
|
+
export declare class InvalidFileError extends WordToMarkdownError {
|
|
68
|
+
constructor(filePath?: string);
|
|
69
|
+
}
|
|
70
|
+
export declare class FilePermissionError extends WordToMarkdownError {
|
|
71
|
+
constructor(filePath?: string);
|
|
72
|
+
}
|
|
73
|
+
export declare class ConversionError extends WordToMarkdownError {
|
|
74
|
+
constructor(message: string, originalError?: Error);
|
|
75
|
+
}
|
|
76
|
+
export declare function validateFileExtension(filePath: string): void;
|
|
77
|
+
export declare function htmlToMd(html: string, options?: object, keepTags?: string[]): string;
|
|
78
|
+
export declare function extractDocumentProperties(input: string | ArrayBuffer): Promise<DocumentProperties>;
|
|
79
|
+
export declare function generateWarnings(properties: DocumentProperties): string[];
|
|
80
|
+
interface MammothMessage {
|
|
81
|
+
type: string;
|
|
82
|
+
message: string;
|
|
83
|
+
}
|
|
84
|
+
export declare function extensionForContentType(contentType: string): string;
|
|
85
|
+
export declare function extractMammothWarnings(messages: readonly MammothMessage[]): string[];
|
|
86
|
+
export declare function convertWithWarnings(input: string | ArrayBuffer, options?: ConvertOptions): Promise<ConvertResult>;
|
|
87
|
+
export default function convert(input: string | ArrayBuffer, options?: ConvertOptions): Promise<string>;
|
|
88
|
+
export {};
|