word-to-markdown 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -1,6 +1,11 @@
1
1
  # Word to Markdown
2
2
 
3
- Convert Word documents to beautiful Markdown. Via command line or in your browser. An even better version of the original [`word-to-markdown`](https://github.com/benbalter/word-to-markdown).
3
+ [![npm version](https://img.shields.io/npm/v/word-to-markdown.svg)](https://www.npmjs.com/package/word-to-markdown)
4
+ [![npm downloads](https://img.shields.io/npm/dm/word-to-markdown.svg)](https://www.npmjs.com/package/word-to-markdown)
5
+ [![CI](https://github.com/benbalter/word-to-markdown-js/actions/workflows/ci.yml/badge.svg)](https://github.com/benbalter/word-to-markdown-js/actions/workflows/ci.yml)
6
+ [![License: Apache-2.0](https://img.shields.io/npm/l/word-to-markdown.svg)](https://github.com/benbalter/word-to-markdown-js/blob/main/LICENSE)
7
+
8
+ Convert Word documents to beautiful Markdown. Via command line, as a Node library, or in your browser. An even better version of the original [`word-to-markdown`](https://github.com/benbalter/word-to-markdown).
4
9
 
5
10
  Try it in your browser at [word2md.com](https://word2md.com), or use it from the command line — no clone required:
6
11
 
@@ -8,22 +13,46 @@ Try it in your browser at [word2md.com](https://word2md.com), or use it from the
8
13
  npx word-to-markdown input.docx > output.md
9
14
  ```
10
15
 
11
- ## Supports
16
+ ## What it converts
12
17
 
13
- - Paragraphs
14
- - Numbered lists
15
- - Bullet lists
16
- - Nested Lists
18
+ - Paragraphs and line breaks
17
19
  - Headings
18
- - Lists
20
+ - Bold, italic, and strikethrough
21
+ - Superscript and subscript — preserved as inline `<sup>`/`<sub>` tags
22
+ - Bullet lists, and numbered lists (kept as `1./2./3.` by default)
23
+ - Nested lists
19
24
  - Tables
20
- - Footnotes and endnotes
21
- - Images
22
- - Bold, italics, underlines, strikethrough, superscript and subscript.
23
25
  - Links
24
- - Line breaks
25
- - Text boxes
26
- - Comments
26
+ - Footnotes and endnotes — converted to GFM/Pandoc `[^1]` footnotes
27
+ - Images — embedded inline as base64 data URIs, or extracted to files
28
+
29
+ ### Notes and limitations
30
+
31
+ - **Numbered lists** are kept as `1./2./3.` ordered lists. Pass
32
+ `{ numberedLists: 'bullets' }` (library) or `--bullet-lists` (CLI) to convert
33
+ them to bullet lists instead (matching the original word-to-markdown).
34
+ - **Images** are inlined as base64 data URIs by default. To extract them to
35
+ files with relative links instead, use `{ images: 'extract' }` (library — the
36
+ bytes come back on `ConvertResult.images`) or `--image-dir <dir>` (CLI). Drop
37
+ them entirely with `{ images: 'strip' }` / `--strip-images`. On the web,
38
+ documents with images offer a **Download .zip** (Markdown + an `images/`
39
+ folder). For full control, pass a custom [Mammoth image handler](https://github.com/mwilliamson/mammoth.js/#images)
40
+ via `options.mammoth`.
41
+ - **Underline** is dropped by default (Mammoth's default, since underlines are
42
+ easily confused with links). Pass `{ underline: 'preserve' }` (library) or
43
+ `--underline` (CLI) to keep it as an inline `<u>` tag.
44
+ - **Footnotes and endnotes** become standard GFM/Pandoc footnotes — a `[^1]`
45
+ reference in the body and a `[^1]: …` definition at the end — which render on
46
+ GitHub and in the web preview. Pass `{ footnotes: 'preserve' }` (library) or
47
+ `--preserve-footnotes` (CLI) to instead keep Mammoth's raw `<sup>` links and
48
+ numbered note list (useful for CommonMark targets that lack footnote support).
49
+ A note whose body spans multiple paragraphs (a rare Word construct) is left in
50
+ Mammoth's raw form rather than converted, so its reference and body stay
51
+ linked.
52
+ - **Comments, text boxes, and equations are not converted** — Mammoth drops
53
+ them during the `.docx` → HTML step. When content is dropped this way,
54
+ `convertWithWarnings` surfaces a warning.
55
+ - Heading levels come from Word's paragraph styles, not from font size.
27
56
 
28
57
  ## How is this different from the original?
29
58
 
@@ -56,8 +85,28 @@ npm install -g word-to-markdown
56
85
  w2m path/to/your/file.docx > output.md
57
86
  ```
58
87
 
88
+ The converted Markdown is written to **stdout** and any document warnings (encryption, sensitivity labels, and the like) to **stderr**, so a redirect captures only the Markdown. The command exits with a non-zero status and a friendly message if the file is missing, unreadable, or not a valid `.docx`. Only `.docx` is supported — re-save older `.doc` files as `.docx` first.
89
+
90
+ Options:
91
+
92
+ - `-o, --output <file>` — write the Markdown to `<file>` instead of stdout
93
+ (warnings still go to stderr).
94
+ - `--bullet-lists` — convert numbered lists to bullets instead of keeping `1./2./3.`.
95
+ - `--underline` — preserve underlined text as inline `<u>` tags (dropped by default).
96
+ - `--strip-images` — remove images instead of embedding them as base64 data URIs.
97
+ - `--image-dir <dir>` — extract images to `<dir>` and link them relatively, instead
98
+ of embedding base64. Links resolve relative to where you save the Markdown, e.g.
99
+ `w2m --image-dir images report.docx > report.md`, or with `-o`,
100
+ `w2m -o out/report.md --image-dir images report.docx` (images land in
101
+ `out/images/`).
102
+ - `--preserve-footnotes` — keep Word footnotes as raw `<sup>` links and a
103
+ numbered note list instead of GFM `[^1]` footnotes.
104
+ - `-V, --version` — print the version.
105
+
59
106
  ## Use as a library
60
107
 
108
+ Published to npm as [`word-to-markdown`](https://www.npmjs.com/package/word-to-markdown). It ships as an ES module with TypeScript declarations and requires Node 22.13 or later.
109
+
61
110
  ```console
62
111
  npm install word-to-markdown
63
112
  ```
@@ -74,7 +123,63 @@ const { markdown, warnings } = await convertWithWarnings(
74
123
  );
75
124
  ```
76
125
 
77
- In the browser, pass an `ArrayBuffer` instead of a file path.
126
+ Both functions accept either a file-path string (Node) or an `ArrayBuffer` (Node or the browser):
127
+
128
+ ```js
129
+ const { markdown } = await convertWithWarnings(arrayBuffer);
130
+ ```
131
+
132
+ ### API
133
+
134
+ - **`convert(input, options?): Promise<string>`** — resolves to the Markdown.
135
+ - **`convertWithWarnings(input, options?): Promise<{ markdown: string; warnings: string[]; images? }>`** — also returns human-readable warnings for encrypted, protected, or sensitivity-labeled documents, and (in `extract` mode) the extracted images.
136
+
137
+ `input` is a file-path `string` (Node) or an `ArrayBuffer`. `options` (type `ConvertOptions`) is optional:
138
+
139
+ - **`images`** — `'inline'` (default) embeds images as base64 data URIs; `'strip'` removes them; `'extract'` replaces each with a relative `![](imageDir/imageN.ext)` link and returns the bytes on `ConvertResult.images` (use `convertWithWarnings` to retrieve them).
140
+ - **`imageDir`** — link/path prefix for extracted images (default `'images'`); only applies with `images: 'extract'`.
141
+ - **`numberedLists`** — `'ordered'` (default) keeps `1./2./3.`; `'bullets'` converts numbered lists to bullets.
142
+ - **`underline`** — `'ignore'` (default) drops underlines; `'preserve'` keeps them as inline `<u>` tags.
143
+ - **`footnotes`** — `'gfm'` (default) converts footnotes and endnotes to `[^1]` references and definitions; `'preserve'` keeps Mammoth's raw `<sup>` links and numbered note list.
144
+ - **`mammoth`** / **`turndown`** — escape hatches forwarded to [Mammoth](https://github.com/mwilliamson/mammoth.js/) and [Turndown](https://github.com/mixmark-io/turndown) respectively.
145
+
146
+ ```js
147
+ // Extract images to files and write them out yourself:
148
+ const { markdown, images } = await convertWithWarnings('file.docx', {
149
+ images: 'extract',
150
+ });
151
+ // markdown → ![](images/image1.png); images → [{ path, contentType, bytes }]
152
+ ```
153
+
154
+ ### Error handling
155
+
156
+ Conversion throws typed errors so you can respond to each failure precisely. All of them extend `WordToMarkdownError`, so `error instanceof WordToMarkdownError` catches any of them.
157
+
158
+ ```js
159
+ import convert, {
160
+ UnsupportedFileError,
161
+ FileNotFoundError,
162
+ InvalidFileError,
163
+ FilePermissionError,
164
+ ConversionError,
165
+ } from 'word-to-markdown';
166
+
167
+ try {
168
+ const markdown = await convert('path/to/your/file.docx');
169
+ } catch (error) {
170
+ if (error instanceof UnsupportedFileError) {
171
+ // a .doc or password-protected file — only unprotected .docx is supported
172
+ } else if (error instanceof FileNotFoundError) {
173
+ // the path doesn't exist (or runs through something that isn't a directory)
174
+ } else if (error instanceof InvalidFileError) {
175
+ // not a valid or parseable .docx (or the path is a directory)
176
+ } else if (error instanceof FilePermissionError) {
177
+ // the file couldn't be read, or is in a blocked system directory
178
+ } else if (error instanceof ConversionError) {
179
+ // something failed mid-conversion — see error.cause
180
+ }
181
+ }
182
+ ```
78
183
 
79
184
  ## Running Locally
80
185
 
@@ -98,7 +203,7 @@ To self-host the static site using Docker Compose:
98
203
 
99
204
  1. Clone the repository
100
205
  2. Run `npm install && npm run build`
101
- 3. Run `docker-compose up -d`
206
+ 3. Run `docker compose up -d`
102
207
  4. Access at http://localhost:3000
103
208
 
104
209
  ## More context
@@ -111,12 +216,12 @@ See the README of [the original Word to Markdown](https://github.com/benbalter/w
111
216
 
112
217
  1. Use [LibreOffice](https://www.libreoffice.org/) to convert the Word document to HTML.
113
218
  2. Use a bunch of RegEx to clean up the HTML
114
- 3. User [Premailer](https://github.com/premailer/premailer) to inline the CSS
219
+ 3. Use [Premailer](https://github.com/premailer/premailer) to inline the CSS
115
220
  4. Use [Nokogiri](https://nokogiri.org) to manipulate the HTML further
116
221
  5. Use [Reverse Markdown](https://github.com/xijo/reverse_markdown) to convert the HTML to Markdown
117
222
  6. Use a bunch of RegEx to clean up the Markdown
118
223
 
119
- Not only did this process require installing and shelling out to a huge binary (LibreOffice), but it was very fragile, and key projects like Reverse Markdown are no longer maintained. I tried experimenting with Pandoc, but it had many of the same limitation.
224
+ Not only did this process require installing and shelling out to a huge binary (LibreOffice), but it was very fragile, and key projects like Reverse Markdown are no longer maintained. I tried experimenting with Pandoc, but it had many of the same limitations.
120
225
 
121
226
  ### The new way
122
227
 
@@ -126,6 +231,6 @@ Not only did this process require installing and shelling out to a huge binary (
126
231
 
127
232
  All three of these projects are actively maintained and heavily used, and allows us to convert the document faster, and entirely in JavaScript. Heck, I think theoretically, this could run in the browser for added privacy.
128
233
 
129
- It's still in beta, but so far, I've found the output to be better, with much less manual cleanup required. Notice something is off? Please [open an issue](https://github.com/benbalter/word-to-markdown-js/issues/new).
234
+ It's still young, but so far, I've found the output to be better, with much less manual cleanup required. Notice something is off? Please [open an issue](https://github.com/benbalter/word-to-markdown-js/issues/new).
130
235
 
131
236
  One note: This project does not yet attempt to guess heading levels based on font size. It could, but it's not yet implemented.
package/build/cli.js CHANGED
@@ -1,16 +1,59 @@
1
1
  #!/usr/bin/env node
2
- import { __awaiter } from "tslib";
3
2
  import { Command } from 'commander';
4
- import { convertWithWarnings, UnsupportedFileError, FileNotFoundError, InvalidFileError, FilePermissionError, ConversionError, } from './main.js';
3
+ import { createRequire } from 'module';
4
+ import { mkdir, writeFile } from 'fs/promises';
5
+ import path from 'path';
6
+ import { convertWithWarnings } from './main.js';
7
+ // Read our own version from package.json. createRequire resolves relative to
8
+ // this module, so `../package.json` points at the package root both from the
9
+ // TypeScript source (src/) and the compiled CLI (build/) after publish. This
10
+ // avoids JSON import attributes, which node16 module resolution would require.
11
+ const require = createRequire(import.meta.url);
12
+ const { version } = require('../package.json');
5
13
  const program = new Command();
6
14
  program.name('w2m');
7
15
  program.description('Convert Word documents to beautiful Markdown');
16
+ program.version(version);
8
17
  program
9
18
  .command('convert', { isDefault: true })
10
19
  .argument('<file>', 'The Word document to convert')
11
- .action((file) => __awaiter(void 0, void 0, void 0, function* () {
20
+ .option('-o, --output <file>', 'Write the Markdown to <file> instead of stdout. Warnings still print to ' +
21
+ 'stderr.')
22
+ .option('--strip-images', 'Remove images instead of embedding them as base64 data URIs')
23
+ .option('--image-dir <dir>', 'Extract images to <dir> and link them relatively, instead of embedding ' +
24
+ 'them as base64. Links resolve relative to where you save the Markdown.')
25
+ .option('--bullet-lists', 'Convert numbered lists to bullets rather than keeping them as 1./2./3.')
26
+ .option('--underline', 'Preserve underlined text as inline <u> tags (dropped by default)')
27
+ .option('--preserve-footnotes', 'Keep Word footnotes as raw <sup> links and a numbered note list instead ' +
28
+ 'of converting them to GFM [^1] footnotes')
29
+ .action(async (file, options) => {
12
30
  try {
13
- const result = yield convertWithWarnings(file);
31
+ // --image-dir (extract) takes precedence over --strip-images.
32
+ if (options.imageDir && options.stripImages) {
33
+ console.error('Ignoring --strip-images because --image-dir is set.');
34
+ }
35
+ const images = options.imageDir
36
+ ? 'extract'
37
+ : options.stripImages
38
+ ? 'strip'
39
+ : 'inline';
40
+ const result = await convertWithWarnings(file, {
41
+ images,
42
+ imageDir: options.imageDir,
43
+ numberedLists: options.bulletLists ? 'bullets' : 'ordered',
44
+ underline: options.underline ? 'preserve' : 'ignore',
45
+ footnotes: options.preserveFootnotes ? 'preserve' : 'gfm',
46
+ });
47
+ // Write extracted images to disk before emitting the Markdown that links
48
+ // them. The links are relative to the Markdown file, so resolve them
49
+ // against its directory (the working directory when writing to stdout).
50
+ if (result.images && result.images.length > 0) {
51
+ const markdownDir = options.output ? path.dirname(options.output) : '.';
52
+ const imageDir = path.resolve(markdownDir, options.imageDir);
53
+ await mkdir(imageDir, { recursive: true });
54
+ await Promise.all(result.images.map((image) => writeFile(path.resolve(markdownDir, image.path), image.bytes)));
55
+ console.error(`Wrote ${result.images.length} image(s) to ${path.relative('.', imageDir) || '.'}/`);
56
+ }
14
57
  // Display warnings to stderr if any
15
58
  if (result.warnings.length > 0) {
16
59
  result.warnings.forEach((warning) => {
@@ -18,23 +61,21 @@ program
18
61
  });
19
62
  console.error(''); // Empty line for separation
20
63
  }
21
- // Output markdown to stdout
22
- console.log(result.markdown);
64
+ // Write the Markdown to the requested file, or stdout by default.
65
+ if (options.output) {
66
+ await mkdir(path.dirname(options.output), { recursive: true });
67
+ await writeFile(options.output, result.markdown);
68
+ console.error(`Wrote Markdown to ${options.output}`);
69
+ }
70
+ else {
71
+ console.log(result.markdown);
72
+ }
23
73
  }
24
74
  catch (error) {
25
- // Handle our custom errors with user-friendly messages
26
- if (error instanceof UnsupportedFileError ||
27
- error instanceof FileNotFoundError ||
28
- error instanceof InvalidFileError ||
29
- error instanceof FilePermissionError ||
30
- error instanceof ConversionError) {
31
- console.error(`Error: ${error.message}`);
32
- process.exit(1);
33
- }
34
- // Handle unexpected errors (including non-Error objects)
75
+ // Converter errors carry user-friendly messages; print anything else as-is
35
76
  console.error('Error:', error instanceof Error ? error.message : String(error));
36
77
  process.exit(1);
37
78
  }
38
- }));
39
- program.parse();
79
+ });
80
+ await program.parseAsync();
40
81
  //# sourceMappingURL=cli.js.map
package/build/cli.js.map CHANGED
@@ -1 +1 @@
1
- {"version":3,"file":"cli.js","sourceRoot":"","sources":["../src/cli.ts"],"names":[],"mappings":";;AAEA,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AACpC,OAAO,EACL,mBAAmB,EACnB,oBAAoB,EACpB,iBAAiB,EACjB,gBAAgB,EAChB,mBAAmB,EACnB,eAAe,GAChB,MAAM,WAAW,CAAC;AAEnB,MAAM,OAAO,GAAG,IAAI,OAAO,EAAE,CAAC;AAC9B,OAAO,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;AACpB,OAAO,CAAC,WAAW,CAAC,8CAA8C,CAAC,CAAC;AACpE,OAAO;KACJ,OAAO,CAAC,SAAS,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC;KACvC,QAAQ,CAAC,QAAQ,EAAE,8BAA8B,CAAC;KAClD,MAAM,CAAC,CAAO,IAAI,EAAE,EAAE;IACrB,IAAI,CAAC;QACH,MAAM,MAAM,GAAG,MAAM,mBAAmB,CAAC,IAAI,CAAC,CAAC;QAE/C,oCAAoC;QACpC,IAAI,MAAM,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC/B,MAAM,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC,OAAO,EAAE,EAAE;gBAClC,OAAO,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC;YACzB,CAAC,CAAC,CAAC;YACH,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC,CAAC,4BAA4B;QACjD,CAAC;QAED,4BAA4B;QAC5B,OAAO,CAAC,GAAG,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC;IAC/B,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,uDAAuD;QACvD,IACE,KAAK,YAAY,oBAAoB;YACrC,KAAK,YAAY,iBAAiB;YAClC,KAAK,YAAY,gBAAgB;YACjC,KAAK,YAAY,mBAAmB;YACpC,KAAK,YAAY,eAAe,EAChC,CAAC;YACD,OAAO,CAAC,KAAK,CAAC,UAAU,KAAK,CAAC,OAAO,EAAE,CAAC,CAAC;YACzC,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;QAClB,CAAC;QACD,yDAAyD;QACzD,OAAO,CAAC,KAAK,CACX,QAAQ,EACR,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CACvD,CAAC;QACF,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IAClB,CAAC;AACH,CAAC,CAAA,CAAC,CAAC;AAEL,OAAO,CAAC,KAAK,EAAE,CAAC"}
1
+ {"version":3,"file":"cli.js","sourceRoot":"","sources":["../src/cli.ts"],"names":[],"mappings":";AAEA,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AACpC,OAAO,EAAE,aAAa,EAAE,MAAM,QAAQ,CAAC;AACvC,OAAO,EAAE,KAAK,EAAE,SAAS,EAAE,MAAM,aAAa,CAAC;AAC/C,OAAO,IAAI,MAAM,MAAM,CAAC;AACxB,OAAO,EAAE,mBAAmB,EAAE,MAAM,WAAW,CAAC;AAEhD,6EAA6E;AAC7E,6EAA6E;AAC7E,6EAA6E;AAC7E,+EAA+E;AAC/E,MAAM,OAAO,GAAG,aAAa,CAAC,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAC/C,MAAM,EAAE,OAAO,EAAE,GAAG,OAAO,CAAC,iBAAiB,CAAwB,CAAC;AAEtE,MAAM,OAAO,GAAG,IAAI,OAAO,EAAE,CAAC;AAC9B,OAAO,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;AACpB,OAAO,CAAC,WAAW,CAAC,8CAA8C,CAAC,CAAC;AACpE,OAAO,CAAC,OAAO,CAAC,OAAO,CAAC,CAAC;AACzB,OAAO;KACJ,OAAO,CAAC,SAAS,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC;KACvC,QAAQ,CAAC,QAAQ,EAAE,8BAA8B,CAAC;KAClD,MAAM,CACL,qBAAqB,EACrB,0EAA0E;IACxE,SAAS,CACZ;KACA,MAAM,CACL,gBAAgB,EAChB,6DAA6D,CAC9D;KACA,MAAM,CACL,mBAAmB,EACnB,yEAAyE;IACvE,wEAAwE,CAC3E;KACA,MAAM,CACL,gBAAgB,EAChB,wEAAwE,CACzE;KACA,MAAM,CACL,aAAa,EACb,kEAAkE,CACnE;KACA,MAAM,CACL,sBAAsB,EACtB,0EAA0E;IACxE,0CAA0C,CAC7C;KACA,MAAM,CAAC,KAAK,EAAE,IAAI,EAAE,OAAO,EAAE,EAAE;IAC9B,IAAI,CAAC;QACH,8DAA8D;QAC9D,IAAI,OAAO,CAAC,QAAQ,IAAI,OAAO,CAAC,WAAW,EAAE,CAAC;YAC5C,OAAO,CAAC,KAAK,CAAC,qDAAqD,CAAC,CAAC;QACvE,CAAC;QACD,MAAM,MAAM,GAAG,OAAO,CAAC,QAAQ;YAC7B,CAAC,CAAC,SAAS;YACX,CAAC,CAAC,OAAO,CAAC,WAAW;gBACnB,CAAC,CAAC,OAAO;gBACT,CAAC,CAAC,QAAQ,CAAC;QACf,MAAM,MAAM,GAAG,MAAM,mBAAmB,CAAC,IAAI,EAAE;YAC7C,MAAM;YACN,QAAQ,EAAE,OAAO,CAAC,QAAQ;YAC1B,aAAa,EAAE,OAAO,CAAC,WAAW,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,SAAS;YAC1D,SAAS,EAAE,OAAO,CAAC,SAAS,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,QAAQ;YACpD,SAAS,EAAE,OAAO,CAAC,iBAAiB,CAAC,CAAC,CAAC,UAAU,CAAC,CAAC,CAAC,KAAK;SAC1D,CAAC,CAAC;QAEH,yEAAyE;QACzE,qEAAqE;QACrE,wEAAwE;QACxE,IAAI,MAAM,CAAC,MAAM,IAAI,MAAM,CAAC,MAAM,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC9C,MAAM,WAAW,GAAG,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,MAAM,CAAC,CAAC,CAAC,CAAC,GAAG,CAAC;YACxE,MAAM,QAAQ,GAAG,IAAI,CAAC,OAAO,CAAC,WAAW,EAAE,OAAO,CAAC,QAAQ,CAAC,CAAC;YAC7D,MAAM,KAAK,CAAC,QAAQ,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC3C,MAAM,OAAO,CAAC,GAAG,CACf,MAAM,CAAC,MAAM,CAAC,GAAG,CAAC,CAAC,KAAK,EAAE,EAAE,CAC1B,SAAS,CAAC,IAAI,CAAC,OAAO,CAAC,WAAW,EAAE,KAAK,CAAC,IAAI,CAAC,EAAE,KAAK,CAAC,KAAK,CAAC,CAC9D,CACF,CAAC;YACF,OAAO,CAAC,KAAK,CACX,SAAS,MAAM,CAAC,MAAM,CAAC,MAAM,gBAAgB,IAAI,CAAC,QAAQ,CAAC,GAAG,EAAE,QAAQ,CAAC,IAAI,GAAG,GAAG,CACpF,CAAC;QACJ,CAAC;QAED,oCAAoC;QACpC,IAAI,MAAM,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;YAC/B,MAAM,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC,OAAO,EAAE,EAAE;gBAClC,OAAO,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC;YACzB,CAAC,CAAC,CAAC;YACH,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC,CAAC,4BAA4B;QACjD,CAAC;QAED,kEAAkE;QAClE,IAAI,OAAO,CAAC,MAAM,EAAE,CAAC;YACnB,MAAM,KAAK,CAAC,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,MAAM,CAAC,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC/D,MAAM,SAAS,CAAC,OAAO,CAAC,MAAM,EAAE,MAAM,CAAC,QAAQ,CAAC,CAAC;YACjD,OAAO,CAAC,KAAK,CAAC,qBAAqB,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;QACvD,CAAC;aAAM,CAAC;YACN,OAAO,CAAC,GAAG,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC;QAC/B,CAAC;IACH,CAAC;IAAC,OAAO,KAAK,EAAE,CAAC;QACf,2EAA2E;QAC3E,OAAO,CAAC,KAAK,CACX,QAAQ,EACR,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CACvD,CAAC;QACF,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC;IAClB,CAAC;AACH,CAAC,CAAC,CAAC;AAEL,MAAM,OAAO,CAAC,UAAU,EAAE,CAAC","sourcesContent":["#!/usr/bin/env node\n\nimport { Command } from 'commander';\nimport { createRequire } from 'module';\nimport { mkdir, writeFile } from 'fs/promises';\nimport path from 'path';\nimport { convertWithWarnings } from './main.js';\n\n// Read our own version from package.json. createRequire resolves relative to\n// this module, so `../package.json` points at the package root both from the\n// TypeScript source (src/) and the compiled CLI (build/) after publish. This\n// avoids JSON import attributes, which node16 module resolution would require.\nconst require = createRequire(import.meta.url);\nconst { version } = require('../package.json') as { version: string };\n\nconst program = new Command();\nprogram.name('w2m');\nprogram.description('Convert Word documents to beautiful Markdown');\nprogram.version(version);\nprogram\n .command('convert', { isDefault: true })\n .argument('<file>', 'The Word document to convert')\n .option(\n '-o, --output <file>',\n 'Write the Markdown to <file> instead of stdout. Warnings still print to ' +\n 'stderr.',\n )\n .option(\n '--strip-images',\n 'Remove images instead of embedding them as base64 data URIs',\n )\n .option(\n '--image-dir <dir>',\n 'Extract images to <dir> and link them relatively, instead of embedding ' +\n 'them as base64. Links resolve relative to where you save the Markdown.',\n )\n .option(\n '--bullet-lists',\n 'Convert numbered lists to bullets rather than keeping them as 1./2./3.',\n )\n .option(\n '--underline',\n 'Preserve underlined text as inline <u> tags (dropped by default)',\n )\n .option(\n '--preserve-footnotes',\n 'Keep Word footnotes as raw <sup> links and a numbered note list instead ' +\n 'of converting them to GFM [^1] footnotes',\n )\n .action(async (file, options) => {\n try {\n // --image-dir (extract) takes precedence over --strip-images.\n if (options.imageDir && options.stripImages) {\n console.error('Ignoring --strip-images because --image-dir is set.');\n }\n const images = options.imageDir\n ? 'extract'\n : options.stripImages\n ? 'strip'\n : 'inline';\n const result = await convertWithWarnings(file, {\n images,\n imageDir: options.imageDir,\n numberedLists: options.bulletLists ? 'bullets' : 'ordered',\n underline: options.underline ? 'preserve' : 'ignore',\n footnotes: options.preserveFootnotes ? 'preserve' : 'gfm',\n });\n\n // Write extracted images to disk before emitting the Markdown that links\n // them. The links are relative to the Markdown file, so resolve them\n // against its directory (the working directory when writing to stdout).\n if (result.images && result.images.length > 0) {\n const markdownDir = options.output ? path.dirname(options.output) : '.';\n const imageDir = path.resolve(markdownDir, options.imageDir);\n await mkdir(imageDir, { recursive: true });\n await Promise.all(\n result.images.map((image) =>\n writeFile(path.resolve(markdownDir, image.path), image.bytes),\n ),\n );\n console.error(\n `Wrote ${result.images.length} image(s) to ${path.relative('.', imageDir) || '.'}/`,\n );\n }\n\n // Display warnings to stderr if any\n if (result.warnings.length > 0) {\n result.warnings.forEach((warning) => {\n console.error(warning);\n });\n console.error(''); // Empty line for separation\n }\n\n // Write the Markdown to the requested file, or stdout by default.\n if (options.output) {\n await mkdir(path.dirname(options.output), { recursive: true });\n await writeFile(options.output, result.markdown);\n console.error(`Wrote Markdown to ${options.output}`);\n } else {\n console.log(result.markdown);\n }\n } catch (error) {\n // Converter errors carry user-friendly messages; print anything else as-is\n console.error(\n 'Error:',\n error instanceof Error ? error.message : String(error),\n );\n process.exit(1);\n }\n });\n\nawait program.parseAsync();\n"]}
@@ -0,0 +1,88 @@
1
+ export interface ConvertOptions {
2
+ mammoth?: object;
3
+ turndown?: object;
4
+ /**
5
+ * How to handle images. `'inline'` (default) keeps Mammoth's base64 data
6
+ * URIs; `'strip'` removes images entirely (useful to avoid multi-MB output
7
+ * from image-heavy documents); `'extract'` replaces each image with a
8
+ * relative `![](imageDir/imageN.ext)` link and returns the image bytes on
9
+ * `ConvertResult.images` (use `convertWithWarnings` to retrieve them).
10
+ */
11
+ images?: 'inline' | 'strip' | 'extract';
12
+ /**
13
+ * Directory prefix used for extracted image links and paths (default
14
+ * `'images'`). Only applies when `images` is `'extract'`. The same value is
15
+ * used for the Markdown link (`![](imageDir/imageN.ext)`) and the returned
16
+ * `ExtractedImage.path`, so links and files always agree.
17
+ */
18
+ imageDir?: string;
19
+ /**
20
+ * How to render Word's numbered lists. `'ordered'` (default) keeps them as
21
+ * `1.`/`2.`/… ordered lists; `'bullets'` converts them to bullet lists
22
+ * (matching the classic word-to-markdown behavior).
23
+ */
24
+ numberedLists?: 'bullets' | 'ordered';
25
+ /**
26
+ * How to handle underlined text. `'ignore'` (default) drops the underline —
27
+ * Mammoth's default, since underlines are easily confused with links in HTML.
28
+ * `'preserve'` keeps it as an inline `<u>…</u>` tag (rendered by GitHub-flavored
29
+ * Markdown). Note that superscript and subscript are always preserved as
30
+ * `<sup>`/`<sub>` and need no option.
31
+ */
32
+ underline?: 'ignore' | 'preserve';
33
+ /**
34
+ * How to render Word's footnotes and endnotes. `'gfm'` (default) rewrites
35
+ * Mammoth's superscript reference links plus trailing note list into standard
36
+ * GitHub-flavored/Pandoc footnote syntax (`[^1]` references and `[^1]:`
37
+ * definitions). `'preserve'` keeps Mammoth's raw `<sup>` links and numbered
38
+ * note list (useful for CommonMark targets that don't support `[^1]`).
39
+ */
40
+ footnotes?: 'gfm' | 'preserve';
41
+ }
42
+ export interface ExtractedImage {
43
+ path: string;
44
+ contentType: string;
45
+ bytes: Uint8Array;
46
+ }
47
+ export interface ConvertResult {
48
+ markdown: string;
49
+ warnings: string[];
50
+ /** Present (possibly empty) when converting with `images: 'extract'`. */
51
+ images?: ExtractedImage[];
52
+ }
53
+ export interface DocumentProperties {
54
+ sensitivity?: string;
55
+ confidentiality?: string;
56
+ encryption?: boolean;
57
+ protection?: boolean;
58
+ }
59
+ export declare class WordToMarkdownError extends Error {
60
+ }
61
+ export declare class UnsupportedFileError extends WordToMarkdownError {
62
+ constructor(message: string);
63
+ }
64
+ export declare class FileNotFoundError extends WordToMarkdownError {
65
+ constructor(filePath?: string);
66
+ }
67
+ export declare class InvalidFileError extends WordToMarkdownError {
68
+ constructor(filePath?: string);
69
+ }
70
+ export declare class FilePermissionError extends WordToMarkdownError {
71
+ constructor(filePath?: string);
72
+ }
73
+ export declare class ConversionError extends WordToMarkdownError {
74
+ constructor(message: string, originalError?: Error);
75
+ }
76
+ export declare function validateFileExtension(filePath: string): void;
77
+ export declare function htmlToMd(html: string, options?: object, keepTags?: string[]): string;
78
+ export declare function extractDocumentProperties(input: string | ArrayBuffer): Promise<DocumentProperties>;
79
+ export declare function generateWarnings(properties: DocumentProperties): string[];
80
+ interface MammothMessage {
81
+ type: string;
82
+ message: string;
83
+ }
84
+ export declare function extensionForContentType(contentType: string): string;
85
+ export declare function extractMammothWarnings(messages: readonly MammothMessage[]): string[];
86
+ export declare function convertWithWarnings(input: string | ArrayBuffer, options?: ConvertOptions): Promise<ConvertResult>;
87
+ export default function convert(input: string | ArrayBuffer, options?: ConvertOptions): Promise<string>;
88
+ export {};