reamkit 1.23.0 → 1.25.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (102) hide show
  1. package/README.md +22 -11
  2. package/dist/esm/core/bmp.d.ts +33 -0
  3. package/dist/esm/core/bmp.js +276 -0
  4. package/dist/esm/core/converter/facade.d.ts +3 -3
  5. package/dist/esm/core/converter/facade.js +12 -0
  6. package/dist/esm/core/converter/ream.d.ts +48 -4
  7. package/dist/esm/core/converter/ream.js +25 -4
  8. package/dist/esm/core/document-model/index.d.ts +1 -1
  9. package/dist/esm/core/document-model/types.d.ts +54 -0
  10. package/dist/esm/core/drawingml/chart-geometry.js +3 -2
  11. package/dist/esm/core/drawingml/diagram/colors.d.ts +52 -0
  12. package/dist/esm/core/drawingml/diagram/colors.js +125 -0
  13. package/dist/esm/core/drawingml/diagram/data-model.d.ts +47 -0
  14. package/dist/esm/core/drawingml/diagram/data-model.js +119 -0
  15. package/dist/esm/core/drawingml/diagram/layout-engine.d.ts +46 -0
  16. package/dist/esm/core/drawingml/diagram/layout-engine.js +948 -0
  17. package/dist/esm/core/drawingml/diagram/run.d.ts +37 -0
  18. package/dist/esm/core/drawingml/diagram/run.js +50 -0
  19. package/dist/esm/core/drawingml/diagram/to-drawing.d.ts +16 -0
  20. package/dist/esm/core/drawingml/diagram/to-drawing.js +86 -0
  21. package/dist/esm/core/drawingml/preset-geometry.js +116 -6
  22. package/dist/esm/core/drawingml/text-warp.d.ts +58 -0
  23. package/dist/esm/core/drawingml/text-warp.js +355 -0
  24. package/dist/esm/core/drawingml/theme-parser.js +9 -3
  25. package/dist/esm/core/font/measure.d.ts +12 -0
  26. package/dist/esm/core/font/measure.js +36 -0
  27. package/dist/esm/core/images.d.ts +30 -2
  28. package/dist/esm/core/images.js +121 -5
  29. package/dist/esm/core/metafile/emf.js +144 -16
  30. package/dist/esm/core/metafile/picture.d.ts +49 -0
  31. package/dist/esm/core/metafile/picture.js +70 -1
  32. package/dist/esm/core/metafile/wmf.js +53 -4
  33. package/dist/esm/core/ole/escher-blip.js +11 -1
  34. package/dist/esm/core/outline.d.ts +17 -0
  35. package/dist/esm/core/outline.js +30 -0
  36. package/dist/esm/core/style-cascade/resolver.js +1 -0
  37. package/dist/esm/core/style-cascade/types.d.ts +3 -1
  38. package/dist/esm/excel/print-model.js +2 -2
  39. package/dist/esm/excel/sheet-drawing.js +1 -1
  40. package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
  41. package/dist/esm/excel/sheet-to-flow.js +30 -3
  42. package/dist/esm/html/html-writer.js +5 -3
  43. package/dist/esm/index.d.ts +3 -0
  44. package/dist/esm/index.js +2 -1
  45. package/dist/esm/layout/page-doc.d.ts +35 -0
  46. package/dist/esm/layout/page-doc.js +14 -1
  47. package/dist/esm/layout/styled-layout.js +217 -30
  48. package/dist/esm/markdown/markdown-writer.d.ts +41 -0
  49. package/dist/esm/markdown/markdown-writer.js +733 -0
  50. package/dist/esm/pdf/shading.js +8 -6
  51. package/dist/esm/pdf/styled-page-emitter.js +62 -4
  52. package/dist/esm/pdf/vector-graphics.js +1 -1
  53. package/dist/esm/pdf-reader/annots.d.ts +24 -0
  54. package/dist/esm/pdf-reader/annots.js +126 -0
  55. package/dist/esm/pdf-reader/content.d.ts +131 -5
  56. package/dist/esm/pdf-reader/content.js +169 -12
  57. package/dist/esm/pdf-reader/display.d.ts +56 -0
  58. package/dist/esm/pdf-reader/display.js +162 -0
  59. package/dist/esm/pdf-reader/document.d.ts +36 -1
  60. package/dist/esm/pdf-reader/document.js +92 -25
  61. package/dist/esm/pdf-reader/embedded-fonts.d.ts +31 -0
  62. package/dist/esm/pdf-reader/embedded-fonts.js +94 -0
  63. package/dist/esm/pdf-reader/flow-build.d.ts +61 -6
  64. package/dist/esm/pdf-reader/flow-build.js +128 -22
  65. package/dist/esm/pdf-reader/font.js +185 -4
  66. package/dist/esm/pdf-reader/image-decode.js +56 -5
  67. package/dist/esm/pdf-reader/images.d.ts +6 -0
  68. package/dist/esm/pdf-reader/images.js +25 -5
  69. package/dist/esm/pdf-reader/jpeg.d.ts +18 -0
  70. package/dist/esm/pdf-reader/jpeg.js +419 -0
  71. package/dist/esm/pdf-reader/layout.d.ts +1 -1
  72. package/dist/esm/pdf-reader/layout.js +221 -32
  73. package/dist/esm/pdf-reader/pattern-tint.d.ts +17 -0
  74. package/dist/esm/pdf-reader/pattern-tint.js +181 -0
  75. package/dist/esm/pdf-reader/reader.d.ts +9 -1
  76. package/dist/esm/pdf-reader/reader.js +22 -6
  77. package/dist/esm/pdf-reader/shading.d.ts +14 -0
  78. package/dist/esm/pdf-reader/shading.js +27 -1
  79. package/dist/esm/pdf-reader/tagged.js +156 -17
  80. package/dist/esm/pdf-reader/text.d.ts +13 -1
  81. package/dist/esm/pdf-reader/text.js +70 -3
  82. package/dist/esm/pdf-reader/vector.d.ts +25 -1
  83. package/dist/esm/pdf-reader/vector.js +168 -12
  84. package/dist/esm/pptx/ppt/ppt-reader.js +95 -27
  85. package/dist/esm/pptx/ppt/ppt-text.d.ts +41 -0
  86. package/dist/esm/pptx/ppt/ppt-text.js +327 -37
  87. package/dist/esm/pptx/pptx-reader.js +90 -10
  88. package/dist/esm/pptx/slide-parser.d.ts +4 -1
  89. package/dist/esm/pptx/slide-parser.js +104 -13
  90. package/dist/esm/pptx/sp-helpers.d.ts +7 -3
  91. package/dist/esm/pptx/sp-helpers.js +45 -7
  92. package/dist/esm/pptx/table-style.d.ts +7 -0
  93. package/dist/esm/pptx/table-style.js +59 -7
  94. package/dist/esm/svg/svg-writer.js +1 -0
  95. package/dist/esm/word/document-parser.d.ts +4 -1
  96. package/dist/esm/word/document-parser.js +4 -1
  97. package/dist/esm/word/docx-reader.js +26 -13
  98. package/dist/esm/word/docx-writer.js +15 -1
  99. package/dist/esm/word/drawing-parser.d.ts +20 -0
  100. package/dist/esm/word/drawing-parser.js +44 -5
  101. package/dist/esm/word/table-parser.js +3 -0
  102. package/package.json +1 -1
@@ -0,0 +1,733 @@
1
+ import { FEATURES } from "../core/ir/features.js";
2
+ import { headingLevelOf } from "../core/outline.js";
3
+ import { detectImageFormat } from "../core/images.js";
4
+ import { effectiveAbstract } from "../core/numbering/state.js";
5
+ import { EMPTY_STYLE_SHEET, resolveParagraphProperties, resolveRunProperties } from "../core/style-cascade/resolver.js";
6
+ import "../core/style-cascade/index.js";
7
+ import { sanitizeHref } from "../core/links.js";
8
+ import { toBase64 } from "../core/bytes.js";
9
+ //#region src/markdown/markdown-writer.ts
10
+ /**
11
+ * Render a {@link FlowDoc} to GitHub-Flavored Markdown (ir-design §7).
12
+ *
13
+ * A flow medium: no pagination, no layout engine and no fonts, so this is a
14
+ * pure, zero-I/O transform. Markdown says far less than the document model
15
+ * does — alignment, indents, colour, font metrics, tab stops and page
16
+ * geometry have no expression at all — so those are dropped and reported as
17
+ * {@link Loss} entries, deduplicated so one recurring omission reports once.
18
+ *
19
+ * @param flow The format-neutral interlayer document tree.
20
+ * @param options Picture handling; see {@link MarkdownWriteOptions}.
21
+ * @returns The encoded Markdown bytes plus the recorded {@link Loss} list.
22
+ */
23
+ function writeMarkdown(flow, options = {}) {
24
+ const ctx = {
25
+ losses: [],
26
+ seen: /* @__PURE__ */ new Set(),
27
+ images: options.images ?? "dataUri",
28
+ pageBreaks: options.pageBreaks ?? "drop",
29
+ list: [],
30
+ inCell: false,
31
+ resources: flow.resources,
32
+ anchors: referencedAnchors(flow.body),
33
+ notes: noteNumbers(flow),
34
+ mediaNames: /* @__PURE__ */ new Map(),
35
+ ...flow.numbering ? { numbering: flow.numbering } : {}
36
+ };
37
+ if (flow.headersFooters && flow.headersFooters.size > 0) lose(ctx, "dropped", FEATURES.headersFooters, "markdown has no pages to band headers onto");
38
+ const blocks = [];
39
+ for (const el of flow.body) emitBlock(blocks, el, ctx);
40
+ emitNoteDefinitions(blocks, flow, ctx);
41
+ return {
42
+ bytes: new TextEncoder().encode(joinBlocks(blocks)),
43
+ losses: ctx.losses
44
+ };
45
+ }
46
+ /**
47
+ * The flow-medium {@link DocumentWriter} adapter (id `'md'`), wrapping
48
+ * {@link writeMarkdown}, with the set of {@link FEATURES} it renders.
49
+ */
50
+ var markdownWriter = {
51
+ id: "md",
52
+ consumes: "flow",
53
+ supports: new Set([
54
+ FEATURES.text,
55
+ FEATURES.lists,
56
+ FEATURES.tables,
57
+ FEATURES.images,
58
+ FEATURES.hyperlinks
59
+ ]),
60
+ write: (doc, opts) => writeMarkdown(doc, opts ?? {})
61
+ };
62
+ /** Record a loss unless an identical one was recorded already. */
63
+ function lose(ctx, severity, feature, detail) {
64
+ const key = `${severity}|${feature}|${detail}`;
65
+ if (ctx.seen.has(key)) return;
66
+ ctx.seen.add(key);
67
+ ctx.losses.push({
68
+ severity,
69
+ feature,
70
+ detail
71
+ });
72
+ }
73
+ /**
74
+ * Join the emitted blocks: one blank line between them (markdown's block
75
+ * separator), no leading blank, exactly one trailing newline.
76
+ *
77
+ * Trailing spaces are stripped from every line: they are invisible, editors
78
+ * and formatters eat them, and two of them are markdown's own hard break —
79
+ * a meaning no document ever asked for. Ream writes a hard break as the
80
+ * backslash GFM also accepts, which survives all three.
81
+ */
82
+ function joinBlocks(blocks) {
83
+ const body = blocks.map(trimHardBreaks).filter((b) => b.length > 0).join("\n\n");
84
+ return body.length > 0 ? `${body.replace(/[ \t]+$/gm, "")}\n` : "";
85
+ }
86
+ /**
87
+ * Drop a hard break sitting at either end of a block: at the end there is no
88
+ * next line to break to, at the start no previous one, and either way the
89
+ * backslash is left standing in the text as itself. Only a break is taken —
90
+ * an escaped literal backslash is `\\` with no newline behind it.
91
+ */
92
+ function trimHardBreaks(block) {
93
+ return block.replace(/^(?:\\\n)+/, "").replace(/\\\n$/, "");
94
+ }
95
+ function emitBlock(out, el, ctx) {
96
+ if (el.kind === "paragraph") {
97
+ emitParagraph(out, el.paragraph, ctx);
98
+ return;
99
+ }
100
+ ctx.list.length = 0;
101
+ if (el.kind === "table") emitTable(out, el.table, ctx);
102
+ else if (el.kind === "image") {
103
+ const img = pictureMarkdown(el.image.resource, el.image.altText, ctx);
104
+ if (img.length > 0) out.push(img);
105
+ } else if (el.kind === "chart") lose(ctx, "dropped", FEATURES.charts, "markdown cannot draw a chart");
106
+ else emitShape(out, el.shape, ctx);
107
+ }
108
+ /**
109
+ * A shape contributes its WORDS. Markdown draws no geometry, no fill and no
110
+ * line, but the text inside a callout or a text box is document content and is
111
+ * emitted as ordinary blocks — including a group's members, however deep.
112
+ */
113
+ function emitShape(out, shape, ctx) {
114
+ lose(ctx, "dropped", FEATURES.shapes, "shape geometry dropped; the text inside it is kept");
115
+ for (const el of shape.text?.content ?? []) emitBlock(out, el, ctx);
116
+ for (const child of shape.children ?? []) emitShape(out, child.shape, ctx);
117
+ }
118
+ function emitParagraph(out, p, ctx) {
119
+ const resolved = resolveParagraphProperties(p.properties, EMPTY_STYLE_SHEET);
120
+ const anchors = bookmarkAnchors(p, ctx);
121
+ const text = inlineRuns(p.runs, p, ctx).replace(/^[ \t]+/, "");
122
+ const inline = anchors + text;
123
+ const marker = markerText(p, ctx);
124
+ reportParagraphLosses(resolved, ctx);
125
+ if (breaksPage(p, resolved)) emitRule(out, ctx);
126
+ const empty = isBlank(text) && anchors.length === 0;
127
+ const level = headingLevelOf(resolved);
128
+ if (level !== void 0) {
129
+ if (empty) return;
130
+ ctx.list.length = 0;
131
+ const oneLine = (marker !== void 0 ? `${marker} ${inline}` : inline).replaceAll("\\\n", " ").replaceAll("\n", " ");
132
+ if (ctx.inCell) {
133
+ lose(ctx, "degraded", FEATURES.tables, "heading inside a cell flattened to plain text");
134
+ out.push(oneLine);
135
+ return;
136
+ }
137
+ out.push(`${"#".repeat(level)} ${oneLine}`);
138
+ return;
139
+ }
140
+ if (marker !== void 0) {
141
+ if (!empty) emitListItem(out, resolved, marker, inline, ctx);
142
+ return;
143
+ }
144
+ if (empty) return;
145
+ ctx.list.length = 0;
146
+ out.push(ctx.inCell ? inline : guardBlockStart(inline));
147
+ }
148
+ /**
149
+ * True when a paragraph's rendered text says nothing at all. Whitespace counts
150
+ * as nothing, and so do the ZERO-WIDTH characters — the `.pptx` reader marks a
151
+ * slide boundary with a U+200B paragraph carrying the page break, and a run of
152
+ * them down the left of a deck is a column of empty lines, not content.
153
+ */
154
+ function isBlank(text) {
155
+ return /^[\s\u200B-\u200D\uFEFF]*$/u.test(text);
156
+ }
157
+ /** Whether a page starts at this paragraph — its own break, or one in a run. */
158
+ function breaksPage(p, resolved) {
159
+ return resolved.pageBreakBefore || p.runs.some((r) => r.pageBreak === true);
160
+ }
161
+ /**
162
+ * A `---` thematic break where a page ends, when the caller asked for one.
163
+ *
164
+ * Never leading and never doubled: a rule before the first block would open
165
+ * the document with a line, and two in a row say nothing the one does not.
166
+ * Never inside a cell or a note either — those hold inline content, where the
167
+ * three hyphens are just three hyphens.
168
+ */
169
+ function emitRule(out, ctx) {
170
+ if (ctx.pageBreaks !== "rule" || ctx.inCell) return;
171
+ if (out.length === 0 || out[out.length - 1] === RULE) return;
172
+ out.push(RULE);
173
+ }
174
+ var RULE = "---";
175
+ /**
176
+ * The marker `applyNumbering` materialized as the paragraph's leading runs
177
+ * (`"1."`, `"•"`, or a picture bullet), stripped of the tab that follows it —
178
+ * or `undefined` when the paragraph is not a list item. The text is taken raw:
179
+ * it is re-rendered as markup, never emitted as content.
180
+ */
181
+ function markerText(p, ctx) {
182
+ const runs = [];
183
+ for (const run of p.runs) {
184
+ if (run.listMarker !== true) break;
185
+ runs.push(run);
186
+ }
187
+ if (runs.length === 0) return void 0;
188
+ if (runs.some((r) => r.inlineImage !== void 0)) lose(ctx, "degraded", FEATURES.lists, "picture bullet rendered as a plain bullet");
189
+ return runs.map((r) => r.text).join("").replaceAll(" ", " ").trim();
190
+ }
191
+ function emitListItem(out, resolved, marker, inline, ctx) {
192
+ const ref = resolved.numbering;
193
+ const numId = ref?.numId ?? "";
194
+ const ilvl = ref?.ilvl ?? 0;
195
+ const stack = ctx.list;
196
+ const wasOpen = stack.length > 0;
197
+ while (stack.length > 0 && stack[stack.length - 1].ilvl > ilvl) stack.pop();
198
+ let top = stack[stack.length - 1];
199
+ if (!top || top.ilvl < ilvl) {
200
+ const parent = top;
201
+ top = {
202
+ ilvl,
203
+ indent: parent ? parent.indent + parent.markerWidth : 0,
204
+ markerWidth: 2,
205
+ counter: 0
206
+ };
207
+ stack.push(top);
208
+ }
209
+ top.counter += 1;
210
+ const bullet = listBullet(marker, numId, ilvl, top.counter, ctx);
211
+ top.markerWidth = bullet.length + 1;
212
+ const pad = " ".repeat(top.indent);
213
+ const cont = " ".repeat(top.indent + top.markerWidth);
214
+ const line = `${pad}${bullet} ${inline.replaceAll("\\\n", `\\\n${cont}`)}`.replace(/\s+$/, "");
215
+ if (wasOpen && out.length > 0) out[out.length - 1] += `\n${line}`;
216
+ else out.push(line);
217
+ }
218
+ /**
219
+ * The markdown marker for an item: `-` for a bullet, `N.` for an ordered list.
220
+ * Ordered lists keep their real number — the one the source's own marker states,
221
+ * so a list starting at 5 (§17.9.28 `w:startOverride`) still starts at 5 —
222
+ * falling back to this level's running count when the marker states no digits.
223
+ */
224
+ function listBullet(marker, numId, ilvl, counter, ctx) {
225
+ const level = numberingLevel(numId, ilvl, ctx.numbering);
226
+ if (!(level ? level.format !== "bullet" && level.format !== "none" : /\d/.test(marker))) return "-";
227
+ if (level && level.format !== "decimal" && level.format !== "decimalZero") lose(ctx, "degraded", FEATURES.lists, `${level.format} list markers render as decimal`);
228
+ const digits = /(\d+)\D*$/.exec(marker);
229
+ if (marker.replace(/\D+/g, "").length > (digits?.[1]?.length ?? 0)) lose(ctx, "degraded", FEATURES.lists, "multi-level list marker flattened to one number");
230
+ return `${digits ? Number(digits[1]) : counter}.`;
231
+ }
232
+ function numberingLevel(numId, ilvl, numbering) {
233
+ if (!numbering) return void 0;
234
+ const instance = numbering.numInstances.get(numId);
235
+ if (!instance) return void 0;
236
+ return effectiveAbstract(numbering, instance)?.levels.get(ilvl);
237
+ }
238
+ /** Everything a paragraph says that markdown has no way to say back. */
239
+ function reportParagraphLosses(r, ctx) {
240
+ if (r.alignment !== "left") lose(ctx, "dropped", FEATURES.text, "paragraph alignment has no markdown expression");
241
+ if (r.indentLeft !== 0 || r.indentRight !== 0 || r.indentFirstLine !== 0) lose(ctx, "dropped", FEATURES.text, "paragraph indents have no markdown expression");
242
+ if (r.pageBreakBefore && ctx.pageBreaks === "drop") lose(ctx, "dropped", FEATURES.sections, "page breaks have no markdown expression");
243
+ }
244
+ /**
245
+ * Backslash-escape a leading character that would otherwise open a block the
246
+ * source never asked for — a heading, a quote, a list item, a setext rule.
247
+ */
248
+ function guardBlockStart(text) {
249
+ return text.replace(/^(\s*)([#>+-]|\d+[.)]|={2,}$)/, "$1\\$2");
250
+ }
251
+ /**
252
+ * Number the notes by the order their references appear in reading order
253
+ * (§17.11: footnotes and endnotes each keep their own counter), and the review
254
+ * comments alongside them.
255
+ *
256
+ * Only ids the package actually holds content for are numbered: an unmatched
257
+ * `[^fn1]` is not a footnote reference at all in GFM — it renders as those
258
+ * six literal characters — so a dangling reference must leave no mark.
259
+ */
260
+ function noteNumbers(flow) {
261
+ const footnotes = /* @__PURE__ */ new Map();
262
+ const endnotes = /* @__PURE__ */ new Map();
263
+ const comments = /* @__PURE__ */ new Map();
264
+ const visitShape = (shape) => {
265
+ if (shape.text) visit(shape.text.content);
266
+ for (const child of shape.children ?? []) visitShape(child.shape);
267
+ };
268
+ const visit = (els) => {
269
+ for (const el of els) if (el.kind === "paragraph") for (const r of el.paragraph.runs) {
270
+ const add = (id, into, content) => {
271
+ if (id === void 0 || into.has(id) || !content?.has(id)) return;
272
+ into.set(id, into.size + 1);
273
+ };
274
+ add(r.footnoteRef, footnotes, flow.footnotes);
275
+ add(r.endnoteRef, endnotes, flow.endnotes);
276
+ add(r.commentRef, comments, flow.comments);
277
+ }
278
+ else if (el.kind === "table") for (const row of el.table.rows) for (const cell of row.cells) visit(cell.content);
279
+ else if (el.kind === "shape") visitShape(el.shape);
280
+ };
281
+ visit(flow.body);
282
+ return {
283
+ footnotes,
284
+ endnotes,
285
+ comments
286
+ };
287
+ }
288
+ /**
289
+ * The GFM footnote definitions every reference above points at, in reference
290
+ * order: the notes first, then the review comments — which markdown has no
291
+ * concept of, and which are carried as footnotes attributed to their author.
292
+ */
293
+ function emitNoteDefinitions(out, flow, ctx) {
294
+ define(out, ctx.notes.footnotes, flow.footnotes, "fn", ctx);
295
+ define(out, ctx.notes.endnotes, flow.endnotes, "en", ctx);
296
+ for (const [id, n] of sorted(ctx.notes.comments)) {
297
+ const comment = flow.comments?.get(id);
298
+ if (!comment) continue;
299
+ const who = comment.author !== void 0 ? `**${escapeInline(comment.author)}:** ` : "";
300
+ out.push(`[^cm${String(n)}]: ${who}${flatten(comment.content, ctx)}`);
301
+ }
302
+ }
303
+ function define(out, numbers, content, prefix, ctx) {
304
+ for (const [id, n] of sorted(numbers)) {
305
+ const blocks = content?.get(id);
306
+ if (!blocks) continue;
307
+ out.push(`[^${prefix}${String(n)}]: ${flatten(blocks, ctx)}`);
308
+ }
309
+ }
310
+ var sorted = (m) => [...m.entries()].sort((a, b) => a[1] - b[1]);
311
+ /**
312
+ * Blocks flattened to the single line a footnote definition occupies. Note
313
+ * content is short by nature; a note that holds several paragraphs keeps them
314
+ * all, joined by the line break the syntax allows.
315
+ */
316
+ function flatten(blocks, ctx) {
317
+ const wasInCell = ctx.inCell;
318
+ const outer = ctx.list.splice(0);
319
+ ctx.inCell = true;
320
+ const rendered = [];
321
+ for (const el of blocks) emitBlock(rendered, el, ctx);
322
+ ctx.inCell = wasInCell;
323
+ ctx.list.length = 0;
324
+ ctx.list.push(...outer);
325
+ return rendered.map(trimHardBreaks).filter((b) => b.length > 0).join("<br>").replaceAll("\\\n", "<br>").replaceAll("\n", "<br>");
326
+ }
327
+ /**
328
+ * Wrap already-rendered inline content in the link it carries. An external
329
+ * target passes the scheme allowlist first (`core/links`) — untrusted input
330
+ * never becomes a clickable link on a scheme markdown viewers will follow.
331
+ */
332
+ function linkify(inner, target, ctx) {
333
+ if (inner.length === 0) return inner;
334
+ if (target.href !== void 0) {
335
+ const safe = sanitizeHref(target.href);
336
+ if (safe === void 0) {
337
+ lose(ctx, "degraded", FEATURES.hyperlinks, "hyperlink target with a disallowed scheme rendered as plain text");
338
+ return inner;
339
+ }
340
+ return `[${inner}](${destination(safe)})`;
341
+ }
342
+ if (target.anchor !== void 0) return `[${inner}](#${bookmarkSlug(target.anchor)})`;
343
+ return inner;
344
+ }
345
+ /**
346
+ * A link destination: bare when it is plain enough, and in the pointy-bracket
347
+ * form CommonMark §6.3 provides when it holds whitespace or unbalanced
348
+ * parentheses that would otherwise end it early.
349
+ */
350
+ function destination(url) {
351
+ if (!/[\s()<>]/.test(url)) return url;
352
+ return `<${url.replaceAll("<", "%3C").replaceAll(">", "%3E")}>`;
353
+ }
354
+ /**
355
+ * The fragment an internal link points at. Markdown has no bookmark of its
356
+ * own, so the name is slugified and planted as an inline `<a id>` on the
357
+ * paragraph it belongs to — the same shape GFM's own heading anchors take.
358
+ */
359
+ function bookmarkSlug(name) {
360
+ const slug = name.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, "-").replace(/^-+|-+$/g, "");
361
+ return slug.length > 0 ? slug : "bookmark";
362
+ }
363
+ /** The bookmark names some run actually links to — the only ones worth an anchor. */
364
+ function referencedAnchors(body) {
365
+ const out = /* @__PURE__ */ new Set();
366
+ const visitShape = (shape) => {
367
+ if (shape.text) visit(shape.text.content);
368
+ for (const child of shape.children ?? []) visitShape(child.shape);
369
+ };
370
+ const visit = (els) => {
371
+ for (const el of els) if (el.kind === "paragraph") {
372
+ for (const r of el.paragraph.runs) if (r.anchor !== void 0) out.add(r.anchor);
373
+ } else if (el.kind === "table") for (const row of el.table.rows) for (const cell of row.cells) visit(cell.content);
374
+ else if (el.kind === "shape") visitShape(el.shape);
375
+ };
376
+ visit(body);
377
+ return out;
378
+ }
379
+ /** The `<a id>` markers a paragraph opens, for the bookmarks something links to. */
380
+ function bookmarkAnchors(p, ctx) {
381
+ return (p.bookmarks ?? []).filter((b) => ctx.anchors.has(b)).map((b) => `<a id="${bookmarkSlug(b)}"></a>`).join("");
382
+ }
383
+ var MIME_BY_FORMAT = {
384
+ jpeg: "image/jpeg",
385
+ png: "image/png",
386
+ jpeg2000: "image/jp2",
387
+ gif: "image/gif",
388
+ tiff: "image/tiff",
389
+ bmp: "image/bmp"
390
+ };
391
+ var EXTENSION_BY_FORMAT = {
392
+ jpeg: "jpg",
393
+ png: "png",
394
+ jpeg2000: "jp2",
395
+ gif: "gif",
396
+ tiff: "tif",
397
+ bmp: "bmp"
398
+ };
399
+ /**
400
+ * A picture as `![alt](…)`, or `''` when there is nothing to point at. The
401
+ * destination follows {@link MarkdownWriteOptions.images}: inlined as a
402
+ * `data:` URI by default, named under `./media/` when the caller means to
403
+ * write the bytes out itself, or dropped.
404
+ */
405
+ function pictureMarkdown(resource, altText, ctx) {
406
+ if (ctx.images === "drop") {
407
+ lose(ctx, "dropped", FEATURES.images, "pictures omitted at the caller’s request");
408
+ return "";
409
+ }
410
+ if (resource === void 0) return "";
411
+ const bytes = ctx.resources.get(resource);
412
+ if (!bytes) return "";
413
+ const format = detectImageFormat(bytes);
414
+ if (!format) {
415
+ lose(ctx, "dropped", FEATURES.images, "picture in a format markdown viewers cannot show");
416
+ return "";
417
+ }
418
+ const alt = escapeInline(altText ?? "").replaceAll("\n", " ");
419
+ if (ctx.images === "link") {
420
+ let name = ctx.mediaNames.get(resource);
421
+ if (name === void 0) {
422
+ name = `./media/image${String(ctx.mediaNames.size + 1)}.${EXTENSION_BY_FORMAT[format]}`;
423
+ ctx.mediaNames.set(resource, name);
424
+ }
425
+ lose(ctx, "degraded", FEATURES.images, "pictures referenced under ./media — a single-file writer does not write their bytes");
426
+ return `![${alt}](${name})`;
427
+ }
428
+ return `![${alt}](data:${MIME_BY_FORMAT[format]};base64,${toBase64(bytes)})`;
429
+ }
430
+ /**
431
+ * A GFM pipe table (§4.10). Markdown's table is a plain grid: it has no
432
+ * merged cells and no headerless form, so a `w:gridSpan` fills its first
433
+ * column and pads the rest empty, a `w:vMerge` continuation stays empty, and
434
+ * the first row is promoted to the header the syntax requires. Each of those
435
+ * is reported once.
436
+ */
437
+ function emitTable(out, table, ctx) {
438
+ const width = columnCount(table);
439
+ if (width === 0 || table.rows.length === 0) return;
440
+ const grid = table.rows.map((row) => rowCells(row, width, ctx));
441
+ const keep = usedRange(grid);
442
+ if (!keep) return;
443
+ const rows = grid.slice(keep.top, keep.bottom + 1).map((row) => row.slice(keep.left, keep.right + 1));
444
+ const header = rows[0];
445
+ if (table.rows[keep.top].properties.isHeader !== true) lose(ctx, "degraded", FEATURES.tables, "first row promoted to the header GFM tables require");
446
+ const line = (cells) => `| ${cells.join(" | ")} |`;
447
+ const bars = delimiters(table, width).slice(keep.left, keep.right + 1);
448
+ const lines = [line(header), line(bars)];
449
+ for (const row of rows.slice(1)) lines.push(line(row));
450
+ out.push(lines.join("\n"));
451
+ }
452
+ /**
453
+ * The rows and columns worth drawing: the blank ones at the EDGES go.
454
+ *
455
+ * A print region carries its whole used range, blank leading rows and trailing
456
+ * columns and all, because a page draws their borders and their fill. Markdown
457
+ * draws neither, so an edge of empty cells is a column of nothing — and a blank
458
+ * first row is worse than nothing, since §4.10 makes it the header. The blank
459
+ * rows INSIDE stay: those separate one group of data from the next, which is
460
+ * the one thing an empty row can still say here.
461
+ */
462
+ function usedRange(grid) {
463
+ const filled = (r, c) => (grid[r]?.[c] ?? "").trim().length > 0;
464
+ const width = grid[0]?.length ?? 0;
465
+ let top = 0;
466
+ let bottom = grid.length - 1;
467
+ let left = 0;
468
+ let right = width - 1;
469
+ const rowFilled = (r) => grid[r].some((_, c) => filled(r, c));
470
+ const colFilled = (c) => grid.some((_, r) => filled(r, c));
471
+ while (top <= bottom && !rowFilled(top)) top++;
472
+ if (top > bottom) return void 0;
473
+ while (bottom > top && !rowFilled(bottom)) bottom--;
474
+ while (left <= right && !colFilled(left)) left++;
475
+ while (right > left && !colFilled(right)) right--;
476
+ return {
477
+ top,
478
+ bottom,
479
+ left,
480
+ right
481
+ };
482
+ }
483
+ /** The widest row wins: a `w:gridSpan` may reach past the declared `w:tblGrid`. */
484
+ function columnCount(table) {
485
+ let widest = table.grid.length;
486
+ for (const row of table.rows) {
487
+ let n = 0;
488
+ for (const cell of row.cells) n += cell.properties.colSpan ?? 1;
489
+ widest = Math.max(widest, n);
490
+ }
491
+ return widest;
492
+ }
493
+ /** One row's cells, spans expanded to empty columns and padded to `cols`. */
494
+ function rowCells(row, cols, ctx) {
495
+ const cells = [];
496
+ for (const cell of row.cells) {
497
+ const span = cell.properties.colSpan ?? 1;
498
+ if (span > 1 || cell.properties.merge !== void 0 && cell.properties.merge !== "start") lose(ctx, "degraded", FEATURES.tables, "merged cells flattened — markdown has no spans");
499
+ cells.push(cell.properties.merge === "middle" || cell.properties.merge === "end" ? "" : cellInline(cell, ctx));
500
+ for (let k = 1; k < span; k++) cells.push("");
501
+ }
502
+ while (cells.length < cols) cells.push("");
503
+ return cells.slice(0, cols);
504
+ }
505
+ /**
506
+ * The delimiter row, carrying each column's alignment when the header cell's
507
+ * first paragraph states one — the only paragraph property a markdown table
508
+ * can keep.
509
+ */
510
+ function delimiters(table, cols) {
511
+ const out = [];
512
+ const first = table.rows[0];
513
+ let col = 0;
514
+ for (const cell of first.cells) {
515
+ const span = cell.properties.colSpan ?? 1;
516
+ const p = cell.content.find((el) => el.kind === "paragraph");
517
+ const align = p ? resolveParagraphProperties(p.paragraph.properties, EMPTY_STYLE_SHEET).alignment : "left";
518
+ const bar = align === "center" ? ":---:" : align === "right" ? "---:" : "---";
519
+ for (let k = 0; k < span && col < cols; k++, col++) out.push(bar);
520
+ }
521
+ while (out.length < cols) out.push("---");
522
+ return out.slice(0, cols);
523
+ }
524
+ /**
525
+ * A cell's blocks flattened to one line: a table row is a single line, so
526
+ * every break between and inside its blocks becomes a `<br>`. The cell's own
527
+ * lists number independently of whatever list surrounds the table.
528
+ *
529
+ * Each block is trimmed at its ends. Word carries a cell's padding as spaces in
530
+ * the text itself, and a pipe table already sets its own — kept, they only
531
+ * widen the source line, since §4.10 strips a cell's outer whitespace before
532
+ * rendering it anyway.
533
+ */
534
+ function cellInline(cell, ctx) {
535
+ const outer = ctx.list.splice(0);
536
+ const wasInCell = ctx.inCell;
537
+ ctx.inCell = true;
538
+ const blocks = [];
539
+ for (const el of cell.content) {
540
+ if (el.kind === "table") {
541
+ lose(ctx, "degraded", FEATURES.tablesNested, "nested table flattened into its cell");
542
+ for (const row of el.table.rows) for (const inner of row.cells) blocks.push(cellInline(inner, ctx));
543
+ continue;
544
+ }
545
+ emitBlock(blocks, el, ctx);
546
+ }
547
+ ctx.inCell = wasInCell;
548
+ ctx.list.length = 0;
549
+ ctx.list.push(...outer);
550
+ return blocks.map((b) => trimHardBreaks(b).trim()).filter((b) => b.length > 0).join("<br>").replaceAll("\\\n", "<br>").replaceAll("\n", "<br>");
551
+ }
552
+ /**
553
+ * The paragraph's runs as one inline string.
554
+ *
555
+ * Two passes, because whether a delimiter may open or close depends on the
556
+ * characters on either side of it (§6.2), and those belong to the NEIGHBOURING
557
+ * spans: the first pass coalesces runs into spans, the second dresses each
558
+ * span knowing what stands beside it.
559
+ */
560
+ function inlineRuns(runs, p, ctx) {
561
+ const pieces = [];
562
+ let pending;
563
+ const flush = () => {
564
+ if (pending && pending.text.length > 0) pieces.push(pending);
565
+ pending = void 0;
566
+ };
567
+ for (const run of runs) {
568
+ const literal = literalRun(run, ctx);
569
+ if (literal !== void 0) {
570
+ flush();
571
+ if (literal.length > 0) pieces.push({ literal });
572
+ continue;
573
+ }
574
+ if (run.text.length === 0) continue;
575
+ const resolved = resolveRunProperties(run.properties, p.properties, EMPTY_STYLE_SHEET);
576
+ reportRunLosses(resolved, ctx);
577
+ const marks = marksOf(resolved, run);
578
+ if (pending && sameMarks(pending.marks, marks)) pending.text += run.text;
579
+ else {
580
+ flush();
581
+ pending = {
582
+ marks,
583
+ text: run.text
584
+ };
585
+ }
586
+ }
587
+ flush();
588
+ let out = "";
589
+ for (let i = 0; i < pieces.length; i++) {
590
+ const piece = pieces[i];
591
+ if ("literal" in piece) {
592
+ out += piece.literal;
593
+ continue;
594
+ }
595
+ out += applyMarks(piece.text, piece.marks, ctx, out.slice(-1), leadOf(pieces, i + 1));
596
+ }
597
+ return out;
598
+ }
599
+ /**
600
+ * The first character the next span will contribute. A dressed span always
601
+ * begins with its delimiter or its html tag — both punctuation — so a marked
602
+ * neighbour answers without being rendered first.
603
+ */
604
+ function leadOf(pieces, from) {
605
+ const next = pieces[from];
606
+ if (!next) return "";
607
+ if ("literal" in next) return next.literal.slice(0, 1);
608
+ if (wrapped(next.marks)) return "*";
609
+ return escapeInline(next.text).slice(0, 1);
610
+ }
611
+ /** Whether a span carries anything that puts a delimiter or a tag in front of it. */
612
+ function wrapped(m) {
613
+ return m.bold || m.italic || m.strike || m.underline || m.vertical !== "baseline" || m.href !== void 0 || m.anchor !== void 0;
614
+ }
615
+ /**
616
+ * The runs that are not a span of styled text: reference markers, inline
617
+ * objects, and the list marker the reader materialized. Returns the literal
618
+ * they contribute (possibly `''`), or `undefined` when the run IS styled text.
619
+ */
620
+ function literalRun(run, ctx) {
621
+ if (run.footnoteRef !== void 0 || run.endnoteRef !== void 0) {
622
+ const foot = run.footnoteRef !== void 0;
623
+ const id = foot ? run.footnoteRef : run.endnoteRef;
624
+ const n = (foot ? ctx.notes.footnotes : ctx.notes.endnotes).get(id);
625
+ return n === void 0 ? "" : `[^${foot ? "fn" : "en"}${String(n)}]`;
626
+ }
627
+ if (run.commentRef !== void 0) {
628
+ const n = ctx.notes.comments.get(run.commentRef);
629
+ if (n === void 0) return "";
630
+ lose(ctx, "degraded", FEATURES.trackedChanges, "review comments rendered as footnotes");
631
+ return `[^cm${String(n)}]`;
632
+ }
633
+ if (run.noteNumber) return "";
634
+ if (run.listMarker) return "";
635
+ if (run.math !== void 0) {
636
+ lose(ctx, "dropped", FEATURES.math, "markdown has no math notation (GFM)");
637
+ return "";
638
+ }
639
+ if (run.inlineImage !== void 0) return linkify(pictureMarkdown(run.inlineImage.resource, void 0, ctx), run, ctx);
640
+ if (run.columnBreak === true || run.pageBreak === true && ctx.pageBreaks === "drop") lose(ctx, "dropped", FEATURES.sections, "page breaks have no markdown expression");
641
+ }
642
+ /** Everything a run says that markdown has no way to say back. */
643
+ function reportRunLosses(r, ctx) {
644
+ if (r.colorHex !== "000000") lose(ctx, "dropped", FEATURES.text, "run colour has no markdown expression");
645
+ if (r.caps || r.smallCaps) lose(ctx, "dropped", FEATURES.text, "w:caps / w:smallCaps has no markdown expression");
646
+ }
647
+ function marksOf(r, run) {
648
+ return {
649
+ bold: r.bold,
650
+ italic: r.italic,
651
+ strike: r.strike,
652
+ underline: r.underline !== "none",
653
+ vertical: r.verticalAlign === "superscript" || r.verticalAlign === "subscript" ? r.verticalAlign : "baseline",
654
+ ...run.href !== void 0 ? { href: run.href } : {},
655
+ ...run.anchor !== void 0 ? { anchor: run.anchor } : {}
656
+ };
657
+ }
658
+ function sameMarks(a, b) {
659
+ return a.bold === b.bold && a.italic === b.italic && a.strike === b.strike && a.underline === b.underline && a.vertical === b.vertical && a.href === b.href && a.anchor === b.anchor;
660
+ }
661
+ /**
662
+ * Escape the text, then dress it: markdown delimiters innermost, then the
663
+ * inline-html wrappers for what markdown cannot say. CommonMark §6.2 requires
664
+ * a delimiter to hug non-whitespace, so any leading or trailing space of the
665
+ * span moves outside the marks.
666
+ *
667
+ * A HARD BREAK at either end moves out too, and it matters more than a space
668
+ * does: a closing `**` that a line break stands in front of is not
669
+ * right-flanking, cannot close, and prints as two asterisks. A `w:br` at the
670
+ * end of a bold title — which every deck writes — did exactly that.
671
+ */
672
+ var SPAN_EDGE = String.raw`(?:\\\n|\s)`;
673
+ function applyMarks(text, marks, ctx, before, after) {
674
+ const escaped = escapeInline(text);
675
+ const lead = new RegExp(`^${SPAN_EDGE}*`).exec(escaped)?.[0] ?? "";
676
+ const trail = new RegExp(`${SPAN_EDGE}*$`).exec(escaped)?.[0] ?? "";
677
+ const core = escaped.slice(lead.length, escaped.length - trail.length);
678
+ if (core.length === 0) return escaped;
679
+ const outerBefore = lead.length > 0 ? lead[lead.length - 1] : before;
680
+ const outerAfter = trail.length > 0 ? trail[0] : after;
681
+ let out = core;
682
+ const levels = [
683
+ [
684
+ marks.italic,
685
+ "*",
686
+ "em"
687
+ ],
688
+ [
689
+ marks.bold,
690
+ "**",
691
+ "strong"
692
+ ],
693
+ [
694
+ marks.strike,
695
+ "~~",
696
+ "del"
697
+ ]
698
+ ];
699
+ for (const [on, delimiter, tag] of levels) {
700
+ if (!on) continue;
701
+ out = canOpen(outerBefore, out[0]) && canClose(out[out.length - 1], outerAfter) ? `${delimiter}${out}${delimiter}` : `<${tag}>${out}</${tag}>`;
702
+ }
703
+ if (marks.underline) out = `<u>${out}</u>`;
704
+ if (marks.vertical !== "baseline") {
705
+ const tag = marks.vertical === "superscript" ? "sup" : "sub";
706
+ out = `<${tag}>${out}</${tag}>`;
707
+ }
708
+ return `${lead}${linkify(out, marks, ctx)}${trail}`;
709
+ }
710
+ var isSpace = (c) => c === "" || /\s/u.test(c);
711
+ var isPunct = (c) => /[\p{P}\p{S}]/u.test(c);
712
+ /** §6.2 — the delimiter run is left-flanking, so it can open emphasis. */
713
+ function canOpen(before, after) {
714
+ return !isSpace(after) && (!isPunct(after) || isSpace(before) || isPunct(before));
715
+ }
716
+ /** §6.2 — the delimiter run is right-flanking, so it can close emphasis. */
717
+ function canClose(before, after) {
718
+ return !isSpace(before) && (!isPunct(before) || isSpace(after) || isPunct(after));
719
+ }
720
+ /**
721
+ * Escape the inline text so nothing in it is read as markup, and map the two
722
+ * control characters the model uses: `\t` (a `w:tab`, which has no medium
723
+ * here) to a space, and `\n` (a `w:br` soft break) to GFM's backslash hard
724
+ * break.
725
+ *
726
+ * `_` is escaped only at a word boundary — GFM does not open intraword
727
+ * emphasis with it, so escaping every one would turn `snake_case` into noise.
728
+ */
729
+ function escapeInline(text) {
730
+ return text.replace(/[\\`*[\]<>|~]/g, (c) => `\\${c}`).replace(/(^|[\s\p{P}])_|_(?=[\s\p{P}]|$)/gu, (m) => m.replace("_", "\\_")).replaceAll(" ", " ").replaceAll("\n", "\\\n");
731
+ }
732
+ //#endregion
733
+ export { markdownWriter, writeMarkdown };