reamkit 1.24.0 → 1.25.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -11
- package/dist/esm/core/converter/facade.d.ts +3 -3
- package/dist/esm/core/converter/facade.js +12 -0
- package/dist/esm/core/converter/ream.d.ts +48 -4
- package/dist/esm/core/converter/ream.js +25 -4
- package/dist/esm/core/document-model/index.d.ts +1 -1
- package/dist/esm/core/document-model/types.d.ts +17 -0
- package/dist/esm/core/outline.d.ts +17 -0
- package/dist/esm/core/outline.js +30 -0
- package/dist/esm/core/style-cascade/resolver.js +1 -0
- package/dist/esm/core/style-cascade/types.d.ts +3 -1
- package/dist/esm/excel/sheet-to-flow.d.ts +10 -0
- package/dist/esm/excel/sheet-to-flow.js +14 -1
- package/dist/esm/html/html-writer.js +3 -2
- package/dist/esm/index.d.ts +3 -0
- package/dist/esm/index.js +2 -1
- package/dist/esm/layout/page-doc.js +1 -1
- package/dist/esm/layout/styled-layout.js +41 -15
- package/dist/esm/markdown/markdown-writer.d.ts +41 -0
- package/dist/esm/markdown/markdown-writer.js +733 -0
- package/dist/esm/pdf/styled-page-emitter.js +20 -1
- package/dist/esm/pdf-reader/annots.d.ts +24 -0
- package/dist/esm/pdf-reader/annots.js +126 -0
- package/dist/esm/pdf-reader/content.d.ts +131 -5
- package/dist/esm/pdf-reader/content.js +169 -12
- package/dist/esm/pdf-reader/display.d.ts +56 -0
- package/dist/esm/pdf-reader/display.js +162 -0
- package/dist/esm/pdf-reader/document.d.ts +36 -1
- package/dist/esm/pdf-reader/document.js +92 -25
- package/dist/esm/pdf-reader/embedded-fonts.d.ts +31 -0
- package/dist/esm/pdf-reader/embedded-fonts.js +94 -0
- package/dist/esm/pdf-reader/flow-build.d.ts +61 -6
- package/dist/esm/pdf-reader/flow-build.js +128 -22
- package/dist/esm/pdf-reader/font.js +185 -4
- package/dist/esm/pdf-reader/image-decode.js +55 -4
- package/dist/esm/pdf-reader/images.d.ts +6 -0
- package/dist/esm/pdf-reader/images.js +25 -5
- package/dist/esm/pdf-reader/jpeg.d.ts +18 -0
- package/dist/esm/pdf-reader/jpeg.js +419 -0
- package/dist/esm/pdf-reader/layout.d.ts +1 -1
- package/dist/esm/pdf-reader/layout.js +221 -32
- package/dist/esm/pdf-reader/pattern-tint.d.ts +17 -0
- package/dist/esm/pdf-reader/pattern-tint.js +181 -0
- package/dist/esm/pdf-reader/reader.d.ts +9 -1
- package/dist/esm/pdf-reader/reader.js +22 -6
- package/dist/esm/pdf-reader/shading.d.ts +14 -0
- package/dist/esm/pdf-reader/shading.js +27 -1
- package/dist/esm/pdf-reader/tagged.js +156 -17
- package/dist/esm/pdf-reader/text.d.ts +13 -1
- package/dist/esm/pdf-reader/text.js +70 -3
- package/dist/esm/pdf-reader/vector.d.ts +25 -1
- package/dist/esm/pdf-reader/vector.js +168 -12
- package/dist/esm/pptx/slide-parser.js +5 -0
- package/dist/esm/word/docx-writer.js +11 -1
- package/dist/esm/word/drawing-parser.js +7 -1
- package/package.json +1 -1
|
@@ -0,0 +1,733 @@
|
|
|
1
|
+
import { FEATURES } from "../core/ir/features.js";
|
|
2
|
+
import { headingLevelOf } from "../core/outline.js";
|
|
3
|
+
import { detectImageFormat } from "../core/images.js";
|
|
4
|
+
import { effectiveAbstract } from "../core/numbering/state.js";
|
|
5
|
+
import { EMPTY_STYLE_SHEET, resolveParagraphProperties, resolveRunProperties } from "../core/style-cascade/resolver.js";
|
|
6
|
+
import "../core/style-cascade/index.js";
|
|
7
|
+
import { sanitizeHref } from "../core/links.js";
|
|
8
|
+
import { toBase64 } from "../core/bytes.js";
|
|
9
|
+
//#region src/markdown/markdown-writer.ts
|
|
10
|
+
/**
|
|
11
|
+
* Render a {@link FlowDoc} to GitHub-Flavored Markdown (ir-design §7).
|
|
12
|
+
*
|
|
13
|
+
* A flow medium: no pagination, no layout engine and no fonts, so this is a
|
|
14
|
+
* pure, zero-I/O transform. Markdown says far less than the document model
|
|
15
|
+
* does — alignment, indents, colour, font metrics, tab stops and page
|
|
16
|
+
* geometry have no expression at all — so those are dropped and reported as
|
|
17
|
+
* {@link Loss} entries, deduplicated so one recurring omission reports once.
|
|
18
|
+
*
|
|
19
|
+
* @param flow The format-neutral interlayer document tree.
|
|
20
|
+
* @param options Picture handling; see {@link MarkdownWriteOptions}.
|
|
21
|
+
* @returns The encoded Markdown bytes plus the recorded {@link Loss} list.
|
|
22
|
+
*/
|
|
23
|
+
function writeMarkdown(flow, options = {}) {
|
|
24
|
+
const ctx = {
|
|
25
|
+
losses: [],
|
|
26
|
+
seen: /* @__PURE__ */ new Set(),
|
|
27
|
+
images: options.images ?? "dataUri",
|
|
28
|
+
pageBreaks: options.pageBreaks ?? "drop",
|
|
29
|
+
list: [],
|
|
30
|
+
inCell: false,
|
|
31
|
+
resources: flow.resources,
|
|
32
|
+
anchors: referencedAnchors(flow.body),
|
|
33
|
+
notes: noteNumbers(flow),
|
|
34
|
+
mediaNames: /* @__PURE__ */ new Map(),
|
|
35
|
+
...flow.numbering ? { numbering: flow.numbering } : {}
|
|
36
|
+
};
|
|
37
|
+
if (flow.headersFooters && flow.headersFooters.size > 0) lose(ctx, "dropped", FEATURES.headersFooters, "markdown has no pages to band headers onto");
|
|
38
|
+
const blocks = [];
|
|
39
|
+
for (const el of flow.body) emitBlock(blocks, el, ctx);
|
|
40
|
+
emitNoteDefinitions(blocks, flow, ctx);
|
|
41
|
+
return {
|
|
42
|
+
bytes: new TextEncoder().encode(joinBlocks(blocks)),
|
|
43
|
+
losses: ctx.losses
|
|
44
|
+
};
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* The flow-medium {@link DocumentWriter} adapter (id `'md'`), wrapping
|
|
48
|
+
* {@link writeMarkdown}, with the set of {@link FEATURES} it renders.
|
|
49
|
+
*/
|
|
50
|
+
var markdownWriter = {
|
|
51
|
+
id: "md",
|
|
52
|
+
consumes: "flow",
|
|
53
|
+
supports: new Set([
|
|
54
|
+
FEATURES.text,
|
|
55
|
+
FEATURES.lists,
|
|
56
|
+
FEATURES.tables,
|
|
57
|
+
FEATURES.images,
|
|
58
|
+
FEATURES.hyperlinks
|
|
59
|
+
]),
|
|
60
|
+
write: (doc, opts) => writeMarkdown(doc, opts ?? {})
|
|
61
|
+
};
|
|
62
|
+
/** Record a loss unless an identical one was recorded already. */
|
|
63
|
+
function lose(ctx, severity, feature, detail) {
|
|
64
|
+
const key = `${severity}|${feature}|${detail}`;
|
|
65
|
+
if (ctx.seen.has(key)) return;
|
|
66
|
+
ctx.seen.add(key);
|
|
67
|
+
ctx.losses.push({
|
|
68
|
+
severity,
|
|
69
|
+
feature,
|
|
70
|
+
detail
|
|
71
|
+
});
|
|
72
|
+
}
|
|
73
|
+
/**
|
|
74
|
+
* Join the emitted blocks: one blank line between them (markdown's block
|
|
75
|
+
* separator), no leading blank, exactly one trailing newline.
|
|
76
|
+
*
|
|
77
|
+
* Trailing spaces are stripped from every line: they are invisible, editors
|
|
78
|
+
* and formatters eat them, and two of them are markdown's own hard break —
|
|
79
|
+
* a meaning no document ever asked for. Ream writes a hard break as the
|
|
80
|
+
* backslash GFM also accepts, which survives all three.
|
|
81
|
+
*/
|
|
82
|
+
function joinBlocks(blocks) {
|
|
83
|
+
const body = blocks.map(trimHardBreaks).filter((b) => b.length > 0).join("\n\n");
|
|
84
|
+
return body.length > 0 ? `${body.replace(/[ \t]+$/gm, "")}\n` : "";
|
|
85
|
+
}
|
|
86
|
+
/**
|
|
87
|
+
* Drop a hard break sitting at either end of a block: at the end there is no
|
|
88
|
+
* next line to break to, at the start no previous one, and either way the
|
|
89
|
+
* backslash is left standing in the text as itself. Only a break is taken —
|
|
90
|
+
* an escaped literal backslash is `\\` with no newline behind it.
|
|
91
|
+
*/
|
|
92
|
+
function trimHardBreaks(block) {
|
|
93
|
+
return block.replace(/^(?:\\\n)+/, "").replace(/\\\n$/, "");
|
|
94
|
+
}
|
|
95
|
+
function emitBlock(out, el, ctx) {
|
|
96
|
+
if (el.kind === "paragraph") {
|
|
97
|
+
emitParagraph(out, el.paragraph, ctx);
|
|
98
|
+
return;
|
|
99
|
+
}
|
|
100
|
+
ctx.list.length = 0;
|
|
101
|
+
if (el.kind === "table") emitTable(out, el.table, ctx);
|
|
102
|
+
else if (el.kind === "image") {
|
|
103
|
+
const img = pictureMarkdown(el.image.resource, el.image.altText, ctx);
|
|
104
|
+
if (img.length > 0) out.push(img);
|
|
105
|
+
} else if (el.kind === "chart") lose(ctx, "dropped", FEATURES.charts, "markdown cannot draw a chart");
|
|
106
|
+
else emitShape(out, el.shape, ctx);
|
|
107
|
+
}
|
|
108
|
+
/**
|
|
109
|
+
* A shape contributes its WORDS. Markdown draws no geometry, no fill and no
|
|
110
|
+
* line, but the text inside a callout or a text box is document content and is
|
|
111
|
+
* emitted as ordinary blocks — including a group's members, however deep.
|
|
112
|
+
*/
|
|
113
|
+
function emitShape(out, shape, ctx) {
|
|
114
|
+
lose(ctx, "dropped", FEATURES.shapes, "shape geometry dropped; the text inside it is kept");
|
|
115
|
+
for (const el of shape.text?.content ?? []) emitBlock(out, el, ctx);
|
|
116
|
+
for (const child of shape.children ?? []) emitShape(out, child.shape, ctx);
|
|
117
|
+
}
|
|
118
|
+
function emitParagraph(out, p, ctx) {
|
|
119
|
+
const resolved = resolveParagraphProperties(p.properties, EMPTY_STYLE_SHEET);
|
|
120
|
+
const anchors = bookmarkAnchors(p, ctx);
|
|
121
|
+
const text = inlineRuns(p.runs, p, ctx).replace(/^[ \t]+/, "");
|
|
122
|
+
const inline = anchors + text;
|
|
123
|
+
const marker = markerText(p, ctx);
|
|
124
|
+
reportParagraphLosses(resolved, ctx);
|
|
125
|
+
if (breaksPage(p, resolved)) emitRule(out, ctx);
|
|
126
|
+
const empty = isBlank(text) && anchors.length === 0;
|
|
127
|
+
const level = headingLevelOf(resolved);
|
|
128
|
+
if (level !== void 0) {
|
|
129
|
+
if (empty) return;
|
|
130
|
+
ctx.list.length = 0;
|
|
131
|
+
const oneLine = (marker !== void 0 ? `${marker} ${inline}` : inline).replaceAll("\\\n", " ").replaceAll("\n", " ");
|
|
132
|
+
if (ctx.inCell) {
|
|
133
|
+
lose(ctx, "degraded", FEATURES.tables, "heading inside a cell flattened to plain text");
|
|
134
|
+
out.push(oneLine);
|
|
135
|
+
return;
|
|
136
|
+
}
|
|
137
|
+
out.push(`${"#".repeat(level)} ${oneLine}`);
|
|
138
|
+
return;
|
|
139
|
+
}
|
|
140
|
+
if (marker !== void 0) {
|
|
141
|
+
if (!empty) emitListItem(out, resolved, marker, inline, ctx);
|
|
142
|
+
return;
|
|
143
|
+
}
|
|
144
|
+
if (empty) return;
|
|
145
|
+
ctx.list.length = 0;
|
|
146
|
+
out.push(ctx.inCell ? inline : guardBlockStart(inline));
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* True when a paragraph's rendered text says nothing at all. Whitespace counts
|
|
150
|
+
* as nothing, and so do the ZERO-WIDTH characters — the `.pptx` reader marks a
|
|
151
|
+
* slide boundary with a U+200B paragraph carrying the page break, and a run of
|
|
152
|
+
* them down the left of a deck is a column of empty lines, not content.
|
|
153
|
+
*/
|
|
154
|
+
function isBlank(text) {
|
|
155
|
+
return /^[\s\u200B-\u200D\uFEFF]*$/u.test(text);
|
|
156
|
+
}
|
|
157
|
+
/** Whether a page starts at this paragraph — its own break, or one in a run. */
|
|
158
|
+
function breaksPage(p, resolved) {
|
|
159
|
+
return resolved.pageBreakBefore || p.runs.some((r) => r.pageBreak === true);
|
|
160
|
+
}
|
|
161
|
+
/**
|
|
162
|
+
* A `---` thematic break where a page ends, when the caller asked for one.
|
|
163
|
+
*
|
|
164
|
+
* Never leading and never doubled: a rule before the first block would open
|
|
165
|
+
* the document with a line, and two in a row say nothing the one does not.
|
|
166
|
+
* Never inside a cell or a note either — those hold inline content, where the
|
|
167
|
+
* three hyphens are just three hyphens.
|
|
168
|
+
*/
|
|
169
|
+
function emitRule(out, ctx) {
|
|
170
|
+
if (ctx.pageBreaks !== "rule" || ctx.inCell) return;
|
|
171
|
+
if (out.length === 0 || out[out.length - 1] === RULE) return;
|
|
172
|
+
out.push(RULE);
|
|
173
|
+
}
|
|
174
|
+
var RULE = "---";
|
|
175
|
+
/**
|
|
176
|
+
* The marker `applyNumbering` materialized as the paragraph's leading runs
|
|
177
|
+
* (`"1."`, `"•"`, or a picture bullet), stripped of the tab that follows it —
|
|
178
|
+
* or `undefined` when the paragraph is not a list item. The text is taken raw:
|
|
179
|
+
* it is re-rendered as markup, never emitted as content.
|
|
180
|
+
*/
|
|
181
|
+
function markerText(p, ctx) {
|
|
182
|
+
const runs = [];
|
|
183
|
+
for (const run of p.runs) {
|
|
184
|
+
if (run.listMarker !== true) break;
|
|
185
|
+
runs.push(run);
|
|
186
|
+
}
|
|
187
|
+
if (runs.length === 0) return void 0;
|
|
188
|
+
if (runs.some((r) => r.inlineImage !== void 0)) lose(ctx, "degraded", FEATURES.lists, "picture bullet rendered as a plain bullet");
|
|
189
|
+
return runs.map((r) => r.text).join("").replaceAll(" ", " ").trim();
|
|
190
|
+
}
|
|
191
|
+
function emitListItem(out, resolved, marker, inline, ctx) {
|
|
192
|
+
const ref = resolved.numbering;
|
|
193
|
+
const numId = ref?.numId ?? "";
|
|
194
|
+
const ilvl = ref?.ilvl ?? 0;
|
|
195
|
+
const stack = ctx.list;
|
|
196
|
+
const wasOpen = stack.length > 0;
|
|
197
|
+
while (stack.length > 0 && stack[stack.length - 1].ilvl > ilvl) stack.pop();
|
|
198
|
+
let top = stack[stack.length - 1];
|
|
199
|
+
if (!top || top.ilvl < ilvl) {
|
|
200
|
+
const parent = top;
|
|
201
|
+
top = {
|
|
202
|
+
ilvl,
|
|
203
|
+
indent: parent ? parent.indent + parent.markerWidth : 0,
|
|
204
|
+
markerWidth: 2,
|
|
205
|
+
counter: 0
|
|
206
|
+
};
|
|
207
|
+
stack.push(top);
|
|
208
|
+
}
|
|
209
|
+
top.counter += 1;
|
|
210
|
+
const bullet = listBullet(marker, numId, ilvl, top.counter, ctx);
|
|
211
|
+
top.markerWidth = bullet.length + 1;
|
|
212
|
+
const pad = " ".repeat(top.indent);
|
|
213
|
+
const cont = " ".repeat(top.indent + top.markerWidth);
|
|
214
|
+
const line = `${pad}${bullet} ${inline.replaceAll("\\\n", `\\\n${cont}`)}`.replace(/\s+$/, "");
|
|
215
|
+
if (wasOpen && out.length > 0) out[out.length - 1] += `\n${line}`;
|
|
216
|
+
else out.push(line);
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* The markdown marker for an item: `-` for a bullet, `N.` for an ordered list.
|
|
220
|
+
* Ordered lists keep their real number — the one the source's own marker states,
|
|
221
|
+
* so a list starting at 5 (§17.9.28 `w:startOverride`) still starts at 5 —
|
|
222
|
+
* falling back to this level's running count when the marker states no digits.
|
|
223
|
+
*/
|
|
224
|
+
function listBullet(marker, numId, ilvl, counter, ctx) {
|
|
225
|
+
const level = numberingLevel(numId, ilvl, ctx.numbering);
|
|
226
|
+
if (!(level ? level.format !== "bullet" && level.format !== "none" : /\d/.test(marker))) return "-";
|
|
227
|
+
if (level && level.format !== "decimal" && level.format !== "decimalZero") lose(ctx, "degraded", FEATURES.lists, `${level.format} list markers render as decimal`);
|
|
228
|
+
const digits = /(\d+)\D*$/.exec(marker);
|
|
229
|
+
if (marker.replace(/\D+/g, "").length > (digits?.[1]?.length ?? 0)) lose(ctx, "degraded", FEATURES.lists, "multi-level list marker flattened to one number");
|
|
230
|
+
return `${digits ? Number(digits[1]) : counter}.`;
|
|
231
|
+
}
|
|
232
|
+
function numberingLevel(numId, ilvl, numbering) {
|
|
233
|
+
if (!numbering) return void 0;
|
|
234
|
+
const instance = numbering.numInstances.get(numId);
|
|
235
|
+
if (!instance) return void 0;
|
|
236
|
+
return effectiveAbstract(numbering, instance)?.levels.get(ilvl);
|
|
237
|
+
}
|
|
238
|
+
/** Everything a paragraph says that markdown has no way to say back. */
|
|
239
|
+
function reportParagraphLosses(r, ctx) {
|
|
240
|
+
if (r.alignment !== "left") lose(ctx, "dropped", FEATURES.text, "paragraph alignment has no markdown expression");
|
|
241
|
+
if (r.indentLeft !== 0 || r.indentRight !== 0 || r.indentFirstLine !== 0) lose(ctx, "dropped", FEATURES.text, "paragraph indents have no markdown expression");
|
|
242
|
+
if (r.pageBreakBefore && ctx.pageBreaks === "drop") lose(ctx, "dropped", FEATURES.sections, "page breaks have no markdown expression");
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* Backslash-escape a leading character that would otherwise open a block the
|
|
246
|
+
* source never asked for — a heading, a quote, a list item, a setext rule.
|
|
247
|
+
*/
|
|
248
|
+
function guardBlockStart(text) {
|
|
249
|
+
return text.replace(/^(\s*)([#>+-]|\d+[.)]|={2,}$)/, "$1\\$2");
|
|
250
|
+
}
|
|
251
|
+
/**
|
|
252
|
+
* Number the notes by the order their references appear in reading order
|
|
253
|
+
* (§17.11: footnotes and endnotes each keep their own counter), and the review
|
|
254
|
+
* comments alongside them.
|
|
255
|
+
*
|
|
256
|
+
* Only ids the package actually holds content for are numbered: an unmatched
|
|
257
|
+
* `[^fn1]` is not a footnote reference at all in GFM — it renders as those
|
|
258
|
+
* six literal characters — so a dangling reference must leave no mark.
|
|
259
|
+
*/
|
|
260
|
+
function noteNumbers(flow) {
|
|
261
|
+
const footnotes = /* @__PURE__ */ new Map();
|
|
262
|
+
const endnotes = /* @__PURE__ */ new Map();
|
|
263
|
+
const comments = /* @__PURE__ */ new Map();
|
|
264
|
+
const visitShape = (shape) => {
|
|
265
|
+
if (shape.text) visit(shape.text.content);
|
|
266
|
+
for (const child of shape.children ?? []) visitShape(child.shape);
|
|
267
|
+
};
|
|
268
|
+
const visit = (els) => {
|
|
269
|
+
for (const el of els) if (el.kind === "paragraph") for (const r of el.paragraph.runs) {
|
|
270
|
+
const add = (id, into, content) => {
|
|
271
|
+
if (id === void 0 || into.has(id) || !content?.has(id)) return;
|
|
272
|
+
into.set(id, into.size + 1);
|
|
273
|
+
};
|
|
274
|
+
add(r.footnoteRef, footnotes, flow.footnotes);
|
|
275
|
+
add(r.endnoteRef, endnotes, flow.endnotes);
|
|
276
|
+
add(r.commentRef, comments, flow.comments);
|
|
277
|
+
}
|
|
278
|
+
else if (el.kind === "table") for (const row of el.table.rows) for (const cell of row.cells) visit(cell.content);
|
|
279
|
+
else if (el.kind === "shape") visitShape(el.shape);
|
|
280
|
+
};
|
|
281
|
+
visit(flow.body);
|
|
282
|
+
return {
|
|
283
|
+
footnotes,
|
|
284
|
+
endnotes,
|
|
285
|
+
comments
|
|
286
|
+
};
|
|
287
|
+
}
|
|
288
|
+
/**
|
|
289
|
+
* The GFM footnote definitions every reference above points at, in reference
|
|
290
|
+
* order: the notes first, then the review comments — which markdown has no
|
|
291
|
+
* concept of, and which are carried as footnotes attributed to their author.
|
|
292
|
+
*/
|
|
293
|
+
function emitNoteDefinitions(out, flow, ctx) {
|
|
294
|
+
define(out, ctx.notes.footnotes, flow.footnotes, "fn", ctx);
|
|
295
|
+
define(out, ctx.notes.endnotes, flow.endnotes, "en", ctx);
|
|
296
|
+
for (const [id, n] of sorted(ctx.notes.comments)) {
|
|
297
|
+
const comment = flow.comments?.get(id);
|
|
298
|
+
if (!comment) continue;
|
|
299
|
+
const who = comment.author !== void 0 ? `**${escapeInline(comment.author)}:** ` : "";
|
|
300
|
+
out.push(`[^cm${String(n)}]: ${who}${flatten(comment.content, ctx)}`);
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
function define(out, numbers, content, prefix, ctx) {
|
|
304
|
+
for (const [id, n] of sorted(numbers)) {
|
|
305
|
+
const blocks = content?.get(id);
|
|
306
|
+
if (!blocks) continue;
|
|
307
|
+
out.push(`[^${prefix}${String(n)}]: ${flatten(blocks, ctx)}`);
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
var sorted = (m) => [...m.entries()].sort((a, b) => a[1] - b[1]);
|
|
311
|
+
/**
|
|
312
|
+
* Blocks flattened to the single line a footnote definition occupies. Note
|
|
313
|
+
* content is short by nature; a note that holds several paragraphs keeps them
|
|
314
|
+
* all, joined by the line break the syntax allows.
|
|
315
|
+
*/
|
|
316
|
+
function flatten(blocks, ctx) {
|
|
317
|
+
const wasInCell = ctx.inCell;
|
|
318
|
+
const outer = ctx.list.splice(0);
|
|
319
|
+
ctx.inCell = true;
|
|
320
|
+
const rendered = [];
|
|
321
|
+
for (const el of blocks) emitBlock(rendered, el, ctx);
|
|
322
|
+
ctx.inCell = wasInCell;
|
|
323
|
+
ctx.list.length = 0;
|
|
324
|
+
ctx.list.push(...outer);
|
|
325
|
+
return rendered.map(trimHardBreaks).filter((b) => b.length > 0).join("<br>").replaceAll("\\\n", "<br>").replaceAll("\n", "<br>");
|
|
326
|
+
}
|
|
327
|
+
/**
|
|
328
|
+
* Wrap already-rendered inline content in the link it carries. An external
|
|
329
|
+
* target passes the scheme allowlist first (`core/links`) — untrusted input
|
|
330
|
+
* never becomes a clickable link on a scheme markdown viewers will follow.
|
|
331
|
+
*/
|
|
332
|
+
function linkify(inner, target, ctx) {
|
|
333
|
+
if (inner.length === 0) return inner;
|
|
334
|
+
if (target.href !== void 0) {
|
|
335
|
+
const safe = sanitizeHref(target.href);
|
|
336
|
+
if (safe === void 0) {
|
|
337
|
+
lose(ctx, "degraded", FEATURES.hyperlinks, "hyperlink target with a disallowed scheme rendered as plain text");
|
|
338
|
+
return inner;
|
|
339
|
+
}
|
|
340
|
+
return `[${inner}](${destination(safe)})`;
|
|
341
|
+
}
|
|
342
|
+
if (target.anchor !== void 0) return `[${inner}](#${bookmarkSlug(target.anchor)})`;
|
|
343
|
+
return inner;
|
|
344
|
+
}
|
|
345
|
+
/**
|
|
346
|
+
* A link destination: bare when it is plain enough, and in the pointy-bracket
|
|
347
|
+
* form CommonMark §6.3 provides when it holds whitespace or unbalanced
|
|
348
|
+
* parentheses that would otherwise end it early.
|
|
349
|
+
*/
|
|
350
|
+
function destination(url) {
|
|
351
|
+
if (!/[\s()<>]/.test(url)) return url;
|
|
352
|
+
return `<${url.replaceAll("<", "%3C").replaceAll(">", "%3E")}>`;
|
|
353
|
+
}
|
|
354
|
+
/**
|
|
355
|
+
* The fragment an internal link points at. Markdown has no bookmark of its
|
|
356
|
+
* own, so the name is slugified and planted as an inline `<a id>` on the
|
|
357
|
+
* paragraph it belongs to — the same shape GFM's own heading anchors take.
|
|
358
|
+
*/
|
|
359
|
+
function bookmarkSlug(name) {
|
|
360
|
+
const slug = name.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, "-").replace(/^-+|-+$/g, "");
|
|
361
|
+
return slug.length > 0 ? slug : "bookmark";
|
|
362
|
+
}
|
|
363
|
+
/** The bookmark names some run actually links to — the only ones worth an anchor. */
|
|
364
|
+
function referencedAnchors(body) {
|
|
365
|
+
const out = /* @__PURE__ */ new Set();
|
|
366
|
+
const visitShape = (shape) => {
|
|
367
|
+
if (shape.text) visit(shape.text.content);
|
|
368
|
+
for (const child of shape.children ?? []) visitShape(child.shape);
|
|
369
|
+
};
|
|
370
|
+
const visit = (els) => {
|
|
371
|
+
for (const el of els) if (el.kind === "paragraph") {
|
|
372
|
+
for (const r of el.paragraph.runs) if (r.anchor !== void 0) out.add(r.anchor);
|
|
373
|
+
} else if (el.kind === "table") for (const row of el.table.rows) for (const cell of row.cells) visit(cell.content);
|
|
374
|
+
else if (el.kind === "shape") visitShape(el.shape);
|
|
375
|
+
};
|
|
376
|
+
visit(body);
|
|
377
|
+
return out;
|
|
378
|
+
}
|
|
379
|
+
/** The `<a id>` markers a paragraph opens, for the bookmarks something links to. */
|
|
380
|
+
function bookmarkAnchors(p, ctx) {
|
|
381
|
+
return (p.bookmarks ?? []).filter((b) => ctx.anchors.has(b)).map((b) => `<a id="${bookmarkSlug(b)}"></a>`).join("");
|
|
382
|
+
}
|
|
383
|
+
var MIME_BY_FORMAT = {
|
|
384
|
+
jpeg: "image/jpeg",
|
|
385
|
+
png: "image/png",
|
|
386
|
+
jpeg2000: "image/jp2",
|
|
387
|
+
gif: "image/gif",
|
|
388
|
+
tiff: "image/tiff",
|
|
389
|
+
bmp: "image/bmp"
|
|
390
|
+
};
|
|
391
|
+
var EXTENSION_BY_FORMAT = {
|
|
392
|
+
jpeg: "jpg",
|
|
393
|
+
png: "png",
|
|
394
|
+
jpeg2000: "jp2",
|
|
395
|
+
gif: "gif",
|
|
396
|
+
tiff: "tif",
|
|
397
|
+
bmp: "bmp"
|
|
398
|
+
};
|
|
399
|
+
/**
|
|
400
|
+
* A picture as ``, or `''` when there is nothing to point at. The
|
|
401
|
+
* destination follows {@link MarkdownWriteOptions.images}: inlined as a
|
|
402
|
+
* `data:` URI by default, named under `./media/` when the caller means to
|
|
403
|
+
* write the bytes out itself, or dropped.
|
|
404
|
+
*/
|
|
405
|
+
function pictureMarkdown(resource, altText, ctx) {
|
|
406
|
+
if (ctx.images === "drop") {
|
|
407
|
+
lose(ctx, "dropped", FEATURES.images, "pictures omitted at the caller’s request");
|
|
408
|
+
return "";
|
|
409
|
+
}
|
|
410
|
+
if (resource === void 0) return "";
|
|
411
|
+
const bytes = ctx.resources.get(resource);
|
|
412
|
+
if (!bytes) return "";
|
|
413
|
+
const format = detectImageFormat(bytes);
|
|
414
|
+
if (!format) {
|
|
415
|
+
lose(ctx, "dropped", FEATURES.images, "picture in a format markdown viewers cannot show");
|
|
416
|
+
return "";
|
|
417
|
+
}
|
|
418
|
+
const alt = escapeInline(altText ?? "").replaceAll("\n", " ");
|
|
419
|
+
if (ctx.images === "link") {
|
|
420
|
+
let name = ctx.mediaNames.get(resource);
|
|
421
|
+
if (name === void 0) {
|
|
422
|
+
name = `./media/image${String(ctx.mediaNames.size + 1)}.${EXTENSION_BY_FORMAT[format]}`;
|
|
423
|
+
ctx.mediaNames.set(resource, name);
|
|
424
|
+
}
|
|
425
|
+
lose(ctx, "degraded", FEATURES.images, "pictures referenced under ./media — a single-file writer does not write their bytes");
|
|
426
|
+
return ``;
|
|
427
|
+
}
|
|
428
|
+
return `})`;
|
|
429
|
+
}
|
|
430
|
+
/**
|
|
431
|
+
* A GFM pipe table (§4.10). Markdown's table is a plain grid: it has no
|
|
432
|
+
* merged cells and no headerless form, so a `w:gridSpan` fills its first
|
|
433
|
+
* column and pads the rest empty, a `w:vMerge` continuation stays empty, and
|
|
434
|
+
* the first row is promoted to the header the syntax requires. Each of those
|
|
435
|
+
* is reported once.
|
|
436
|
+
*/
|
|
437
|
+
function emitTable(out, table, ctx) {
|
|
438
|
+
const width = columnCount(table);
|
|
439
|
+
if (width === 0 || table.rows.length === 0) return;
|
|
440
|
+
const grid = table.rows.map((row) => rowCells(row, width, ctx));
|
|
441
|
+
const keep = usedRange(grid);
|
|
442
|
+
if (!keep) return;
|
|
443
|
+
const rows = grid.slice(keep.top, keep.bottom + 1).map((row) => row.slice(keep.left, keep.right + 1));
|
|
444
|
+
const header = rows[0];
|
|
445
|
+
if (table.rows[keep.top].properties.isHeader !== true) lose(ctx, "degraded", FEATURES.tables, "first row promoted to the header GFM tables require");
|
|
446
|
+
const line = (cells) => `| ${cells.join(" | ")} |`;
|
|
447
|
+
const bars = delimiters(table, width).slice(keep.left, keep.right + 1);
|
|
448
|
+
const lines = [line(header), line(bars)];
|
|
449
|
+
for (const row of rows.slice(1)) lines.push(line(row));
|
|
450
|
+
out.push(lines.join("\n"));
|
|
451
|
+
}
|
|
452
|
+
/**
|
|
453
|
+
* The rows and columns worth drawing: the blank ones at the EDGES go.
|
|
454
|
+
*
|
|
455
|
+
* A print region carries its whole used range, blank leading rows and trailing
|
|
456
|
+
* columns and all, because a page draws their borders and their fill. Markdown
|
|
457
|
+
* draws neither, so an edge of empty cells is a column of nothing — and a blank
|
|
458
|
+
* first row is worse than nothing, since §4.10 makes it the header. The blank
|
|
459
|
+
* rows INSIDE stay: those separate one group of data from the next, which is
|
|
460
|
+
* the one thing an empty row can still say here.
|
|
461
|
+
*/
|
|
462
|
+
function usedRange(grid) {
|
|
463
|
+
const filled = (r, c) => (grid[r]?.[c] ?? "").trim().length > 0;
|
|
464
|
+
const width = grid[0]?.length ?? 0;
|
|
465
|
+
let top = 0;
|
|
466
|
+
let bottom = grid.length - 1;
|
|
467
|
+
let left = 0;
|
|
468
|
+
let right = width - 1;
|
|
469
|
+
const rowFilled = (r) => grid[r].some((_, c) => filled(r, c));
|
|
470
|
+
const colFilled = (c) => grid.some((_, r) => filled(r, c));
|
|
471
|
+
while (top <= bottom && !rowFilled(top)) top++;
|
|
472
|
+
if (top > bottom) return void 0;
|
|
473
|
+
while (bottom > top && !rowFilled(bottom)) bottom--;
|
|
474
|
+
while (left <= right && !colFilled(left)) left++;
|
|
475
|
+
while (right > left && !colFilled(right)) right--;
|
|
476
|
+
return {
|
|
477
|
+
top,
|
|
478
|
+
bottom,
|
|
479
|
+
left,
|
|
480
|
+
right
|
|
481
|
+
};
|
|
482
|
+
}
|
|
483
|
+
/** The widest row wins: a `w:gridSpan` may reach past the declared `w:tblGrid`. */
|
|
484
|
+
function columnCount(table) {
|
|
485
|
+
let widest = table.grid.length;
|
|
486
|
+
for (const row of table.rows) {
|
|
487
|
+
let n = 0;
|
|
488
|
+
for (const cell of row.cells) n += cell.properties.colSpan ?? 1;
|
|
489
|
+
widest = Math.max(widest, n);
|
|
490
|
+
}
|
|
491
|
+
return widest;
|
|
492
|
+
}
|
|
493
|
+
/** One row's cells, spans expanded to empty columns and padded to `cols`. */
|
|
494
|
+
function rowCells(row, cols, ctx) {
|
|
495
|
+
const cells = [];
|
|
496
|
+
for (const cell of row.cells) {
|
|
497
|
+
const span = cell.properties.colSpan ?? 1;
|
|
498
|
+
if (span > 1 || cell.properties.merge !== void 0 && cell.properties.merge !== "start") lose(ctx, "degraded", FEATURES.tables, "merged cells flattened — markdown has no spans");
|
|
499
|
+
cells.push(cell.properties.merge === "middle" || cell.properties.merge === "end" ? "" : cellInline(cell, ctx));
|
|
500
|
+
for (let k = 1; k < span; k++) cells.push("");
|
|
501
|
+
}
|
|
502
|
+
while (cells.length < cols) cells.push("");
|
|
503
|
+
return cells.slice(0, cols);
|
|
504
|
+
}
|
|
505
|
+
/**
|
|
506
|
+
* The delimiter row, carrying each column's alignment when the header cell's
|
|
507
|
+
* first paragraph states one — the only paragraph property a markdown table
|
|
508
|
+
* can keep.
|
|
509
|
+
*/
|
|
510
|
+
function delimiters(table, cols) {
|
|
511
|
+
const out = [];
|
|
512
|
+
const first = table.rows[0];
|
|
513
|
+
let col = 0;
|
|
514
|
+
for (const cell of first.cells) {
|
|
515
|
+
const span = cell.properties.colSpan ?? 1;
|
|
516
|
+
const p = cell.content.find((el) => el.kind === "paragraph");
|
|
517
|
+
const align = p ? resolveParagraphProperties(p.paragraph.properties, EMPTY_STYLE_SHEET).alignment : "left";
|
|
518
|
+
const bar = align === "center" ? ":---:" : align === "right" ? "---:" : "---";
|
|
519
|
+
for (let k = 0; k < span && col < cols; k++, col++) out.push(bar);
|
|
520
|
+
}
|
|
521
|
+
while (out.length < cols) out.push("---");
|
|
522
|
+
return out.slice(0, cols);
|
|
523
|
+
}
|
|
524
|
+
/**
|
|
525
|
+
* A cell's blocks flattened to one line: a table row is a single line, so
|
|
526
|
+
* every break between and inside its blocks becomes a `<br>`. The cell's own
|
|
527
|
+
* lists number independently of whatever list surrounds the table.
|
|
528
|
+
*
|
|
529
|
+
* Each block is trimmed at its ends. Word carries a cell's padding as spaces in
|
|
530
|
+
* the text itself, and a pipe table already sets its own — kept, they only
|
|
531
|
+
* widen the source line, since §4.10 strips a cell's outer whitespace before
|
|
532
|
+
* rendering it anyway.
|
|
533
|
+
*/
|
|
534
|
+
function cellInline(cell, ctx) {
|
|
535
|
+
const outer = ctx.list.splice(0);
|
|
536
|
+
const wasInCell = ctx.inCell;
|
|
537
|
+
ctx.inCell = true;
|
|
538
|
+
const blocks = [];
|
|
539
|
+
for (const el of cell.content) {
|
|
540
|
+
if (el.kind === "table") {
|
|
541
|
+
lose(ctx, "degraded", FEATURES.tablesNested, "nested table flattened into its cell");
|
|
542
|
+
for (const row of el.table.rows) for (const inner of row.cells) blocks.push(cellInline(inner, ctx));
|
|
543
|
+
continue;
|
|
544
|
+
}
|
|
545
|
+
emitBlock(blocks, el, ctx);
|
|
546
|
+
}
|
|
547
|
+
ctx.inCell = wasInCell;
|
|
548
|
+
ctx.list.length = 0;
|
|
549
|
+
ctx.list.push(...outer);
|
|
550
|
+
return blocks.map((b) => trimHardBreaks(b).trim()).filter((b) => b.length > 0).join("<br>").replaceAll("\\\n", "<br>").replaceAll("\n", "<br>");
|
|
551
|
+
}
|
|
552
|
+
/**
|
|
553
|
+
* The paragraph's runs as one inline string.
|
|
554
|
+
*
|
|
555
|
+
* Two passes, because whether a delimiter may open or close depends on the
|
|
556
|
+
* characters on either side of it (§6.2), and those belong to the NEIGHBOURING
|
|
557
|
+
* spans: the first pass coalesces runs into spans, the second dresses each
|
|
558
|
+
* span knowing what stands beside it.
|
|
559
|
+
*/
|
|
560
|
+
function inlineRuns(runs, p, ctx) {
|
|
561
|
+
const pieces = [];
|
|
562
|
+
let pending;
|
|
563
|
+
const flush = () => {
|
|
564
|
+
if (pending && pending.text.length > 0) pieces.push(pending);
|
|
565
|
+
pending = void 0;
|
|
566
|
+
};
|
|
567
|
+
for (const run of runs) {
|
|
568
|
+
const literal = literalRun(run, ctx);
|
|
569
|
+
if (literal !== void 0) {
|
|
570
|
+
flush();
|
|
571
|
+
if (literal.length > 0) pieces.push({ literal });
|
|
572
|
+
continue;
|
|
573
|
+
}
|
|
574
|
+
if (run.text.length === 0) continue;
|
|
575
|
+
const resolved = resolveRunProperties(run.properties, p.properties, EMPTY_STYLE_SHEET);
|
|
576
|
+
reportRunLosses(resolved, ctx);
|
|
577
|
+
const marks = marksOf(resolved, run);
|
|
578
|
+
if (pending && sameMarks(pending.marks, marks)) pending.text += run.text;
|
|
579
|
+
else {
|
|
580
|
+
flush();
|
|
581
|
+
pending = {
|
|
582
|
+
marks,
|
|
583
|
+
text: run.text
|
|
584
|
+
};
|
|
585
|
+
}
|
|
586
|
+
}
|
|
587
|
+
flush();
|
|
588
|
+
let out = "";
|
|
589
|
+
for (let i = 0; i < pieces.length; i++) {
|
|
590
|
+
const piece = pieces[i];
|
|
591
|
+
if ("literal" in piece) {
|
|
592
|
+
out += piece.literal;
|
|
593
|
+
continue;
|
|
594
|
+
}
|
|
595
|
+
out += applyMarks(piece.text, piece.marks, ctx, out.slice(-1), leadOf(pieces, i + 1));
|
|
596
|
+
}
|
|
597
|
+
return out;
|
|
598
|
+
}
|
|
599
|
+
/**
|
|
600
|
+
* The first character the next span will contribute. A dressed span always
|
|
601
|
+
* begins with its delimiter or its html tag — both punctuation — so a marked
|
|
602
|
+
* neighbour answers without being rendered first.
|
|
603
|
+
*/
|
|
604
|
+
function leadOf(pieces, from) {
|
|
605
|
+
const next = pieces[from];
|
|
606
|
+
if (!next) return "";
|
|
607
|
+
if ("literal" in next) return next.literal.slice(0, 1);
|
|
608
|
+
if (wrapped(next.marks)) return "*";
|
|
609
|
+
return escapeInline(next.text).slice(0, 1);
|
|
610
|
+
}
|
|
611
|
+
/** Whether a span carries anything that puts a delimiter or a tag in front of it. */
|
|
612
|
+
function wrapped(m) {
|
|
613
|
+
return m.bold || m.italic || m.strike || m.underline || m.vertical !== "baseline" || m.href !== void 0 || m.anchor !== void 0;
|
|
614
|
+
}
|
|
615
|
+
/**
|
|
616
|
+
* The runs that are not a span of styled text: reference markers, inline
|
|
617
|
+
* objects, and the list marker the reader materialized. Returns the literal
|
|
618
|
+
* they contribute (possibly `''`), or `undefined` when the run IS styled text.
|
|
619
|
+
*/
|
|
620
|
+
function literalRun(run, ctx) {
|
|
621
|
+
if (run.footnoteRef !== void 0 || run.endnoteRef !== void 0) {
|
|
622
|
+
const foot = run.footnoteRef !== void 0;
|
|
623
|
+
const id = foot ? run.footnoteRef : run.endnoteRef;
|
|
624
|
+
const n = (foot ? ctx.notes.footnotes : ctx.notes.endnotes).get(id);
|
|
625
|
+
return n === void 0 ? "" : `[^${foot ? "fn" : "en"}${String(n)}]`;
|
|
626
|
+
}
|
|
627
|
+
if (run.commentRef !== void 0) {
|
|
628
|
+
const n = ctx.notes.comments.get(run.commentRef);
|
|
629
|
+
if (n === void 0) return "";
|
|
630
|
+
lose(ctx, "degraded", FEATURES.trackedChanges, "review comments rendered as footnotes");
|
|
631
|
+
return `[^cm${String(n)}]`;
|
|
632
|
+
}
|
|
633
|
+
if (run.noteNumber) return "";
|
|
634
|
+
if (run.listMarker) return "";
|
|
635
|
+
if (run.math !== void 0) {
|
|
636
|
+
lose(ctx, "dropped", FEATURES.math, "markdown has no math notation (GFM)");
|
|
637
|
+
return "";
|
|
638
|
+
}
|
|
639
|
+
if (run.inlineImage !== void 0) return linkify(pictureMarkdown(run.inlineImage.resource, void 0, ctx), run, ctx);
|
|
640
|
+
if (run.columnBreak === true || run.pageBreak === true && ctx.pageBreaks === "drop") lose(ctx, "dropped", FEATURES.sections, "page breaks have no markdown expression");
|
|
641
|
+
}
|
|
642
|
+
/** Everything a run says that markdown has no way to say back. */
|
|
643
|
+
function reportRunLosses(r, ctx) {
|
|
644
|
+
if (r.colorHex !== "000000") lose(ctx, "dropped", FEATURES.text, "run colour has no markdown expression");
|
|
645
|
+
if (r.caps || r.smallCaps) lose(ctx, "dropped", FEATURES.text, "w:caps / w:smallCaps has no markdown expression");
|
|
646
|
+
}
|
|
647
|
+
function marksOf(r, run) {
|
|
648
|
+
return {
|
|
649
|
+
bold: r.bold,
|
|
650
|
+
italic: r.italic,
|
|
651
|
+
strike: r.strike,
|
|
652
|
+
underline: r.underline !== "none",
|
|
653
|
+
vertical: r.verticalAlign === "superscript" || r.verticalAlign === "subscript" ? r.verticalAlign : "baseline",
|
|
654
|
+
...run.href !== void 0 ? { href: run.href } : {},
|
|
655
|
+
...run.anchor !== void 0 ? { anchor: run.anchor } : {}
|
|
656
|
+
};
|
|
657
|
+
}
|
|
658
|
+
function sameMarks(a, b) {
|
|
659
|
+
return a.bold === b.bold && a.italic === b.italic && a.strike === b.strike && a.underline === b.underline && a.vertical === b.vertical && a.href === b.href && a.anchor === b.anchor;
|
|
660
|
+
}
|
|
661
|
+
/**
|
|
662
|
+
* Escape the text, then dress it: markdown delimiters innermost, then the
|
|
663
|
+
* inline-html wrappers for what markdown cannot say. CommonMark §6.2 requires
|
|
664
|
+
* a delimiter to hug non-whitespace, so any leading or trailing space of the
|
|
665
|
+
* span moves outside the marks.
|
|
666
|
+
*
|
|
667
|
+
* A HARD BREAK at either end moves out too, and it matters more than a space
|
|
668
|
+
* does: a closing `**` that a line break stands in front of is not
|
|
669
|
+
* right-flanking, cannot close, and prints as two asterisks. A `w:br` at the
|
|
670
|
+
* end of a bold title — which every deck writes — did exactly that.
|
|
671
|
+
*/
|
|
672
|
+
var SPAN_EDGE = String.raw`(?:\\\n|\s)`;
|
|
673
|
+
function applyMarks(text, marks, ctx, before, after) {
|
|
674
|
+
const escaped = escapeInline(text);
|
|
675
|
+
const lead = new RegExp(`^${SPAN_EDGE}*`).exec(escaped)?.[0] ?? "";
|
|
676
|
+
const trail = new RegExp(`${SPAN_EDGE}*$`).exec(escaped)?.[0] ?? "";
|
|
677
|
+
const core = escaped.slice(lead.length, escaped.length - trail.length);
|
|
678
|
+
if (core.length === 0) return escaped;
|
|
679
|
+
const outerBefore = lead.length > 0 ? lead[lead.length - 1] : before;
|
|
680
|
+
const outerAfter = trail.length > 0 ? trail[0] : after;
|
|
681
|
+
let out = core;
|
|
682
|
+
const levels = [
|
|
683
|
+
[
|
|
684
|
+
marks.italic,
|
|
685
|
+
"*",
|
|
686
|
+
"em"
|
|
687
|
+
],
|
|
688
|
+
[
|
|
689
|
+
marks.bold,
|
|
690
|
+
"**",
|
|
691
|
+
"strong"
|
|
692
|
+
],
|
|
693
|
+
[
|
|
694
|
+
marks.strike,
|
|
695
|
+
"~~",
|
|
696
|
+
"del"
|
|
697
|
+
]
|
|
698
|
+
];
|
|
699
|
+
for (const [on, delimiter, tag] of levels) {
|
|
700
|
+
if (!on) continue;
|
|
701
|
+
out = canOpen(outerBefore, out[0]) && canClose(out[out.length - 1], outerAfter) ? `${delimiter}${out}${delimiter}` : `<${tag}>${out}</${tag}>`;
|
|
702
|
+
}
|
|
703
|
+
if (marks.underline) out = `<u>${out}</u>`;
|
|
704
|
+
if (marks.vertical !== "baseline") {
|
|
705
|
+
const tag = marks.vertical === "superscript" ? "sup" : "sub";
|
|
706
|
+
out = `<${tag}>${out}</${tag}>`;
|
|
707
|
+
}
|
|
708
|
+
return `${lead}${linkify(out, marks, ctx)}${trail}`;
|
|
709
|
+
}
|
|
710
|
+
var isSpace = (c) => c === "" || /\s/u.test(c);
|
|
711
|
+
var isPunct = (c) => /[\p{P}\p{S}]/u.test(c);
|
|
712
|
+
/** §6.2 — the delimiter run is left-flanking, so it can open emphasis. */
|
|
713
|
+
function canOpen(before, after) {
|
|
714
|
+
return !isSpace(after) && (!isPunct(after) || isSpace(before) || isPunct(before));
|
|
715
|
+
}
|
|
716
|
+
/** §6.2 — the delimiter run is right-flanking, so it can close emphasis. */
|
|
717
|
+
function canClose(before, after) {
|
|
718
|
+
return !isSpace(before) && (!isPunct(before) || isSpace(after) || isPunct(after));
|
|
719
|
+
}
|
|
720
|
+
/**
|
|
721
|
+
* Escape the inline text so nothing in it is read as markup, and map the two
|
|
722
|
+
* control characters the model uses: `\t` (a `w:tab`, which has no medium
|
|
723
|
+
* here) to a space, and `\n` (a `w:br` soft break) to GFM's backslash hard
|
|
724
|
+
* break.
|
|
725
|
+
*
|
|
726
|
+
* `_` is escaped only at a word boundary — GFM does not open intraword
|
|
727
|
+
* emphasis with it, so escaping every one would turn `snake_case` into noise.
|
|
728
|
+
*/
|
|
729
|
+
function escapeInline(text) {
|
|
730
|
+
return text.replace(/[\\`*[\]<>|~]/g, (c) => `\\${c}`).replace(/(^|[\s\p{P}])_|_(?=[\s\p{P}]|$)/gu, (m) => m.replace("_", "\\_")).replaceAll(" ", " ").replaceAll("\n", "\\\n");
|
|
731
|
+
}
|
|
732
|
+
//#endregion
|
|
733
|
+
export { markdownWriter, writeMarkdown };
|