switchroom 0.21.8 → 0.21.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +100 -39
- package/dist/host-control/main.js +1 -1
- package/package.json +2 -2
- package/skills/switchroom-architecture/telegram.md +12 -10
- package/skills/switchroom-cli/SKILL.md +1 -1
- package/telegram-plugin/README.md +3 -1
- package/telegram-plugin/dist/gateway/gateway.js +919 -348
- package/telegram-plugin/format.ts +30 -5
- package/telegram-plugin/gateway/outbound-send-path.ts +7 -0
- package/telegram-plugin/gateway/speech-capture.ts +158 -0
- package/telegram-plugin/package.json +1 -1
- package/telegram-plugin/render/code-segments.ts +38 -4
- package/telegram-plugin/render/dollar-math-guard.ts +16 -1
- package/telegram-plugin/render/html-fold.ts +354 -0
- package/telegram-plugin/render/ir.ts +53 -3
- package/telegram-plugin/render/parse.ts +642 -42
- package/telegram-plugin/render/render.ts +53 -15
- package/telegram-plugin/render/unsupported-token-guard.ts +45 -80
- package/telegram-plugin/rich-send.ts +22 -7
- package/telegram-plugin/shared/bot-runtime.ts +3 -2
- package/telegram-plugin/telegraph.ts +6 -4
- package/telegram-plugin/tests/grammy-rich-message-types.test.ts +199 -0
- package/telegram-plugin/tests/render/dollar-math-guard.test.ts +43 -0
- package/telegram-plugin/tests/render/guard-composition.test.ts +102 -0
- package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +253 -0
- package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
- package/telegram-plugin/tests/render/parse.test.ts +39 -10
- package/telegram-plugin/tests/render/render.test.ts +9 -4
- package/telegram-plugin/tests/render/rich-render.test.ts +46 -5
- package/telegram-plugin/tests/render/tg-entity.test.ts +242 -0
- package/telegram-plugin/tests/render/unsupported-token-guard.test.ts +66 -66
- package/telegram-plugin/tests/send-reply-golden.test.ts +99 -1
- package/telegram-plugin/tests/sent-text-capture.test.ts +3 -3
- package/telegram-plugin/tests/speech-capture.test.ts +296 -0
- package/telegram-plugin/tests/telegraph.test.ts +1 -1
- package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
- package/telegram-plugin/tts-normalize.ts +47 -9
- package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +17 -8
- package/telegram-plugin/voice-normalize-text.ts +48 -9
|
@@ -12,7 +12,11 @@
|
|
|
12
12
|
// `source.slice(node.start, node.end)` round-trips to the source text.
|
|
13
13
|
// - Never lose text. Any mdast node type outside the supported palette
|
|
14
14
|
// degrades to a `plain` inline (or a paragraph wrapping one) carrying the
|
|
15
|
-
// raw source slice, rather than being dropped.
|
|
15
|
+
// raw source slice, rather than being dropped. The ONE deliberate
|
|
16
|
+
// exception is raw-HTML MARKUP: an unrecognised tag has its angle-bracket
|
|
17
|
+
// token dropped while its content is kept and rendered (see the raw-HTML
|
|
18
|
+
// note below and `html-fold.ts`). Content is never lost; unguaranteeable
|
|
19
|
+
// markup is.
|
|
16
20
|
//
|
|
17
21
|
// Underline vs bold (`__…__` vs `**…**`):
|
|
18
22
|
// Telegram's Bot API 10.1 rich markdown reads a `__…__` double-underscore run
|
|
@@ -37,12 +41,18 @@
|
|
|
37
41
|
// own reading of the delimiters.
|
|
38
42
|
//
|
|
39
43
|
// Blockquote expandable handling:
|
|
40
|
-
// The IR carries `expandable: boolean` for
|
|
41
|
-
//
|
|
42
|
-
//
|
|
43
|
-
//
|
|
44
|
-
//
|
|
45
|
-
//
|
|
44
|
+
// The IR carries `expandable: boolean` for the LEGACY switchroom `**> `
|
|
45
|
+
// expandable-quote encoding. `**>` was believed to be the Bot API 10.1
|
|
46
|
+
// expandable-blockquote marker; wire probes (2026-08-13) proved it is
|
|
47
|
+
// MarkdownV2-only syntax that the rich markdown path renders as LITERAL
|
|
48
|
+
// `**>` text, so `render.ts` no longer emits it — an expandable node renders
|
|
49
|
+
// as a plain `> ` quote. Recognition here is kept as INPUT REPAIR: agent
|
|
50
|
+
// output (and Hindsight memories) trained on the old floor card still
|
|
51
|
+
// contains `**> ` quotes, and without this rewrite such a line would reach
|
|
52
|
+
// the wire as a broken literal-`**>` paragraph. micromark does NOT
|
|
53
|
+
// understand `**> ` as a blockquote — the leading `**` makes the line a
|
|
54
|
+
// paragraph with an unclosed strong-emphasis run — so this module
|
|
55
|
+
// pre-transforms each
|
|
46
56
|
// `**>` marker into a plain ` >` marker of IDENTICAL length (`**` → two
|
|
47
57
|
// spaces) before handing the text to mdast. Length preservation keeps every
|
|
48
58
|
// UTF-16 source offset (and therefore the never-lose-text round-trip
|
|
@@ -50,6 +60,25 @@
|
|
|
50
60
|
// marker characters differ, and they are never part of quoted content. The
|
|
51
61
|
// set of line-start offsets that carried the marker is threaded into
|
|
52
62
|
// `foldBlock` so the matching blockquote nodes get `expandable: true`.
|
|
63
|
+
//
|
|
64
|
+
// The rewrite is FENCE-AWARE. Every other fold reads the ORIGINAL `markdown`
|
|
65
|
+
// through `slice()`, so the rewritten bytes never escape — except for the
|
|
66
|
+
// `code` fold, which must use mdast's `node.value` (the dedented, fence-
|
|
67
|
+
// stripped content, which offsets alone cannot reconstruct without
|
|
68
|
+
// re-implementing micromark's fence parser). That made a fenced block
|
|
69
|
+
// containing a line starting `**>` ship silently corrupted as ` >` —
|
|
70
|
+
// anyone documenting the legacy syntax got their code altered. Skipping
|
|
71
|
+
// fenced regions in the pre-pass fixes it at the one place the rewrite is
|
|
72
|
+
// decided, keeps `code.value` trustworthy for every consumer, and preserves
|
|
73
|
+
// the length invariant trivially (a skipped line is copied verbatim).
|
|
74
|
+
//
|
|
75
|
+
// Raw HTML handling:
|
|
76
|
+
// mdast `html` nodes (inline `<b>`/`<a href>` runs and whole HTML blocks)
|
|
77
|
+
// used to fall through to `plain`, which the renderer `escapeMarkdown`s —
|
|
78
|
+
// escaping `=` and shipping `<a href\="…">`. They now go through
|
|
79
|
+
// `html-fold.ts`'s three-bucket policy: fold to the native IR construct,
|
|
80
|
+
// pass through raw (wire-verified allowlist), or drop the markup and keep
|
|
81
|
+
// the content. See that module's header for the rationale.
|
|
53
82
|
|
|
54
83
|
import { fromMarkdown } from "mdast-util-from-markdown";
|
|
55
84
|
import { gfm } from "micromark-extension-gfm";
|
|
@@ -72,6 +101,19 @@ import type {
|
|
|
72
101
|
TableCell,
|
|
73
102
|
TableRow,
|
|
74
103
|
} from "./ir.js";
|
|
104
|
+
import {
|
|
105
|
+
HTML_FOLD_TAGS,
|
|
106
|
+
escapeHtmlLiteral,
|
|
107
|
+
hrefOf,
|
|
108
|
+
isBlockLevelTag,
|
|
109
|
+
isPassthroughTag,
|
|
110
|
+
isVoidTag,
|
|
111
|
+
languageOf,
|
|
112
|
+
matchedTagOffsets,
|
|
113
|
+
tokenizeHtml,
|
|
114
|
+
type HtmlFoldTarget,
|
|
115
|
+
type HtmlTagInfo,
|
|
116
|
+
} from "./html-fold.js";
|
|
75
117
|
|
|
76
118
|
/** Copy UTF-16 offsets off an mdast node. Falls back to 0-length when a
|
|
77
119
|
* synthesized node lacks a position (from-markdown always sets one, but the
|
|
@@ -90,22 +132,71 @@ function slice(source: string, node: MdastNode): string {
|
|
|
90
132
|
return source.slice(start, end);
|
|
91
133
|
}
|
|
92
134
|
|
|
93
|
-
/** The
|
|
94
|
-
* a line (column 0).
|
|
95
|
-
*
|
|
96
|
-
*
|
|
135
|
+
/** The LEGACY switchroom expandable-blockquote marker: `**>` at the very
|
|
136
|
+
* start of a line (column 0). The render path no longer emits it (it is
|
|
137
|
+
* MarkdownV2-only syntax, not rich markdown — see `render.ts`
|
|
138
|
+
* renderBlockquote), but it is still RECOGNISED on input so a legacy `**> `
|
|
139
|
+
* line is repaired into a real blockquote instead of shipping as literal
|
|
140
|
+
* `**>` text. Matching only at column 0 keeps
|
|
97
141
|
* the length-preserving rewrite (`**` → two spaces) inside CommonMark's
|
|
98
142
|
* 3-space blockquote-indent budget — allowing leading indent here would push
|
|
99
143
|
* the rewritten ` >` past 3 spaces and turn it into an indented code block. */
|
|
100
144
|
const EXPANDABLE_MARKER_RE = /^\*\*>/;
|
|
101
145
|
|
|
146
|
+
/** The `tg:` hrefs Telegram's "Rich Markdown style" grammar accepts in IMAGE
|
|
147
|
+
* position — ``. Exactly two are documented
|
|
148
|
+
* (https://core.telegram.org/bots/api): `tg://emoji?id=…` (custom emoji) and
|
|
149
|
+
* `tg://time?unix=…[&format=…]` (the `date_time` entity). Deliberately an
|
|
150
|
+
* ALLOWLIST rather than a bare `tg:` scheme test: an undocumented `tg://…` in
|
|
151
|
+
* image position is not known-good syntax, and demoting it to literal text
|
|
152
|
+
* (the historical behaviour) is safer than shipping a construct Telegram may
|
|
153
|
+
* parse-reject. */
|
|
154
|
+
const TG_INLINE_ENTITY_HREFS = ["tg://emoji", "tg://time"] as const;
|
|
155
|
+
|
|
156
|
+
/** True when an mdast `image` url is one of the documented inline `tg:`
|
|
157
|
+
* entities. Scheme/host comparison is case-insensitive (URLs are), but the
|
|
158
|
+
* ORIGINAL href is what gets re-emitted — we never rewrite the author's bytes. */
|
|
159
|
+
function isTgInlineEntityHref(href: string): boolean {
|
|
160
|
+
const h = href.toLowerCase();
|
|
161
|
+
return TG_INLINE_ENTITY_HREFS.some((base) => h === base || h.startsWith(`${base}?`));
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/** A top-level fenced-code delimiter: 3+ backticks or tildes, indented 0–3
|
|
165
|
+
* spaces. Deeper indentation is not a fence, and a fence nested inside a
|
|
166
|
+
* blockquote / list item is prefixed by its container's markup, so this only
|
|
167
|
+
* ever tracks TOP-LEVEL fences.
|
|
168
|
+
*
|
|
169
|
+
* KNOWN LIMITATION (pre-existing, not a safety property): tracking only
|
|
170
|
+
* top-level fences does NOT establish that every column-0 `**>` line outside
|
|
171
|
+
* one is a real blockquote candidate. Counterexample:
|
|
172
|
+
*
|
|
173
|
+
* > ```
|
|
174
|
+
* **> lazy
|
|
175
|
+
* > ```
|
|
176
|
+
*
|
|
177
|
+
* The `> ``` ` opener is inside a blockquote so it never matches here; the
|
|
178
|
+
* column-0 `**> lazy` line is rewritten to ` > lazy` and the `**` is eaten.
|
|
179
|
+
* CommonMark forbids lazy continuation into fenced code, so that line should
|
|
180
|
+
* CLOSE the blockquote and stay literal. Fixing it needs real container
|
|
181
|
+
* tracking (blockquote/list prefixes), not a wider fence regex. The case is
|
|
182
|
+
* pinned by a test in `tests/render/html-dialect.test.ts` so any future
|
|
183
|
+
* container-aware rewrite has to decide it deliberately. */
|
|
184
|
+
const FENCE_DELIM_RE = /^ {0,3}(`{3,}|~{3,})/;
|
|
185
|
+
|
|
102
186
|
/** Pre-transform expandable-blockquote markers so mdast can parse them as
|
|
103
187
|
* ordinary blockquotes, WITHOUT shifting any source offset. Each line that
|
|
104
|
-
* opens with `**>`
|
|
105
|
-
* (`**>` → ` >`), which micromark reads as
|
|
106
|
-
* 1–3-space-indented) blockquote line. Returns the
|
|
107
|
-
* set of line-start offsets that carried the marker —
|
|
108
|
-
* set to flip `expandable: true` on the produced
|
|
188
|
+
* opens with `**>` — and is NOT inside a fenced code block — has its two `*`
|
|
189
|
+
* characters replaced by two spaces (`**>` → ` >`), which micromark reads as
|
|
190
|
+
* a normal (optionally 1–3-space-indented) blockquote line. Returns the
|
|
191
|
+
* rewritten text plus the set of line-start offsets that carried the marker —
|
|
192
|
+
* `foldBlock` uses that set to flip `expandable: true` on the produced
|
|
193
|
+
* blockquote nodes.
|
|
194
|
+
*
|
|
195
|
+
* Fence tracking mirrors CommonMark: an opener is 3+ backticks/tildes at
|
|
196
|
+
* 0–3 spaces of indent; the block closes on a delimiter of the SAME character
|
|
197
|
+
* that is at least as long and carries nothing but whitespace after it, or at
|
|
198
|
+
* end of document. Inside such a block every line is copied verbatim, so a
|
|
199
|
+
* documented `**> …` example survives byte-identical into `code.value`. */
|
|
109
200
|
function markExpandableQuotes(markdown: string): {
|
|
110
201
|
text: string;
|
|
111
202
|
expandableLineStarts: Set<number>;
|
|
@@ -113,8 +204,38 @@ function markExpandableQuotes(markdown: string): {
|
|
|
113
204
|
const expandableLineStarts = new Set<number>();
|
|
114
205
|
let out = "";
|
|
115
206
|
let offset = 0;
|
|
207
|
+
/** The open fence's delimiter run, or null outside a fenced block. */
|
|
208
|
+
let openFence: string | null = null;
|
|
116
209
|
// Split keeping the trailing newline on each line so offsets are exact.
|
|
117
210
|
for (const line of markdown.split(/(?<=\n)/)) {
|
|
211
|
+
const body = line.replace(/\r?\n$/, "");
|
|
212
|
+
const fence = FENCE_DELIM_RE.exec(body)?.[1] ?? null;
|
|
213
|
+
if (openFence !== null) {
|
|
214
|
+
// Inside a fence: copy verbatim, and close on a matching delimiter.
|
|
215
|
+
out += line;
|
|
216
|
+
if (
|
|
217
|
+
fence !== null &&
|
|
218
|
+
fence[0] === openFence[0] &&
|
|
219
|
+
fence.length >= openFence.length &&
|
|
220
|
+
body.slice(body.indexOf(fence) + fence.length).trim() === ""
|
|
221
|
+
) {
|
|
222
|
+
openFence = null;
|
|
223
|
+
}
|
|
224
|
+
offset += line.length;
|
|
225
|
+
continue;
|
|
226
|
+
}
|
|
227
|
+
if (fence !== null) {
|
|
228
|
+
// A backtick info string may not contain a backtick; such a line is not
|
|
229
|
+
// a fence opener at all (CommonMark), so only treat it as one when the
|
|
230
|
+
// rest of the line is clean.
|
|
231
|
+
const rest = body.slice(body.indexOf(fence) + fence.length);
|
|
232
|
+
if (!(fence[0] === "`" && rest.includes("`"))) {
|
|
233
|
+
openFence = fence;
|
|
234
|
+
out += line;
|
|
235
|
+
offset += line.length;
|
|
236
|
+
continue;
|
|
237
|
+
}
|
|
238
|
+
}
|
|
118
239
|
if (EXPANDABLE_MARKER_RE.test(line)) {
|
|
119
240
|
expandableLineStarts.add(offset);
|
|
120
241
|
// Replace the leading two `*` chars with two spaces; the `>` and
|
|
@@ -142,7 +263,7 @@ function foldAlign(a: AlignType | undefined): "left" | "center" | "right" | null
|
|
|
142
263
|
return a ?? null;
|
|
143
264
|
}
|
|
144
265
|
|
|
145
|
-
function foldInline(node: PhrasingContent, source: string): Inline {
|
|
266
|
+
function foldInline(node: PhrasingContent, source: string, ctx: FoldCtx): Inline {
|
|
146
267
|
switch (node.type) {
|
|
147
268
|
case "text":
|
|
148
269
|
return { type: "plain", text: node.value, ...pos(node) };
|
|
@@ -153,25 +274,47 @@ function foldInline(node: PhrasingContent, source: string): Inline {
|
|
|
153
274
|
const underline = source.slice(p.start, p.start + 2) === "__";
|
|
154
275
|
return {
|
|
155
276
|
type: underline ? "underline" : "bold",
|
|
156
|
-
children: foldInlineChildren(node, source),
|
|
277
|
+
children: foldInlineChildren(node, source, ctx),
|
|
157
278
|
...p,
|
|
158
279
|
};
|
|
159
280
|
}
|
|
160
281
|
case "emphasis":
|
|
161
|
-
return { type: "italic", children: foldInlineChildren(node, source), ...pos(node) };
|
|
282
|
+
return { type: "italic", children: foldInlineChildren(node, source, ctx), ...pos(node) };
|
|
162
283
|
case "delete":
|
|
163
|
-
return { type: "strike", children: foldInlineChildren(node, source), ...pos(node) };
|
|
284
|
+
return { type: "strike", children: foldInlineChildren(node, source, ctx), ...pos(node) };
|
|
164
285
|
case "inlineCode":
|
|
165
286
|
return { type: "code", text: node.value, ...pos(node) };
|
|
166
287
|
case "link":
|
|
167
288
|
return {
|
|
168
289
|
type: "link",
|
|
169
290
|
href: node.url,
|
|
170
|
-
children: foldInlineChildren(node, source),
|
|
291
|
+
children: foldInlineChildren(node, source, ctx),
|
|
171
292
|
...pos(node),
|
|
172
293
|
};
|
|
173
|
-
|
|
174
|
-
|
|
294
|
+
case "image": {
|
|
295
|
+
// GFM's image syntax doubles as Telegram's INLINE-entity syntax:
|
|
296
|
+
// `` (date_time) and
|
|
297
|
+
// `` (custom emoji). Fold those two into a
|
|
298
|
+
// `tg-entity` node so the renderer re-emits the construct verbatim
|
|
299
|
+
// instead of escaping the brackets to literal text. Every OTHER image
|
|
300
|
+
// url — notably the http(s) MEDIA forms, which Telegram accepts only as
|
|
301
|
+
// a SEPARATE block — falls through to the demote-to-`plain` default
|
|
302
|
+
// below, unchanged.
|
|
303
|
+
if (isTgInlineEntityHref(node.url)) {
|
|
304
|
+
return { type: "tg-entity", label: node.alt ?? "", href: node.url, ...pos(node) };
|
|
305
|
+
}
|
|
306
|
+
return { type: "plain", text: slice(source, node), ...pos(node) };
|
|
307
|
+
}
|
|
308
|
+
// GFM footnote reference marker (`[^1]`): natively supported by Telegram's
|
|
309
|
+
// rich markdown path (wire-verified 2026-08-13 — renders as the full
|
|
310
|
+
// superscript + anchor + reference_link machinery). The source bytes ARE
|
|
311
|
+
// the wire syntax, so fold to a `raw` node the renderer emits verbatim;
|
|
312
|
+
// a `plain` node would be escapeMarkdown'd (`\[^1\]`) and break the
|
|
313
|
+
// construct on the wire.
|
|
314
|
+
case "footnoteReference":
|
|
315
|
+
return { type: "raw", text: slice(source, node), ...pos(node) };
|
|
316
|
+
// Not in the palette (break, non-`tg:` image, html, …): keep the raw
|
|
317
|
+
// source text so no content is lost.
|
|
175
318
|
default:
|
|
176
319
|
return { type: "plain", text: slice(source, node), ...pos(node) };
|
|
177
320
|
}
|
|
@@ -242,42 +385,444 @@ function expandPlainNode(node: PlainNode, source: string): Inline[] {
|
|
|
242
385
|
return out;
|
|
243
386
|
}
|
|
244
387
|
|
|
388
|
+
// ---------------------------------------------------------------------------
|
|
389
|
+
// Raw-HTML tag folding (see html-fold.ts for the policy and its rationale)
|
|
390
|
+
// ---------------------------------------------------------------------------
|
|
391
|
+
|
|
392
|
+
/** One element of the stream the HTML matcher walks: either a classified HTML
|
|
393
|
+
* token or an already-folded IR inline. */
|
|
394
|
+
type HtmlFoldItem =
|
|
395
|
+
| { kind: "tag"; tag: HtmlTagInfo; start: number; end: number }
|
|
396
|
+
| { kind: "inline"; node: Inline };
|
|
397
|
+
|
|
398
|
+
/** Index of the token that closes `name`, honouring same-name nesting, or -1. */
|
|
399
|
+
function findClosingTag(items: HtmlFoldItem[], from: number, name: string): number {
|
|
400
|
+
let depth = 0;
|
|
401
|
+
for (let i = from; i < items.length; i++) {
|
|
402
|
+
const it = items[i];
|
|
403
|
+
if (it.kind !== "tag" || it.tag.name !== name) continue;
|
|
404
|
+
if (it.tag.kind === "open") depth++;
|
|
405
|
+
else if (it.tag.kind === "close") {
|
|
406
|
+
if (depth === 0) return i;
|
|
407
|
+
depth--;
|
|
408
|
+
}
|
|
409
|
+
}
|
|
410
|
+
return -1;
|
|
411
|
+
}
|
|
412
|
+
|
|
413
|
+
/** Flatten a folded run to literal text, or null when any child carries
|
|
414
|
+
* structure a code span cannot represent. Used for `<code>`. Deliberately
|
|
415
|
+
* accepts ONLY `plain`: an inner `code` node's own backticks are not in its
|
|
416
|
+
* `text`, so folding it in would silently DELETE them
|
|
417
|
+
* (`<code>a `b` c</code>` -> `` `a b c` ``), and a `raw` node's wire bytes
|
|
418
|
+
* would be swallowed into a literal span. Both degrade to the children
|
|
419
|
+
* instead, which preserves every byte. */
|
|
420
|
+
function inlineLiteralText(nodes: Inline[]): string | null {
|
|
421
|
+
let out = "";
|
|
422
|
+
for (const n of nodes) {
|
|
423
|
+
if (n.type !== "plain") return null;
|
|
424
|
+
out += n.text;
|
|
425
|
+
}
|
|
426
|
+
return out;
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
/** Where a folded run is being emitted. `noBreaks` marks a container whose
|
|
430
|
+
* content must stay on ONE line — a heading or a table cell — so a `<br>` or
|
|
431
|
+
* a structural separator has to degrade to a space instead of a line break,
|
|
432
|
+
* and a `<pre>` to an inline code span instead of a fence. Without this a
|
|
433
|
+
* `# a<br>b` heading emitted `# a \nb`, which ends the heading block and
|
|
434
|
+
* drops `b` out of it. */
|
|
435
|
+
interface FoldCtx {
|
|
436
|
+
noBreaks?: boolean;
|
|
437
|
+
/** Source offsets of every HTML token the DOCUMENT-level pass found a
|
|
438
|
+
* partner for (`matchedTagOffsets`). A token missing from this set is
|
|
439
|
+
* unmatched — prose, not markup. Empty means "no HTML in this document",
|
|
440
|
+
* in which case nothing consults it. */
|
|
441
|
+
matched: ReadonlySet<number>;
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
/** Separator nodes this module synthesized while degrading structural markup.
|
|
445
|
+
* Tracked by identity so `normalizeSeparators` can collapse and trim OUR
|
|
446
|
+
* separators without touching a hard break the author actually wrote with
|
|
447
|
+
* `<br>` (`a<br><br>b` must keep both). */
|
|
448
|
+
const structuralSeparators = new WeakSet<Inline>();
|
|
449
|
+
|
|
450
|
+
/** The whitespace a degraded block-level tag leaves behind. A GFM HARD break
|
|
451
|
+
* (two spaces + newline), not a bare `\n`: the renderer runs AFTER
|
|
452
|
+
* `normalizeParagraphBreaks`, so a lone `\n` would never be promoted and
|
|
453
|
+
* Telegram collapses it to a space. */
|
|
454
|
+
function mkSeparator(ctx: FoldCtx, start: number, end: number): Inline {
|
|
455
|
+
const node: Inline = {
|
|
456
|
+
type: "raw",
|
|
457
|
+
text: ctx.noBreaks === true ? " " : " \n",
|
|
458
|
+
start,
|
|
459
|
+
end,
|
|
460
|
+
};
|
|
461
|
+
structuralSeparators.add(node);
|
|
462
|
+
return node;
|
|
463
|
+
}
|
|
464
|
+
|
|
465
|
+
/** Drop leading/trailing synthesized separators and collapse runs of them, so
|
|
466
|
+
* a degrade never contributes stray whitespace at the edges of a block
|
|
467
|
+
* (`<div></div>` must still render to nothing, and a `<pre>` fence must not
|
|
468
|
+
* leave a dangling newline). */
|
|
469
|
+
function normalizeSeparators(nodes: Inline[]): Inline[] {
|
|
470
|
+
const out: Inline[] = [];
|
|
471
|
+
for (const node of nodes) {
|
|
472
|
+
if (!structuralSeparators.has(node)) {
|
|
473
|
+
out.push(node);
|
|
474
|
+
continue;
|
|
475
|
+
}
|
|
476
|
+
if (out.length === 0) continue;
|
|
477
|
+
if (structuralSeparators.has(out[out.length - 1])) continue;
|
|
478
|
+
out.push(node);
|
|
479
|
+
}
|
|
480
|
+
while (out.length > 0 && structuralSeparators.has(out[out.length - 1])) out.pop();
|
|
481
|
+
return out;
|
|
482
|
+
}
|
|
483
|
+
|
|
484
|
+
/** An unmatched marker or unclassifiable token kept as LITERAL text, with its
|
|
485
|
+
* angle brackets HTML-entity-escaped so the wire sees prose, not a tag. */
|
|
486
|
+
function literalTag(tag: HtmlTagInfo, start: number, end: number): Inline {
|
|
487
|
+
return { type: "plain", text: escapeHtmlLiteral(tag.raw), start, end };
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
/**
|
|
491
|
+
* Apply the HTML dialect policy over a mixed tag/inline stream.
|
|
492
|
+
*
|
|
493
|
+
* `active` carries the constructs already open in an ANCESTOR fold. Markdown
|
|
494
|
+
* cannot nest same-kind emphasis — `**a **b** c**` is asterisk soup on the
|
|
495
|
+
* reader's screen, not nested bold — so an emphasis tag whose target is
|
|
496
|
+
* already active degrades to its children. Combined with the single-child
|
|
497
|
+
* unwrap in `buildFoldedTag`, `<b>a <b>b</b> c</b>` and `<b>**already**</b>`
|
|
498
|
+
* both come out as ONE bold run. A nested `<a>` is NOT unwrapped that way: its
|
|
499
|
+
* href is content, so it degrades to escaped literal markup instead of being
|
|
500
|
+
* silently deleted.
|
|
501
|
+
*
|
|
502
|
+
* See `html-fold.ts` for the bucket policy and the matched-vs-unmatched
|
|
503
|
+
* discriminator this implements.
|
|
504
|
+
*/
|
|
505
|
+
function foldHtmlItems(
|
|
506
|
+
items: HtmlFoldItem[],
|
|
507
|
+
active: ReadonlySet<HtmlFoldTarget>,
|
|
508
|
+
ctx: FoldCtx,
|
|
509
|
+
): Inline[] {
|
|
510
|
+
const out: Inline[] = [];
|
|
511
|
+
let i = 0;
|
|
512
|
+
while (i < items.length) {
|
|
513
|
+
const it = items[i];
|
|
514
|
+
if (it.kind === "inline") {
|
|
515
|
+
out.push(it.node);
|
|
516
|
+
i++;
|
|
517
|
+
continue;
|
|
518
|
+
}
|
|
519
|
+
const tag = it.tag;
|
|
520
|
+
|
|
521
|
+
// A comment renders nothing anywhere: dropping it loses no content.
|
|
522
|
+
if (tag.kind === "comment") {
|
|
523
|
+
i++;
|
|
524
|
+
continue;
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
// `<br>` / `<br/>` — a real line break (a space where breaks are illegal).
|
|
528
|
+
if (HTML_FOLD_TAGS[tag.name] === "break" && tag.kind !== "close") {
|
|
529
|
+
out.push(
|
|
530
|
+
ctx.noBreaks === true
|
|
531
|
+
? { type: "raw", text: " ", start: it.start, end: it.end }
|
|
532
|
+
: { type: "raw", text: " \n", start: it.start, end: it.end },
|
|
533
|
+
);
|
|
534
|
+
i++;
|
|
535
|
+
continue;
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
// Void / self-closing tags delimit no content, so dropping their markup
|
|
539
|
+
// cannot lose text. Block-level ones still leave a separator.
|
|
540
|
+
if (tag.kind === "selfclose" || (tag.kind === "open" && isVoidTag(tag.name))) {
|
|
541
|
+
if (isPassthroughTag(tag)) {
|
|
542
|
+
out.push({ type: "raw", text: tag.raw, start: it.start, end: it.end });
|
|
543
|
+
} else if (isBlockLevelTag(tag.name)) {
|
|
544
|
+
out.push(mkSeparator(ctx, it.start, it.end));
|
|
545
|
+
}
|
|
546
|
+
i++;
|
|
547
|
+
continue;
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
// Bucket 2: the wire-verified allowlist, now BALANCE-CHECKED. Emitting an
|
|
551
|
+
// unmatched `<u>` raw is precisely the `unclosed start tag` 400 the fold
|
|
552
|
+
// policy exists to prevent — and the fallback resends the whole message as
|
|
553
|
+
// PLAIN TEXT, so one stray marker in relayed content degrades the entire
|
|
554
|
+
// reply. The check is document-level because a `<details>` open and close
|
|
555
|
+
// legitimately land in different mdast blocks.
|
|
556
|
+
if (isPassthroughTag(tag) && (tag.kind === "open" || tag.kind === "close")) {
|
|
557
|
+
out.push(
|
|
558
|
+
ctx.matched.has(it.start)
|
|
559
|
+
? { type: "raw", text: tag.raw, start: it.start, end: it.end }
|
|
560
|
+
: literalTag(tag, it.start, it.end),
|
|
561
|
+
);
|
|
562
|
+
i++;
|
|
563
|
+
continue;
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
// Everything else must be BALANCED to count as markup.
|
|
567
|
+
if (tag.kind === "open") {
|
|
568
|
+
const close = findClosingTag(items, i + 1, tag.name);
|
|
569
|
+
const closeItem = close >= 0 ? items[close] : undefined;
|
|
570
|
+
if (closeItem !== undefined && closeItem.kind === "tag") {
|
|
571
|
+
out.push(
|
|
572
|
+
...foldMatchedTag(tag, closeItem, items.slice(i + 1, close), active, ctx, {
|
|
573
|
+
start: it.start,
|
|
574
|
+
end: closeItem.end,
|
|
575
|
+
}),
|
|
576
|
+
);
|
|
577
|
+
i = close + 1;
|
|
578
|
+
continue;
|
|
579
|
+
}
|
|
580
|
+
}
|
|
581
|
+
|
|
582
|
+
// Matched somewhere ELSE in the document (a `<div>` whose `</div>` is a
|
|
583
|
+
// separate mdast block) — markup, so drop it, but leave a separator for a
|
|
584
|
+
// block-level name so the neighbouring blocks do not glue.
|
|
585
|
+
if (ctx.matched.has(it.start) && tag.kind !== "other") {
|
|
586
|
+
if (isBlockLevelTag(tag.name)) out.push(mkSeparator(ctx, it.start, it.end));
|
|
587
|
+
i++;
|
|
588
|
+
continue;
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
// Unmatched open, stray close, or an unclassifiable token: PROSE.
|
|
592
|
+
out.push(literalTag(tag, it.start, it.end));
|
|
593
|
+
i++;
|
|
594
|
+
}
|
|
595
|
+
return normalizeSeparators(out);
|
|
596
|
+
}
|
|
597
|
+
|
|
598
|
+
/** Emit a tag whose closing partner was found. */
|
|
599
|
+
function foldMatchedTag(
|
|
600
|
+
tag: HtmlTagInfo,
|
|
601
|
+
closeTag: HtmlFoldItem & { kind: "tag" },
|
|
602
|
+
innerItems: HtmlFoldItem[],
|
|
603
|
+
active: ReadonlySet<HtmlFoldTarget>,
|
|
604
|
+
ctx: FoldCtx,
|
|
605
|
+
span: Pos,
|
|
606
|
+
): Inline[] {
|
|
607
|
+
const target = HTML_FOLD_TAGS[tag.name];
|
|
608
|
+
|
|
609
|
+
// `<pre>` — the one BLOCK-level fold (see html-fold.ts).
|
|
610
|
+
if (target === "pre") {
|
|
611
|
+
const folded = buildPreBlock(innerItems, ctx, span);
|
|
612
|
+
if (folded !== null) return folded;
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
if (target !== undefined && target !== "break" && target !== "pre") {
|
|
616
|
+
const nested = new Set(active);
|
|
617
|
+
nested.add(target);
|
|
618
|
+
const inner = foldHtmlItems(innerItems, nested, ctx);
|
|
619
|
+
// A LINK cannot nest in markdown, and its href is content — degrading to
|
|
620
|
+
// the children would silently delete the inner URL. Keep the markup as
|
|
621
|
+
// escaped literal text instead.
|
|
622
|
+
if (active.has(target)) {
|
|
623
|
+
return target === "link"
|
|
624
|
+
? [
|
|
625
|
+
literalTag(tag, span.start, span.start + tag.raw.length),
|
|
626
|
+
...inner,
|
|
627
|
+
literalTag(closeTag.tag, closeTag.start, closeTag.end),
|
|
628
|
+
]
|
|
629
|
+
: inner;
|
|
630
|
+
}
|
|
631
|
+
const folded = buildFoldedTag(target, tag, inner, span.start, span.end);
|
|
632
|
+
// A tag we cannot represent faithfully (an `<a>` with no href, a `<code>`
|
|
633
|
+
// wrapping structure) degrades to its CONTENT — nothing is lost.
|
|
634
|
+
return folded !== null ? [folded] : inner;
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
// Unknown but BALANCED: markup dropped, content kept. A block-level name
|
|
638
|
+
// leaves a separator so its neighbours cannot glue together.
|
|
639
|
+
const inner = foldHtmlItems(innerItems, active, ctx);
|
|
640
|
+
if (!isBlockLevelTag(tag.name)) return inner;
|
|
641
|
+
return [
|
|
642
|
+
mkSeparator(ctx, span.start, span.start + tag.raw.length),
|
|
643
|
+
...inner,
|
|
644
|
+
mkSeparator(ctx, closeTag.start, closeTag.end),
|
|
645
|
+
];
|
|
646
|
+
}
|
|
647
|
+
|
|
648
|
+
/** Build the fenced-code equivalent of a matched `<pre>…</pre>`, or null when
|
|
649
|
+
* the content is not literal text a fence can carry (in which case the caller
|
|
650
|
+
* falls back to the ordinary degrade). An optional single `<code …>` wrapper
|
|
651
|
+
* is unwrapped, and its `class="language-…"` becomes the fence info string —
|
|
652
|
+
* exactly the shape Telegram's own Rich HTML reference documents. */
|
|
653
|
+
function buildPreBlock(
|
|
654
|
+
innerItems: HtmlFoldItem[],
|
|
655
|
+
ctx: FoldCtx,
|
|
656
|
+
span: Pos,
|
|
657
|
+
): Inline[] | null {
|
|
658
|
+
let body = innerItems;
|
|
659
|
+
let language: string | null = null;
|
|
660
|
+
const first = body[0];
|
|
661
|
+
const last = body[body.length - 1];
|
|
662
|
+
if (
|
|
663
|
+
body.length >= 2 &&
|
|
664
|
+
first?.kind === "tag" &&
|
|
665
|
+
first.tag.kind === "open" &&
|
|
666
|
+
first.tag.name === "code" &&
|
|
667
|
+
last?.kind === "tag" &&
|
|
668
|
+
last.tag.kind === "close" &&
|
|
669
|
+
last.tag.name === "code"
|
|
670
|
+
) {
|
|
671
|
+
language = languageOf(first.tag);
|
|
672
|
+
body = body.slice(1, -1);
|
|
673
|
+
}
|
|
674
|
+
let text = "";
|
|
675
|
+
for (const item of body) {
|
|
676
|
+
if (item.kind !== "inline" || item.node.type !== "plain") return null;
|
|
677
|
+
text += item.node.text;
|
|
678
|
+
}
|
|
679
|
+
// `<pre>` content routinely starts on the line after the tag and ends on the
|
|
680
|
+
// line before the close; those layout newlines are not content.
|
|
681
|
+
text = text.replace(/^\r?\n/, "").replace(/[ \t]*\r?\n[ \t]*$/, "");
|
|
682
|
+
// A fence cannot survive inside a heading or a table cell — degrade to an
|
|
683
|
+
// inline code span there, with the newlines flattened.
|
|
684
|
+
if (ctx.noBreaks === true) {
|
|
685
|
+
return [{ type: "code", text: text.replace(/\s*\r?\n\s*/g, " "), ...span }];
|
|
686
|
+
}
|
|
687
|
+
const fence = "```";
|
|
688
|
+
return [
|
|
689
|
+
mkSeparator(ctx, span.start, span.start),
|
|
690
|
+
{
|
|
691
|
+
type: "raw",
|
|
692
|
+
text: `${fence}${language ?? ""}\n${text}\n${fence}`,
|
|
693
|
+
...span,
|
|
694
|
+
},
|
|
695
|
+
mkSeparator(ctx, span.end, span.end),
|
|
696
|
+
];
|
|
697
|
+
}
|
|
698
|
+
|
|
699
|
+
function buildFoldedTag(
|
|
700
|
+
target: HtmlFoldTarget,
|
|
701
|
+
tag: HtmlTagInfo,
|
|
702
|
+
children: Inline[],
|
|
703
|
+
start: number,
|
|
704
|
+
end: number,
|
|
705
|
+
): Inline | null {
|
|
706
|
+
switch (target) {
|
|
707
|
+
case "bold":
|
|
708
|
+
case "italic":
|
|
709
|
+
case "strike":
|
|
710
|
+
// Already exactly that construct (`<b>**already**</b>`) — re-wrapping
|
|
711
|
+
// would emit `****already****`, which reads as literal asterisks.
|
|
712
|
+
if (children.length === 1 && children[0].type === target) return children[0];
|
|
713
|
+
return { type: target, children, start, end };
|
|
714
|
+
case "code": {
|
|
715
|
+
const text = inlineLiteralText(children);
|
|
716
|
+
return text === null ? null : { type: "code", text, start, end };
|
|
717
|
+
}
|
|
718
|
+
case "link": {
|
|
719
|
+
const href = hrefOf(tag);
|
|
720
|
+
return href === null ? null : { type: "link", href, children, start, end };
|
|
721
|
+
}
|
|
722
|
+
default:
|
|
723
|
+
return null;
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
/** Turn a raw HTML string into fold items: tokens stay tokens, the text
|
|
728
|
+
* between them becomes `plain` inlines (with spoiler/highlight recognition,
|
|
729
|
+
* matching how prose is treated everywhere else). */
|
|
730
|
+
function htmlStringToItems(raw: string, base: number, source: string): HtmlFoldItem[] {
|
|
731
|
+
return tokenizeHtml(raw, base).flatMap<HtmlFoldItem>((piece) =>
|
|
732
|
+
piece.kind === "tag"
|
|
733
|
+
? [{ kind: "tag", tag: piece.tag, start: piece.start, end: piece.end }]
|
|
734
|
+
: expandPlainNode(mkPlain(piece.text, piece.start, piece.end), source).map(
|
|
735
|
+
(node) => ({ kind: "inline", node }) as HtmlFoldItem,
|
|
736
|
+
),
|
|
737
|
+
);
|
|
738
|
+
}
|
|
739
|
+
|
|
245
740
|
/** Fold a run of mdast phrasing children into IR inlines, then run the
|
|
246
|
-
* post-parse spoiler/highlight recognition over the produced `plain` nodes
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
741
|
+
* post-parse spoiler/highlight recognition over the produced `plain` nodes,
|
|
742
|
+
* then apply the raw-HTML tag policy across the whole run (open/close markers
|
|
743
|
+
* are SIBLING mdast `html` nodes, never a parent, so matching has to happen
|
|
744
|
+
* at this level). */
|
|
745
|
+
function buildInlines(
|
|
746
|
+
children: ReadonlyArray<MdastNode>,
|
|
747
|
+
source: string,
|
|
748
|
+
ctx: FoldCtx,
|
|
749
|
+
): Inline[] {
|
|
750
|
+
const items: HtmlFoldItem[] = [];
|
|
751
|
+
let sawHtml = false;
|
|
752
|
+
for (const child of children) {
|
|
753
|
+
if ((child as PhrasingContent).type === "html") {
|
|
754
|
+
sawHtml = true;
|
|
755
|
+
const p = pos(child);
|
|
756
|
+
items.push(...htmlStringToItems(slice(source, child), p.start, source));
|
|
757
|
+
continue;
|
|
758
|
+
}
|
|
759
|
+
const folded = foldInline(child as PhrasingContent, source, ctx);
|
|
760
|
+
const expanded = folded.type === "plain" ? expandPlainNode(folded, source) : [folded];
|
|
761
|
+
for (const node of expanded) items.push({ kind: "inline", node });
|
|
762
|
+
}
|
|
763
|
+
if (!sawHtml) return items.map((it) => (it as { node: Inline }).node);
|
|
764
|
+
return foldHtmlItems(items, new Set(), ctx);
|
|
251
765
|
}
|
|
252
766
|
|
|
253
|
-
function foldInlineChildren(
|
|
254
|
-
|
|
767
|
+
function foldInlineChildren(
|
|
768
|
+
node: MdastParent,
|
|
769
|
+
source: string,
|
|
770
|
+
ctx: FoldCtx,
|
|
771
|
+
): Inline[] {
|
|
772
|
+
return buildInlines(node.children, source, ctx);
|
|
255
773
|
}
|
|
256
774
|
|
|
257
775
|
function foldTableRow(
|
|
258
776
|
row: Extract<RootContent, { type: "tableRow" }>,
|
|
259
777
|
source: string,
|
|
778
|
+
ctx: FoldCtx,
|
|
260
779
|
): TableRow {
|
|
261
780
|
const cells: TableCell[] = row.children.map((cell) => ({
|
|
262
|
-
|
|
781
|
+
// A table cell is one line on the wire — a `<br>` or a structural degrade
|
|
782
|
+
// inside it must not emit a newline (it would end the row).
|
|
783
|
+
children: buildInlines(cell.children, source, { ...ctx, noBreaks: true }),
|
|
263
784
|
...pos(cell),
|
|
264
785
|
}));
|
|
265
786
|
return { cells, ...pos(row) };
|
|
266
787
|
}
|
|
267
788
|
|
|
789
|
+
/** True for a block that would render to the empty string. Only the raw-HTML
|
|
790
|
+
* fold can produce one — a block that is nothing but markup we dropped
|
|
791
|
+
* (`<!-- comment -->`, `<hr/>`, `<div></div>` on their own line). Left in,
|
|
792
|
+
* each would contribute a spurious `\n\n` separator to the rendered document,
|
|
793
|
+
* so they are filtered out at the point blocks are collected. */
|
|
794
|
+
function isEmptyBlock(block: Block): boolean {
|
|
795
|
+
return block.type === "paragraph" && block.children.length === 0;
|
|
796
|
+
}
|
|
797
|
+
|
|
798
|
+
/** Fold a run of mdast block nodes, dropping any that render to nothing. */
|
|
799
|
+
function foldBlocks(
|
|
800
|
+
nodes: ReadonlyArray<RootContent>,
|
|
801
|
+
source: string,
|
|
802
|
+
expandableLineStarts: Set<number>,
|
|
803
|
+
ctx: FoldCtx,
|
|
804
|
+
): Block[] {
|
|
805
|
+
return nodes
|
|
806
|
+
.map((n) => foldBlock(n, source, expandableLineStarts, ctx))
|
|
807
|
+
.filter((b) => !isEmptyBlock(b));
|
|
808
|
+
}
|
|
809
|
+
|
|
268
810
|
function foldBlock(
|
|
269
811
|
node: RootContent,
|
|
270
812
|
source: string,
|
|
271
813
|
expandableLineStarts: Set<number>,
|
|
814
|
+
ctx: FoldCtx,
|
|
272
815
|
): Block {
|
|
273
816
|
switch (node.type) {
|
|
274
817
|
case "paragraph":
|
|
275
|
-
return { type: "paragraph", children: foldInlineChildren(node, source), ...pos(node) };
|
|
818
|
+
return { type: "paragraph", children: foldInlineChildren(node, source, ctx), ...pos(node) };
|
|
276
819
|
case "heading":
|
|
277
820
|
return {
|
|
278
821
|
type: "heading",
|
|
279
822
|
level: node.depth,
|
|
280
|
-
|
|
823
|
+
// A heading is a single `#…` LINE: an embedded newline would end the
|
|
824
|
+
// block and orphan the rest of the text (`# a<br>b`).
|
|
825
|
+
children: foldInlineChildren(node, source, { ...ctx, noBreaks: true }),
|
|
281
826
|
...pos(node),
|
|
282
827
|
};
|
|
283
828
|
case "blockquote": {
|
|
@@ -287,7 +832,7 @@ function foldBlock(
|
|
|
287
832
|
const expandable = expandableLineStarts.has(lineStart(source, p.start));
|
|
288
833
|
return {
|
|
289
834
|
type: "blockquote",
|
|
290
|
-
children: node.children
|
|
835
|
+
children: foldBlocks(node.children, source, expandableLineStarts, ctx),
|
|
291
836
|
expandable,
|
|
292
837
|
...p,
|
|
293
838
|
};
|
|
@@ -301,7 +846,7 @@ function foldBlock(
|
|
|
301
846
|
};
|
|
302
847
|
case "list": {
|
|
303
848
|
const items: ListItem[] = node.children.map((li) => ({
|
|
304
|
-
children: li.children
|
|
849
|
+
children: foldBlocks(li.children, source, expandableLineStarts, ctx),
|
|
305
850
|
checked: li.checked ?? null,
|
|
306
851
|
// mdast `listItem.spread`: whether this item's block children are
|
|
307
852
|
// separated by a blank line. Tight (false) keeps a paragraph and its
|
|
@@ -326,14 +871,44 @@ function foldBlock(
|
|
|
326
871
|
const [headerRow, ...bodyRows] = rows;
|
|
327
872
|
return {
|
|
328
873
|
type: "table",
|
|
329
|
-
header: foldTableRow(headerRow, source),
|
|
330
|
-
rows: bodyRows.map((r) => foldTableRow(r, source)),
|
|
874
|
+
header: foldTableRow(headerRow, source, ctx),
|
|
875
|
+
rows: bodyRows.map((r) => foldTableRow(r, source, ctx)),
|
|
331
876
|
align: (node.align ?? []).map(foldAlign),
|
|
332
877
|
...pos(node),
|
|
333
878
|
};
|
|
334
879
|
}
|
|
335
|
-
//
|
|
336
|
-
//
|
|
880
|
+
// GFM footnote DEFINITION (`[^1]: body`): natively supported on the wire
|
|
881
|
+
// (2026-08-13 probe — pairs with the reference marker into footer/anchor
|
|
882
|
+
// nodes). Emit the raw source slice VERBATIM via a `raw` inline: a `plain`
|
|
883
|
+
// fold would escapeMarkdown the `[`/`]` (`\[^1\]: body`) and orphan the
|
|
884
|
+
// reference.
|
|
885
|
+
case "footnoteDefinition":
|
|
886
|
+
return {
|
|
887
|
+
type: "paragraph",
|
|
888
|
+
children: [{ type: "raw", text: slice(source, node), ...pos(node) }],
|
|
889
|
+
...pos(node),
|
|
890
|
+
};
|
|
891
|
+
// A raw HTML BLOCK (`<details>…`, `<div>…`, a bare `<b>alone</b>` line).
|
|
892
|
+
// mdast hands the whole block over as one opaque string, so tokenize it
|
|
893
|
+
// and run the same three-bucket policy the inline path uses: the
|
|
894
|
+
// wire-verified allowlist passes through raw (which is also what stops
|
|
895
|
+
// `escapeMarkdown` mangling `<details open="x">` into `open\="x"`),
|
|
896
|
+
// markdown-equivalent tags fold, everything else drops its markup and
|
|
897
|
+
// keeps its content.
|
|
898
|
+
case "html": {
|
|
899
|
+
const p = pos(node);
|
|
900
|
+
return {
|
|
901
|
+
type: "paragraph",
|
|
902
|
+
children: foldHtmlItems(
|
|
903
|
+
htmlStringToItems(slice(source, node), p.start, source),
|
|
904
|
+
new Set(),
|
|
905
|
+
ctx,
|
|
906
|
+
),
|
|
907
|
+
...p,
|
|
908
|
+
};
|
|
909
|
+
}
|
|
910
|
+
// Not in the palette (definition, …): degrade to a paragraph
|
|
911
|
+
// carrying the raw source slice so no content is dropped.
|
|
337
912
|
default:
|
|
338
913
|
return {
|
|
339
914
|
type: "paragraph",
|
|
@@ -356,8 +931,33 @@ export function parse(markdown: string): Document {
|
|
|
356
931
|
mdastExtensions: [gfmFromMarkdown()],
|
|
357
932
|
});
|
|
358
933
|
return {
|
|
359
|
-
blocks: tree.children
|
|
360
|
-
|
|
361
|
-
),
|
|
934
|
+
blocks: foldBlocks(tree.children, markdown, expandableLineStarts, {
|
|
935
|
+
matched: matchedTagOffsets(collectHtmlTokens(tree.children, markdown)),
|
|
936
|
+
}),
|
|
362
937
|
};
|
|
363
938
|
}
|
|
939
|
+
|
|
940
|
+
/** Every HTML token in the tree, in document order, for the balance pass.
|
|
941
|
+
* Walks mdast `html` nodes only — a tag inside a code fence or a code span is
|
|
942
|
+
* not markup and must not balance anything. */
|
|
943
|
+
function collectHtmlTokens(
|
|
944
|
+
nodes: ReadonlyArray<MdastNode>,
|
|
945
|
+
source: string,
|
|
946
|
+
): { tag: HtmlTagInfo; start: number }[] {
|
|
947
|
+
const out: { tag: HtmlTagInfo; start: number }[] = [];
|
|
948
|
+
const walk = (list: ReadonlyArray<MdastNode>): void => {
|
|
949
|
+
for (const node of list) {
|
|
950
|
+
if (node.type === "html") {
|
|
951
|
+
const p = pos(node);
|
|
952
|
+
for (const piece of tokenizeHtml(slice(source, node), p.start)) {
|
|
953
|
+
if (piece.kind === "tag") out.push({ tag: piece.tag, start: piece.start });
|
|
954
|
+
}
|
|
955
|
+
continue;
|
|
956
|
+
}
|
|
957
|
+
const children = (node as MdastParent).children;
|
|
958
|
+
if (Array.isArray(children)) walk(children);
|
|
959
|
+
}
|
|
960
|
+
};
|
|
961
|
+
walk(nodes);
|
|
962
|
+
return out;
|
|
963
|
+
}
|