switchroom 0.21.8 → 0.21.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +100 -39
- package/dist/host-control/main.js +1 -1
- package/package.json +2 -2
- package/skills/switchroom-architecture/telegram.md +12 -10
- package/skills/switchroom-cli/SKILL.md +1 -1
- package/telegram-plugin/README.md +3 -1
- package/telegram-plugin/dist/gateway/gateway.js +919 -348
- package/telegram-plugin/format.ts +30 -5
- package/telegram-plugin/gateway/outbound-send-path.ts +7 -0
- package/telegram-plugin/gateway/speech-capture.ts +158 -0
- package/telegram-plugin/package.json +1 -1
- package/telegram-plugin/render/code-segments.ts +38 -4
- package/telegram-plugin/render/dollar-math-guard.ts +16 -1
- package/telegram-plugin/render/html-fold.ts +354 -0
- package/telegram-plugin/render/ir.ts +53 -3
- package/telegram-plugin/render/parse.ts +642 -42
- package/telegram-plugin/render/render.ts +53 -15
- package/telegram-plugin/render/unsupported-token-guard.ts +45 -80
- package/telegram-plugin/rich-send.ts +22 -7
- package/telegram-plugin/shared/bot-runtime.ts +3 -2
- package/telegram-plugin/telegraph.ts +6 -4
- package/telegram-plugin/tests/grammy-rich-message-types.test.ts +199 -0
- package/telegram-plugin/tests/render/dollar-math-guard.test.ts +43 -0
- package/telegram-plugin/tests/render/guard-composition.test.ts +102 -0
- package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +253 -0
- package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
- package/telegram-plugin/tests/render/parse.test.ts +39 -10
- package/telegram-plugin/tests/render/render.test.ts +9 -4
- package/telegram-plugin/tests/render/rich-render.test.ts +46 -5
- package/telegram-plugin/tests/render/tg-entity.test.ts +242 -0
- package/telegram-plugin/tests/render/unsupported-token-guard.test.ts +66 -66
- package/telegram-plugin/tests/send-reply-golden.test.ts +99 -1
- package/telegram-plugin/tests/sent-text-capture.test.ts +3 -3
- package/telegram-plugin/tests/speech-capture.test.ts +296 -0
- package/telegram-plugin/tests/telegraph.test.ts +1 -1
- package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
- package/telegram-plugin/tts-normalize.ts +47 -9
- package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +17 -8
- package/telegram-plugin/voice-normalize-text.ts +48 -9
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
// Raw-HTML dialect handling for the Bot API 10.1 rich-markdown render path.
|
|
2
|
+
//
|
|
3
|
+
// ── Why this module exists ────────────────────────────────────────────────
|
|
4
|
+
// Agents habitually emit raw HTML tags in prose (`<b>`, `<i>`, `<a href>`).
|
|
5
|
+
// Before this module those tags reached the wire byte-verbatim, which is two
|
|
6
|
+
// distinct hazards at once:
|
|
7
|
+
//
|
|
8
|
+
// 1. Nothing establishes that Telegram's rich markdown parser accepts them.
|
|
9
|
+
// The wire-verified allowlist (raw `sendRichMessage` probes, 2026-08-13)
|
|
10
|
+
// covers exactly `<u>`, `<sub>`, `<sup>`, `<details>`/`<summary>` and
|
|
11
|
+
// `<aside>`/`<cite>`. `isParseEntitiesError` (`rich-send.ts`) matches
|
|
12
|
+
// `unsupported start tag` / `unclosed start tag`, so the wire CAN 400 on
|
|
13
|
+
// an unknown tag — and that fallback resends the body as PLAIN TEXT,
|
|
14
|
+
// where the reader then sees literal `<b>` markup.
|
|
15
|
+
// 2. mdast `html` nodes degrade to `plain` in `parse.ts`, and `plain` is
|
|
16
|
+
// `escapeMarkdown`'d on render. `escapeMarkdown` escapes `=`, so
|
|
17
|
+
// `<a href="https://example.com">` shipped as `<a href\="…">` — an
|
|
18
|
+
// attribute no parser can read.
|
|
19
|
+
//
|
|
20
|
+
// ── The policy (three buckets, degrade by TYPE, never silently) ───────────
|
|
21
|
+
// • FOLD — a tag with an exact native markdown equivalent is folded
|
|
22
|
+
// into the IR node for that construct, so it renders as the
|
|
23
|
+
// markdown the wire actually understands:
|
|
24
|
+
// <b>/<strong> -> bold **…**
|
|
25
|
+
// <i>/<em> -> italic *…*
|
|
26
|
+
// <s>/<del>/<strike> -> strike ~~…~~
|
|
27
|
+
// <code> -> code span `…`
|
|
28
|
+
// <a href="URL"> -> link [label](URL)
|
|
29
|
+
// <br> -> a line break
|
|
30
|
+
// <pre> -> fenced code block ```…```
|
|
31
|
+
// • PASSTHROUGH — the wire-verified allowlist above is emitted RAW and
|
|
32
|
+
// UNESCAPED (an IR `raw` inline), which is also what stops
|
|
33
|
+
// `escapeMarkdown` from mangling their attributes. Requires a
|
|
34
|
+
// MATCHED close: an unbalanced `<u>` emitted raw is exactly
|
|
35
|
+
// the `unclosed start tag` 400 this module exists to prevent.
|
|
36
|
+
// • DEGRADE — everything else, split by whether the token delimits
|
|
37
|
+
// content. The discriminator is MATCHING, not the tag name:
|
|
38
|
+
// – A MATCHED pair (`<marquee>…</marquee>`, `<div>…</div>`)
|
|
39
|
+
// is markup wrapped around content. The markup is dropped
|
|
40
|
+
// and the content kept; a block-level name additionally
|
|
41
|
+
// leaves a hard line break, so `<li>one</li><li>two</li>`
|
|
42
|
+
// cannot glue into `onetwo`.
|
|
43
|
+
// – A comment, or a void / self-closing tag (`<hr/>`,
|
|
44
|
+
// `<img …/>`), delimits nothing. Markup dropped; a
|
|
45
|
+
// block-level one leaves the same separator.
|
|
46
|
+
// – An UNMATCHED open/close marker is not markup at all — it
|
|
47
|
+
// is PROSE. `<service>/<key>`, `<agent>`, "the `<b>` tag",
|
|
48
|
+
// `run switchroom vault get <key>` are pervasive in this
|
|
49
|
+
// project's own agent output, and DELETING them silently
|
|
50
|
+
// mangles the agent's own reply. Such a token is kept as
|
|
51
|
+
// LITERAL TEXT with `&`/`<`/`>` HTML-entity-escaped (see
|
|
52
|
+
// `escapeHtmlLiteral`). Bot API rich markdown "can contain
|
|
53
|
+
// arbitrary HTML … parsed as described in Rich HTML
|
|
54
|
+
// style", and Rich HTML documents `<`, `>` and
|
|
55
|
+
// `&` among its supported named entities — so the
|
|
56
|
+
// entity form is the DOCUMENTED way to ship a literal
|
|
57
|
+
// angle bracket without tripping `unsupported start tag`.
|
|
58
|
+
//
|
|
59
|
+
// ── The invariant ─────────────────────────────────────────────────────────
|
|
60
|
+
// CONTENT is never lost. Only MARKUP is dropped, and only where dropping it
|
|
61
|
+
// loses no reader-visible text: a matched pair, a comment, a void tag.
|
|
62
|
+
// Anything not demonstrably markup survives to the wire as escaped literal
|
|
63
|
+
// text. (Known residual, NOT introduced by this module and not fixed here: a
|
|
64
|
+
// `<` in prose that mdast hands over as a TEXT node rather than an `html` node
|
|
65
|
+
// — the decoded `<c>`, or `response in <1s` — never reaches this module
|
|
66
|
+
// and still ships unescaped. That is a `plain`-node / `escapeMarkdown`
|
|
67
|
+
// concern.)
|
|
68
|
+
//
|
|
69
|
+
// This is a deterministic code-level guarantee, deliberately not a prompt
|
|
70
|
+
// instruction to the agents that emit the tags.
|
|
71
|
+
|
|
72
|
+
/** How a raw HTML token reads. `other` covers anything the tokenizer matched
|
|
73
|
+
* but could not classify (it degrades like an unknown tag). */
|
|
74
|
+
export type HtmlTagKind = "open" | "close" | "selfclose" | "comment" | "other";
|
|
75
|
+
|
|
76
|
+
export interface HtmlTagInfo {
|
|
77
|
+
kind: HtmlTagKind;
|
|
78
|
+
/** Lowercased tag name (`""` for a comment). */
|
|
79
|
+
name: string;
|
|
80
|
+
/** The raw source bytes of the token, verbatim. */
|
|
81
|
+
raw: string;
|
|
82
|
+
/** Raw attribute text between the tag name and the closing `>`. */
|
|
83
|
+
attrs: string;
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
/** The IR construct a foldable tag maps onto. `pre` is BLOCK-level (a fenced
|
|
87
|
+
* code block); every other target is inline. */
|
|
88
|
+
export type HtmlFoldTarget =
|
|
89
|
+
| "bold"
|
|
90
|
+
| "italic"
|
|
91
|
+
| "strike"
|
|
92
|
+
| "code"
|
|
93
|
+
| "link"
|
|
94
|
+
| "break"
|
|
95
|
+
| "pre";
|
|
96
|
+
|
|
97
|
+
/** Tags with an exact native markdown equivalent on the Bot API 10.1 rich
|
|
98
|
+
* path. Folding them means the wire sees markdown it definitely parses
|
|
99
|
+
* instead of HTML it may reject. */
|
|
100
|
+
export const HTML_FOLD_TAGS: Readonly<Record<string, HtmlFoldTarget>> = {
|
|
101
|
+
b: "bold",
|
|
102
|
+
strong: "bold",
|
|
103
|
+
i: "italic",
|
|
104
|
+
em: "italic",
|
|
105
|
+
s: "strike",
|
|
106
|
+
del: "strike",
|
|
107
|
+
strike: "strike",
|
|
108
|
+
code: "code",
|
|
109
|
+
a: "link",
|
|
110
|
+
br: "break",
|
|
111
|
+
// `<pre>` is the one BLOCK-level fold. Telegram's own Rich HTML reference
|
|
112
|
+
// pairs `<pre><code class="language-…">` with the ```` ```lang ```` fence,
|
|
113
|
+
// so the fenced block is the exact native equivalent. Without it `<pre>`
|
|
114
|
+
// dropped its markup and the inner `<code>` folded to an INLINE span
|
|
115
|
+
// wrapping a newline — a construct Telegram will not parse, leaving the
|
|
116
|
+
// reader literal backticks around broken text.
|
|
117
|
+
pre: "pre",
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
/** HTML elements that carry no content of their own (void elements) plus the
|
|
121
|
+
* XHTML self-closing spelling. Dropping such a token's markup cannot lose
|
|
122
|
+
* text, so it degrades silently rather than surviving as literal prose.
|
|
123
|
+
* `br` is excluded — it FOLDS to a line break. */
|
|
124
|
+
export const HTML_VOID_TAGS: ReadonlySet<string> = new Set([
|
|
125
|
+
"area",
|
|
126
|
+
"base",
|
|
127
|
+
"col",
|
|
128
|
+
"embed",
|
|
129
|
+
"hr",
|
|
130
|
+
"img",
|
|
131
|
+
"input",
|
|
132
|
+
"link",
|
|
133
|
+
"meta",
|
|
134
|
+
"param",
|
|
135
|
+
"source",
|
|
136
|
+
"track",
|
|
137
|
+
"wbr",
|
|
138
|
+
]);
|
|
139
|
+
|
|
140
|
+
/** Structural / block-level element names. A degrade of one of these leaves a
|
|
141
|
+
* hard line break behind so its neighbours do not glue together
|
|
142
|
+
* (`<ul><li>one</li><li>two</li></ul>` -> `onetwo` was the defect). The
|
|
143
|
+
* passthrough allowlist (`details`, `summary`, `aside`, `cite`) is
|
|
144
|
+
* deliberately absent — those never degrade. */
|
|
145
|
+
export const HTML_BLOCK_LEVEL_TAGS: ReadonlySet<string> = new Set([
|
|
146
|
+
"address",
|
|
147
|
+
"article",
|
|
148
|
+
"blockquote",
|
|
149
|
+
"center",
|
|
150
|
+
"dd",
|
|
151
|
+
"div",
|
|
152
|
+
"dl",
|
|
153
|
+
"dt",
|
|
154
|
+
"fieldset",
|
|
155
|
+
"figcaption",
|
|
156
|
+
"figure",
|
|
157
|
+
"footer",
|
|
158
|
+
"form",
|
|
159
|
+
"h1",
|
|
160
|
+
"h2",
|
|
161
|
+
"h3",
|
|
162
|
+
"h4",
|
|
163
|
+
"h5",
|
|
164
|
+
"h6",
|
|
165
|
+
"header",
|
|
166
|
+
"hr",
|
|
167
|
+
"li",
|
|
168
|
+
"main",
|
|
169
|
+
"nav",
|
|
170
|
+
"ol",
|
|
171
|
+
"p",
|
|
172
|
+
"section",
|
|
173
|
+
"table",
|
|
174
|
+
"tbody",
|
|
175
|
+
"td",
|
|
176
|
+
"tfoot",
|
|
177
|
+
"th",
|
|
178
|
+
"thead",
|
|
179
|
+
"tr",
|
|
180
|
+
"ul",
|
|
181
|
+
]);
|
|
182
|
+
|
|
183
|
+
/** True when a degraded token should leave a line break behind. */
|
|
184
|
+
export function isBlockLevelTag(name: string): boolean {
|
|
185
|
+
return HTML_BLOCK_LEVEL_TAGS.has(name);
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** True for a tag that delimits no content of its own. */
|
|
189
|
+
export function isVoidTag(name: string): boolean {
|
|
190
|
+
return HTML_VOID_TAGS.has(name);
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* HTML-entity-escape a run of literal text so its angle brackets reach the
|
|
195
|
+
* reader instead of being parsed as a tag (or 400ing as an unsupported one).
|
|
196
|
+
*
|
|
197
|
+
* Bot API rich markdown "can contain arbitrary HTML … parsed as described in
|
|
198
|
+
* Rich HTML style", and Rich HTML's supported named entities are documented as
|
|
199
|
+
* exactly `< > & " ' … — –
|
|
200
|
+
* ‘ ’ “ ”` (https://core.telegram.org/bots/api). Only
|
|
201
|
+
* the first three are needed here. `&` is escaped FIRST so an already-entity
|
|
202
|
+
* -looking run cannot be double-decoded.
|
|
203
|
+
*/
|
|
204
|
+
export function escapeHtmlLiteral(text: string): string {
|
|
205
|
+
return text.replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">");
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
/** The `class="language-xxx"` hint on a `<pre><code …>` block, or null. */
|
|
209
|
+
const LANGUAGE_CLASS_RE = /(?:^|\s)class\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))/i;
|
|
210
|
+
|
|
211
|
+
/** Pull the fenced-block language out of a `<code class="language-python">`
|
|
212
|
+
* tag's attributes, or null. Telegram's reference spells the language hint
|
|
213
|
+
* exactly this way; anything else yields a plain fence. */
|
|
214
|
+
export function languageOf(tag: HtmlTagInfo): string | null {
|
|
215
|
+
const m = LANGUAGE_CLASS_RE.exec(tag.attrs);
|
|
216
|
+
if (m == null) return null;
|
|
217
|
+
const cls = m[1] ?? m[2] ?? m[3] ?? "";
|
|
218
|
+
const lang = cls
|
|
219
|
+
.split(/\s+/)
|
|
220
|
+
.map((c) => /^language-(.+)$/.exec(c)?.[1])
|
|
221
|
+
.find((c): c is string => c != null && c.length > 0);
|
|
222
|
+
// A fence info string cannot contain a backtick (it would close the fence).
|
|
223
|
+
return lang != null && !lang.includes("`") ? lang : null;
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/** Tags PROVEN on the wire by raw `sendRichMessage` probes (2026-08-13) to
|
|
227
|
+
* render as native typed nodes. These pass through raw and unescaped.
|
|
228
|
+
* Deliberately an allowlist: an unprobed tag is not known-good syntax. */
|
|
229
|
+
export const HTML_PASSTHROUGH_TAGS: ReadonlySet<string> = new Set([
|
|
230
|
+
"u",
|
|
231
|
+
"sub",
|
|
232
|
+
"sup",
|
|
233
|
+
"details",
|
|
234
|
+
"summary",
|
|
235
|
+
"aside",
|
|
236
|
+
"cite",
|
|
237
|
+
]);
|
|
238
|
+
|
|
239
|
+
/** Matches one HTML comment or one start/end tag. Attribute values containing
|
|
240
|
+
* a raw `>` are not supported (they are invalid unquoted HTML and vanishingly
|
|
241
|
+
* rare in agent prose); such a token simply degrades. */
|
|
242
|
+
export const HTML_TOKEN_RE =
|
|
243
|
+
/<!--[\s\S]*?-->|<\/?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\/?>/g;
|
|
244
|
+
|
|
245
|
+
const TAG_RE = /^<(\/?)([A-Za-z][A-Za-z0-9-]*)((?:\s[^<>]*?)?)(\/?)>$/;
|
|
246
|
+
const HREF_RE = /(?:^|\s)href\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'>]+))/i;
|
|
247
|
+
|
|
248
|
+
/** Classify a single raw HTML token. */
|
|
249
|
+
export function classifyHtmlTag(raw: string): HtmlTagInfo {
|
|
250
|
+
if (raw.startsWith("<!--")) return { kind: "comment", name: "", raw, attrs: "" };
|
|
251
|
+
const m = TAG_RE.exec(raw);
|
|
252
|
+
if (m == null) return { kind: "other", name: "", raw, attrs: "" };
|
|
253
|
+
const [, slash, name, attrs, selfClose] = m;
|
|
254
|
+
const kind: HtmlTagKind =
|
|
255
|
+
slash === "/" ? "close" : selfClose === "/" ? "selfclose" : "open";
|
|
256
|
+
return { kind, name: name.toLowerCase(), raw, attrs: attrs ?? "" };
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/** Pull the `href` value out of a tag's attribute text, or null when absent
|
|
260
|
+
* or empty. The ORIGINAL bytes are returned — never rewritten. */
|
|
261
|
+
export function hrefOf(tag: HtmlTagInfo): string | null {
|
|
262
|
+
const m = HREF_RE.exec(tag.attrs);
|
|
263
|
+
if (m == null) return null;
|
|
264
|
+
const value = m[1] ?? m[2] ?? m[3] ?? "";
|
|
265
|
+
return value.length > 0 ? value : null;
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
/** True when this tag is emitted raw and unescaped (wire-verified allowlist). */
|
|
269
|
+
export function isPassthroughTag(tag: HtmlTagInfo): boolean {
|
|
270
|
+
return tag.name.length > 0 && HTML_PASSTHROUGH_TAGS.has(tag.name);
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Decide which HTML tokens in a DOCUMENT are balanced markup and which are
|
|
275
|
+
* bare prose, given every token in document order.
|
|
276
|
+
*
|
|
277
|
+
* Matching cannot be decided inside a single mdast `html` node: micromark
|
|
278
|
+
* splits `<details open>…\n\nbody\n\n</details>` into THREE nodes, so the open
|
|
279
|
+
* and close markers arrive in different streams. A document-level pass is the
|
|
280
|
+
* only place the question is answerable. Feeding it mdast `html` nodes (rather
|
|
281
|
+
* than the raw source) also means tags inside a code fence or a code span are
|
|
282
|
+
* never counted — they are not markup and must not balance anything.
|
|
283
|
+
*
|
|
284
|
+
* Standard HTML-ish matching: opens are pushed on a stack and a close pops to
|
|
285
|
+
* the nearest same-named open, leaving anything above it unmatched. Void and
|
|
286
|
+
* self-closing tags are balanced by definition and are not tracked.
|
|
287
|
+
*
|
|
288
|
+
* Returns the SOURCE OFFSETS of every token that has a partner; a token whose
|
|
289
|
+
* offset is absent is unmatched — prose, per the module policy above.
|
|
290
|
+
*/
|
|
291
|
+
export function matchedTagOffsets(
|
|
292
|
+
tokens: ReadonlyArray<{ tag: HtmlTagInfo; start: number }>,
|
|
293
|
+
): Set<number> {
|
|
294
|
+
const matched = new Set<number>();
|
|
295
|
+
const stack: { name: string; start: number }[] = [];
|
|
296
|
+
for (const { tag, start } of tokens) {
|
|
297
|
+
if (tag.kind === "open" && !isVoidTag(tag.name)) {
|
|
298
|
+
stack.push({ name: tag.name, start });
|
|
299
|
+
continue;
|
|
300
|
+
}
|
|
301
|
+
if (tag.kind !== "close") continue;
|
|
302
|
+
for (let i = stack.length - 1; i >= 0; i--) {
|
|
303
|
+
if (stack[i].name !== tag.name) continue;
|
|
304
|
+
matched.add(stack[i].start);
|
|
305
|
+
matched.add(start);
|
|
306
|
+
stack.length = i;
|
|
307
|
+
break;
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
return matched;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
/** A piece of a raw-HTML string: either an HTML token or a run of text. */
|
|
314
|
+
export type HtmlPiece =
|
|
315
|
+
| { kind: "tag"; tag: HtmlTagInfo; start: number; end: number }
|
|
316
|
+
| { kind: "text"; text: string; start: number; end: number };
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Split a raw string into HTML tokens and the text runs between them.
|
|
320
|
+
* `base` is the absolute source offset the string starts at, so every piece
|
|
321
|
+
* carries usable UTF-16 offsets into the original markdown.
|
|
322
|
+
*/
|
|
323
|
+
export function tokenizeHtml(raw: string, base: number): HtmlPiece[] {
|
|
324
|
+
const pieces: HtmlPiece[] = [];
|
|
325
|
+
const re = new RegExp(HTML_TOKEN_RE.source, "g");
|
|
326
|
+
let last = 0;
|
|
327
|
+
let m: RegExpExecArray | null;
|
|
328
|
+
while ((m = re.exec(raw)) != null) {
|
|
329
|
+
if (m.index > last) {
|
|
330
|
+
pieces.push({
|
|
331
|
+
kind: "text",
|
|
332
|
+
text: raw.slice(last, m.index),
|
|
333
|
+
start: base + last,
|
|
334
|
+
end: base + m.index,
|
|
335
|
+
});
|
|
336
|
+
}
|
|
337
|
+
pieces.push({
|
|
338
|
+
kind: "tag",
|
|
339
|
+
tag: classifyHtmlTag(m[0]),
|
|
340
|
+
start: base + m.index,
|
|
341
|
+
end: base + m.index + m[0].length,
|
|
342
|
+
});
|
|
343
|
+
last = m.index + m[0].length;
|
|
344
|
+
}
|
|
345
|
+
if (last < raw.length) {
|
|
346
|
+
pieces.push({
|
|
347
|
+
kind: "text",
|
|
348
|
+
text: raw.slice(last),
|
|
349
|
+
start: base + last,
|
|
350
|
+
end: base + raw.length,
|
|
351
|
+
});
|
|
352
|
+
}
|
|
353
|
+
return pieces;
|
|
354
|
+
}
|
|
@@ -30,11 +30,21 @@
|
|
|
30
30
|
// highlight -> `==…==` (Bot API 10.1 marked entity)
|
|
31
31
|
// code -> `` `…` ``
|
|
32
32
|
// link -> `[…](…)`
|
|
33
|
+
// tg-entity -> `` (Bot API date_time / custom-emoji entity)
|
|
34
|
+
// raw -> source bytes verbatim (never escaped) — footnote markers
|
|
35
|
+
// `[^1]` and definition lines `[^1]: …`, which Telegram's
|
|
36
|
+
// rich parser renders natively and escapeMarkdown would break
|
|
33
37
|
//
|
|
34
38
|
// Block
|
|
35
39
|
// paragraph -> children joined; blocks separated by "\n\n"
|
|
36
40
|
// heading -> `#`…`######` line
|
|
37
|
-
// blockquote -> `> …`
|
|
41
|
+
// blockquote -> `> …` on every line. `expandable === true` records that
|
|
42
|
+
// the SOURCE carried the legacy `**> ` marker, but it is
|
|
43
|
+
// NOT a distinct wire style: `**>` is MarkdownV2-only
|
|
44
|
+
// syntax that the rich markdown path renders as literal
|
|
45
|
+
// text (wire-verified 2026-08-13), so the renderer
|
|
46
|
+
// degrades it to a plain quote. Authors wanting a real
|
|
47
|
+
// collapsible use `<details><summary>…</summary>…</details>`.
|
|
38
48
|
// code-block -> ```` ```lang … ``` ````
|
|
39
49
|
// list -> line-per-item with `-`/`1.` markers
|
|
40
50
|
// thematic-break -> `---` thematic break
|
|
@@ -109,6 +119,40 @@ export interface LinkNode extends Pos {
|
|
|
109
119
|
children: Inline[];
|
|
110
120
|
}
|
|
111
121
|
|
|
122
|
+
/** A Telegram rich-markdown INLINE entity written in mdast IMAGE position.
|
|
123
|
+
* The "Rich Markdown style" grammar (https://core.telegram.org/bots/api,
|
|
124
|
+
* quoted in `reference/telegram-formatting-guide.md`) lists exactly two:
|
|
125
|
+
*
|
|
126
|
+
*  custom emoji
|
|
127
|
+
*  date_time
|
|
128
|
+
*
|
|
129
|
+
* (the `date_time` MessageEntity is Bot API 9.5, March 1 2026; the rich-message
|
|
130
|
+
* `RichTextDateTime` class is 10.1, June 11 2026 — both in the Bot API
|
|
131
|
+
* changelog.) `parse.ts` folds ONLY those two `tg:` hrefs into this node;
|
|
132
|
+
* every other image url keeps the historical demote-to-`plain` fallback,
|
|
133
|
+
* because an http(s) `` is a Telegram MEDIA block — "Media can be
|
|
134
|
+
* specified only as a separate block" (same doc) — not an inline entity, and
|
|
135
|
+
* switchroom does not emit media blocks.
|
|
136
|
+
*
|
|
137
|
+
* `label` is the DECODED alternative text (mdast `image.alt`, empty for the
|
|
138
|
+
* emoji form); `href` is the `tg:` URL. Both are re-escaped on render, same
|
|
139
|
+
* as `LinkNode`. */
|
|
140
|
+
export interface TgEntityNode extends Pos {
|
|
141
|
+
type: "tg-entity";
|
|
142
|
+
label: string;
|
|
143
|
+
href: string;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** Verbatim wire passthrough: the node's SOURCE bytes are already the exact
|
|
147
|
+
* syntax Telegram's rich parser expects, so the renderer must emit them
|
|
148
|
+
* unescaped (escapeMarkdown would corrupt them). Used for GFM footnote
|
|
149
|
+
* reference markers (`[^1]`) and footnote definition lines (`[^1]: …`) —
|
|
150
|
+
* both natively supported on the rich path (wire-verified 2026-08-13). */
|
|
151
|
+
export interface RawNode extends Pos {
|
|
152
|
+
type: "raw";
|
|
153
|
+
text: string;
|
|
154
|
+
}
|
|
155
|
+
|
|
112
156
|
export type Inline =
|
|
113
157
|
| PlainNode
|
|
114
158
|
| BoldNode
|
|
@@ -118,7 +162,9 @@ export type Inline =
|
|
|
118
162
|
| SpoilerNode
|
|
119
163
|
| HighlightNode
|
|
120
164
|
| CodeNode
|
|
121
|
-
| LinkNode
|
|
165
|
+
| LinkNode
|
|
166
|
+
| TgEntityNode
|
|
167
|
+
| RawNode;
|
|
122
168
|
|
|
123
169
|
// ---------------------------------------------------------------------------
|
|
124
170
|
// Block nodes
|
|
@@ -139,7 +185,11 @@ export interface HeadingNode extends Pos {
|
|
|
139
185
|
export interface BlockquoteNode extends Pos {
|
|
140
186
|
type: "blockquote";
|
|
141
187
|
children: Block[];
|
|
142
|
-
/**
|
|
188
|
+
/** True when the source carried the LEGACY switchroom `**> ` expandable
|
|
189
|
+
* marker (see parse.ts markExpandableQuotes). Records authoring intent
|
|
190
|
+
* only: `**>` is MarkdownV2 syntax with no rich-markdown equivalent
|
|
191
|
+
* (wire-verified 2026-08-13), so the renderer emits a plain `> ` quote
|
|
192
|
+
* either way. */
|
|
143
193
|
expandable: boolean;
|
|
144
194
|
}
|
|
145
195
|
|