switchroom 0.21.8 → 0.21.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +100 -39
- package/dist/host-control/main.js +1 -1
- package/package.json +2 -2
- package/skills/switchroom-architecture/telegram.md +12 -10
- package/skills/switchroom-cli/SKILL.md +1 -1
- package/telegram-plugin/README.md +3 -1
- package/telegram-plugin/dist/gateway/gateway.js +919 -348
- package/telegram-plugin/format.ts +30 -5
- package/telegram-plugin/gateway/outbound-send-path.ts +7 -0
- package/telegram-plugin/gateway/speech-capture.ts +158 -0
- package/telegram-plugin/package.json +1 -1
- package/telegram-plugin/render/code-segments.ts +38 -4
- package/telegram-plugin/render/dollar-math-guard.ts +16 -1
- package/telegram-plugin/render/html-fold.ts +354 -0
- package/telegram-plugin/render/ir.ts +53 -3
- package/telegram-plugin/render/parse.ts +642 -42
- package/telegram-plugin/render/render.ts +53 -15
- package/telegram-plugin/render/unsupported-token-guard.ts +45 -80
- package/telegram-plugin/rich-send.ts +22 -7
- package/telegram-plugin/shared/bot-runtime.ts +3 -2
- package/telegram-plugin/telegraph.ts +6 -4
- package/telegram-plugin/tests/grammy-rich-message-types.test.ts +199 -0
- package/telegram-plugin/tests/render/dollar-math-guard.test.ts +43 -0
- package/telegram-plugin/tests/render/guard-composition.test.ts +102 -0
- package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +253 -0
- package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
- package/telegram-plugin/tests/render/parse.test.ts +39 -10
- package/telegram-plugin/tests/render/render.test.ts +9 -4
- package/telegram-plugin/tests/render/rich-render.test.ts +46 -5
- package/telegram-plugin/tests/render/tg-entity.test.ts +242 -0
- package/telegram-plugin/tests/render/unsupported-token-guard.test.ts +66 -66
- package/telegram-plugin/tests/send-reply-golden.test.ts +99 -1
- package/telegram-plugin/tests/sent-text-capture.test.ts +3 -3
- package/telegram-plugin/tests/speech-capture.test.ts +296 -0
- package/telegram-plugin/tests/telegraph.test.ts +1 -1
- package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
- package/telegram-plugin/tts-normalize.ts +47 -9
- package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +17 -8
- package/telegram-plugin/voice-normalize-text.ts +48 -9
|
@@ -218,17 +218,21 @@ describe("parse: nested inline", () => {
|
|
|
218
218
|
});
|
|
219
219
|
|
|
220
220
|
describe("parse: unsupported construct fallback", () => {
|
|
221
|
-
it("degrades
|
|
222
|
-
// Raw HTML blocks (mdast `html`) are outside the palette
|
|
223
|
-
//
|
|
221
|
+
it("degrades an UNRECOGNISED raw HTML block to its content, markup dropped", () => {
|
|
222
|
+
// Raw HTML blocks (mdast `html`) are outside the palette. An unrecognised
|
|
223
|
+
// tag is NOT passed through verbatim: nothing establishes that Telegram's
|
|
224
|
+
// rich parser accepts `<div>`, and `isParseEntitiesError` (rich-send.ts)
|
|
225
|
+
// shows the wire can 400 with `unsupported start tag` — whose fallback
|
|
226
|
+
// resends the body as PLAIN TEXT, putting literal `<div class="x">` on the
|
|
227
|
+
// reader's screen. Degrade by type: markup dropped, content kept.
|
|
224
228
|
const md = "<div class=\"x\">raw</div>";
|
|
225
229
|
const doc = parse(md);
|
|
226
230
|
const block = doc.blocks[0] as any;
|
|
227
231
|
expect(block.type).toBe("paragraph");
|
|
228
232
|
const plain = block.children[0];
|
|
229
233
|
expect(plain.type).toBe("plain");
|
|
230
|
-
expect(plain.text).toBe(
|
|
231
|
-
expect(md.slice(plain.start, plain.end)).toBe(
|
|
234
|
+
expect(plain.text).toBe("raw");
|
|
235
|
+
expect(md.slice(plain.start, plain.end)).toBe("raw");
|
|
232
236
|
});
|
|
233
237
|
});
|
|
234
238
|
|
|
@@ -336,7 +340,7 @@ describe("parse: nested lists", () => {
|
|
|
336
340
|
});
|
|
337
341
|
});
|
|
338
342
|
|
|
339
|
-
describe("expandable blockquote (
|
|
343
|
+
describe("expandable blockquote (LEGACY switchroom `**>` marker — input repair only)", () => {
|
|
340
344
|
it("parses a single-line `**>` quote into an expandable blockquote", () => {
|
|
341
345
|
const md = "**> a collapsible line";
|
|
342
346
|
const doc = parse(md);
|
|
@@ -374,10 +378,10 @@ describe("expandable blockquote (Bot API 10.1 `**>` marker)", () => {
|
|
|
374
378
|
});
|
|
375
379
|
|
|
376
380
|
it("only recognises the marker at column 0 (leading indent is not expandable)", () => {
|
|
377
|
-
// The
|
|
378
|
-
// variant is deliberately NOT treated as an expandable
|
|
379
|
-
// would push the length-preserving rewrite past the
|
|
380
|
-
// budget into indented-code-block territory).
|
|
381
|
+
// The legacy encoding only ever placed `**>` at column 0; a
|
|
382
|
+
// leading-indented variant is deliberately NOT treated as an expandable
|
|
383
|
+
// quote (matching it would push the length-preserving rewrite past the
|
|
384
|
+
// 3-space blockquote budget into indented-code-block territory).
|
|
381
385
|
const md = " **> indented";
|
|
382
386
|
const doc = parse(md);
|
|
383
387
|
const bq = doc.blocks[0] as any;
|
|
@@ -391,3 +395,28 @@ describe("expandable blockquote (Bot API 10.1 `**>` marker)", () => {
|
|
|
391
395
|
expect(doc.blocks[0].type).toBe("paragraph");
|
|
392
396
|
});
|
|
393
397
|
});
|
|
398
|
+
|
|
399
|
+
describe("footnotes fold to verbatim `raw` nodes (native Telegram construct)", () => {
|
|
400
|
+
it("a footnote reference marker folds to a raw inline carrying its source bytes", () => {
|
|
401
|
+
// GFM only recognises a reference when a matching definition exists in
|
|
402
|
+
// the document (a lone `[^n1]` is ordinary text).
|
|
403
|
+
const doc = parse("claim[^n1] more\n\n[^n1]: body");
|
|
404
|
+
const para = doc.blocks[0] as Extract<Block, { type: "paragraph" }>;
|
|
405
|
+
const raw = para.children.find((c: Inline) => c.type === "raw") as
|
|
406
|
+
| Extract<Inline, { type: "raw" }>
|
|
407
|
+
| undefined;
|
|
408
|
+
expect(raw).toBeDefined();
|
|
409
|
+
expect(raw!.text).toBe("[^n1]");
|
|
410
|
+
});
|
|
411
|
+
|
|
412
|
+
it("a footnote definition folds to a paragraph with one raw inline (verbatim slice)", () => {
|
|
413
|
+
const doc = parse("[^n1]: body text");
|
|
414
|
+
const para = doc.blocks[0] as Extract<Block, { type: "paragraph" }>;
|
|
415
|
+
expect(para.type).toBe("paragraph");
|
|
416
|
+
expect(para.children).toHaveLength(1);
|
|
417
|
+
expect(para.children[0].type).toBe("raw");
|
|
418
|
+
expect((para.children[0] as Extract<Inline, { type: "raw" }>).text).toBe(
|
|
419
|
+
"[^n1]: body text",
|
|
420
|
+
);
|
|
421
|
+
});
|
|
422
|
+
});
|
|
@@ -195,7 +195,10 @@ describe("render: block palette", () => {
|
|
|
195
195
|
it("plain blockquote", () => {
|
|
196
196
|
expect(render(parse("> quoted line"))).toBe("> quoted line");
|
|
197
197
|
});
|
|
198
|
-
it("expandable
|
|
198
|
+
it("expandable IR flag renders as a PLAIN quote — the retired `**>` marker is never emitted", () => {
|
|
199
|
+
// `**>` is MarkdownV2-only syntax; the rich markdown path renders it as
|
|
200
|
+
// LITERAL `**>` text (wire-proved 2026-08-13 via raw sendRichMessage
|
|
201
|
+
// probes), so the renderer degrades an expandable node to a plain quote.
|
|
199
202
|
const doc: Document = {
|
|
200
203
|
blocks: [
|
|
201
204
|
{
|
|
@@ -214,9 +217,10 @@ describe("render: block palette", () => {
|
|
|
214
217
|
},
|
|
215
218
|
],
|
|
216
219
|
};
|
|
217
|
-
expect(render(doc)).toBe("
|
|
220
|
+
expect(render(doc)).toBe("> hidden gem");
|
|
221
|
+
expect(render(doc)).not.toContain("**>");
|
|
218
222
|
});
|
|
219
|
-
it("multi-line expandable blockquote
|
|
223
|
+
it("multi-line expandable blockquote renders every line with a plain > marker", () => {
|
|
220
224
|
const doc: Document = {
|
|
221
225
|
blocks: [
|
|
222
226
|
{
|
|
@@ -242,8 +246,9 @@ describe("render: block palette", () => {
|
|
|
242
246
|
],
|
|
243
247
|
};
|
|
244
248
|
const out = render(doc);
|
|
249
|
+
expect(out).not.toContain("**>");
|
|
245
250
|
const lines = out.split("\n");
|
|
246
|
-
expect(lines[0]).toBe("
|
|
251
|
+
expect(lines[0]).toBe("> line one");
|
|
247
252
|
for (const line of lines.slice(1)) {
|
|
248
253
|
expect(line.startsWith("> ") || line === ">").toBe(true);
|
|
249
254
|
}
|
|
@@ -5,6 +5,7 @@ import {
|
|
|
5
5
|
renderOutbound,
|
|
6
6
|
maybeRenderOutbound,
|
|
7
7
|
} from "../../render/rich-render.js";
|
|
8
|
+
import { guardAccidentalFormatting } from "../../rich-send.js";
|
|
8
9
|
|
|
9
10
|
describe("parseRichRenderEnabled", () => {
|
|
10
11
|
it("defaults ON when unset (escape hatch, not opt-in)", () => {
|
|
@@ -52,8 +53,10 @@ describe("maybeRenderOutbound", () => {
|
|
|
52
53
|
it("default (env unset) routes through parse -> renderSafe", () => {
|
|
53
54
|
const r = maybeRenderOutbound("**> collapsible", {} as NodeJS.ProcessEnv);
|
|
54
55
|
expect(r.mode).toBe("markdown");
|
|
55
|
-
// The expandable
|
|
56
|
-
|
|
56
|
+
// The legacy expandable marker is REPAIRED to a plain quote: `**>` is
|
|
57
|
+
// MarkdownV2-only syntax the rich path renders as literal text
|
|
58
|
+
// (wire-proved 2026-08-13), so the renderer must never re-emit it.
|
|
59
|
+
expect(r.text).toBe("> collapsible");
|
|
57
60
|
});
|
|
58
61
|
|
|
59
62
|
it("default preserves plain prose through the round-trip", () => {
|
|
@@ -84,15 +87,53 @@ describe("maybeRenderOutbound", () => {
|
|
|
84
87
|
SWITCHROOM_RICH_RENDER: "maybe",
|
|
85
88
|
} as NodeJS.ProcessEnv);
|
|
86
89
|
expect(r.mode).toBe("markdown");
|
|
87
|
-
expect(r.text).
|
|
90
|
+
expect(r.text).toBe("> collapsible");
|
|
88
91
|
});
|
|
89
92
|
});
|
|
90
93
|
|
|
91
94
|
describe("renderOutbound (flag-independent)", () => {
|
|
92
|
-
it("
|
|
95
|
+
it("repairs a legacy `**>` quote to a plain quote end to end", () => {
|
|
93
96
|
const r = renderOutbound("**> hidden line one\n> hidden line two");
|
|
94
97
|
expect(r.mode).toBe("markdown");
|
|
95
|
-
|
|
98
|
+
// One coherent quote, no retired marker: on the pre-fix renderer the
|
|
99
|
+
// first line came back as `**> hidden line one`, which Telegram rendered
|
|
100
|
+
// as LITERAL `**>` paragraph text (wire-proved 2026-08-13).
|
|
101
|
+
expect(r.text).not.toContain("**>");
|
|
102
|
+
expect(r.text.split("\n")[0]).toBe("> hidden line one");
|
|
103
|
+
expect(r.text).toContain("> hidden line two");
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it("passes native constructs through: <details>, $math$, footnotes", () => {
|
|
107
|
+
// Wire-verified natives (2026-08-13) must survive parse -> renderSafe
|
|
108
|
+
// BYTE-IDENTICAL. On the pre-fix pipeline the footnote case failed:
|
|
109
|
+
// escapeMarkdown turned `[^n1]` into `\[^n1\]`, breaking the construct.
|
|
110
|
+
const details =
|
|
111
|
+
"<details open><summary>S</summary>\n\nbody\n\n</details>";
|
|
112
|
+
expect(renderOutbound(details).text).toBe(details);
|
|
113
|
+
|
|
114
|
+
const math = "inline $x^2+y^2$ done";
|
|
115
|
+
expect(renderOutbound(math).text).toBe(math);
|
|
116
|
+
|
|
117
|
+
const footnotes = "claim[^n1] more\n\n[^n1]: body text";
|
|
118
|
+
expect(renderOutbound(footnotes).text).toBe(footnotes);
|
|
119
|
+
});
|
|
120
|
+
|
|
121
|
+
it("a tg:// inline entity and a footnote survive TOGETHER in one message", () => {
|
|
122
|
+
// Cross-feature composition (#4683 x #4685): the `tg-entity` fold (mdast
|
|
123
|
+
// image position) and the footnote `raw` fold (footnoteReference /
|
|
124
|
+
// footnoteDefinition) land in the SAME foldInline/foldBlock walk — this
|
|
125
|
+
// pins that neither eats the other. On #4683 alone the footnote came back
|
|
126
|
+
// as `claim\[^n1\]`; on #4685 alone (pre-rebase) the entity came back as
|
|
127
|
+
// `!\[now\](tg://time?unix\=…)` literal text.
|
|
128
|
+
const combined =
|
|
129
|
+
"meet at  as promised[^n1]\n\n[^n1]: agreed yesterday";
|
|
130
|
+
expect(renderOutbound(combined).text).toBe(combined);
|
|
131
|
+
|
|
132
|
+
// Same pair inside ONE paragraph plus the guard seam on top: the composed
|
|
133
|
+
// wire body (renderOutbound then the richMessage guard, i.e. the full
|
|
134
|
+
// streamed send path) is still byte-identical.
|
|
135
|
+
const guarded = guardAccidentalFormatting(renderOutbound(combined).text);
|
|
136
|
+
expect(guarded).toBe(combined);
|
|
96
137
|
});
|
|
97
138
|
|
|
98
139
|
it("falls back to plain mode for oversized atomic content", () => {
|
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
// Telegram inline `tg:` entities in mdast IMAGE position survive the whole
|
|
2
|
+
// outbound pipeline.
|
|
3
|
+
//
|
|
4
|
+
// Bot API grammar (https://core.telegram.org/bots/api, "Rich Markdown style"):
|
|
5
|
+
//
|
|
6
|
+
//  custom emoji
|
|
7
|
+
//  date_time
|
|
8
|
+
//
|
|
9
|
+
// (the `date_time` MessageEntity landed in Bot API 9.5, 2026-03-01; the
|
|
10
|
+
// rich-message `RichTextDateTime` class in 10.1, 2026-06-11 — Bot API
|
|
11
|
+
// changelog.) grammy 1.44.0 speaks 10.1, so the wire has accepted this since
|
|
12
|
+
// #2669 — but the RENDER path escaped the brackets to literal text, so the
|
|
13
|
+
// syntax was dead on the streamed send path.
|
|
14
|
+
//
|
|
15
|
+
// Two DISTINCT outbound surfaces are pinned here, because they do not share a
|
|
16
|
+
// pipeline:
|
|
17
|
+
// - the STREAMED path (stream-controller.ts:272) → `renderOutboundChunks` →
|
|
18
|
+
// `parse → renderSafe`. This is the surface the escaping bug lived on.
|
|
19
|
+
// - the REPLY-TOOL path (outbound-send-path.ts) → `computeReplyChunks` →
|
|
20
|
+
// `richMessage()`, which NEVER runs the renderer — only the
|
|
21
|
+
// accidental-formatting guards. It was already correct; these tests pin it
|
|
22
|
+
// so a future guard can't silently start eating the syntax.
|
|
23
|
+
|
|
24
|
+
import { describe, it, expect } from "vitest";
|
|
25
|
+
import { parse } from "../../render/parse.js";
|
|
26
|
+
import { renderSafe, SUPPORTED_INLINE } from "../../render/render.js";
|
|
27
|
+
import { renderOutboundChunks } from "../../render/rich-render.js";
|
|
28
|
+
import { guardAccidentalFormatting, richMessage } from "../../rich-send.js";
|
|
29
|
+
import { computeReplyChunks } from "../../gateway/outbound-send-path.js";
|
|
30
|
+
import { splitMarkdownChunks, RICH_MESSAGE_MAX_CHARS } from "../../format.js";
|
|
31
|
+
import type { Inline } from "../../render/ir.js";
|
|
32
|
+
|
|
33
|
+
/** The doc's own date_time example, verbatim. */
|
|
34
|
+
const DATE_TIME = "";
|
|
35
|
+
/** The doc's own custom-emoji example, verbatim (empty label). */
|
|
36
|
+
const EMOJI = "";
|
|
37
|
+
|
|
38
|
+
/** parse → renderSafe, the streamed path's core. */
|
|
39
|
+
const roundTrip = (md: string): string => renderSafe(parse(md), md).text;
|
|
40
|
+
|
|
41
|
+
/** Flatten every inline node in the rendered IR of `md`. */
|
|
42
|
+
function inlines(md: string): Inline[] {
|
|
43
|
+
const out: Inline[] = [];
|
|
44
|
+
const walk = (n: Inline): void => {
|
|
45
|
+
out.push(n);
|
|
46
|
+
for (const c of (n as { children?: Inline[] }).children ?? []) walk(c);
|
|
47
|
+
};
|
|
48
|
+
for (const b of parse(md).blocks) {
|
|
49
|
+
for (const n of (b as { children?: Inline[] }).children ?? []) walk(n);
|
|
50
|
+
}
|
|
51
|
+
return out;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
describe("tg:// inline entities: parse → renderSafe round-trip", () => {
|
|
55
|
+
// THE bug this PR fixes: at HEAD the image node demoted to `plain` and
|
|
56
|
+
// `escapeMarkdown` backslash-escaped `[`, `]` and `=`, shipping
|
|
57
|
+
// `!\[22:45 tomorrow\](tg://time?unix\=…&format\=wDT)` — literal text, no
|
|
58
|
+
// entity. Byte-exact preservation is the whole contract.
|
|
59
|
+
it("preserves a date_time entity byte-exact", () => {
|
|
60
|
+
const body = `Meeting at ${DATE_TIME} sharp.`;
|
|
61
|
+
expect(roundTrip(body)).toBe(body);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
it("preserves the empty-label custom-emoji entity byte-exact", () => {
|
|
65
|
+
expect(roundTrip(`hi ${EMOJI} there`)).toBe(`hi ${EMOJI} there`);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
it("preserves a date_time entity with no `format` parameter", () => {
|
|
69
|
+
const body = "";
|
|
70
|
+
expect(roundTrip(body)).toBe(body);
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
it("folds the entity into a `tg-entity` IR node, not `plain`", () => {
|
|
74
|
+
const node = inlines(`x ${DATE_TIME}`).find((n) => n.type === "tg-entity");
|
|
75
|
+
expect(node).toMatchObject({
|
|
76
|
+
type: "tg-entity",
|
|
77
|
+
label: "22:45 tomorrow",
|
|
78
|
+
href: "tg://time?unix=1647531900&format=wDT",
|
|
79
|
+
});
|
|
80
|
+
});
|
|
81
|
+
|
|
82
|
+
it("survives inside a blockquote, a heading, a list item and a table cell", () => {
|
|
83
|
+
for (const body of [
|
|
84
|
+
`> when: ${DATE_TIME}`,
|
|
85
|
+
`## Due ${DATE_TIME}`,
|
|
86
|
+
`- due ${DATE_TIME}`,
|
|
87
|
+
`| when | who |\n| --- | --- |\n| ${DATE_TIME} | me |`,
|
|
88
|
+
]) {
|
|
89
|
+
expect(roundTrip(body)).toContain(DATE_TIME);
|
|
90
|
+
}
|
|
91
|
+
});
|
|
92
|
+
|
|
93
|
+
it("matches the `tg:` href case-insensitively (URL schemes are)", () => {
|
|
94
|
+
const body = "";
|
|
95
|
+
// Folded as an entity, and the author's ORIGINAL bytes are re-emitted —
|
|
96
|
+
// the parser never rewrites the href.
|
|
97
|
+
expect(inlines(body).some((n) => n.type === "tg-entity")).toBe(true);
|
|
98
|
+
expect(roundTrip(body)).toBe(body);
|
|
99
|
+
});
|
|
100
|
+
|
|
101
|
+
it("lists `tg-entity` in the renderer's construct allowlist", () => {
|
|
102
|
+
expect(SUPPORTED_INLINE).toContain("tg-entity");
|
|
103
|
+
});
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
describe("tg:// inline entities: label escaping (no bracket breakout)", () => {
|
|
107
|
+
// A model-authored label carrying `]` would close the label early and
|
|
108
|
+
// smuggle raw bracket syntax past the renderer if the label were re-emitted
|
|
109
|
+
// undecorated. It is prose, so it gets `escapeMarkdown` exactly like a
|
|
110
|
+
// `plain` node.
|
|
111
|
+
it("escapes brackets in the label instead of letting them close it early", () => {
|
|
112
|
+
const out = roundTrip("![see [22:45]](tg://time?unix=1&format=t)");
|
|
113
|
+
expect(out).toBe("![see \\[22:45\\]](tg://time?unix=1&format=t)");
|
|
114
|
+
// No BARE `]` before the destination — the only unescaped one closes the label.
|
|
115
|
+
expect(out.slice(0, out.indexOf("](")).includes("\\]")).toBe(true);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
it("escapes emphasis delimiters in the label", () => {
|
|
119
|
+
expect(roundTrip("")).toBe(
|
|
120
|
+
"",
|
|
121
|
+
);
|
|
122
|
+
});
|
|
123
|
+
|
|
124
|
+
it("collapses a soft line break in the label so the construct stays on one line", () => {
|
|
125
|
+
const out = roundTrip("");
|
|
126
|
+
expect(out).toBe("");
|
|
127
|
+
expect(out).not.toContain("\n");
|
|
128
|
+
});
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
describe("tg:// inline entities: no regression of the image demotion", () => {
|
|
132
|
+
// Deliberate, unchanged: an http(s) `` is a Telegram MEDIA block —
|
|
133
|
+
// "Media can be specified only as a separate block" (Rich Markdown style) —
|
|
134
|
+
// not an inline entity, and switchroom emits no media blocks. It keeps
|
|
135
|
+
// degrading to escaped literal text.
|
|
136
|
+
it("still degrades an http(s) image link to escaped plain text", () => {
|
|
137
|
+
expect(roundTrip("")).toBe(
|
|
138
|
+
"!\\[alt\\](https://x.example/y.png)",
|
|
139
|
+
);
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
it("still degrades an UNDOCUMENTED tg:// image href to escaped plain text", () => {
|
|
143
|
+
// The allowlist is deliberate: only `tg://time` and `tg://emoji` are
|
|
144
|
+
// documented in image position, so anything else stays literal rather than
|
|
145
|
+
// shipping syntax Telegram may parse-reject.
|
|
146
|
+
expect(roundTrip("")).toBe("!\\[t\\](tg://photo?id\\=1)");
|
|
147
|
+
expect(roundTrip("")).toBe("!\\[t\\](tg://user?id\\=1)");
|
|
148
|
+
});
|
|
149
|
+
|
|
150
|
+
it("leaves an ordinary `[label](tg://user?id=…)` mention link alone", () => {
|
|
151
|
+
const body = "ping [Ken](tg://user?id=123456789)";
|
|
152
|
+
expect(roundTrip(body)).toBe(body);
|
|
153
|
+
});
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
describe("tg:// inline entities: code context stays literal", () => {
|
|
157
|
+
it("keeps the syntax verbatim inside an inline code span", () => {
|
|
158
|
+
const body = "use `` for that";
|
|
159
|
+
expect(roundTrip(body)).toBe(body);
|
|
160
|
+
// and it is a `code` node, NOT a folded entity
|
|
161
|
+
expect(inlines(body).some((n) => n.type === "tg-entity")).toBe(false);
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
it("keeps the syntax verbatim inside a fenced code block", () => {
|
|
165
|
+
const body = "```\n\n```";
|
|
166
|
+
expect(roundTrip(body)).toBe(body);
|
|
167
|
+
expect(inlines(body).some((n) => n.type === "tg-entity")).toBe(false);
|
|
168
|
+
});
|
|
169
|
+
});
|
|
170
|
+
|
|
171
|
+
describe("tg:// inline entities: the accidental-formatting guards are a no-op", () => {
|
|
172
|
+
// `guardAccidentalFormatting` is the universal seam (installed as a grammy
|
|
173
|
+
// API transformer in shared/bot-runtime.ts), so it runs over EVERY outbound
|
|
174
|
+
// body — rendered or hand-built card alike. Its documented trigger set
|
|
175
|
+
// (`_ * > . ~ == || $ #`) does not intersect this syntax, but nothing
|
|
176
|
+
// asserted that until now.
|
|
177
|
+
const bodies = [
|
|
178
|
+
`Meeting at ${DATE_TIME} sharp.`,
|
|
179
|
+
`${DATE_TIME} at the very start of the line`,
|
|
180
|
+
EMOJI,
|
|
181
|
+
`- due ${DATE_TIME}\n- and ${DATE_TIME}`,
|
|
182
|
+
`> when: ${DATE_TIME}`,
|
|
183
|
+
`| when |\n| --- |\n| ${DATE_TIME} |`,
|
|
184
|
+
"![see \\[22:45\\]](tg://time?unix=1&format=t)",
|
|
185
|
+
];
|
|
186
|
+
|
|
187
|
+
it("leaves every tg:// entity body byte-identical", () => {
|
|
188
|
+
for (const body of bodies) expect(guardAccidentalFormatting(body)).toBe(body);
|
|
189
|
+
});
|
|
190
|
+
|
|
191
|
+
it("is idempotent over the rendered form too", () => {
|
|
192
|
+
for (const body of bodies) {
|
|
193
|
+
const rendered = roundTrip(body);
|
|
194
|
+
expect(guardAccidentalFormatting(rendered)).toBe(rendered);
|
|
195
|
+
}
|
|
196
|
+
});
|
|
197
|
+
});
|
|
198
|
+
|
|
199
|
+
describe("tg:// inline entities: live outbound seams", () => {
|
|
200
|
+
it("survives the STREAMED seam (renderOutboundChunks)", () => {
|
|
201
|
+
const body = `Standup starts ${DATE_TIME}.`;
|
|
202
|
+
const pieces = renderOutboundChunks(body);
|
|
203
|
+
expect(pieces).toHaveLength(1);
|
|
204
|
+
expect(pieces[0].mode).toBe("markdown");
|
|
205
|
+
expect(pieces[0].text).toBe(body);
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
it("survives the REPLY-TOOL seam (computeReplyChunks → richMessage)", () => {
|
|
209
|
+
// This surface never touches the renderer — it is guards-only. Pinned so a
|
|
210
|
+
// future guard cannot start escaping the syntax unnoticed.
|
|
211
|
+
const body = `Standup starts ${DATE_TIME}.`;
|
|
212
|
+
const chunks = computeReplyChunks({
|
|
213
|
+
effectiveText: body,
|
|
214
|
+
literalText: false,
|
|
215
|
+
limit: RICH_MESSAGE_MAX_CHARS,
|
|
216
|
+
chunkMode: "length",
|
|
217
|
+
});
|
|
218
|
+
expect(chunks.map((c) => richMessage(c).markdown).join("")).toContain(DATE_TIME);
|
|
219
|
+
});
|
|
220
|
+
});
|
|
221
|
+
|
|
222
|
+
describe("tg:// inline entities: chunk boundaries never bisect the construct", () => {
|
|
223
|
+
// `INLINE_SPAN_PATTERNS`' link pattern used to start at the `[`, leaving the
|
|
224
|
+
// leading `!` outside the protected span: a cut inside the construct
|
|
225
|
+
// retreated only to the `[`, stranding `!` on the previous chunk and
|
|
226
|
+
// silently demoting a date_time entity to an ordinary link.
|
|
227
|
+
it("keeps a cap-straddling date_time entity whole in one chunk", () => {
|
|
228
|
+
const body = `${"x".repeat(80)} ${DATE_TIME}`;
|
|
229
|
+
const chunks = splitMarkdownChunks(body, 110);
|
|
230
|
+
expect(chunks.length).toBeGreaterThan(1);
|
|
231
|
+
// Exactly one chunk carries the entity, whole.
|
|
232
|
+
expect(chunks.filter((c) => c.includes(DATE_TIME))).toHaveLength(1);
|
|
233
|
+
// And no chunk ends on the stranded `!`.
|
|
234
|
+
for (const c of chunks) expect(c.endsWith("!")).toBe(false);
|
|
235
|
+
});
|
|
236
|
+
|
|
237
|
+
it("still keeps an ordinary link whole (unchanged behaviour)", () => {
|
|
238
|
+
const link = "[a fairly long link label here](https://example.com/some/path)";
|
|
239
|
+
const chunks = splitMarkdownChunks(`${"x".repeat(80)} ${link}`, 110);
|
|
240
|
+
expect(chunks.filter((c) => c.includes(link))).toHaveLength(1);
|
|
241
|
+
});
|
|
242
|
+
});
|
|
@@ -3,44 +3,61 @@ import { guardUnsupportedTokens } from "../../render/unsupported-token-guard.js"
|
|
|
3
3
|
import { richMessage } from "../../rich-send.js";
|
|
4
4
|
|
|
5
5
|
describe("guardUnsupportedTokens — deterministic send-time repair", () => {
|
|
6
|
-
|
|
6
|
+
// ── Natively-supported constructs pass through BYTE-IDENTICAL ────────────
|
|
7
|
+
// Wire-verified 2026-08-13: raw sendRichMessage probes showed `<details>`,
|
|
8
|
+
// footnotes, `<sub>`/`<sup>`/`<u>`, `<aside>`, `tg://time` and task lists
|
|
9
|
+
// all parse into real typed nodes on Telegram's rich markdown path. An
|
|
10
|
+
// earlier revision of this guard "repaired" `<details>` into a `**> `
|
|
11
|
+
// expandable blockquote (MarkdownV2-only syntax that renders as LITERAL
|
|
12
|
+
// `**>` text on this path) and deleted footnote markers — both conversions
|
|
13
|
+
// destroyed supported constructs and are deleted. These tests would FAIL on
|
|
14
|
+
// the old guard.
|
|
15
|
+
it("passes a <details><summary> block through untouched (native construct)", () => {
|
|
7
16
|
const input =
|
|
8
|
-
"Here is the trace:\n<details><summary>Stack trace</summary>\nline 1\nline 2\n</details>\ndone";
|
|
9
|
-
|
|
10
|
-
// Anti-tautology: the raw HTML tags MUST be gone from the wire body.
|
|
11
|
-
expect(out).not.toContain("<details>");
|
|
12
|
-
expect(out).not.toContain("</details>");
|
|
13
|
-
expect(out).not.toContain("<summary>");
|
|
14
|
-
// Summary becomes the expandable-blockquote first line (`**> ` marker).
|
|
15
|
-
expect(out).toContain("**> Stack trace");
|
|
16
|
-
// Body lines become plain `> ` continuation lines.
|
|
17
|
-
expect(out).toContain("> line 1");
|
|
18
|
-
expect(out).toContain("> line 2");
|
|
17
|
+
"Here is the trace:\n<details open><summary>Stack trace</summary>\n\nline 1\nline 2\n\n</details>\ndone";
|
|
18
|
+
expect(guardUnsupportedTokens(input)).toBe(input);
|
|
19
19
|
});
|
|
20
20
|
|
|
21
|
-
it("
|
|
22
|
-
const
|
|
23
|
-
expect(
|
|
24
|
-
expect(out).toContain("**> hidden body text");
|
|
21
|
+
it("passes a <details> without a <summary> through untouched", () => {
|
|
22
|
+
const input = "<details>hidden body text</details>";
|
|
23
|
+
expect(guardUnsupportedTokens(input)).toBe(input);
|
|
25
24
|
});
|
|
26
25
|
|
|
27
|
-
it("
|
|
28
|
-
|
|
29
|
-
"
|
|
30
|
-
);
|
|
31
|
-
expect(guardUnsupportedTokens("a ^highlighted^ word")).toBe(
|
|
32
|
-
"a highlighted word",
|
|
26
|
+
it("never converts <details> into the retired `**>` marker", () => {
|
|
27
|
+
const out = guardUnsupportedTokens(
|
|
28
|
+
"<details><summary>More</summary>\nbody\n</details>",
|
|
33
29
|
);
|
|
30
|
+
expect(out).not.toContain("**>");
|
|
31
|
+
expect(out).toContain("<details>");
|
|
34
32
|
});
|
|
35
33
|
|
|
36
|
-
it("
|
|
34
|
+
it("keeps footnote reference markers AND definition lines (native construct)", () => {
|
|
37
35
|
expect(guardUnsupportedTokens("see the note[^1] here")).toBe(
|
|
38
|
-
"see the note here",
|
|
36
|
+
"see the note[^1] here",
|
|
39
37
|
);
|
|
40
|
-
// A `[^1]:` definition line is left intact (the negative lookahead).
|
|
41
38
|
expect(guardUnsupportedTokens("[^1]: the definition")).toBe(
|
|
42
39
|
"[^1]: the definition",
|
|
43
40
|
);
|
|
41
|
+
// Alphanumeric ids too — the pair renders as real footnote machinery.
|
|
42
|
+
expect(guardUnsupportedTokens("claim[^note] holds\n\n[^note]: body")).toBe(
|
|
43
|
+
"claim[^note] holds\n\n[^note]: body",
|
|
44
|
+
);
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
it("passes <sub>/<sup>/<u>/<aside> HTML tags through untouched", () => {
|
|
48
|
+
const input =
|
|
49
|
+
"H<sub>2</sub>O and x<sup>2</sup> and <u>under</u>\n<aside>pull\n<cite>credit</cite></aside>";
|
|
50
|
+
expect(guardUnsupportedTokens(input)).toBe(input);
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
// ── The one genuinely-unsupported token: caret pairs ─────────────────────
|
|
54
|
+
it("strips caret highlight / superscript pairs to their inner text", () => {
|
|
55
|
+
expect(guardUnsupportedTokens("energy is x^2^ joules")).toBe(
|
|
56
|
+
"energy is x2 joules",
|
|
57
|
+
);
|
|
58
|
+
expect(guardUnsupportedTokens("a ^highlighted^ word")).toBe(
|
|
59
|
+
"a highlighted word",
|
|
60
|
+
);
|
|
44
61
|
});
|
|
45
62
|
|
|
46
63
|
it("leaves unpaired carets scattered across prose intact (no interior space)", () => {
|
|
@@ -71,27 +88,18 @@ describe("guardUnsupportedTokens — deterministic send-time repair", () => {
|
|
|
71
88
|
expect(guardUnsupportedTokens("x^2^ metres")).toBe("x2 metres");
|
|
72
89
|
});
|
|
73
90
|
|
|
74
|
-
it("
|
|
75
|
-
//
|
|
76
|
-
//
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
expect(guardUnsupportedTokens(
|
|
81
|
-
"as shown above",
|
|
82
|
-
);
|
|
83
|
-
expect(guardUnsupportedTokens("point[^fn1] made")).toBe("point made");
|
|
84
|
-
// A single-letter id is a footnote too.
|
|
85
|
-
expect(guardUnsupportedTokens("claim[^a] holds")).toBe("claim holds");
|
|
86
|
-
// Definition lines with an alphanumeric id are still left intact (`(?!:)`).
|
|
87
|
-
expect(guardUnsupportedTokens("[^note]: the definition")).toBe(
|
|
88
|
-
"[^note]: the definition",
|
|
89
|
-
);
|
|
91
|
+
it("never strips carets inside a $…$ math span (protected segment)", () => {
|
|
92
|
+
// `$x^2y^2$` contains an alphanumeric-only caret pair (`^2y^`) that the
|
|
93
|
+
// caret regex WOULD strip in bare prose — but the compact math span is a
|
|
94
|
+
// protected segment (code-segments.ts), so the wire bytes survive intact
|
|
95
|
+
// for Telegram to typeset as a mathematical_expression node.
|
|
96
|
+
const math = "inline $x^2y^2$ done";
|
|
97
|
+
expect(guardUnsupportedTokens(math)).toBe(math);
|
|
90
98
|
});
|
|
91
99
|
|
|
92
100
|
it("leaves regex-literal negated char classes intact", () => {
|
|
93
|
-
//
|
|
94
|
-
//
|
|
101
|
+
// The guard no longer touches `[^…]` at all (footnotes are supported),
|
|
102
|
+
// so every negated-char-class shape survives verbatim.
|
|
95
103
|
expect(guardUnsupportedTokens("use [^/] to match")).toBe(
|
|
96
104
|
"use [^/] to match",
|
|
97
105
|
);
|
|
@@ -99,23 +107,14 @@ describe("guardUnsupportedTokens — deterministic send-time repair", () => {
|
|
|
99
107
|
"strip [^a-z] chars",
|
|
100
108
|
);
|
|
101
109
|
expect(guardUnsupportedTokens('match [^"] here')).toBe('match [^"] here');
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
// which splitProtectedSegments masks (see code-span test below).
|
|
106
|
-
expect(guardUnsupportedTokens("array[^index] lookup")).toBe("array lookup");
|
|
107
|
-
// …but inside a code span the same literal is untouched.
|
|
110
|
+
expect(guardUnsupportedTokens("array[^index] lookup")).toBe(
|
|
111
|
+
"array[^index] lookup",
|
|
112
|
+
);
|
|
108
113
|
expect(guardUnsupportedTokens("`array[^index]` lookup")).toBe(
|
|
109
114
|
"`array[^index]` lookup",
|
|
110
115
|
);
|
|
111
116
|
});
|
|
112
117
|
|
|
113
|
-
it("repairs a real numeric footnote marker", () => {
|
|
114
|
-
expect(guardUnsupportedTokens("see the note[^1] here")).toBe(
|
|
115
|
-
"see the note here",
|
|
116
|
-
);
|
|
117
|
-
});
|
|
118
|
-
|
|
119
118
|
it("is a strict no-op for clean markdown (no target tokens)", () => {
|
|
120
119
|
const clean =
|
|
121
120
|
"**Answer:** the `config.yaml` file. See [docs](https://example.com/x).";
|
|
@@ -139,18 +138,19 @@ describe("guardUnsupportedTokens — deterministic send-time repair", () => {
|
|
|
139
138
|
expect(guardUnsupportedTokens(once)).toBe(once);
|
|
140
139
|
});
|
|
141
140
|
|
|
142
|
-
it("OUTCOME: the composed richMessage wire body
|
|
143
|
-
// End-to-end through the real send-path composition
|
|
144
|
-
// anti-tautology anchor:
|
|
145
|
-
//
|
|
146
|
-
//
|
|
141
|
+
it("OUTCOME: the composed richMessage wire body preserves supported constructs and repairs carets", () => {
|
|
142
|
+
// End-to-end through the real send-path composition (the FULL
|
|
143
|
+
// guardAccidentalFormatting pipeline). This is the anti-tautology anchor:
|
|
144
|
+
// on the pre-fix pipeline, `<details>` was folded into the unsupported
|
|
145
|
+
// `**>` marker and `[^1]` was deleted — every assertion below would FAIL.
|
|
147
146
|
const { markdown } = richMessage(
|
|
148
|
-
"note[^1]\n<details><summary>More</summary>\ndetail line\n</details>\nx^2^",
|
|
147
|
+
"note[^1]\n<details><summary>More</summary>\ndetail line\n</details>\nx^2^\n\n[^1]: the note",
|
|
149
148
|
);
|
|
150
|
-
expect(markdown).
|
|
151
|
-
expect(markdown).
|
|
152
|
-
expect(markdown).toContain("
|
|
153
|
-
expect(markdown).toContain("
|
|
154
|
-
expect(markdown).toContain("
|
|
149
|
+
expect(markdown).toContain("<details><summary>More</summary>");
|
|
150
|
+
expect(markdown).toContain("</details>");
|
|
151
|
+
expect(markdown).toContain("note[^1]");
|
|
152
|
+
expect(markdown).toContain("[^1]: the note");
|
|
153
|
+
expect(markdown).not.toContain("**>");
|
|
154
|
+
expect(markdown).toContain("x2"); // caret pair repaired
|
|
155
155
|
});
|
|
156
156
|
});
|