switchroom 0.21.8 → 0.21.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/cli/switchroom.js +100 -39
  2. package/dist/host-control/main.js +1 -1
  3. package/package.json +2 -2
  4. package/skills/switchroom-architecture/telegram.md +12 -10
  5. package/skills/switchroom-cli/SKILL.md +1 -1
  6. package/telegram-plugin/README.md +3 -1
  7. package/telegram-plugin/dist/gateway/gateway.js +919 -348
  8. package/telegram-plugin/format.ts +30 -5
  9. package/telegram-plugin/gateway/outbound-send-path.ts +7 -0
  10. package/telegram-plugin/gateway/speech-capture.ts +158 -0
  11. package/telegram-plugin/package.json +1 -1
  12. package/telegram-plugin/render/code-segments.ts +38 -4
  13. package/telegram-plugin/render/dollar-math-guard.ts +16 -1
  14. package/telegram-plugin/render/html-fold.ts +354 -0
  15. package/telegram-plugin/render/ir.ts +53 -3
  16. package/telegram-plugin/render/parse.ts +642 -42
  17. package/telegram-plugin/render/render.ts +53 -15
  18. package/telegram-plugin/render/unsupported-token-guard.ts +45 -80
  19. package/telegram-plugin/rich-send.ts +22 -7
  20. package/telegram-plugin/shared/bot-runtime.ts +3 -2
  21. package/telegram-plugin/telegraph.ts +6 -4
  22. package/telegram-plugin/tests/grammy-rich-message-types.test.ts +199 -0
  23. package/telegram-plugin/tests/render/dollar-math-guard.test.ts +43 -0
  24. package/telegram-plugin/tests/render/guard-composition.test.ts +102 -0
  25. package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +253 -0
  26. package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
  27. package/telegram-plugin/tests/render/parse.test.ts +39 -10
  28. package/telegram-plugin/tests/render/render.test.ts +9 -4
  29. package/telegram-plugin/tests/render/rich-render.test.ts +46 -5
  30. package/telegram-plugin/tests/render/tg-entity.test.ts +242 -0
  31. package/telegram-plugin/tests/render/unsupported-token-guard.test.ts +66 -66
  32. package/telegram-plugin/tests/send-reply-golden.test.ts +99 -1
  33. package/telegram-plugin/tests/sent-text-capture.test.ts +3 -3
  34. package/telegram-plugin/tests/speech-capture.test.ts +296 -0
  35. package/telegram-plugin/tests/telegraph.test.ts +1 -1
  36. package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
  37. package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
  38. package/telegram-plugin/tts-normalize.ts +47 -9
  39. package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +17 -8
  40. package/telegram-plugin/voice-normalize-text.ts +48 -9
@@ -218,17 +218,21 @@ describe("parse: nested inline", () => {
218
218
  });
219
219
 
220
220
  describe("parse: unsupported construct fallback", () => {
221
- it("degrades a raw HTML block to plain text without dropping content", () => {
222
- // Raw HTML blocks (mdast `html`) are outside the palette; they must
223
- // survive as plain source text.
221
+ it("degrades an UNRECOGNISED raw HTML block to its content, markup dropped", () => {
222
+ // Raw HTML blocks (mdast `html`) are outside the palette. An unrecognised
223
+ // tag is NOT passed through verbatim: nothing establishes that Telegram's
224
+ // rich parser accepts `<div>`, and `isParseEntitiesError` (rich-send.ts)
225
+ // shows the wire can 400 with `unsupported start tag` — whose fallback
226
+ // resends the body as PLAIN TEXT, putting literal `<div class="x">` on the
227
+ // reader's screen. Degrade by type: markup dropped, content kept.
224
228
  const md = "<div class=\"x\">raw</div>";
225
229
  const doc = parse(md);
226
230
  const block = doc.blocks[0] as any;
227
231
  expect(block.type).toBe("paragraph");
228
232
  const plain = block.children[0];
229
233
  expect(plain.type).toBe("plain");
230
- expect(plain.text).toBe(md);
231
- expect(md.slice(plain.start, plain.end)).toBe(md);
234
+ expect(plain.text).toBe("raw");
235
+ expect(md.slice(plain.start, plain.end)).toBe("raw");
232
236
  });
233
237
  });
234
238
 
@@ -336,7 +340,7 @@ describe("parse: nested lists", () => {
336
340
  });
337
341
  });
338
342
 
339
- describe("expandable blockquote (Bot API 10.1 `**>` marker)", () => {
343
+ describe("expandable blockquote (LEGACY switchroom `**>` marker — input repair only)", () => {
340
344
  it("parses a single-line `**>` quote into an expandable blockquote", () => {
341
345
  const md = "**> a collapsible line";
342
346
  const doc = parse(md);
@@ -374,10 +378,10 @@ describe("expandable blockquote (Bot API 10.1 `**>` marker)", () => {
374
378
  });
375
379
 
376
380
  it("only recognises the marker at column 0 (leading indent is not expandable)", () => {
377
- // The renderer only ever emits `**>` at column 0; a leading-indented
378
- // variant is deliberately NOT treated as an expandable quote (matching it
379
- // would push the length-preserving rewrite past the 3-space blockquote
380
- // budget into indented-code-block territory).
381
+ // The legacy encoding only ever placed `**>` at column 0; a
382
+ // leading-indented variant is deliberately NOT treated as an expandable
383
+ // quote (matching it would push the length-preserving rewrite past the
384
+ // 3-space blockquote budget into indented-code-block territory).
381
385
  const md = " **> indented";
382
386
  const doc = parse(md);
383
387
  const bq = doc.blocks[0] as any;
@@ -391,3 +395,28 @@ describe("expandable blockquote (Bot API 10.1 `**>` marker)", () => {
391
395
  expect(doc.blocks[0].type).toBe("paragraph");
392
396
  });
393
397
  });
398
+
399
+ describe("footnotes fold to verbatim `raw` nodes (native Telegram construct)", () => {
400
+ it("a footnote reference marker folds to a raw inline carrying its source bytes", () => {
401
+ // GFM only recognises a reference when a matching definition exists in
402
+ // the document (a lone `[^n1]` is ordinary text).
403
+ const doc = parse("claim[^n1] more\n\n[^n1]: body");
404
+ const para = doc.blocks[0] as Extract<Block, { type: "paragraph" }>;
405
+ const raw = para.children.find((c: Inline) => c.type === "raw") as
406
+ | Extract<Inline, { type: "raw" }>
407
+ | undefined;
408
+ expect(raw).toBeDefined();
409
+ expect(raw!.text).toBe("[^n1]");
410
+ });
411
+
412
+ it("a footnote definition folds to a paragraph with one raw inline (verbatim slice)", () => {
413
+ const doc = parse("[^n1]: body text");
414
+ const para = doc.blocks[0] as Extract<Block, { type: "paragraph" }>;
415
+ expect(para.type).toBe("paragraph");
416
+ expect(para.children).toHaveLength(1);
417
+ expect(para.children[0].type).toBe("raw");
418
+ expect((para.children[0] as Extract<Inline, { type: "raw" }>).text).toBe(
419
+ "[^n1]: body text",
420
+ );
421
+ });
422
+ });
@@ -195,7 +195,10 @@ describe("render: block palette", () => {
195
195
  it("plain blockquote", () => {
196
196
  expect(render(parse("> quoted line"))).toBe("> quoted line");
197
197
  });
198
- it("expandable blockquote uses **> on the first line", () => {
198
+ it("expandable IR flag renders as a PLAIN quote — the retired `**>` marker is never emitted", () => {
199
+ // `**>` is MarkdownV2-only syntax; the rich markdown path renders it as
200
+ // LITERAL `**>` text (wire-proved 2026-08-13 via raw sendRichMessage
201
+ // probes), so the renderer degrades an expandable node to a plain quote.
199
202
  const doc: Document = {
200
203
  blocks: [
201
204
  {
@@ -214,9 +217,10 @@ describe("render: block palette", () => {
214
217
  },
215
218
  ],
216
219
  };
217
- expect(render(doc)).toBe("**> hidden gem");
220
+ expect(render(doc)).toBe("> hidden gem");
221
+ expect(render(doc)).not.toContain("**>");
218
222
  });
219
- it("multi-line expandable blockquote continues with a plain > marker", () => {
223
+ it("multi-line expandable blockquote renders every line with a plain > marker", () => {
220
224
  const doc: Document = {
221
225
  blocks: [
222
226
  {
@@ -242,8 +246,9 @@ describe("render: block palette", () => {
242
246
  ],
243
247
  };
244
248
  const out = render(doc);
249
+ expect(out).not.toContain("**>");
245
250
  const lines = out.split("\n");
246
- expect(lines[0]).toBe("**> line one");
251
+ expect(lines[0]).toBe("> line one");
247
252
  for (const line of lines.slice(1)) {
248
253
  expect(line.startsWith("> ") || line === ">").toBe(true);
249
254
  }
@@ -5,6 +5,7 @@ import {
5
5
  renderOutbound,
6
6
  maybeRenderOutbound,
7
7
  } from "../../render/rich-render.js";
8
+ import { guardAccidentalFormatting } from "../../rich-send.js";
8
9
 
9
10
  describe("parseRichRenderEnabled", () => {
10
11
  it("defaults ON when unset (escape hatch, not opt-in)", () => {
@@ -52,8 +53,10 @@ describe("maybeRenderOutbound", () => {
52
53
  it("default (env unset) routes through parse -> renderSafe", () => {
53
54
  const r = maybeRenderOutbound("**> collapsible", {} as NodeJS.ProcessEnv);
54
55
  expect(r.mode).toBe("markdown");
55
- // The expandable blockquote round-trips back to the `**> ` marker.
56
- expect(r.text).toContain("**> ");
56
+ // The legacy expandable marker is REPAIRED to a plain quote: `**>` is
57
+ // MarkdownV2-only syntax the rich path renders as literal text
58
+ // (wire-proved 2026-08-13), so the renderer must never re-emit it.
59
+ expect(r.text).toBe("> collapsible");
57
60
  });
58
61
 
59
62
  it("default preserves plain prose through the round-trip", () => {
@@ -84,15 +87,53 @@ describe("maybeRenderOutbound", () => {
84
87
  SWITCHROOM_RICH_RENDER: "maybe",
85
88
  } as NodeJS.ProcessEnv);
86
89
  expect(r.mode).toBe("markdown");
87
- expect(r.text).toContain("**> ");
90
+ expect(r.text).toBe("> collapsible");
88
91
  });
89
92
  });
90
93
 
91
94
  describe("renderOutbound (flag-independent)", () => {
92
- it("renders an expandable blockquote round-trip end to end", () => {
95
+ it("repairs a legacy `**>` quote to a plain quote end to end", () => {
93
96
  const r = renderOutbound("**> hidden line one\n> hidden line two");
94
97
  expect(r.mode).toBe("markdown");
95
- expect(r.text.split("\n")[0]).toMatch(/^\*\*> /);
98
+ // One coherent quote, no retired marker: on the pre-fix renderer the
99
+ // first line came back as `**> hidden line one`, which Telegram rendered
100
+ // as LITERAL `**>` paragraph text (wire-proved 2026-08-13).
101
+ expect(r.text).not.toContain("**>");
102
+ expect(r.text.split("\n")[0]).toBe("> hidden line one");
103
+ expect(r.text).toContain("> hidden line two");
104
+ });
105
+
106
+ it("passes native constructs through: <details>, $math$, footnotes", () => {
107
+ // Wire-verified natives (2026-08-13) must survive parse -> renderSafe
108
+ // BYTE-IDENTICAL. On the pre-fix pipeline the footnote case failed:
109
+ // escapeMarkdown turned `[^n1]` into `\[^n1\]`, breaking the construct.
110
+ const details =
111
+ "<details open><summary>S</summary>\n\nbody\n\n</details>";
112
+ expect(renderOutbound(details).text).toBe(details);
113
+
114
+ const math = "inline $x^2+y^2$ done";
115
+ expect(renderOutbound(math).text).toBe(math);
116
+
117
+ const footnotes = "claim[^n1] more\n\n[^n1]: body text";
118
+ expect(renderOutbound(footnotes).text).toBe(footnotes);
119
+ });
120
+
121
+ it("a tg:// inline entity and a footnote survive TOGETHER in one message", () => {
122
+ // Cross-feature composition (#4683 x #4685): the `tg-entity` fold (mdast
123
+ // image position) and the footnote `raw` fold (footnoteReference /
124
+ // footnoteDefinition) land in the SAME foldInline/foldBlock walk — this
125
+ // pins that neither eats the other. On #4683 alone the footnote came back
126
+ // as `claim\[^n1\]`; on #4685 alone (pre-rebase) the entity came back as
127
+ // `!\[now\](tg://time?unix\=…)` literal text.
128
+ const combined =
129
+ "meet at ![now](tg://time?unix=1755000000&format=wDT) as promised[^n1]\n\n[^n1]: agreed yesterday";
130
+ expect(renderOutbound(combined).text).toBe(combined);
131
+
132
+ // Same pair inside ONE paragraph plus the guard seam on top: the composed
133
+ // wire body (renderOutbound then the richMessage guard, i.e. the full
134
+ // streamed send path) is still byte-identical.
135
+ const guarded = guardAccidentalFormatting(renderOutbound(combined).text);
136
+ expect(guarded).toBe(combined);
96
137
  });
97
138
 
98
139
  it("falls back to plain mode for oversized atomic content", () => {
@@ -0,0 +1,242 @@
1
+ // Telegram inline `tg:` entities in mdast IMAGE position survive the whole
2
+ // outbound pipeline.
3
+ //
4
+ // Bot API grammar (https://core.telegram.org/bots/api, "Rich Markdown style"):
5
+ //
6
+ // ![](tg://emoji?id=5368324170671202286) custom emoji
7
+ // ![22:45 tomorrow](tg://time?unix=1647531900&format=wDT) date_time
8
+ //
9
+ // (the `date_time` MessageEntity landed in Bot API 9.5, 2026-03-01; the
10
+ // rich-message `RichTextDateTime` class in 10.1, 2026-06-11 — Bot API
11
+ // changelog.) grammy 1.44.0 speaks 10.1, so the wire has accepted this since
12
+ // #2669 — but the RENDER path escaped the brackets to literal text, so the
13
+ // syntax was dead on the streamed send path.
14
+ //
15
+ // Two DISTINCT outbound surfaces are pinned here, because they do not share a
16
+ // pipeline:
17
+ // - the STREAMED path (stream-controller.ts:272) → `renderOutboundChunks` →
18
+ // `parse → renderSafe`. This is the surface the escaping bug lived on.
19
+ // - the REPLY-TOOL path (outbound-send-path.ts) → `computeReplyChunks` →
20
+ // `richMessage()`, which NEVER runs the renderer — only the
21
+ // accidental-formatting guards. It was already correct; these tests pin it
22
+ // so a future guard can't silently start eating the syntax.
23
+
24
+ import { describe, it, expect } from "vitest";
25
+ import { parse } from "../../render/parse.js";
26
+ import { renderSafe, SUPPORTED_INLINE } from "../../render/render.js";
27
+ import { renderOutboundChunks } from "../../render/rich-render.js";
28
+ import { guardAccidentalFormatting, richMessage } from "../../rich-send.js";
29
+ import { computeReplyChunks } from "../../gateway/outbound-send-path.js";
30
+ import { splitMarkdownChunks, RICH_MESSAGE_MAX_CHARS } from "../../format.js";
31
+ import type { Inline } from "../../render/ir.js";
32
+
33
+ /** The doc's own date_time example, verbatim. */
34
+ const DATE_TIME = "![22:45 tomorrow](tg://time?unix=1647531900&format=wDT)";
35
+ /** The doc's own custom-emoji example, verbatim (empty label). */
36
+ const EMOJI = "![](tg://emoji?id=5368324170671202286)";
37
+
38
+ /** parse → renderSafe, the streamed path's core. */
39
+ const roundTrip = (md: string): string => renderSafe(parse(md), md).text;
40
+
41
+ /** Flatten every inline node in the rendered IR of `md`. */
42
+ function inlines(md: string): Inline[] {
43
+ const out: Inline[] = [];
44
+ const walk = (n: Inline): void => {
45
+ out.push(n);
46
+ for (const c of (n as { children?: Inline[] }).children ?? []) walk(c);
47
+ };
48
+ for (const b of parse(md).blocks) {
49
+ for (const n of (b as { children?: Inline[] }).children ?? []) walk(n);
50
+ }
51
+ return out;
52
+ }
53
+
54
+ describe("tg:// inline entities: parse → renderSafe round-trip", () => {
55
+ // THE bug this PR fixes: at HEAD the image node demoted to `plain` and
56
+ // `escapeMarkdown` backslash-escaped `[`, `]` and `=`, shipping
57
+ // `!\[22:45 tomorrow\](tg://time?unix\=…&format\=wDT)` — literal text, no
58
+ // entity. Byte-exact preservation is the whole contract.
59
+ it("preserves a date_time entity byte-exact", () => {
60
+ const body = `Meeting at ${DATE_TIME} sharp.`;
61
+ expect(roundTrip(body)).toBe(body);
62
+ });
63
+
64
+ it("preserves the empty-label custom-emoji entity byte-exact", () => {
65
+ expect(roundTrip(`hi ${EMOJI} there`)).toBe(`hi ${EMOJI} there`);
66
+ });
67
+
68
+ it("preserves a date_time entity with no `format` parameter", () => {
69
+ const body = "![22:45 tomorrow](tg://time?unix=1647531900)";
70
+ expect(roundTrip(body)).toBe(body);
71
+ });
72
+
73
+ it("folds the entity into a `tg-entity` IR node, not `plain`", () => {
74
+ const node = inlines(`x ${DATE_TIME}`).find((n) => n.type === "tg-entity");
75
+ expect(node).toMatchObject({
76
+ type: "tg-entity",
77
+ label: "22:45 tomorrow",
78
+ href: "tg://time?unix=1647531900&format=wDT",
79
+ });
80
+ });
81
+
82
+ it("survives inside a blockquote, a heading, a list item and a table cell", () => {
83
+ for (const body of [
84
+ `> when: ${DATE_TIME}`,
85
+ `## Due ${DATE_TIME}`,
86
+ `- due ${DATE_TIME}`,
87
+ `| when | who |\n| --- | --- |\n| ${DATE_TIME} | me |`,
88
+ ]) {
89
+ expect(roundTrip(body)).toContain(DATE_TIME);
90
+ }
91
+ });
92
+
93
+ it("matches the `tg:` href case-insensitively (URL schemes are)", () => {
94
+ const body = "![t](TG://Time?unix=1647531900)";
95
+ // Folded as an entity, and the author's ORIGINAL bytes are re-emitted —
96
+ // the parser never rewrites the href.
97
+ expect(inlines(body).some((n) => n.type === "tg-entity")).toBe(true);
98
+ expect(roundTrip(body)).toBe(body);
99
+ });
100
+
101
+ it("lists `tg-entity` in the renderer's construct allowlist", () => {
102
+ expect(SUPPORTED_INLINE).toContain("tg-entity");
103
+ });
104
+ });
105
+
106
+ describe("tg:// inline entities: label escaping (no bracket breakout)", () => {
107
+ // A model-authored label carrying `]` would close the label early and
108
+ // smuggle raw bracket syntax past the renderer if the label were re-emitted
109
+ // undecorated. It is prose, so it gets `escapeMarkdown` exactly like a
110
+ // `plain` node.
111
+ it("escapes brackets in the label instead of letting them close it early", () => {
112
+ const out = roundTrip("![see [22:45]](tg://time?unix=1&format=t)");
113
+ expect(out).toBe("![see \\[22:45\\]](tg://time?unix=1&format=t)");
114
+ // No BARE `]` before the destination — the only unescaped one closes the label.
115
+ expect(out.slice(0, out.indexOf("](")).includes("\\]")).toBe(true);
116
+ });
117
+
118
+ it("escapes emphasis delimiters in the label", () => {
119
+ expect(roundTrip("![a*b_c~d](tg://time?unix=1&format=t)")).toBe(
120
+ "![a\\*b\\_c\\~d](tg://time?unix=1&format=t)",
121
+ );
122
+ });
123
+
124
+ it("collapses a soft line break in the label so the construct stays on one line", () => {
125
+ const out = roundTrip("![22:45\ntomorrow](tg://time?unix=1&format=t)");
126
+ expect(out).toBe("![22:45 tomorrow](tg://time?unix=1&format=t)");
127
+ expect(out).not.toContain("\n");
128
+ });
129
+ });
130
+
131
+ describe("tg:// inline entities: no regression of the image demotion", () => {
132
+ // Deliberate, unchanged: an http(s) `![](…)` is a Telegram MEDIA block —
133
+ // "Media can be specified only as a separate block" (Rich Markdown style) —
134
+ // not an inline entity, and switchroom emits no media blocks. It keeps
135
+ // degrading to escaped literal text.
136
+ it("still degrades an http(s) image link to escaped plain text", () => {
137
+ expect(roundTrip("![alt](https://x.example/y.png)")).toBe(
138
+ "!\\[alt\\](https://x.example/y.png)",
139
+ );
140
+ });
141
+
142
+ it("still degrades an UNDOCUMENTED tg:// image href to escaped plain text", () => {
143
+ // The allowlist is deliberate: only `tg://time` and `tg://emoji` are
144
+ // documented in image position, so anything else stays literal rather than
145
+ // shipping syntax Telegram may parse-reject.
146
+ expect(roundTrip("![t](tg://photo?id=1)")).toBe("!\\[t\\](tg://photo?id\\=1)");
147
+ expect(roundTrip("![t](tg://user?id=1)")).toBe("!\\[t\\](tg://user?id\\=1)");
148
+ });
149
+
150
+ it("leaves an ordinary `[label](tg://user?id=…)` mention link alone", () => {
151
+ const body = "ping [Ken](tg://user?id=123456789)";
152
+ expect(roundTrip(body)).toBe(body);
153
+ });
154
+ });
155
+
156
+ describe("tg:// inline entities: code context stays literal", () => {
157
+ it("keeps the syntax verbatim inside an inline code span", () => {
158
+ const body = "use `![22:45](tg://time?unix=1&format=t)` for that";
159
+ expect(roundTrip(body)).toBe(body);
160
+ // and it is a `code` node, NOT a folded entity
161
+ expect(inlines(body).some((n) => n.type === "tg-entity")).toBe(false);
162
+ });
163
+
164
+ it("keeps the syntax verbatim inside a fenced code block", () => {
165
+ const body = "```\n![22:45](tg://time?unix=1&format=t)\n```";
166
+ expect(roundTrip(body)).toBe(body);
167
+ expect(inlines(body).some((n) => n.type === "tg-entity")).toBe(false);
168
+ });
169
+ });
170
+
171
+ describe("tg:// inline entities: the accidental-formatting guards are a no-op", () => {
172
+ // `guardAccidentalFormatting` is the universal seam (installed as a grammy
173
+ // API transformer in shared/bot-runtime.ts), so it runs over EVERY outbound
174
+ // body — rendered or hand-built card alike. Its documented trigger set
175
+ // (`_ * > . ~ == || $ #`) does not intersect this syntax, but nothing
176
+ // asserted that until now.
177
+ const bodies = [
178
+ `Meeting at ${DATE_TIME} sharp.`,
179
+ `${DATE_TIME} at the very start of the line`,
180
+ EMOJI,
181
+ `- due ${DATE_TIME}\n- and ${DATE_TIME}`,
182
+ `> when: ${DATE_TIME}`,
183
+ `| when |\n| --- |\n| ${DATE_TIME} |`,
184
+ "![see \\[22:45\\]](tg://time?unix=1&format=t)",
185
+ ];
186
+
187
+ it("leaves every tg:// entity body byte-identical", () => {
188
+ for (const body of bodies) expect(guardAccidentalFormatting(body)).toBe(body);
189
+ });
190
+
191
+ it("is idempotent over the rendered form too", () => {
192
+ for (const body of bodies) {
193
+ const rendered = roundTrip(body);
194
+ expect(guardAccidentalFormatting(rendered)).toBe(rendered);
195
+ }
196
+ });
197
+ });
198
+
199
+ describe("tg:// inline entities: live outbound seams", () => {
200
+ it("survives the STREAMED seam (renderOutboundChunks)", () => {
201
+ const body = `Standup starts ${DATE_TIME}.`;
202
+ const pieces = renderOutboundChunks(body);
203
+ expect(pieces).toHaveLength(1);
204
+ expect(pieces[0].mode).toBe("markdown");
205
+ expect(pieces[0].text).toBe(body);
206
+ });
207
+
208
+ it("survives the REPLY-TOOL seam (computeReplyChunks → richMessage)", () => {
209
+ // This surface never touches the renderer — it is guards-only. Pinned so a
210
+ // future guard cannot start escaping the syntax unnoticed.
211
+ const body = `Standup starts ${DATE_TIME}.`;
212
+ const chunks = computeReplyChunks({
213
+ effectiveText: body,
214
+ literalText: false,
215
+ limit: RICH_MESSAGE_MAX_CHARS,
216
+ chunkMode: "length",
217
+ });
218
+ expect(chunks.map((c) => richMessage(c).markdown).join("")).toContain(DATE_TIME);
219
+ });
220
+ });
221
+
222
+ describe("tg:// inline entities: chunk boundaries never bisect the construct", () => {
223
+ // `INLINE_SPAN_PATTERNS`' link pattern used to start at the `[`, leaving the
224
+ // leading `!` outside the protected span: a cut inside the construct
225
+ // retreated only to the `[`, stranding `!` on the previous chunk and
226
+ // silently demoting a date_time entity to an ordinary link.
227
+ it("keeps a cap-straddling date_time entity whole in one chunk", () => {
228
+ const body = `${"x".repeat(80)} ${DATE_TIME}`;
229
+ const chunks = splitMarkdownChunks(body, 110);
230
+ expect(chunks.length).toBeGreaterThan(1);
231
+ // Exactly one chunk carries the entity, whole.
232
+ expect(chunks.filter((c) => c.includes(DATE_TIME))).toHaveLength(1);
233
+ // And no chunk ends on the stranded `!`.
234
+ for (const c of chunks) expect(c.endsWith("!")).toBe(false);
235
+ });
236
+
237
+ it("still keeps an ordinary link whole (unchanged behaviour)", () => {
238
+ const link = "[a fairly long link label here](https://example.com/some/path)";
239
+ const chunks = splitMarkdownChunks(`${"x".repeat(80)} ${link}`, 110);
240
+ expect(chunks.filter((c) => c.includes(link))).toHaveLength(1);
241
+ });
242
+ });
@@ -3,44 +3,61 @@ import { guardUnsupportedTokens } from "../../render/unsupported-token-guard.js"
3
3
  import { richMessage } from "../../rich-send.js";
4
4
 
5
5
  describe("guardUnsupportedTokens — deterministic send-time repair", () => {
6
- it("folds <details><summary> into a Telegram expandable blockquote", () => {
6
+ // ── Natively-supported constructs pass through BYTE-IDENTICAL ────────────
7
+ // Wire-verified 2026-08-13: raw sendRichMessage probes showed `<details>`,
8
+ // footnotes, `<sub>`/`<sup>`/`<u>`, `<aside>`, `tg://time` and task lists
9
+ // all parse into real typed nodes on Telegram's rich markdown path. An
10
+ // earlier revision of this guard "repaired" `<details>` into a `**> `
11
+ // expandable blockquote (MarkdownV2-only syntax that renders as LITERAL
12
+ // `**>` text on this path) and deleted footnote markers — both conversions
13
+ // destroyed supported constructs and are deleted. These tests would FAIL on
14
+ // the old guard.
15
+ it("passes a <details><summary> block through untouched (native construct)", () => {
7
16
  const input =
8
- "Here is the trace:\n<details><summary>Stack trace</summary>\nline 1\nline 2\n</details>\ndone";
9
- const out = guardUnsupportedTokens(input);
10
- // Anti-tautology: the raw HTML tags MUST be gone from the wire body.
11
- expect(out).not.toContain("<details>");
12
- expect(out).not.toContain("</details>");
13
- expect(out).not.toContain("<summary>");
14
- // Summary becomes the expandable-blockquote first line (`**> ` marker).
15
- expect(out).toContain("**> Stack trace");
16
- // Body lines become plain `> ` continuation lines.
17
- expect(out).toContain("> line 1");
18
- expect(out).toContain("> line 2");
17
+ "Here is the trace:\n<details open><summary>Stack trace</summary>\n\nline 1\nline 2\n\n</details>\ndone";
18
+ expect(guardUnsupportedTokens(input)).toBe(input);
19
19
  });
20
20
 
21
- it("folds a <details> without a <summary> into an expandable blockquote", () => {
22
- const out = guardUnsupportedTokens("<details>hidden body text</details>");
23
- expect(out).not.toContain("<details");
24
- expect(out).toContain("**> hidden body text");
21
+ it("passes a <details> without a <summary> through untouched", () => {
22
+ const input = "<details>hidden body text</details>";
23
+ expect(guardUnsupportedTokens(input)).toBe(input);
25
24
  });
26
25
 
27
- it("strips caret highlight / superscript pairs to their inner text", () => {
28
- expect(guardUnsupportedTokens("energy is x^2^ joules")).toBe(
29
- "energy is x2 joules",
30
- );
31
- expect(guardUnsupportedTokens("a ^highlighted^ word")).toBe(
32
- "a highlighted word",
26
+ it("never converts <details> into the retired `**>` marker", () => {
27
+ const out = guardUnsupportedTokens(
28
+ "<details><summary>More</summary>\nbody\n</details>",
33
29
  );
30
+ expect(out).not.toContain("**>");
31
+ expect(out).toContain("<details>");
34
32
  });
35
33
 
36
- it("removes footnote reference markers but keeps definition lines", () => {
34
+ it("keeps footnote reference markers AND definition lines (native construct)", () => {
37
35
  expect(guardUnsupportedTokens("see the note[^1] here")).toBe(
38
- "see the note here",
36
+ "see the note[^1] here",
39
37
  );
40
- // A `[^1]:` definition line is left intact (the negative lookahead).
41
38
  expect(guardUnsupportedTokens("[^1]: the definition")).toBe(
42
39
  "[^1]: the definition",
43
40
  );
41
+ // Alphanumeric ids too — the pair renders as real footnote machinery.
42
+ expect(guardUnsupportedTokens("claim[^note] holds\n\n[^note]: body")).toBe(
43
+ "claim[^note] holds\n\n[^note]: body",
44
+ );
45
+ });
46
+
47
+ it("passes <sub>/<sup>/<u>/<aside> HTML tags through untouched", () => {
48
+ const input =
49
+ "H<sub>2</sub>O and x<sup>2</sup> and <u>under</u>\n<aside>pull\n<cite>credit</cite></aside>";
50
+ expect(guardUnsupportedTokens(input)).toBe(input);
51
+ });
52
+
53
+ // ── The one genuinely-unsupported token: caret pairs ─────────────────────
54
+ it("strips caret highlight / superscript pairs to their inner text", () => {
55
+ expect(guardUnsupportedTokens("energy is x^2^ joules")).toBe(
56
+ "energy is x2 joules",
57
+ );
58
+ expect(guardUnsupportedTokens("a ^highlighted^ word")).toBe(
59
+ "a highlighted word",
60
+ );
44
61
  });
45
62
 
46
63
  it("leaves unpaired carets scattered across prose intact (no interior space)", () => {
@@ -71,27 +88,18 @@ describe("guardUnsupportedTokens — deterministic send-time repair", () => {
71
88
  expect(guardUnsupportedTokens("x^2^ metres")).toBe("x2 metres");
72
89
  });
73
90
 
74
- it("repairs alphanumeric footnote markers, not just numeric", () => {
75
- // POLISH-4: widened from digits-only so `[^note]`/`[^ref]` no longer ship
76
- // as literal bracket noise. Each would survive UNTOUCHED on origin/main.
77
- expect(guardUnsupportedTokens("see the note[^note] here")).toBe(
78
- "see the note here",
79
- );
80
- expect(guardUnsupportedTokens("as shown[^ref] above")).toBe(
81
- "as shown above",
82
- );
83
- expect(guardUnsupportedTokens("point[^fn1] made")).toBe("point made");
84
- // A single-letter id is a footnote too.
85
- expect(guardUnsupportedTokens("claim[^a] holds")).toBe("claim holds");
86
- // Definition lines with an alphanumeric id are still left intact (`(?!:)`).
87
- expect(guardUnsupportedTokens("[^note]: the definition")).toBe(
88
- "[^note]: the definition",
89
- );
91
+ it("never strips carets inside a $…$ math span (protected segment)", () => {
92
+ // `$x^2y^2$` contains an alphanumeric-only caret pair (`^2y^`) that the
93
+ // caret regex WOULD strip in bare prose — but the compact math span is a
94
+ // protected segment (code-segments.ts), so the wire bytes survive intact
95
+ // for Telegram to typeset as a mathematical_expression node.
96
+ const math = "inline $x^2y^2$ done";
97
+ expect(guardUnsupportedTokens(math)).toBe(math);
90
98
  });
91
99
 
92
100
  it("leaves regex-literal negated char classes intact", () => {
93
- // Real negated char classes contain punctuation / ranges / escapes, so the
94
- // alphanumeric-only id requirement excludes them — these stay verbatim.
101
+ // The guard no longer touches `[^…]` at all (footnotes are supported),
102
+ // so every negated-char-class shape survives verbatim.
95
103
  expect(guardUnsupportedTokens("use [^/] to match")).toBe(
96
104
  "use [^/] to match",
97
105
  );
@@ -99,23 +107,14 @@ describe("guardUnsupportedTokens — deterministic send-time repair", () => {
99
107
  "strip [^a-z] chars",
100
108
  );
101
109
  expect(guardUnsupportedTokens('match [^"] here')).toBe('match [^"] here');
102
- // Accepted tradeoff (task POLISH-4): a PURE-alphanumeric negated class in
103
- // BARE prose like `array[^index]` is now treated as a footnote and stripped.
104
- // This is rare — regex/index literals in real prose live in code spans,
105
- // which splitProtectedSegments masks (see code-span test below).
106
- expect(guardUnsupportedTokens("array[^index] lookup")).toBe("array lookup");
107
- // …but inside a code span the same literal is untouched.
110
+ expect(guardUnsupportedTokens("array[^index] lookup")).toBe(
111
+ "array[^index] lookup",
112
+ );
108
113
  expect(guardUnsupportedTokens("`array[^index]` lookup")).toBe(
109
114
  "`array[^index]` lookup",
110
115
  );
111
116
  });
112
117
 
113
- it("repairs a real numeric footnote marker", () => {
114
- expect(guardUnsupportedTokens("see the note[^1] here")).toBe(
115
- "see the note here",
116
- );
117
- });
118
-
119
118
  it("is a strict no-op for clean markdown (no target tokens)", () => {
120
119
  const clean =
121
120
  "**Answer:** the `config.yaml` file. See [docs](https://example.com/x).";
@@ -139,18 +138,19 @@ describe("guardUnsupportedTokens — deterministic send-time repair", () => {
139
138
  expect(guardUnsupportedTokens(once)).toBe(once);
140
139
  });
141
140
 
142
- it("OUTCOME: the composed richMessage wire body has unsupported tokens repaired", () => {
143
- // End-to-end through the real send-path composition. This is the
144
- // anti-tautology anchor: without guardUnsupportedTokens wired into
145
- // guardAccidentalFormatting, the raw `<details>` / `^` / `[^1]` tokens would
146
- // reach the wire and this assertion would FAIL.
141
+ it("OUTCOME: the composed richMessage wire body preserves supported constructs and repairs carets", () => {
142
+ // End-to-end through the real send-path composition (the FULL
143
+ // guardAccidentalFormatting pipeline). This is the anti-tautology anchor:
144
+ // on the pre-fix pipeline, `<details>` was folded into the unsupported
145
+ // `**>` marker and `[^1]` was deleted — every assertion below would FAIL.
147
146
  const { markdown } = richMessage(
148
- "note[^1]\n<details><summary>More</summary>\ndetail line\n</details>\nx^2^",
147
+ "note[^1]\n<details><summary>More</summary>\ndetail line\n</details>\nx^2^\n\n[^1]: the note",
149
148
  );
150
- expect(markdown).not.toContain("<details>");
151
- expect(markdown).not.toContain("[^1]");
152
- expect(markdown).toContain("**> More");
153
- expect(markdown).toContain("> detail line");
154
- expect(markdown).toContain("x2");
149
+ expect(markdown).toContain("<details><summary>More</summary>");
150
+ expect(markdown).toContain("</details>");
151
+ expect(markdown).toContain("note[^1]");
152
+ expect(markdown).toContain("[^1]: the note");
153
+ expect(markdown).not.toContain("**>");
154
+ expect(markdown).toContain("x2"); // caret pair repaired
155
155
  });
156
156
  });