switchroom 0.21.9 → 0.21.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/dist/cli/switchroom.js +143 -27
  2. package/dist/host-control/main.js +1 -1
  3. package/package.json +3 -2
  4. package/telegram-plugin/dist/gateway/gateway.js +1152 -296
  5. package/telegram-plugin/format.ts +74 -1
  6. package/telegram-plugin/gateway/answer-route-overrides.ts +163 -0
  7. package/telegram-plugin/gateway/answer-thread-resolve.test.ts +175 -1
  8. package/telegram-plugin/gateway/answer-thread-resolve.ts +58 -6
  9. package/telegram-plugin/gateway/escalation-staleness.ts +526 -0
  10. package/telegram-plugin/gateway/gateway.ts +61 -68
  11. package/telegram-plugin/gateway/obligation-wiring.ts +91 -3
  12. package/telegram-plugin/gateway/outbound-send-path.ts +62 -1
  13. package/telegram-plugin/gateway/reply-route-log.test.ts +134 -0
  14. package/telegram-plugin/gateway/reply-route-log.ts +118 -0
  15. package/telegram-plugin/gateway/speech-capture.ts +158 -0
  16. package/telegram-plugin/gateway/stream-render.ts +1 -1
  17. package/telegram-plugin/history.ts +21 -0
  18. package/telegram-plugin/registry/subagents-bugs.test.ts +3 -3
  19. package/telegram-plugin/render/html-fold.ts +372 -0
  20. package/telegram-plugin/render/parse.ts +578 -29
  21. package/telegram-plugin/render/render.ts +14 -13
  22. package/telegram-plugin/tests/answer-route-side-effect.test.ts +111 -0
  23. package/telegram-plugin/tests/catch-all-forwarded-history.test.ts +3 -3
  24. package/telegram-plugin/tests/escalation-staleness.test.ts +1275 -0
  25. package/telegram-plugin/tests/forwarded-rich-message-coalesce.test.ts +6 -6
  26. package/telegram-plugin/tests/forwarded-rich-message.test.ts +8 -8
  27. package/telegram-plugin/tests/history.test.ts +78 -0
  28. package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +29 -1
  29. package/telegram-plugin/tests/orphaned-db-sweep.test.ts +17 -1
  30. package/telegram-plugin/tests/render/html-dialect-content-loss.test.ts +326 -0
  31. package/telegram-plugin/tests/render/html-dialect.test.ts +283 -0
  32. package/telegram-plugin/tests/render/parse.test.ts +9 -5
  33. package/telegram-plugin/tests/send-reply-golden.test.ts +290 -3
  34. package/telegram-plugin/tests/speech-capture.test.ts +296 -0
  35. package/telegram-plugin/tests/status-pin.test.ts +2 -2
  36. package/telegram-plugin/tests/subagent-handback-inbound-builder.test.ts +2 -2
  37. package/telegram-plugin/tests/subagent-progress-inbound-builder.test.ts +2 -2
  38. package/telegram-plugin/tests/telegram-format.test.ts +52 -0
  39. package/telegram-plugin/tests/tts-normalize.test.ts +114 -0
  40. package/telegram-plugin/tests/turn-supersede-finalizes-prior-card.test.ts +1 -1
  41. package/telegram-plugin/tests/voice-normalize-text.test.ts +89 -0
  42. package/telegram-plugin/tests/worker-origin-gap-dispatch.test.ts +1 -1
  43. package/telegram-plugin/tts-normalize.ts +47 -9
  44. package/telegram-plugin/uat/scenarios/jtbd-supergroup-reply-channel.test.ts +1 -1
  45. package/telegram-plugin/voice-normalize-text.ts +48 -9
  46. package/telegram-plugin/worker-activity-feed.ts +2 -2
@@ -12,7 +12,11 @@
12
12
  // `source.slice(node.start, node.end)` round-trips to the source text.
13
13
  // - Never lose text. Any mdast node type outside the supported palette
14
14
  // degrades to a `plain` inline (or a paragraph wrapping one) carrying the
15
- // raw source slice, rather than being dropped.
15
+ // raw source slice, rather than being dropped. The ONE deliberate
16
+ // exception is raw-HTML MARKUP: an unrecognised tag has its angle-bracket
17
+ // token dropped while its content is kept and rendered (see the raw-HTML
18
+ // note below and `html-fold.ts`). Content is never lost; unguaranteeable
19
+ // markup is.
16
20
  //
17
21
  // Underline vs bold (`__…__` vs `**…**`):
18
22
  // Telegram's Bot API 10.1 rich markdown reads a `__…__` double-underscore run
@@ -56,6 +60,25 @@
56
60
  // marker characters differ, and they are never part of quoted content. The
57
61
  // set of line-start offsets that carried the marker is threaded into
58
62
  // `foldBlock` so the matching blockquote nodes get `expandable: true`.
63
+ //
64
+ // The rewrite is FENCE-AWARE. Every other fold reads the ORIGINAL `markdown`
65
+ // through `slice()`, so the rewritten bytes never escape — except for the
66
+ // `code` fold, which must use mdast's `node.value` (the dedented, fence-
67
+ // stripped content, which offsets alone cannot reconstruct without
68
+ // re-implementing micromark's fence parser). That made a fenced block
69
+ // containing a line starting `**>` ship silently corrupted as ` >` —
70
+ // anyone documenting the legacy syntax got their code altered. Skipping
71
+ // fenced regions in the pre-pass fixes it at the one place the rewrite is
72
+ // decided, keeps `code.value` trustworthy for every consumer, and preserves
73
+ // the length invariant trivially (a skipped line is copied verbatim).
74
+ //
75
+ // Raw HTML handling:
76
+ // mdast `html` nodes (inline `<b>`/`<a href>` runs and whole HTML blocks)
77
+ // used to fall through to `plain`, which the renderer `escapeMarkdown`s —
78
+ // escaping `=` and shipping `<a href\="…">`. They now go through
79
+ // `html-fold.ts`'s three-bucket policy: fold to the native IR construct,
80
+ // pass through raw (wire-verified allowlist), or drop the markup and keep
81
+ // the content. See that module's header for the rationale.
59
82
 
60
83
  import { fromMarkdown } from "mdast-util-from-markdown";
61
84
  import { gfm } from "micromark-extension-gfm";
@@ -78,6 +101,20 @@ import type {
78
101
  TableCell,
79
102
  TableRow,
80
103
  } from "./ir.js";
104
+ import {
105
+ HTML_FOLD_TAGS,
106
+ escapeHtmlLiteral,
107
+ hrefOf,
108
+ isBlockLevelTag,
109
+ isPassthroughTag,
110
+ isVoidTag,
111
+ languageOf,
112
+ matchedTagOffsets,
113
+ tokenizeHtml,
114
+ type HtmlFoldTarget,
115
+ type HtmlTagInfo,
116
+ } from "./html-fold.js";
117
+ import { codeFenceFor } from "../format.js";
81
118
 
82
119
  /** Copy UTF-16 offsets off an mdast node. Falls back to 0-length when a
83
120
  * synthesized node lacks a position (from-markdown always sets one, but the
@@ -125,13 +162,42 @@ function isTgInlineEntityHref(href: string): boolean {
125
162
  return TG_INLINE_ENTITY_HREFS.some((base) => h === base || h.startsWith(`${base}?`));
126
163
  }
127
164
 
165
+ /** A top-level fenced-code delimiter: 3+ backticks or tildes, indented 0–3
166
+ * spaces. Deeper indentation is not a fence, and a fence nested inside a
167
+ * blockquote / list item is prefixed by its container's markup, so this only
168
+ * ever tracks TOP-LEVEL fences.
169
+ *
170
+ * KNOWN LIMITATION (pre-existing, not a safety property): tracking only
171
+ * top-level fences does NOT establish that every column-0 `**>` line outside
172
+ * one is a real blockquote candidate. Counterexample:
173
+ *
174
+ * > ```
175
+ * **> lazy
176
+ * > ```
177
+ *
178
+ * The `> ``` ` opener is inside a blockquote so it never matches here; the
179
+ * column-0 `**> lazy` line is rewritten to ` > lazy` and the `**` is eaten.
180
+ * CommonMark forbids lazy continuation into fenced code, so that line should
181
+ * CLOSE the blockquote and stay literal. Fixing it needs real container
182
+ * tracking (blockquote/list prefixes), not a wider fence regex. The case is
183
+ * pinned by a test in `tests/render/html-dialect.test.ts` so any future
184
+ * container-aware rewrite has to decide it deliberately. */
185
+ const FENCE_DELIM_RE = /^ {0,3}(`{3,}|~{3,})/;
186
+
128
187
  /** Pre-transform expandable-blockquote markers so mdast can parse them as
129
188
  * ordinary blockquotes, WITHOUT shifting any source offset. Each line that
130
- * opens with `**>` has its two `*` characters replaced by two spaces
131
- * (`**>` → ` >`), which micromark reads as a normal (optionally
132
- * 1–3-space-indented) blockquote line. Returns the rewritten text plus the
133
- * set of line-start offsets that carried the marker — `foldBlock` uses that
134
- * set to flip `expandable: true` on the produced blockquote nodes. */
189
+ * opens with `**>` — and is NOT inside a fenced code block — has its two `*`
190
+ * characters replaced by two spaces (`**>` → ` >`), which micromark reads as
191
+ * a normal (optionally 1–3-space-indented) blockquote line. Returns the
192
+ * rewritten text plus the set of line-start offsets that carried the marker —
193
+ * `foldBlock` uses that set to flip `expandable: true` on the produced
194
+ * blockquote nodes.
195
+ *
196
+ * Fence tracking mirrors CommonMark: an opener is 3+ backticks/tildes at
197
+ * 0–3 spaces of indent; the block closes on a delimiter of the SAME character
198
+ * that is at least as long and carries nothing but whitespace after it, or at
199
+ * end of document. Inside such a block every line is copied verbatim, so a
200
+ * documented `**> …` example survives byte-identical into `code.value`. */
135
201
  function markExpandableQuotes(markdown: string): {
136
202
  text: string;
137
203
  expandableLineStarts: Set<number>;
@@ -139,8 +205,38 @@ function markExpandableQuotes(markdown: string): {
139
205
  const expandableLineStarts = new Set<number>();
140
206
  let out = "";
141
207
  let offset = 0;
208
+ /** The open fence's delimiter run, or null outside a fenced block. */
209
+ let openFence: string | null = null;
142
210
  // Split keeping the trailing newline on each line so offsets are exact.
143
211
  for (const line of markdown.split(/(?<=\n)/)) {
212
+ const body = line.replace(/\r?\n$/, "");
213
+ const fence = FENCE_DELIM_RE.exec(body)?.[1] ?? null;
214
+ if (openFence !== null) {
215
+ // Inside a fence: copy verbatim, and close on a matching delimiter.
216
+ out += line;
217
+ if (
218
+ fence !== null &&
219
+ fence[0] === openFence[0] &&
220
+ fence.length >= openFence.length &&
221
+ body.slice(body.indexOf(fence) + fence.length).trim() === ""
222
+ ) {
223
+ openFence = null;
224
+ }
225
+ offset += line.length;
226
+ continue;
227
+ }
228
+ if (fence !== null) {
229
+ // A backtick info string may not contain a backtick; such a line is not
230
+ // a fence opener at all (CommonMark), so only treat it as one when the
231
+ // rest of the line is clean.
232
+ const rest = body.slice(body.indexOf(fence) + fence.length);
233
+ if (!(fence[0] === "`" && rest.includes("`"))) {
234
+ openFence = fence;
235
+ out += line;
236
+ offset += line.length;
237
+ continue;
238
+ }
239
+ }
144
240
  if (EXPANDABLE_MARKER_RE.test(line)) {
145
241
  expandableLineStarts.add(offset);
146
242
  // Replace the leading two `*` chars with two spaces; the `>` and
@@ -168,7 +264,7 @@ function foldAlign(a: AlignType | undefined): "left" | "center" | "right" | null
168
264
  return a ?? null;
169
265
  }
170
266
 
171
- function foldInline(node: PhrasingContent, source: string): Inline {
267
+ function foldInline(node: PhrasingContent, source: string, ctx: FoldCtx): Inline {
172
268
  switch (node.type) {
173
269
  case "text":
174
270
  return { type: "plain", text: node.value, ...pos(node) };
@@ -179,21 +275,21 @@ function foldInline(node: PhrasingContent, source: string): Inline {
179
275
  const underline = source.slice(p.start, p.start + 2) === "__";
180
276
  return {
181
277
  type: underline ? "underline" : "bold",
182
- children: foldInlineChildren(node, source),
278
+ children: foldInlineChildren(node, source, ctx),
183
279
  ...p,
184
280
  };
185
281
  }
186
282
  case "emphasis":
187
- return { type: "italic", children: foldInlineChildren(node, source), ...pos(node) };
283
+ return { type: "italic", children: foldInlineChildren(node, source, ctx), ...pos(node) };
188
284
  case "delete":
189
- return { type: "strike", children: foldInlineChildren(node, source), ...pos(node) };
285
+ return { type: "strike", children: foldInlineChildren(node, source, ctx), ...pos(node) };
190
286
  case "inlineCode":
191
287
  return { type: "code", text: node.value, ...pos(node) };
192
288
  case "link":
193
289
  return {
194
290
  type: "link",
195
291
  href: node.url,
196
- children: foldInlineChildren(node, source),
292
+ children: foldInlineChildren(node, source, ctx),
197
293
  ...pos(node),
198
294
  };
199
295
  case "image": {
@@ -290,42 +386,451 @@ function expandPlainNode(node: PlainNode, source: string): Inline[] {
290
386
  return out;
291
387
  }
292
388
 
389
+ // ---------------------------------------------------------------------------
390
+ // Raw-HTML tag folding (see html-fold.ts for the policy and its rationale)
391
+ // ---------------------------------------------------------------------------
392
+
393
+ /** One element of the stream the HTML matcher walks: either a classified HTML
394
+ * token or an already-folded IR inline. */
395
+ type HtmlFoldItem =
396
+ | { kind: "tag"; tag: HtmlTagInfo; start: number; end: number }
397
+ | { kind: "inline"; node: Inline };
398
+
399
+ /** Index of the token that closes `name`, honouring same-name nesting, or -1. */
400
+ function findClosingTag(items: HtmlFoldItem[], from: number, name: string): number {
401
+ let depth = 0;
402
+ for (let i = from; i < items.length; i++) {
403
+ const it = items[i];
404
+ if (it.kind !== "tag" || it.tag.name !== name) continue;
405
+ if (it.tag.kind === "open") depth++;
406
+ else if (it.tag.kind === "close") {
407
+ if (depth === 0) return i;
408
+ depth--;
409
+ }
410
+ }
411
+ return -1;
412
+ }
413
+
414
+ /** Flatten a folded run to literal text, or null when any child carries
415
+ * structure a code span cannot represent. Used for `<code>`. Deliberately
416
+ * accepts ONLY `plain`: an inner `code` node's own backticks are not in its
417
+ * `text`, so folding it in would silently DELETE them
418
+ * (`<code>a `b` c</code>` -> `` `a b c` ``), and a `raw` node's wire bytes
419
+ * would be swallowed into a literal span. Both degrade to the children
420
+ * instead, which preserves every byte. */
421
+ function inlineLiteralText(nodes: Inline[]): string | null {
422
+ let out = "";
423
+ for (const n of nodes) {
424
+ if (n.type !== "plain") return null;
425
+ out += n.text;
426
+ }
427
+ return out;
428
+ }
429
+
430
+ /** Where a folded run is being emitted. `noBreaks` marks a container whose
431
+ * content must stay on ONE line — a heading or a table cell — so a `<br>` or
432
+ * a structural separator has to degrade to a space instead of a line break,
433
+ * and a `<pre>` to an inline code span instead of a fence. Without this a
434
+ * `# a<br>b` heading emitted `# a \nb`, which ends the heading block and
435
+ * drops `b` out of it. */
436
+ interface FoldCtx {
437
+ noBreaks?: boolean;
438
+ /** Source offsets of every HTML token the DOCUMENT-level pass found a
439
+ * partner for (`matchedTagOffsets`). A token missing from this set is
440
+ * unmatched — prose, not markup. Empty means "no HTML in this document",
441
+ * in which case nothing consults it. */
442
+ matched: ReadonlySet<number>;
443
+ }
444
+
445
+ /** Separator nodes this module synthesized while degrading structural markup.
446
+ * Tracked by identity so `normalizeSeparators` can collapse and trim OUR
447
+ * separators without touching a hard break the author actually wrote with
448
+ * `<br>` (`a<br><br>b` must keep both). */
449
+ const structuralSeparators = new WeakSet<Inline>();
450
+
451
+ /** The whitespace a degraded block-level tag leaves behind. A GFM HARD break
452
+ * (two spaces + newline), not a bare `\n`: the renderer runs AFTER
453
+ * `normalizeParagraphBreaks`, so a lone `\n` would never be promoted and
454
+ * Telegram collapses it to a space. */
455
+ function mkSeparator(ctx: FoldCtx, start: number, end: number): Inline {
456
+ const node: Inline = {
457
+ type: "raw",
458
+ text: ctx.noBreaks === true ? " " : " \n",
459
+ start,
460
+ end,
461
+ };
462
+ structuralSeparators.add(node);
463
+ return node;
464
+ }
465
+
466
+ /** Drop leading/trailing synthesized separators and collapse runs of them, so
467
+ * a degrade never contributes stray whitespace at the edges of a block
468
+ * (`<div></div>` must still render to nothing, and a `<pre>` fence must not
469
+ * leave a dangling newline). */
470
+ function normalizeSeparators(nodes: Inline[]): Inline[] {
471
+ const out: Inline[] = [];
472
+ for (const node of nodes) {
473
+ if (!structuralSeparators.has(node)) {
474
+ out.push(node);
475
+ continue;
476
+ }
477
+ if (out.length === 0) continue;
478
+ if (structuralSeparators.has(out[out.length - 1])) continue;
479
+ out.push(node);
480
+ }
481
+ while (out.length > 0 && structuralSeparators.has(out[out.length - 1])) out.pop();
482
+ return out;
483
+ }
484
+
485
+ /** An unmatched marker or unclassifiable token kept as LITERAL text, with its
486
+ * angle brackets HTML-entity-escaped so the wire sees prose, not a tag. */
487
+ function literalTag(tag: HtmlTagInfo, start: number, end: number): Inline {
488
+ return { type: "plain", text: escapeHtmlLiteral(tag.raw), start, end };
489
+ }
490
+
491
+ /**
492
+ * Apply the HTML dialect policy over a mixed tag/inline stream.
493
+ *
494
+ * `active` carries the constructs already open in an ANCESTOR fold. Markdown
495
+ * cannot nest same-kind emphasis — `**a **b** c**` is asterisk soup on the
496
+ * reader's screen, not nested bold — so an emphasis tag whose target is
497
+ * already active degrades to its children. Combined with the single-child
498
+ * unwrap in `buildFoldedTag`, `<b>a <b>b</b> c</b>` and `<b>**already**</b>`
499
+ * both come out as ONE bold run. A nested `<a>` is NOT unwrapped that way: its
500
+ * href is content, so it degrades to escaped literal markup instead of being
501
+ * silently deleted.
502
+ *
503
+ * See `html-fold.ts` for the bucket policy and the matched-vs-unmatched
504
+ * discriminator this implements.
505
+ */
506
+ function foldHtmlItems(
507
+ items: HtmlFoldItem[],
508
+ active: ReadonlySet<HtmlFoldTarget>,
509
+ ctx: FoldCtx,
510
+ ): Inline[] {
511
+ const out: Inline[] = [];
512
+ let i = 0;
513
+ while (i < items.length) {
514
+ const it = items[i];
515
+ if (it.kind === "inline") {
516
+ out.push(it.node);
517
+ i++;
518
+ continue;
519
+ }
520
+ const tag = it.tag;
521
+
522
+ // A comment renders nothing anywhere: dropping it loses no content.
523
+ if (tag.kind === "comment") {
524
+ i++;
525
+ continue;
526
+ }
527
+
528
+ // `<br>` / `<br/>` — a real line break (a space where breaks are illegal).
529
+ if (HTML_FOLD_TAGS[tag.name] === "break" && tag.kind !== "close") {
530
+ out.push(
531
+ ctx.noBreaks === true
532
+ ? { type: "raw", text: " ", start: it.start, end: it.end }
533
+ : { type: "raw", text: " \n", start: it.start, end: it.end },
534
+ );
535
+ i++;
536
+ continue;
537
+ }
538
+
539
+ // Void / self-closing tags delimit no content, so dropping their markup
540
+ // cannot lose text. Block-level ones still leave a separator.
541
+ if (tag.kind === "selfclose" || (tag.kind === "open" && isVoidTag(tag.name))) {
542
+ if (isPassthroughTag(tag)) {
543
+ out.push({ type: "raw", text: tag.raw, start: it.start, end: it.end });
544
+ } else if (isBlockLevelTag(tag.name)) {
545
+ out.push(mkSeparator(ctx, it.start, it.end));
546
+ }
547
+ i++;
548
+ continue;
549
+ }
550
+
551
+ // Bucket 2: the wire-verified allowlist, now BALANCE-CHECKED. Emitting an
552
+ // unmatched `<u>` raw is precisely the `unclosed start tag` 400 the fold
553
+ // policy exists to prevent — and the fallback resends the whole message as
554
+ // PLAIN TEXT, so one stray marker in relayed content degrades the entire
555
+ // reply. The check is document-level because a `<details>` open and close
556
+ // legitimately land in different mdast blocks.
557
+ if (isPassthroughTag(tag) && (tag.kind === "open" || tag.kind === "close")) {
558
+ out.push(
559
+ ctx.matched.has(it.start)
560
+ ? { type: "raw", text: tag.raw, start: it.start, end: it.end }
561
+ : literalTag(tag, it.start, it.end),
562
+ );
563
+ i++;
564
+ continue;
565
+ }
566
+
567
+ // Everything else must be BALANCED to count as markup.
568
+ if (tag.kind === "open") {
569
+ const close = findClosingTag(items, i + 1, tag.name);
570
+ const closeItem = close >= 0 ? items[close] : undefined;
571
+ if (closeItem !== undefined && closeItem.kind === "tag") {
572
+ out.push(
573
+ ...foldMatchedTag(tag, closeItem, items.slice(i + 1, close), active, ctx, {
574
+ start: it.start,
575
+ end: closeItem.end,
576
+ }),
577
+ );
578
+ i = close + 1;
579
+ continue;
580
+ }
581
+ }
582
+
583
+ // Matched somewhere ELSE in the document (a `<div>` whose `</div>` is a
584
+ // separate mdast block) — markup, so drop it, but leave a separator for a
585
+ // block-level name so the neighbouring blocks do not glue.
586
+ if (ctx.matched.has(it.start) && tag.kind !== "other") {
587
+ if (isBlockLevelTag(tag.name)) out.push(mkSeparator(ctx, it.start, it.end));
588
+ i++;
589
+ continue;
590
+ }
591
+
592
+ // Unmatched open, stray close, or an unclassifiable token: PROSE.
593
+ out.push(literalTag(tag, it.start, it.end));
594
+ i++;
595
+ }
596
+ return normalizeSeparators(out);
597
+ }
598
+
599
+ /** Emit a tag whose closing partner was found. */
600
+ function foldMatchedTag(
601
+ tag: HtmlTagInfo,
602
+ closeTag: HtmlFoldItem & { kind: "tag" },
603
+ innerItems: HtmlFoldItem[],
604
+ active: ReadonlySet<HtmlFoldTarget>,
605
+ ctx: FoldCtx,
606
+ span: Pos,
607
+ ): Inline[] {
608
+ const target = HTML_FOLD_TAGS[tag.name];
609
+
610
+ // `<pre>` — the one BLOCK-level fold (see html-fold.ts).
611
+ if (target === "pre") {
612
+ const folded = buildPreBlock(innerItems, ctx, span);
613
+ if (folded !== null) return folded;
614
+ }
615
+
616
+ if (target !== undefined && target !== "break" && target !== "pre") {
617
+ const nested = new Set(active);
618
+ nested.add(target);
619
+ const inner = foldHtmlItems(innerItems, nested, ctx);
620
+ // A LINK cannot nest in markdown, and its href is content — degrading to
621
+ // the children would silently delete the inner URL. Keep the markup as
622
+ // escaped literal text instead.
623
+ if (active.has(target)) {
624
+ return target === "link"
625
+ ? [
626
+ literalTag(tag, span.start, span.start + tag.raw.length),
627
+ ...inner,
628
+ literalTag(closeTag.tag, closeTag.start, closeTag.end),
629
+ ]
630
+ : inner;
631
+ }
632
+ const folded = buildFoldedTag(target, tag, inner, span.start, span.end);
633
+ // A tag we cannot represent faithfully (an `<a>` with no href, a `<code>`
634
+ // wrapping structure) degrades to its CONTENT — nothing is lost.
635
+ return folded !== null ? [folded] : inner;
636
+ }
637
+
638
+ // Unknown but BALANCED: markup dropped, content kept. A block-level name
639
+ // leaves a separator so its neighbours cannot glue together.
640
+ const inner = foldHtmlItems(innerItems, active, ctx);
641
+ if (!isBlockLevelTag(tag.name)) return inner;
642
+ return [
643
+ mkSeparator(ctx, span.start, span.start + tag.raw.length),
644
+ ...inner,
645
+ mkSeparator(ctx, closeTag.start, closeTag.end),
646
+ ];
647
+ }
648
+
649
+ /** Build the fenced-code equivalent of a matched `<pre>…</pre>`, or null when
650
+ * the content is not literal text a fence can carry (in which case the caller
651
+ * falls back to the ordinary degrade). An optional single `<code …>` wrapper
652
+ * is unwrapped, and its `class="language-…"` becomes the fence info string —
653
+ * exactly the shape Telegram's own Rich HTML reference documents. */
654
+ function buildPreBlock(
655
+ innerItems: HtmlFoldItem[],
656
+ ctx: FoldCtx,
657
+ span: Pos,
658
+ ): Inline[] | null {
659
+ let body = innerItems;
660
+ let language: string | null = null;
661
+ const first = body[0];
662
+ const last = body[body.length - 1];
663
+ if (
664
+ body.length >= 2 &&
665
+ first?.kind === "tag" &&
666
+ first.tag.kind === "open" &&
667
+ first.tag.name === "code" &&
668
+ last?.kind === "tag" &&
669
+ last.tag.kind === "close" &&
670
+ last.tag.name === "code"
671
+ ) {
672
+ language = languageOf(first.tag);
673
+ body = body.slice(1, -1);
674
+ }
675
+ let text = "";
676
+ for (const item of body) {
677
+ if (item.kind !== "inline" || item.node.type !== "plain") return null;
678
+ text += item.node.text;
679
+ }
680
+ // `<pre>` content routinely starts on the line after the tag and ends on the
681
+ // line before the close; those layout newlines are not content.
682
+ text = text.replace(/^\r?\n/, "").replace(/[ \t]*\r?\n[ \t]*$/, "");
683
+ // A fence cannot survive inside a heading or a table cell — degrade to an
684
+ // inline code span there, with the newlines flattened.
685
+ if (ctx.noBreaks === true) {
686
+ return [{ type: "code", text: text.replace(/\s*\r?\n\s*/g, " "), ...span }];
687
+ }
688
+ // The fence ships as a `raw` node — verbatim wire passthrough, so nothing
689
+ // downstream will widen it for us. A hardcoded ``` around a body carrying
690
+ // its own ``` earns `can't find end of Pre entity` and a plain-text resend
691
+ // of the whole message. Width comes from the same shared rule
692
+ // `renderCodeBlock` uses (`codeFenceFor`, format.ts): `buildPreBlock`
693
+ // returns `Inline[]` and so cannot route through `renderCodeBlock` (which
694
+ // takes a `Block`), but the width rule is SHARED, not copied.
695
+ const fence = codeFenceFor(text);
696
+ return [
697
+ mkSeparator(ctx, span.start, span.start),
698
+ {
699
+ type: "raw",
700
+ text: `${fence}${language ?? ""}\n${text}\n${fence}`,
701
+ ...span,
702
+ },
703
+ mkSeparator(ctx, span.end, span.end),
704
+ ];
705
+ }
706
+
707
+ function buildFoldedTag(
708
+ target: HtmlFoldTarget,
709
+ tag: HtmlTagInfo,
710
+ children: Inline[],
711
+ start: number,
712
+ end: number,
713
+ ): Inline | null {
714
+ switch (target) {
715
+ case "bold":
716
+ case "italic":
717
+ case "strike":
718
+ // Already exactly that construct (`<b>**already**</b>`) — re-wrapping
719
+ // would emit `****already****`, which reads as literal asterisks.
720
+ if (children.length === 1 && children[0].type === target) return children[0];
721
+ return { type: target, children, start, end };
722
+ case "code": {
723
+ const text = inlineLiteralText(children);
724
+ return text === null ? null : { type: "code", text, start, end };
725
+ }
726
+ case "link": {
727
+ const href = hrefOf(tag);
728
+ return href === null ? null : { type: "link", href, children, start, end };
729
+ }
730
+ default:
731
+ return null;
732
+ }
733
+ }
734
+
735
+ /** Turn a raw HTML string into fold items: tokens stay tokens, the text
736
+ * between them becomes `plain` inlines (with spoiler/highlight recognition,
737
+ * matching how prose is treated everywhere else). */
738
+ function htmlStringToItems(raw: string, base: number, source: string): HtmlFoldItem[] {
739
+ return tokenizeHtml(raw, base).flatMap<HtmlFoldItem>((piece) =>
740
+ piece.kind === "tag"
741
+ ? [{ kind: "tag", tag: piece.tag, start: piece.start, end: piece.end }]
742
+ : expandPlainNode(mkPlain(piece.text, piece.start, piece.end), source).map(
743
+ (node) => ({ kind: "inline", node }) as HtmlFoldItem,
744
+ ),
745
+ );
746
+ }
747
+
293
748
  /** Fold a run of mdast phrasing children into IR inlines, then run the
294
- * post-parse spoiler/highlight recognition over the produced `plain` nodes. */
295
- function buildInlines(children: ReadonlyArray<MdastNode>, source: string): Inline[] {
296
- return children
297
- .map((c) => foldInline(c as PhrasingContent, source))
298
- .flatMap((n) => (n.type === "plain" ? expandPlainNode(n, source) : [n]));
749
+ * post-parse spoiler/highlight recognition over the produced `plain` nodes,
750
+ * then apply the raw-HTML tag policy across the whole run (open/close markers
751
+ * are SIBLING mdast `html` nodes, never a parent, so matching has to happen
752
+ * at this level). */
753
+ function buildInlines(
754
+ children: ReadonlyArray<MdastNode>,
755
+ source: string,
756
+ ctx: FoldCtx,
757
+ ): Inline[] {
758
+ const items: HtmlFoldItem[] = [];
759
+ let sawHtml = false;
760
+ for (const child of children) {
761
+ if ((child as PhrasingContent).type === "html") {
762
+ sawHtml = true;
763
+ const p = pos(child);
764
+ items.push(...htmlStringToItems(slice(source, child), p.start, source));
765
+ continue;
766
+ }
767
+ const folded = foldInline(child as PhrasingContent, source, ctx);
768
+ const expanded = folded.type === "plain" ? expandPlainNode(folded, source) : [folded];
769
+ for (const node of expanded) items.push({ kind: "inline", node });
770
+ }
771
+ if (!sawHtml) return items.map((it) => (it as { node: Inline }).node);
772
+ return foldHtmlItems(items, new Set(), ctx);
299
773
  }
300
774
 
301
- function foldInlineChildren(node: MdastParent, source: string): Inline[] {
302
- return buildInlines(node.children, source);
775
+ function foldInlineChildren(
776
+ node: MdastParent,
777
+ source: string,
778
+ ctx: FoldCtx,
779
+ ): Inline[] {
780
+ return buildInlines(node.children, source, ctx);
303
781
  }
304
782
 
305
783
  function foldTableRow(
306
784
  row: Extract<RootContent, { type: "tableRow" }>,
307
785
  source: string,
786
+ ctx: FoldCtx,
308
787
  ): TableRow {
309
788
  const cells: TableCell[] = row.children.map((cell) => ({
310
- children: buildInlines(cell.children, source),
789
+ // A table cell is one line on the wire — a `<br>` or a structural degrade
790
+ // inside it must not emit a newline (it would end the row).
791
+ children: buildInlines(cell.children, source, { ...ctx, noBreaks: true }),
311
792
  ...pos(cell),
312
793
  }));
313
794
  return { cells, ...pos(row) };
314
795
  }
315
796
 
797
+ /** True for a block that would render to the empty string. Only the raw-HTML
798
+ * fold can produce one — a block that is nothing but markup we dropped
799
+ * (`<!-- comment -->`, `<hr/>`, `<div></div>` on their own line). Left in,
800
+ * each would contribute a spurious `\n\n` separator to the rendered document,
801
+ * so they are filtered out at the point blocks are collected. */
802
+ function isEmptyBlock(block: Block): boolean {
803
+ return block.type === "paragraph" && block.children.length === 0;
804
+ }
805
+
806
+ /** Fold a run of mdast block nodes, dropping any that render to nothing. */
807
+ function foldBlocks(
808
+ nodes: ReadonlyArray<RootContent>,
809
+ source: string,
810
+ expandableLineStarts: Set<number>,
811
+ ctx: FoldCtx,
812
+ ): Block[] {
813
+ return nodes
814
+ .map((n) => foldBlock(n, source, expandableLineStarts, ctx))
815
+ .filter((b) => !isEmptyBlock(b));
816
+ }
817
+
316
818
  function foldBlock(
317
819
  node: RootContent,
318
820
  source: string,
319
821
  expandableLineStarts: Set<number>,
822
+ ctx: FoldCtx,
320
823
  ): Block {
321
824
  switch (node.type) {
322
825
  case "paragraph":
323
- return { type: "paragraph", children: foldInlineChildren(node, source), ...pos(node) };
826
+ return { type: "paragraph", children: foldInlineChildren(node, source, ctx), ...pos(node) };
324
827
  case "heading":
325
828
  return {
326
829
  type: "heading",
327
830
  level: node.depth,
328
- children: foldInlineChildren(node, source),
831
+ // A heading is a single `#…` LINE: an embedded newline would end the
832
+ // block and orphan the rest of the text (`# a<br>b`).
833
+ children: foldInlineChildren(node, source, { ...ctx, noBreaks: true }),
329
834
  ...pos(node),
330
835
  };
331
836
  case "blockquote": {
@@ -335,7 +840,7 @@ function foldBlock(
335
840
  const expandable = expandableLineStarts.has(lineStart(source, p.start));
336
841
  return {
337
842
  type: "blockquote",
338
- children: node.children.map((c) => foldBlock(c, source, expandableLineStarts)),
843
+ children: foldBlocks(node.children, source, expandableLineStarts, ctx),
339
844
  expandable,
340
845
  ...p,
341
846
  };
@@ -349,7 +854,7 @@ function foldBlock(
349
854
  };
350
855
  case "list": {
351
856
  const items: ListItem[] = node.children.map((li) => ({
352
- children: li.children.map((c) => foldBlock(c, source, expandableLineStarts)),
857
+ children: foldBlocks(li.children, source, expandableLineStarts, ctx),
353
858
  checked: li.checked ?? null,
354
859
  // mdast `listItem.spread`: whether this item's block children are
355
860
  // separated by a blank line. Tight (false) keeps a paragraph and its
@@ -374,8 +879,8 @@ function foldBlock(
374
879
  const [headerRow, ...bodyRows] = rows;
375
880
  return {
376
881
  type: "table",
377
- header: foldTableRow(headerRow, source),
378
- rows: bodyRows.map((r) => foldTableRow(r, source)),
882
+ header: foldTableRow(headerRow, source, ctx),
883
+ rows: bodyRows.map((r) => foldTableRow(r, source, ctx)),
379
884
  align: (node.align ?? []).map(foldAlign),
380
885
  ...pos(node),
381
886
  };
@@ -391,7 +896,26 @@ function foldBlock(
391
896
  children: [{ type: "raw", text: slice(source, node), ...pos(node) }],
392
897
  ...pos(node),
393
898
  };
394
- // Not in the palette (html, definition, …): degrade to a paragraph
899
+ // A raw HTML BLOCK (`<details>…`, `<div>…`, a bare `<b>alone</b>` line).
900
+ // mdast hands the whole block over as one opaque string, so tokenize it
901
+ // and run the same three-bucket policy the inline path uses: the
902
+ // wire-verified allowlist passes through raw (which is also what stops
903
+ // `escapeMarkdown` mangling `<details open="x">` into `open\="x"`),
904
+ // markdown-equivalent tags fold, everything else drops its markup and
905
+ // keeps its content.
906
+ case "html": {
907
+ const p = pos(node);
908
+ return {
909
+ type: "paragraph",
910
+ children: foldHtmlItems(
911
+ htmlStringToItems(slice(source, node), p.start, source),
912
+ new Set(),
913
+ ctx,
914
+ ),
915
+ ...p,
916
+ };
917
+ }
918
+ // Not in the palette (definition, …): degrade to a paragraph
395
919
  // carrying the raw source slice so no content is dropped.
396
920
  default:
397
921
  return {
@@ -415,8 +939,33 @@ export function parse(markdown: string): Document {
415
939
  mdastExtensions: [gfmFromMarkdown()],
416
940
  });
417
941
  return {
418
- blocks: tree.children.map((child) =>
419
- foldBlock(child, markdown, expandableLineStarts),
420
- ),
942
+ blocks: foldBlocks(tree.children, markdown, expandableLineStarts, {
943
+ matched: matchedTagOffsets(collectHtmlTokens(tree.children, markdown)),
944
+ }),
945
+ };
946
+ }
947
+
948
+ /** Every HTML token in the tree, in document order, for the balance pass.
949
+ * Walks mdast `html` nodes only — a tag inside a code fence or a code span is
950
+ * not markup and must not balance anything. */
951
+ function collectHtmlTokens(
952
+ nodes: ReadonlyArray<MdastNode>,
953
+ source: string,
954
+ ): { tag: HtmlTagInfo; start: number }[] {
955
+ const out: { tag: HtmlTagInfo; start: number }[] = [];
956
+ const walk = (list: ReadonlyArray<MdastNode>): void => {
957
+ for (const node of list) {
958
+ if (node.type === "html") {
959
+ const p = pos(node);
960
+ for (const piece of tokenizeHtml(slice(source, node), p.start)) {
961
+ if (piece.kind === "tag") out.push({ tag: piece.tag, start: piece.start });
962
+ }
963
+ continue;
964
+ }
965
+ const children = (node as MdastParent).children;
966
+ if (Array.isArray(children)) walk(children);
967
+ }
421
968
  };
969
+ walk(nodes);
970
+ return out;
422
971
  }