switchroom 0.18.31 → 0.18.33

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (159) hide show
  1. package/dist/agent-scheduler/index.js +4 -2
  2. package/dist/auth-broker/index.js +21 -3
  3. package/dist/cli/notion-write-pretool.mjs +4 -2
  4. package/dist/cli/switchroom.js +1410 -852
  5. package/dist/host-control/main.js +22 -4
  6. package/dist/vault/approvals/kernel-server.js +21 -3
  7. package/dist/vault/broker/server.js +48 -4
  8. package/package.json +4 -3
  9. package/profiles/_base/start.sh.hbs +148 -23
  10. package/telegram-plugin/dist/gateway/gateway.js +62726 -58282
  11. package/telegram-plugin/gateway/agent-button-callback-handler.ts +237 -0
  12. package/telegram-plugin/gateway/ask-callback-handler.ts +92 -0
  13. package/telegram-plugin/gateway/attachment-message-handlers.ts +152 -0
  14. package/telegram-plugin/gateway/backstop-delivery.ts +223 -23
  15. package/telegram-plugin/gateway/boot-card.ts +169 -1
  16. package/telegram-plugin/gateway/bot-commands-model-effort.ts +209 -0
  17. package/telegram-plugin/gateway/bot-commands-start-info.ts +108 -0
  18. package/telegram-plugin/gateway/callback-query-handlers.ts +124 -0
  19. package/telegram-plugin/gateway/captured-answer-resume.ts +259 -0
  20. package/telegram-plugin/gateway/card-approval-keyboards.test.ts +28 -0
  21. package/telegram-plugin/gateway/card-tool-handlers.ts +639 -0
  22. package/telegram-plugin/gateway/checklist-message-handler.ts +107 -0
  23. package/telegram-plugin/gateway/delivery-confirm-wiring.ts +133 -0
  24. package/telegram-plugin/gateway/disconnect-flush.ts +6 -44
  25. package/telegram-plugin/gateway/gateway-import-clean.test.ts +188 -0
  26. package/telegram-plugin/gateway/gateway.ts +6403 -13456
  27. package/telegram-plugin/gateway/inbound-delivery-machine-dispatch.ts +7 -15
  28. package/telegram-plugin/gateway/inbound-delivery-machine-shadow.ts +35 -68
  29. package/telegram-plugin/gateway/inbound-interceptors.ts +1133 -0
  30. package/telegram-plugin/gateway/inbound-router.ts +400 -0
  31. package/telegram-plugin/gateway/liveness-wiring.ts +440 -0
  32. package/telegram-plugin/gateway/media-message-handlers.ts +256 -0
  33. package/telegram-plugin/gateway/mental-model-propose-card.ts +16 -0
  34. package/telegram-plugin/gateway/model-command.ts +23 -0
  35. package/telegram-plugin/gateway/narrative-lane.ts +865 -0
  36. package/telegram-plugin/gateway/obligation-ledger.ts +42 -0
  37. package/telegram-plugin/gateway/obligation-store.ts +37 -1
  38. package/telegram-plugin/gateway/obligation-wiring.ts +333 -0
  39. package/telegram-plugin/gateway/outbound-send-path.ts +2012 -0
  40. package/telegram-plugin/gateway/photo-message-handler.ts +80 -0
  41. package/telegram-plugin/gateway/pinned-message-handler.ts +86 -0
  42. package/telegram-plugin/gateway/secret-request-card.test.ts +46 -0
  43. package/telegram-plugin/gateway/secret-request-card.ts +45 -0
  44. package/telegram-plugin/gateway/stream-render.ts +2166 -0
  45. package/telegram-plugin/gateway/turn-end.ts +606 -0
  46. package/telegram-plugin/gateway/turn-start-surfaces.ts +298 -0
  47. package/telegram-plugin/gateway/vault-request-access-card.ts +16 -0
  48. package/telegram-plugin/gateway/vault-request-save-card.test.ts +49 -0
  49. package/telegram-plugin/gateway/vault-request-save-card.ts +52 -0
  50. package/telegram-plugin/gateway/voice-message-handler.ts +123 -0
  51. package/telegram-plugin/gateway/voice-ondemand-callback-handler.ts +204 -0
  52. package/telegram-plugin/gateway/worker-feed-dispatch.ts +40 -0
  53. package/telegram-plugin/narrative-dedup.ts +24 -1
  54. package/telegram-plugin/narrative-flush.ts +2 -2
  55. package/telegram-plugin/pending-user-notice.ts +59 -13
  56. package/telegram-plugin/render/render.ts +25 -1
  57. package/telegram-plugin/status-no-truncate.ts +13 -0
  58. package/telegram-plugin/subagent-watcher.ts +297 -31
  59. package/telegram-plugin/tests/activity-card-wiring.test.ts +8 -3
  60. package/telegram-plugin/tests/activity-ever-opened-sticky.test.ts +18 -3
  61. package/telegram-plugin/tests/agent-button-callback-handler.test.ts +149 -0
  62. package/telegram-plugin/tests/ask-callback-handler.test.ts +118 -0
  63. package/telegram-plugin/tests/attachment-message-handlers.test.ts +135 -0
  64. package/telegram-plugin/tests/backstop-delivery.test.ts +167 -0
  65. package/telegram-plugin/tests/backstop-readback-probe.test.ts +144 -0
  66. package/telegram-plugin/tests/boot-card-routing.test.ts +139 -0
  67. package/telegram-plugin/tests/bot-commands-model-effort.test.ts +189 -0
  68. package/telegram-plugin/tests/bot-commands-start-info.test.ts +240 -0
  69. package/telegram-plugin/tests/buffer-gate-broadened.test.ts +28 -9
  70. package/telegram-plugin/tests/busy-ack-wiring.test.ts +6 -1
  71. package/telegram-plugin/tests/button-tap-turn-gated.test.ts +21 -12
  72. package/telegram-plugin/tests/callback-query-handlers.test.ts +101 -0
  73. package/telegram-plugin/tests/captured-answer-resume.test.ts +358 -0
  74. package/telegram-plugin/tests/card-tool-handlers.test.ts +497 -0
  75. package/telegram-plugin/tests/catch-all-unhandled-message.test.ts +5 -2
  76. package/telegram-plugin/tests/checklist-message-handler.test.ts +160 -0
  77. package/telegram-plugin/tests/emission-authority-facade.test.ts +76 -29
  78. package/telegram-plugin/tests/emission-authority-ping-gate.test.ts +4 -1
  79. package/telegram-plugin/tests/emission-determinism-wiring.test.ts +45 -16
  80. package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +30 -7
  81. package/telegram-plugin/tests/gateway-boot-side-effect-gating.test.ts +270 -0
  82. package/telegram-plugin/tests/gateway-boot-smoke.test.ts +150 -0
  83. package/telegram-plugin/tests/gateway-bot-construction-deferral.test.ts +251 -0
  84. package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +5 -128
  85. package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +299 -0
  86. package/telegram-plugin/tests/gateway-loopback-paste-redact.test.ts +44 -29
  87. package/telegram-plugin/tests/gateway-outbound-redact.test.ts +18 -5
  88. package/telegram-plugin/tests/gateway-request-secret.test.ts +7 -3
  89. package/telegram-plugin/tests/gateway-secret-detect.test.ts +20 -10
  90. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +8 -2
  91. package/telegram-plugin/tests/inbound-delivery-cutover-flip.test.ts +54 -150
  92. package/telegram-plugin/tests/inbound-delivery-cutover-gate.test.ts +10 -14
  93. package/telegram-plugin/tests/inbound-delivery-dispatch-equivalence.test.ts +6 -7
  94. package/telegram-plugin/tests/inbound-delivery-machine-dispatch.test.ts +0 -16
  95. package/telegram-plugin/tests/inbound-emit-after-intercepts.test.ts +18 -7
  96. package/telegram-plugin/tests/inbound-message-types.test.ts +52 -16
  97. package/telegram-plugin/tests/litellm-proxy-auth-misconfig.test.ts +69 -14
  98. package/telegram-plugin/tests/media-message-handlers.test.ts +276 -0
  99. package/telegram-plugin/tests/mental-model-propose-callback-gate.test.ts +8 -4
  100. package/telegram-plugin/tests/model-command.test.ts +30 -0
  101. package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +36 -12
  102. package/telegram-plugin/tests/narrative-dedup.test.ts +32 -0
  103. package/telegram-plugin/tests/narrative-flush.test.ts +6 -2
  104. package/telegram-plugin/tests/narrative-lane-golden.test.ts +458 -0
  105. package/telegram-plugin/tests/no-reply-bounded-drain.test.ts +14 -3
  106. package/telegram-plugin/tests/obligation-ledger.test.ts +40 -0
  107. package/telegram-plugin/tests/obligation-store.test.ts +43 -0
  108. package/telegram-plugin/tests/pending-card-durability-wiring.test.ts +18 -8
  109. package/telegram-plugin/tests/per-topic-current-turn.test.ts +32 -8
  110. package/telegram-plugin/tests/permission-no-repeat-wiring.test.ts +9 -6
  111. package/telegram-plugin/tests/photo-message-handler.test.ts +114 -0
  112. package/telegram-plugin/tests/photo-reroute-wiring.test.ts +5 -2
  113. package/telegram-plugin/tests/pinned-message-handler.test.ts +108 -0
  114. package/telegram-plugin/tests/render/render.test.ts +42 -0
  115. package/telegram-plugin/tests/reply-terminal-reaction.test.ts +6 -2
  116. package/telegram-plugin/tests/secret-detect-delete-must-surface-failures.test.ts +8 -4
  117. package/telegram-plugin/tests/secret-detect-fail-closed.test.ts +38 -28
  118. package/telegram-plugin/tests/secret-detect-oauth-code.test.ts +31 -20
  119. package/telegram-plugin/tests/send-reply-golden.test.ts +571 -0
  120. package/telegram-plugin/tests/silence-liveness-wiring.test.ts +22 -8
  121. package/telegram-plugin/tests/status-pin-service-message-suppression.test.ts +42 -49
  122. package/telegram-plugin/tests/stop-command.test.ts +22 -12
  123. package/telegram-plugin/tests/stream-render-golden.test.ts +424 -0
  124. package/telegram-plugin/tests/subagent-watcher-boot-skip-dead.test.ts +218 -0
  125. package/telegram-plugin/tests/subagent-watcher-resume-reregister.test.ts +305 -0
  126. package/telegram-plugin/tests/subagent-watcher-resurrection.test.ts +32 -0
  127. package/telegram-plugin/tests/subagent-watcher.test.ts +35 -3
  128. package/telegram-plugin/tests/turn-end-gate-backstop.test.ts +8 -12
  129. package/telegram-plugin/tests/turn-flush-safety.test.ts +191 -11
  130. package/telegram-plugin/tests/turn-flush-suppression-wiring.test.ts +117 -0
  131. package/telegram-plugin/tests/vault-approval-posture.test.ts +8 -2
  132. package/telegram-plugin/tests/vault-grant-inbound-builders.test.ts +2 -2
  133. package/telegram-plugin/tests/vault-grant-union.test.ts +4 -1
  134. package/telegram-plugin/tests/vault-key-regex-allows-slash.test.ts +16 -5
  135. package/telegram-plugin/tests/vault-request-access-tool.test.ts +10 -5
  136. package/telegram-plugin/tests/vault-request-access-unlock-resume.test.ts +4 -1
  137. package/telegram-plugin/tests/vault-subcommands.test.ts +6 -1
  138. package/telegram-plugin/tests/voice-message-handler.test.ts +111 -0
  139. package/telegram-plugin/tests/voice-ondemand-callback-handler.test.ts +140 -0
  140. package/telegram-plugin/tests/worker-activity-feed.test.ts +86 -19
  141. package/telegram-plugin/tests/worker-feed-coalesce.test.ts +236 -20
  142. package/telegram-plugin/tests/worker-feed-resume-guard.test.ts +86 -0
  143. package/telegram-plugin/tool-activity-summary.ts +110 -38
  144. package/telegram-plugin/turn-flush-safety.ts +80 -14
  145. package/telegram-plugin/uat/restart-capability.ts +76 -0
  146. package/telegram-plugin/uat/scenarios/bg-sub-agent-dispatch-dm.test.ts +14 -4
  147. package/telegram-plugin/uat/scenarios/bridge-flap-resilience-dm.test.ts +11 -1
  148. package/telegram-plugin/uat/scenarios/cross-turn-pending-progress-dm.test.ts +19 -2
  149. package/telegram-plugin/uat/scenarios/jtbd-always-on-after-restart-dm.test.ts +6 -12
  150. package/telegram-plugin/uat/scenarios/jtbd-deliberate-restart-resumes-dm.test.ts +6 -12
  151. package/telegram-plugin/uat/scenarios/jtbd-interrupted-turn-resumes-dm.test.ts +6 -12
  152. package/telegram-plugin/uat/scenarios/jtbd-multipart-render-dm.test.ts +47 -13
  153. package/telegram-plugin/worker-activity-feed.ts +34 -4
  154. package/telegram-plugin/gateway/busy-key-reaper.ts +0 -113
  155. package/telegram-plugin/gateway/gate-parity-probe.ts +0 -102
  156. package/telegram-plugin/tests/busy-key-reaper.test.ts +0 -192
  157. package/telegram-plugin/tests/fixtures/cutover-killswitch-probe.ts +0 -75
  158. package/telegram-plugin/tests/gate-parity-probe.test.ts +0 -171
  159. package/telegram-plugin/tests/parallel-turns-deadlock-fix.test.ts +0 -217
@@ -0,0 +1,86 @@
1
+ /**
2
+ * Issue #3373 follow-up (PR #3376 adversarial review) — guards for the
3
+ * SendMessage-resume → feed re-surface seam.
4
+ *
5
+ * The #3376 fix has three links: the watcher fires `onResume` on jsonl growth
6
+ * past a genuine terminal (pinned by subagent-watcher-resume-reregister.test.ts),
7
+ * the gateway's `onResume` handler delegates to `handleWorkerResume`, and
8
+ * `handleWorkerResume` calls `feed.resurrect(agentId)` to clear the terminal
9
+ * `finalized` latch. The review found the middle + last links untested: deleting
10
+ * the gateway handler (or gutting its body) left every test green. This file
11
+ * closes both:
12
+ *
13
+ * 1. UNIT — `handleWorkerResume` calls `feed.resurrect` with the agentId,
14
+ * never throws when resurrect throws or the feed is absent, and always
15
+ * emits the re-surface audit line.
16
+ * 2. SOURCE SCAN (mirrors permission-verdict-resume-guard.test.ts) — the
17
+ * gateway wires an `onResume:` callback into the watcher config AND that
18
+ * callback delegates to `handleWorkerResume`. Deleting the handler, or
19
+ * replacing the delegation with a no-op body, reds this suite.
20
+ */
21
+
22
+ import { describe, it, expect } from 'vitest'
23
+ import { readFileSync } from 'node:fs'
24
+ import { fileURLToPath } from 'node:url'
25
+ import { dirname, resolve } from 'node:path'
26
+ import { handleWorkerResume } from '../gateway/worker-feed-dispatch.js'
27
+
28
+ describe('handleWorkerResume (issue #3373 seam unit)', () => {
29
+ it('calls feed.resurrect with the agentId and logs the re-surface line', () => {
30
+ const calls: string[] = []
31
+ const logs: string[] = []
32
+ handleWorkerResume({ resurrect: (id) => calls.push(id) }, 'w1', (m) => logs.push(m))
33
+ expect(calls).toEqual(['w1'])
34
+ expect(logs.some((l) => l.includes('w1') && l.includes('RE-SURFACED'))).toBe(true)
35
+ })
36
+
37
+ it('a throwing resurrect is logged, never propagated (watcher poll loop safety)', () => {
38
+ const logs: string[] = []
39
+ expect(() =>
40
+ handleWorkerResume(
41
+ { resurrect: () => { throw new Error('boom') } },
42
+ 'w2',
43
+ (m) => logs.push(m),
44
+ ),
45
+ ).not.toThrow()
46
+ expect(logs.some((l) => l.includes('boom') && l.includes('w2'))).toBe(true)
47
+ // The audit line still lands after the failure.
48
+ expect(logs.some((l) => l.includes('RE-SURFACED'))).toBe(true)
49
+ })
50
+
51
+ it('a null/undefined feed (feed disabled) is a safe no-op that still audits', () => {
52
+ const logs: string[] = []
53
+ expect(() => handleWorkerResume(null, 'w3', (m) => logs.push(m))).not.toThrow()
54
+ expect(logs.some((l) => l.includes('w3') && l.includes('RE-SURFACED'))).toBe(true)
55
+ })
56
+ })
57
+
58
+ describe('gateway onResume wiring (issue #3373 source-scan guard)', () => {
59
+ const __dirname = dirname(fileURLToPath(import.meta.url))
60
+ const GATEWAY_SRC = readFileSync(
61
+ resolve(__dirname, '..', 'gateway', 'gateway.ts'),
62
+ 'utf8',
63
+ )
64
+
65
+ it('the watcher config carries an onResume callback', () => {
66
+ expect(/\bonResume:\s*\(/.test(GATEWAY_SRC)).toBe(true)
67
+ })
68
+
69
+ it('the onResume callback delegates to handleWorkerResume (not an inline no-op)', () => {
70
+ // Match the `onResume: (…) => { … }` callback body and require the
71
+ // delegation call inside it. A gutted handler (empty body / dropped
72
+ // resurrect) fails here even though tsc stays green.
73
+ const m = GATEWAY_SRC.match(/\bonResume:\s*\([^)]*\)\s*=>\s*\{([\s\S]*?)\n\s{14}\},/)
74
+ expect(m, 'onResume callback not found in gateway.ts').not.toBeNull()
75
+ expect(m![1]).toMatch(/\bhandleWorkerResume\s*\(/)
76
+ expect(m![1]).toMatch(/\bworkerActivityFeed\b/)
77
+ })
78
+
79
+ it('handleWorkerResume is imported from worker-feed-dispatch (the tested seam)', () => {
80
+ expect(
81
+ /import\s*\{[^}]*\bhandleWorkerResume\b[^}]*\}\s*from\s*'\.\/worker-feed-dispatch\.js'/.test(
82
+ GATEWAY_SRC,
83
+ ),
84
+ ).toBe(true)
85
+ })
86
+ })
@@ -86,6 +86,7 @@ export function describeToolUse(
86
86
  import {
87
87
  STATUS_CARD_CHAR_BUDGET,
88
88
  STATUS_ROLLING_LINES,
89
+ WORKER_HISTORY_MAX,
89
90
  STATUS_LINE_MAX,
90
91
  NESTED_PREFIX,
91
92
  } from './status-no-truncate.js'
@@ -286,7 +287,8 @@ function escapeStepLine(raw: string): string {
286
287
 
287
288
  /**
288
289
  * Shared step-feed emitter. Appends `✓`/`→` bullet lines to `out` for the
289
- * given ALREADY-ESCAPED step strings, windowing to STATUS_ROLLING_LINES and
290
+ * given ALREADY-ESCAPED step strings, windowing to `window` (default
291
+ * STATUS_ROLLING_LINES; the worker surfaces pass a deeper window) and
290
292
  * prepending a `+N earlier…` header when the feed overflows the window (on
291
293
  * BOTH surfaces now). The worker feed imports this directly.
292
294
  *
@@ -301,9 +303,10 @@ export function renderStepFeed(
301
303
  steps: string[],
302
304
  allDone: boolean,
303
305
  liveSuffix = '',
306
+ window: number = STATUS_ROLLING_LINES,
304
307
  ): void {
305
308
  if (steps.length === 0) return
306
- const shown = steps.slice(-STATUS_ROLLING_LINES)
309
+ const shown = steps.slice(-Math.max(1, window))
307
310
  const hidden = steps.length - shown.length
308
311
  if (hidden > 0) out.push(`_✓ +${hidden} earlier…_`)
309
312
  const lastIdx = shown.length - 1
@@ -351,6 +354,13 @@ export interface StatusCardOpts {
351
354
  stepCount?: number
352
355
  /** Optional terminal result block (worker recap), already-cleaned text + emoji. */
353
356
  result?: { emoji: string; text: string }
357
+ /**
358
+ * How many trailing step/child lines the rolling window shows. Defaults to
359
+ * STATUS_ROLLING_LINES (the 🤖 agent card). The 🛠 single-worker card passes
360
+ * a deeper window (`workerHistoryDepth(1)` = 6) so a lone worker can show its
361
+ * full recent trail — see WORKER_HISTORY_MAX.
362
+ */
363
+ historyWindow?: number
354
364
  }
355
365
 
356
366
  /**
@@ -364,6 +374,7 @@ export interface StatusCardOpts {
364
374
  */
365
375
  export function renderStatusCard(opts: StatusCardOpts): string | null {
366
376
  const { header, final = false, liveSuffix = '', stepCount, result } = opts
377
+ const window = Math.max(1, opts.historyWindow ?? STATUS_ROLLING_LINES)
367
378
  const rawSteps = opts.steps.filter((s) => s != null)
368
379
  const rawChildren = (opts.childSteps ?? []).map((s) => s.trim()).filter((s) => s.length > 0)
369
380
  const hasChildren = rawChildren.length > 0
@@ -393,12 +404,12 @@ export function renderStatusCard(opts: StatusCardOpts): string | null {
393
404
 
394
405
  if (hasChildren) {
395
406
  // Parent lines all render done — the live → step lives in the nested block.
396
- const shownParent = steps.slice(-STATUS_ROLLING_LINES)
407
+ const shownParent = steps.slice(-window)
397
408
  const hiddenParent = steps.length - shownParent.length
398
409
  if (hiddenParent > 0) out.push(`_✓ +${hiddenParent} earlier…_`)
399
410
  for (const s of shownParent) out.push(`~~_✓ ${s}_~~`)
400
411
  // Child block.
401
- const shownChild = children.slice(-STATUS_ROLLING_LINES)
412
+ const shownChild = children.slice(-window)
402
413
  const hiddenChild = children.length - shownChild.length
403
414
  if (hiddenChild > 0) out.push(`${NESTED_PREFIX}_+${hiddenChild} earlier…_`)
404
415
  const lastChildIdx = shownChild.length - 1
@@ -410,7 +421,7 @@ export function renderStatusCard(opts: StatusCardOpts): string | null {
410
421
  )
411
422
  })
412
423
  } else {
413
- renderStepFeed(out, steps, final, liveSuffix)
424
+ renderStepFeed(out, steps, final, liveSuffix, window)
414
425
  }
415
426
 
416
427
  if (final && stepCount != null && stepCount > 0) {
@@ -646,54 +657,96 @@ export interface CombinedWorkerRow {
646
657
  /** Running total tokens for this worker — rendered as `· {N} tok` on the row
647
658
  * header. Omitted (0/undefined) → no token segment. */
648
659
  totalTokens?: number
660
+ /**
661
+ * Stable per-card ordinal (1-based), assigned when the worker joins its feed
662
+ * group and KEPT for the card's lifetime — survivors keep their numbers when
663
+ * an earlier worker finishes (the card may show `2.`/`3.` with no `1.`; the
664
+ * `N running` chrome carries the count). Rendered as a `{ordinal}. ` prefix
665
+ * inside the bold header when the card has 2+ rows. Omitted → unnumbered
666
+ * (back-compat for direct callers).
667
+ */
668
+ ordinal?: number
649
669
  }
650
670
 
651
671
  export interface CombinedWorkerFeedOpts {
652
672
  /** Max worker rows rendered before the `+M more working…` spill line.
653
- * The overflow is ordered oldest-hidden-first (the newest/most-recently
654
- * active workers stay visible). */
673
+ * Rows are kept head-first (`rows.slice(0, visibleCount)`): the earliest
674
+ * supplied (oldest dispatch-order) workers stay visible and the
675
+ * newest/trailing rows spill. */
655
676
  maxRows: number
656
677
  }
657
678
 
679
+ /** Each visible worker costs one header line before any history. */
680
+ const PER_WORKER_HEADER_COST = 1
681
+
682
+ /**
683
+ * Floor of the per-worker depth curve — every SHOWN worker renders at least
684
+ * this many recent steps, no matter how large the fan-out (Ken's "4+ → 3 each").
685
+ */
686
+ const MIN_WORKER_DEPTH = 3
687
+
688
+ /**
689
+ * Design fan-out: the largest concurrent-worker count whose full Ken curve is
690
+ * allowed to render before the total-line backstop starts collapsing the
691
+ * NEWEST (trailing) rows into `+M more working…`. Chosen at 6 — beyond six live workers a
692
+ * per-worker trail is no longer a glanceable card, so extra rows spill rather
693
+ * than every shown worker losing depth. Every worker that IS shown keeps its
694
+ * full curve depth; the ceiling trims row COUNT, never per-worker depth.
695
+ */
696
+ const DESIGN_FANOUT = 6
697
+
658
698
  /**
659
699
  * Total per-worker BODY line budget for the combined feed — the sum, across all
660
700
  * visible workers, of (one header line + that worker's history lines). The top
661
701
  * `🛠 Workers · N running` line and the `+M more working…` spill are OUTSIDE
662
- * this budget (fixed chrome). 13 is chosen so the card stays a compact glance,
663
- * not a wall: at 2 workers it yields the full 5-line history each (2·1 header +
664
- * 2·5 history = 12 ≤ 13), and it degrades to a single history line each by ~6
665
- * workers — matching the pre-adaptive one-line-per-worker floor while never
666
- * letting a 2–3 worker fan-out lose its narrative trail.
702
+ * this budget (fixed chrome).
703
+ *
704
+ * Re-derived (#3349) to fit Ken's per-worker curve `max(3, 7 − w)` rather than
705
+ * to DRIVE the depth: at the flat tail (w ≥ 4, depth = MIN_WORKER_DEPTH) each
706
+ * worker costs `PER_WORKER_HEADER_COST + MIN_WORKER_DEPTH` = 4 lines, so a
707
+ * DESIGN_FANOUT of 6 fully-rendered workers needs 6 × 4 = 24 lines. That fully
708
+ * fits every fan-out through 6 workers (w1=7, w2=12, w3=15, w4=16, w5=20,
709
+ * w6=24); a 7th/8th concurrent worker overflows and the backstop drops the
710
+ * newest (trailing) visible rows to the spill — bounding the card at 24 body lines so a
711
+ * big swarm never explodes it. The per-worker curve wins on DEPTH; this ceiling
712
+ * wins on ROW COUNT.
667
713
  */
668
- const MAX_COMBINED_BODY_LINES = 13
669
- /** Each visible worker costs one header line before any history. */
670
- const PER_WORKER_HEADER_COST = 1
714
+ const MAX_COMBINED_BODY_LINES = DESIGN_FANOUT * (PER_WORKER_HEADER_COST + MIN_WORKER_DEPTH)
671
715
 
672
716
  /**
673
717
  * Deterministic per-worker history depth for `w` visible workers:
674
- * clamp( floor( (BUDGET − headerCost·w) / w ), 1, STATUS_ROLLING_LINES )
675
- * So 2 workers → 5 lines each, 3 → 3, 4 → 2, ≥6 → 1 (today's single-line floor
676
- * is the graceful-degradation floor, never below it). Pure function of the
677
- * visible worker count — no model input, consistent with deterministic controls.
718
+ * clamp( max(MIN_WORKER_DEPTH, 7 − w), 1, WORKER_HISTORY_MAX )
719
+ * So 1 worker → 6, 2 → 5, 3 → 4, 4 → 3, ≥4 → 3 (Ken's 6/5/4/3 curve, #3349).
720
+ * Pure function of the visible worker count — no model input, consistent with
721
+ * deterministic controls. Replaces the former budget-driven divide (which
722
+ * yielded 5/5/3/2/… and could never reach 6 for a lone worker).
678
723
  */
679
724
  export function combinedHistoryDepth(w: number): number {
680
725
  if (w <= 0) return 1
681
- const raw = Math.floor((MAX_COMBINED_BODY_LINES - PER_WORKER_HEADER_COST * w) / w)
682
- return Math.max(1, Math.min(STATUS_ROLLING_LINES, raw))
726
+ return Math.min(WORKER_HISTORY_MAX, Math.max(MIN_WORKER_DEPTH, 7 - w))
683
727
  }
684
728
 
729
+ /** Alias for readability at the single-worker call site — same curve. */
730
+ export const workerHistoryDepth = combinedHistoryDepth
731
+
685
732
  /**
686
733
  * Render N≥1 live workers into ONE combined feed body (ready Telegram
687
734
  * markdown; callers send verbatim — do NOT re-escape). Layout:
688
735
  *
689
736
  * 🛠 **Workers** · _N running_
690
- * **{desc1}** _· {elapsed} · {n} tools_
737
+ * **1. {desc1}** _· {elapsed} · {n} tools_
691
738
  * ~~_✓ {earlier step}_~~
692
739
  * **→ {newest step}**
693
- * **{desc2}** _· {elapsed} · {n} tools_
740
+ * **2. {desc2}** _· {elapsed} · {n} tools_
694
741
  * **→ {newest step}**
695
742
  * _+M more working…_
696
743
  *
744
+ * NUMBERING (#3298): when the card tracks 2+ rows AND a row carries `ordinal`,
745
+ * its header gets a stable `{ordinal}. ` prefix. Ordinals are assigned by the
746
+ * caller at dispatch and kept for the card's life — after an earlier worker
747
+ * finishes the survivors keep their numbers (`2.`, `3.` with no `1.`). A lone
748
+ * row, or rows without ordinals, render unnumbered.
749
+ *
697
750
  * ADAPTIVE DENSITY: each visible worker renders its last-K narrative lines as a
698
751
  * `✓`/`→` trail (prior steps struck, newest bold in-progress) — the single-
699
752
  * worker card's idiom — where K = `combinedHistoryDepth(visibleCount)` splits a
@@ -705,7 +758,7 @@ export function combinedHistoryDepth(w: number): number {
705
758
  * Pure. Rows are rendered in the order supplied (the manager passes them
706
759
  * dispatch-order, oldest first). `maxRows` caps the visible rows; the hidden
707
760
  * remainder collapses to a single `+M more working…` line. A total-budget
708
- * backstop drops the OLDEST visible rows one at a time (growing the spill)
761
+ * backstop drops the NEWEST (trailing) visible rows one at a time (growing the spill)
709
762
  * until the body fits STATUS_CARD_CHAR_BUDGET, so a burst of long descriptions
710
763
  * can never overflow the wire limit. Returns null only when `rows` is empty.
711
764
  */
@@ -716,6 +769,11 @@ export function renderCombinedWorkerFeed(
716
769
  if (rows.length === 0) return null
717
770
  const maxRows = Math.max(1, Math.floor(opts.maxRows))
718
771
 
772
+ // Number the workers only when the CARD tracks 2+ (a lone worker stays
773
+ // unnumbered). Uses the total row count, not the visible count, so ordinals
774
+ // don't appear/vanish as the overflow backstop shrinks the visible set.
775
+ const numbered = rows.length >= 2
776
+
719
777
  const rowHeader = (r: CombinedWorkerRow): string => {
720
778
  const desc = escapeMarkdown(
721
779
  truncate(stripMarkdown(r.description).replace(/\s+/g, ' ').trim() || 'background task', COMBINED_ROW_DESC_MAX),
@@ -724,7 +782,11 @@ export function renderCombinedWorkerFeed(
724
782
  const tokPart = tokenSegment(r.totalTokens)
725
783
  const modelLabel = formatModelLabel(r.model)
726
784
  const modelPart = modelLabel != null ? ` · ${escapeMarkdown(modelLabel)}` : ''
727
- return `**${desc}** _· ${formatFeedElapsed(r.elapsedMs)} · ${r.toolCount} ${toolWord}${tokPart}${modelPart}_`
785
+ // Stable ordinal prefix INSIDE the bold span, before the already-escaped
786
+ // description — no new escaping surface, and the gateway md→HTML conversion
787
+ // has no ordered-list auto-formatting on bolded text.
788
+ const num = numbered && r.ordinal != null ? `${r.ordinal}. ` : ''
789
+ return `**${num}${desc}** _· ${formatFeedElapsed(r.elapsedMs)} · ${r.toolCount} ${toolWord}${tokPart}${modelPart}_`
728
790
  }
729
791
 
730
792
  // Raw (unescaped) history for a worker, oldest→newest, empty lines stripped.
@@ -734,37 +796,47 @@ export function renderCombinedWorkerFeed(
734
796
  return src.filter((s) => s != null && stripMarkdown(s).replace(/\s+/g, ' ').trim().length > 0)
735
797
  }
736
798
 
737
- const compose = (visibleCount: number): string => {
799
+ const compose = (visibleCount: number): { body: string; bodyLines: number } => {
738
800
  const shown = rows.slice(0, visibleCount)
739
801
  const hidden = rows.length - shown.length
740
- // Adaptive depth: split the fixed body-line budget across the VISIBLE
741
- // workers so the card stays bounded regardless of fan-out.
802
+ // Per-worker depth follows Ken's deterministic curve max(3, 7−w) (#3349):
803
+ // the curve drives DEPTH; the total-line budget below drives ROW COUNT.
742
804
  const depth = combinedHistoryDepth(shown.length)
743
- const out: string[] = [`🛠 **Workers** · _${rows.length} running_`]
805
+ const chrome: string[] = [`🛠 **Workers** · _${rows.length} running_`]
806
+ const bodyOut: string[] = []
744
807
  for (const r of shown) {
745
- out.push(rowHeader(r))
808
+ bodyOut.push(rowHeader(r))
746
809
  const hist = rowHistory(r)
747
810
  if (hist.length === 0) {
748
- out.push('→ _starting…_')
811
+ bodyOut.push('→ _starting…_')
749
812
  continue
750
813
  }
751
814
  // Paint the last-K history lines with the SAME `✓`/`→` idiom as the
752
815
  // single-worker card: escape each raw line through the shared per-line
753
816
  // pipeline (escapeStepLine), then renderStepFeed strikes the prior steps
754
- // and bolds the newest in-progress step.
817
+ // and bolds the newest in-progress step. The window equals the depth so a
818
+ // per-worker `+N earlier…` marker never appears inside the combined feed.
755
819
  const esc = hist.slice(-depth).map(escapeStepLine)
756
- renderStepFeed(out, esc, false)
820
+ renderStepFeed(bodyOut, esc, false, '', depth)
757
821
  }
822
+ const out = [...chrome, ...bodyOut]
758
823
  if (hidden > 0) out.push(`_+${hidden} more working…_`)
759
- return stackCardLines(out)
824
+ return { body: stackCardLines(out), bodyLines: bodyOut.length }
760
825
  }
761
826
 
762
- // Cap to maxRows first, then shrink further only if the char budget demands.
827
+ // Cap to maxRows first, then shrink the visible set while EITHER the total
828
+ // body-line budget (#3349: bounds a big swarm without stealing depth from the
829
+ // shown workers) OR the wire char budget is exceeded. Newest (trailing) rows
830
+ // collapse into the `+M more working…` spill (`rows.slice(0, visibleCount)`
831
+ // keeps the head of the list).
763
832
  let visible = Math.min(rows.length, maxRows)
764
- let body = compose(visible)
765
- while (body.length > STATUS_CARD_CHAR_BUDGET && visible > 1) {
833
+ let { body, bodyLines } = compose(visible)
834
+ while (
835
+ (bodyLines > MAX_COMBINED_BODY_LINES || body.length > STATUS_CARD_CHAR_BUDGET) &&
836
+ visible > 1
837
+ ) {
766
838
  visible -= 1
767
- body = compose(visible)
839
+ ;({ body, bodyLines } = compose(visible))
768
840
  }
769
841
  return body
770
842
  }
@@ -155,28 +155,80 @@ export const FLUSH_SUBSTANTIVE_MIN_CHARS = 200
155
155
  * `[verboseNarration(250), realAnswer(150)]` a reversed length scan returns the
156
156
  * 250-char narration and DROPS the 150-char real answer. Instead we take the
157
157
  * last non-empty block as the answer and strip only the EARLIER blocks — and
158
- * only when they look like intent-narration (short, or the classic "Let me…" /
159
- * "I'll…" openers). If the earlier blocks are themselves substantial (a genuine
160
- * multi-paragraph answer written as several blocks) we keep the whole thing
161
- * joined, so we never truncate a real long answer down to its last paragraph.
158
+ * only when they look like intent-narration. If the earlier blocks are
159
+ * themselves substantial (a genuine multi-paragraph answer written as several
160
+ * blocks) we keep the whole thing joined, so we never truncate a real long
161
+ * answer down to its last paragraph.
162
+ *
163
+ * #3237 — the STRUCTURAL discriminator. The opener heuristic below
164
+ * (`isNarrationBlock`) cannot tell a narration preamble ("Let me pull the
165
+ * numbers…" followed by a separate reply) from a real answer paragraph that
166
+ * merely OPENS with "Let me explain…": both match the same regex, and length
167
+ * cannot separate them (the observed narration was itself ≥200 chars). The one
168
+ * signal that DOES separate them is structural — did a `tool_use` follow this
169
+ * block in the model's actual message? A narration preamble is drafted, then
170
+ * the model ACTS (a tool call follows it); a terminal answer paragraph is not
171
+ * followed by any tool call. That per-block flag (`lastInMessage`) is computed
172
+ * upstream by `projectAssistantTextBlocks` (session-tail.ts) and, when the
173
+ * caller plumbs it through as `followedByToolUse`, we consult STRUCTURE, but
174
+ * asymmetrically:
175
+ * - present-and-FALSE (no tool_use followed) is purely additive — it can only
176
+ * RESCUE a block the opener regex would have mis-stripped (a terminal answer
177
+ * opening "Let me explain…"); it never drops a block the heuristic kept.
178
+ * - present-and-TRUE (a tool_use followed) is NOT taken as narration on its
179
+ * own: a substantial real-content paragraph can precede a tool call, so the
180
+ * strip is GATED by the substance check (narration only if it also matches
181
+ * the opener/trailer heuristic OR falls below the substantive floor). This
182
+ * is deliberately not "strictly additive" over the opener-only strip — it
183
+ * both rescues real content the old flag-alone path would have dropped and
184
+ * stays truncation-free.
185
+ * Where the structural flag is ABSENT for a block (legacy `string[]` caller, or
186
+ * the accumulator lost provenance) we fall back to the opener/trailer heuristic
187
+ * for that block.
188
+ *
189
+ * `followedByToolUse` is a parallel array aligned to `blocks` (index `i` ⇒
190
+ * `blocks[i]`); it is zipped BEFORE the empty-block filter so alignment holds
191
+ * even if the caller passes empty/whitespace blocks. `undefined` at an index
192
+ * (or a missing/short array) means "no reliable structural signal for this
193
+ * block" → opener-heuristic fallback.
162
194
  *
163
195
  * `blocks` are already trimmed/non-empty candidates (silent markers removed by
164
196
  * the caller's guards). Returns the chosen delivery text.
165
197
  */
166
- export function selectFlushDeliveryText(blocks: string[]): string {
198
+ export function selectFlushDeliveryText(
199
+ blocks: string[],
200
+ followedByToolUse?: ReadonlyArray<boolean | undefined>,
201
+ ): string {
167
202
  const candidates = blocks
168
- .map(b => b.trim())
169
- .filter(b => b.length > 0)
203
+ .map((b, i) => ({ text: b.trim(), followedByToolUse: followedByToolUse?.[i] }))
204
+ .filter(c => c.text.length > 0)
170
205
  if (candidates.length === 0) return ''
171
- if (candidates.length === 1) return candidates[0]
172
- const answer = candidates[candidates.length - 1]
206
+ if (candidates.length === 1) return candidates[0].text
207
+ const answer = candidates[candidates.length - 1].text
173
208
  const preceding = candidates.slice(0, -1)
174
209
  // Deliver only the terminal answer when every earlier block is
175
- // intent-narration (a short block, or a "Let me…/I'll…/I'm going to…" opener).
210
+ // intent-narration. Per block:
211
+ // - structural flag TRUE ⇒ a tool_use followed this block, but that alone
212
+ // is NOT sufficient to drop it: a substantial real-content paragraph can
213
+ // legitimately precede a tool call (the model writes a real answer, then
214
+ // calls a memory/verify tool, then a short wrap-up). So gate the
215
+ // structural strip with the substantive floor — narration only if the
216
+ // block ALSO looks like narration OR is below the substantive floor.
217
+ // - structural flag FALSE ⇒ present-and-false definitively overrides the
218
+ // opener regex: no tool_use followed, so it is a terminal-style block,
219
+ // never narration (#3237).
220
+ // - structural flag ABSENT ⇒ fall back to the opener/trailer heuristic.
176
221
  // Otherwise the earlier blocks carry real content — keep the full joined text
177
- // so a legitimate multi-block answer is never truncated to its last paragraph.
178
- const allNarration = preceding.every(isNarrationBlock)
179
- return allNarration ? answer : candidates.join('\n\n')
222
+ // so a legitimate multi-block answer is never truncated to its last
223
+ // paragraph (#3237).
224
+ const allNarration = preceding.every(c =>
225
+ c.followedByToolUse === true
226
+ ? isNarrationBlock(c.text) || c.text.trim().length < FLUSH_SUBSTANTIVE_MIN_CHARS
227
+ : c.followedByToolUse === false
228
+ ? false
229
+ : isNarrationBlock(c.text),
230
+ )
231
+ return allNarration ? answer : candidates.map(c => c.text).join('\n\n')
180
232
  }
181
233
 
182
234
  /**
@@ -244,6 +296,17 @@ export interface FlushDecisionInput {
244
296
  * is false — once the model has called reply / stream_reply the turn is
245
297
  * served and trailing terminal text is dropped (see `decideTurnFlush`). */
246
298
  capturedText: string[]
299
+ /** Optional per-block structural provenance, aligned to `capturedText`
300
+ * (index `i` describes `capturedText[i]`). `true` ⇒ a `tool_use` followed
301
+ * this text block in its assistant message (the draft-then-send narration
302
+ * signal — the negation of `projectAssistantTextBlocks`' `lastInMessage`).
303
+ * Consumed by `selectFlushDeliveryText` to separate a narration preamble from
304
+ * a real answer paragraph that merely opens with a narration phrase (#3237).
305
+ * When absent (legacy caller / lost provenance) the strip falls back to the
306
+ * opener/trailer heuristic. Present-and-false is additive (only rescues a
307
+ * mis-stripped terminal answer); present-and-true is gated by the substantive
308
+ * floor rather than trusted alone. */
309
+ capturedBlockMeta?: boolean[]
247
310
  /** Feature flag — defaults to true. Pass `false` to force skip everywhere. */
248
311
  flushEnabled?: boolean
249
312
  }
@@ -316,7 +379,10 @@ export function decideTurnFlush(input: FlushDecisionInput): FlushDecision {
316
379
  // blob (see `selectFlushDeliveryText`). The silent-marker / empty guards above
317
380
  // still run on the full `joined` string so a partly-silent turn is classified
318
381
  // correctly; only the DELIVERED text is narrowed to the answer.
319
- return { kind: 'flush', text: selectFlushDeliveryText(input.capturedText) }
382
+ return {
383
+ kind: 'flush',
384
+ text: selectFlushDeliveryText(input.capturedText, input.capturedBlockMeta),
385
+ }
320
386
  }
321
387
 
322
388
  /**
@@ -0,0 +1,76 @@
1
+ /**
2
+ * Shared restart-capability probe + loud-skip announcer for the
3
+ * restart/resume UAT scenarios.
4
+ *
5
+ * These scenarios (`jtbd-deliberate-restart-resumes-dm`,
6
+ * `jtbd-interrupted-turn-resumes-dm`, `jtbd-always-on-after-restart-dm`)
7
+ * can only exercise a REAL restart when the runner has NOPASSWD `sudo` and
8
+ * the `switchroom` CLI on PATH. On a sandboxed runner that lacks those,
9
+ * `vitest`'s `describe.skip` marks them skipped — but a bare skip reads as
10
+ * a plain green in a summarised board, which is exactly how the v0.18.32
11
+ * UAT canary over-claimed "#3315 restart/resume validated" when in fact
12
+ * these three self-skipped (#3334 item d).
13
+ *
14
+ * The durable fix has two parts. Part 2 (a runner that can actually
15
+ * restart) is infrastructure and is tracked separately. Part 1 — this file
16
+ * — makes the skip LOUD and unmistakable: a single-source-of-truth probe
17
+ * plus a banner printed to stderr at collection time, so a green that is
18
+ * really a skip can never be silently indistinguishable from a green that
19
+ * ran live.
20
+ */
21
+
22
+ import { execSync } from "node:child_process";
23
+
24
+ /**
25
+ * True only when the runner can drive a real agent restart: NOPASSWD
26
+ * `sudo` is available (the scenarios shell out to
27
+ * `sudo -n switchroom agent restart …`). A sandboxed CI runner returns
28
+ * false, which routes the scenario to `describe.skip`.
29
+ */
30
+ export function canShellSudo(): boolean {
31
+ try {
32
+ execSync("sudo -n true", { stdio: "ignore", timeout: 2_000 });
33
+ return true;
34
+ } catch {
35
+ return false;
36
+ }
37
+ }
38
+
39
+ /**
40
+ * Print a loud, unmissable banner to stderr when a restart/resume scenario
41
+ * is about to self-skip because the runner cannot exercise a real restart.
42
+ * Call this at module-collection time (top level of the scenario file) so
43
+ * the banner lands in CI stdout/stderr regardless of the reporter's
44
+ * skip-count rendering.
45
+ *
46
+ * The banner deliberately spells out that a GREEN here is a SKIP, not live
47
+ * proof — the exact confusion #3334 item d flags as "the dangerous one".
48
+ */
49
+ export function announceRestartSkip(scenarioTitle: string): void {
50
+ console.warn(
51
+ "\n" +
52
+ "════════════════════════════════════════════════════════════════════\n" +
53
+ " ⚠️ UAT SKIPPED — NOT LIVE-VALIDATED (restart capability absent)\n" +
54
+ ` scenario: ${scenarioTitle}\n` +
55
+ " reason: this runner has no NOPASSWD sudo + switchroom CLI, so\n" +
56
+ " it cannot exercise a real agent restart. The scenario\n" +
57
+ " self-skips. Its GREEN is a SKIP, not live proof.\n" +
58
+ " see: #3334 item (d) / #3315 — restart/resume assurance from\n" +
59
+ " this run rests on the unit/vitest layer, not this UAT.\n" +
60
+ "════════════════════════════════════════════════════════════════════\n",
61
+ );
62
+ }
63
+
64
+ /**
65
+ * Convenience: probe capability once and, when absent, emit the loud-skip
66
+ * banner. Returns the capability boolean so the caller can gate
67
+ * `describe` vs `describe.skip`:
68
+ *
69
+ * const sudoOk = restartCapableOrAnnounceSkip(TITLE);
70
+ * (sudoOk ? describe : describe.skip)(TITLE, () => { … });
71
+ */
72
+ export function restartCapableOrAnnounceSkip(scenarioTitle: string): boolean {
73
+ const ok = canShellSudo();
74
+ if (!ok) announceRestartSkip(scenarioTitle);
75
+ return ok;
76
+ }
@@ -89,7 +89,17 @@ const BG_DISPATCH_PROMPT =
89
89
  `brief reply saying you've kicked off the background worker so I can ` +
90
90
  `watch the progress feed.`;
91
91
 
92
- const WORKER_RUNNING_RE = /running\s*·/i;
92
+ // Single-worker RUNNING headers show NO literal "running" — they render
93
+ // `<elapsed> · <n> tools[ · <tok> tok][ · <model>]`
94
+ // (`tool-activity-summary.ts` renderActivityHeader, running branch); the
95
+ // word "running" only appears on the MULTI-worker combined header
96
+ // (`🛠 Workers · N running`). The old `/running\s*·/i` oracle matched only
97
+ // by accident when the multi-worker header happened to render (failed live
98
+ // 2026-07-18, ci-uat run 29634400115, against a healthy in-flight card).
99
+ // In-flight signal = either header shape's live metric, paired with the
100
+ // `not.toMatch(WORKER_DONE_RE)` terminal exclusion below — a skeleton that
101
+ // paints only `🛠 Worker · <name>` with no metric line still fails.
102
+ const WORKER_RUNNING_RE = /\brunning\b|·\s*\d+\s+tools?\b/i;
93
103
  const WORKER_DONE_RE = /finished\s*·\s*(completed|failed)/i;
94
104
 
95
105
  describe("uat: background sub-agent visibility (#709/#776/#782/#788)", () => {
@@ -117,9 +127,9 @@ describe("uat: background sub-agent visibility (#709/#776/#782/#788)", () => {
117
127
  expect(feed.messageId).toBeGreaterThan(0);
118
128
  expect(feed.text).toMatch(WORKER_FEED_RE);
119
129
 
120
- // AC-2 step 1: feed body MUST show "running ·" (the in-flight
121
- // status), NOT the terminal "finished ·" — the worker hasn't
122
- // completed yet.
130
+ // AC-2 step 1: feed body MUST show an in-flight signal (live
131
+ // metric line, or multi-worker "N running"), NOT the terminal
132
+ // "finished ·" — the worker hasn't completed yet.
123
133
  expect(feed.text).toMatch(WORKER_RUNNING_RE);
124
134
  expect(feed.text).not.toMatch(WORKER_DONE_RE);
125
135
 
@@ -123,8 +123,18 @@ describe("uat: bridge-flap resilience — agent stays responsive, gateway does n
123
123
  `overall deadline hit before DM ${i} — earlier turns were too slow`,
124
124
  ).toBeGreaterThan(0);
125
125
 
126
+ // Skip empty-text observations: the Bot API cannot send an
127
+ // empty text message (Telegram rejects it), so an empty-text
128
+ // fromBot observation is by construction a SERVICE message —
129
+ // e.g. the `[pinned_message]` event from the progress-card pin.
130
+ // One of those latched here as "the reply" on 2026-07-18
131
+ // (ci-uat run 29634400115: pinChatMessage 06:54:36.420Z → rx
132
+ // [pinned_message] 06:54:37.012Z, exactly at the DM-2 failure).
133
+ // A genuinely eaten turn_end still fails loudly: no non-empty
134
+ // reply arrives and this expectMessage times out.
126
135
  const reply = await sc.expectMessage(
127
- (m: ObservedMessage) => m.fromBot && !m.edited,
136
+ (m: ObservedMessage) =>
137
+ m.fromBot && !m.edited && m.text.length > 0,
128
138
  { from: "bot", timeout: remaining },
129
139
  );
130
140
  expect(