switchroom 0.18.15 → 0.18.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/agent-scheduler/index.js +16 -0
  2. package/dist/auth-broker/index.js +445 -10
  3. package/dist/cli/notion-write-pretool.mjs +16 -0
  4. package/dist/cli/switchroom.js +654 -479
  5. package/dist/host-control/main.js +20 -1
  6. package/dist/vault/approvals/kernel-server.js +16 -0
  7. package/dist/vault/broker/server.js +16 -0
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/bridge/bridge.ts +7 -1
  11. package/telegram-plugin/dist/bridge/bridge.js +26 -1
  12. package/telegram-plugin/dist/gateway/gateway.js +1758 -661
  13. package/telegram-plugin/dist/server.js +26 -1
  14. package/telegram-plugin/draft-stream.ts +78 -3
  15. package/telegram-plugin/fleet-fallback-resume.ts +26 -3
  16. package/telegram-plugin/gateway/approval-hold.ts +49 -0
  17. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +64 -22
  18. package/telegram-plugin/gateway/effort-command.ts +9 -7
  19. package/telegram-plugin/gateway/gateway.ts +627 -291
  20. package/telegram-plugin/gateway/linear-activity.ts +20 -4
  21. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  22. package/telegram-plugin/gateway/model-command.ts +96 -18
  23. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  24. package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
  25. package/telegram-plugin/gateway/session-model-file.ts +141 -172
  26. package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
  27. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
  28. package/telegram-plugin/litellm-local-notice.ts +189 -0
  29. package/telegram-plugin/llm-error-present.ts +436 -0
  30. package/telegram-plugin/operator-events.ts +7 -1
  31. package/telegram-plugin/permission-title.ts +172 -10
  32. package/telegram-plugin/premium-recovery.ts +101 -0
  33. package/telegram-plugin/quota-watch.ts +16 -4
  34. package/telegram-plugin/raw-error-scrub.ts +73 -0
  35. package/telegram-plugin/retry-api-call.ts +8 -2
  36. package/telegram-plugin/runtime-metrics.ts +16 -0
  37. package/telegram-plugin/send-gate-degraded.test.ts +161 -8
  38. package/telegram-plugin/send-gate-observability.test.ts +140 -0
  39. package/telegram-plugin/send-gate-observability.ts +65 -20
  40. package/telegram-plugin/send-gate.test.ts +143 -1
  41. package/telegram-plugin/send-gate.ts +246 -23
  42. package/telegram-plugin/session-tail.ts +16 -0
  43. package/telegram-plugin/shared/local-time.ts +69 -0
  44. package/telegram-plugin/stream-controller.ts +143 -20
  45. package/telegram-plugin/stream-reply-handler.ts +12 -2
  46. package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
  47. package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
  48. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  49. package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
  50. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  51. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  52. package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
  53. package/telegram-plugin/tests/flood-windows-persistence.test.ts +5 -4
  54. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  55. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  56. package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
  57. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  58. package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
  59. package/telegram-plugin/tests/model-command.test.ts +84 -1
  60. package/telegram-plugin/tests/permission-title.test.ts +167 -4
  61. package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
  62. package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
  63. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  64. package/telegram-plugin/tests/reaction-gate-routing.test.ts +8 -3
  65. package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
  66. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  67. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  68. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  69. package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
  70. package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
  71. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
  72. package/telegram-plugin/tests/worker-activity-feed.test.ts +212 -2
  73. package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
  74. package/telegram-plugin/tier-downgrade.ts +198 -0
  75. package/telegram-plugin/tool-activity-summary.ts +99 -0
  76. package/telegram-plugin/worker-activity-feed.ts +543 -368
@@ -0,0 +1,380 @@
1
+ /**
2
+ * llm-error-present.test.ts — outcome-asserting tests for the humanized,
3
+ * cross-surface-deduped LLM-error surfacing (no raw JSON passthrough).
4
+ *
5
+ * Each block is written to FAIL against pre-fix behaviour:
6
+ * - golden fan-out: pre-fix, the raw synthetic-error text fanned out to the
7
+ * reply + done-card surfaces (≥2 messages, raw bytes) — this asserts EXACTLY
8
+ * ONE JSON-free message across all three.
9
+ * - operator-card sanitize: pre-fix, renderOperatorEvent embedded the raw
10
+ * detail verbatim — this asserts the bytes are gone.
11
+ */
12
+
13
+ import { describe, it, expect, beforeEach } from 'vitest'
14
+ import {
15
+ parseLlmError,
16
+ renderLlmError,
17
+ stripRawErrorBytes,
18
+ extractRequestId,
19
+ formatResetLocal,
20
+ decideErrorSurface,
21
+ isActionableKind,
22
+ ErrorPresenceGate,
23
+ errorPresenceGate,
24
+ ERROR_COLLAPSE_WINDOW_MS,
25
+ type LlmErrorRetryState,
26
+ } from '../llm-error-present.js'
27
+ import { truncateDetailPreservingRequestId } from '../raw-error-scrub.js'
28
+ import { projectTranscriptLine, detectErrorInTranscriptLine } from '../session-tail.js'
29
+ import { renderOperatorEvent, type OperatorEvent } from '../operator-events.js'
30
+
31
+ // A raw byte-blob every surface must scrub.
32
+ const RAW_BYTES = `b'{"type":"error","error":{"type":"rate_limit_error","message":"rate limit"},"request_id":"req_abc123"}'`
33
+
34
+ function assertNoRawBytes(text: string): void {
35
+ expect(text).not.toContain("b'{")
36
+ expect(text).not.toContain('{"type":"error"')
37
+ expect(text.toLowerCase()).not.toContain('api error:')
38
+ }
39
+
40
+ // ─── stripRawErrorBytes ──────────────────────────────────────────────────────
41
+
42
+ describe('stripRawErrorBytes', () => {
43
+ it('removes the Python byte-blob, JSON error object, and API Error prefix', () => {
44
+ const raw = `API Error: 429 Server is temporarily limiting requests · ${RAW_BYTES}`
45
+ const out = stripRawErrorBytes(raw)
46
+ assertNoRawBytes(out)
47
+ expect(out).toContain('Server is temporarily limiting requests')
48
+ })
49
+
50
+ it('passes a clean human string through unchanged (modulo whitespace)', () => {
51
+ expect(stripRawErrorBytes('Invalid API key')).toBe('Invalid API key')
52
+ expect(stripRawErrorBytes('Rate limit exceeded')).toBe('Rate limit exceeded')
53
+ expect(stripRawErrorBytes('')).toBe('')
54
+ })
55
+
56
+ it('strips a bare JSON blob down to empty', () => {
57
+ expect(stripRawErrorBytes('{"type":"error","error":{"type":"overloaded_error"}}')).toBe('')
58
+ })
59
+ })
60
+
61
+ // ─── extractRequestId ────────────────────────────────────────────────────────
62
+
63
+ describe('extractRequestId', () => {
64
+ it('pulls an Anthropic request_id from a raw JSON string', () => {
65
+ expect(extractRequestId(RAW_BYTES)).toBe('req_abc123')
66
+ expect(extractRequestId('no id here')).toBeUndefined()
67
+ })
68
+ })
69
+
70
+ // ─── parser table ────────────────────────────────────────────────────────────
71
+
72
+ describe('parseLlmError — classification table', () => {
73
+ it('rate_limit (Anthropic account throttle)', () => {
74
+ const p = parseLlmError(
75
+ "This request would exceed your account's rate limit. Please try again later.",
76
+ )
77
+ expect(p.kind).toBe('rate_limit')
78
+ expect(p.source).toBe('anthropic')
79
+ assertNoRawBytes(p.coreText)
80
+ })
81
+
82
+ it('overload_529', () => {
83
+ const p = parseLlmError('overloaded_error: Anthropic 529 overloaded')
84
+ expect(p.kind).toBe('overload_529')
85
+ expect(p.source).toBe('anthropic')
86
+ assertNoRawBytes(p.coreText)
87
+ })
88
+
89
+ it('quota_wall with a parsed reset', () => {
90
+ const p = parseLlmError("You've hit your limit · resets 8:50am (Australia/Melbourne)")
91
+ expect(p.kind).toBe('quota_wall')
92
+ expect(p.source).toBe('anthropic')
93
+ expect(p.resetAt).toBeInstanceOf(Date)
94
+ expect(p.terminal).toBe(true)
95
+ assertNoRawBytes(p.coreText)
96
+ })
97
+
98
+ it('litellm-local proxy 429', () => {
99
+ const p = parseLlmError(
100
+ 'litellm.RateLimitError: Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241',
101
+ )
102
+ expect(p.kind).toBe('rate_limit')
103
+ expect(p.source).toBe('litellm-local')
104
+ assertNoRawBytes(p.coreText)
105
+ })
106
+
107
+ it('auth (credentials expired)', () => {
108
+ const p = parseLlmError('authentication_error: OAuth token expired, please refresh')
109
+ expect(p.kind).toBe('auth')
110
+ expect(p.source).toBe('anthropic')
111
+ expect(p.terminal).toBe(true)
112
+ expect(p.autoRetrying).toBe(false)
113
+ })
114
+
115
+ it('network', () => {
116
+ const p = parseLlmError('fetch failed: ECONNREFUSED getaddrinfo ENOTFOUND api.anthropic.com')
117
+ expect(p.kind).toBe('transient')
118
+ expect(p.source).toBe('network')
119
+ assertNoRawBytes(p.coreText)
120
+ })
121
+
122
+ it('unknown falls through cleanly', () => {
123
+ const p = parseLlmError('something entirely unrecognizable happened')
124
+ expect(p.kind).toBe('unknown')
125
+ assertNoRawBytes(p.coreText)
126
+ })
127
+
128
+ it('extracts model + requestId, and coreText is JSON-free even from a raw line', () => {
129
+ const raw = `Server is temporarily limiting requests · model=claude-opus-4-8 · ${RAW_BYTES}`
130
+ const p = parseLlmError(raw)
131
+ expect(p.kind).toBe('rate_limit')
132
+ expect(p.model).toBe('claude-opus-4-8')
133
+ expect(p.requestId).toBe('req_abc123')
134
+ assertNoRawBytes(p.coreText)
135
+ })
136
+ })
137
+
138
+ // ─── local-time render ───────────────────────────────────────────────────────
139
+
140
+ describe('renderLlmError / formatResetLocal — local-time rendering', () => {
141
+ const tz = 'Australia/Melbourne'
142
+ // Pin now to a fixed instant. 2026-07-13T06:14:00Z = 4:14pm AEST (winter, +10).
143
+ const now = new Date('2026-07-13T06:14:00Z')
144
+
145
+ it('renders the reset in local wall-clock + relative tail, DST-correct', () => {
146
+ // resetAt = 2026-07-13T06:52:00Z = 4:52pm AEST, ~38m out.
147
+ const reset = new Date('2026-07-13T06:52:00Z')
148
+ const line = formatResetLocal(reset, tz, now)
149
+ expect(line).toContain('4:52pm')
150
+ expect(line).toContain('AEST')
151
+ expect(line).toContain('~in 38m')
152
+ })
153
+
154
+ it('renderLlmError produces a JSON-free card with the local reset', () => {
155
+ const parsed = parseLlmError(
156
+ `Server is temporarily limiting requests · resets 4:52pm (Australia/Melbourne) · ${RAW_BYTES}`,
157
+ )
158
+ const { text } = renderLlmError(parsed, 'gymbro', tz, now)
159
+ assertNoRawBytes(text)
160
+ expect(text).toContain('gymbro')
161
+ expect(text).toContain('AEST')
162
+ })
163
+
164
+ it('auth + quota_wall cards carry action buttons', () => {
165
+ const auth = renderLlmError(parseLlmError('authentication_error: token expired'), 'a', tz, now)
166
+ expect(auth.keyboard?.inline_keyboard.flat().some((b) => b.callback_data?.includes('reauth'))).toBe(true)
167
+ const quota = renderLlmError(
168
+ parseLlmError("You've hit your limit · resets 5pm"),
169
+ 'a',
170
+ tz,
171
+ now,
172
+ )
173
+ expect(quota.keyboard?.inline_keyboard.flat().length).toBeGreaterThan(0)
174
+ })
175
+ })
176
+
177
+ // ─── dedup key ───────────────────────────────────────────────────────────────
178
+
179
+ describe('ErrorPresenceGate — dedup key semantics', () => {
180
+ let gate: ErrorPresenceGate
181
+ beforeEach(() => {
182
+ gate = new ErrorPresenceGate()
183
+ })
184
+
185
+ it('same requestId → collapses to 1', () => {
186
+ const p = { kind: 'rate_limit' as const, requestId: 'req_x' }
187
+ const now = 1_000_000
188
+ const k = gate.keyFor(p, 'agent', now)
189
+ expect(gate.claim(k, now)).toBe(true)
190
+ expect(gate.claim(gate.keyFor(p, 'agent', now + 5_000), now + 5_000)).toBe(false)
191
+ })
192
+
193
+ it('different requestId, same kind → 2 distinct claims', () => {
194
+ const now = 1_000_000
195
+ expect(gate.claim(gate.keyFor({ kind: 'rate_limit', requestId: 'a' }, 'g', now), now)).toBe(true)
196
+ expect(gate.claim(gate.keyFor({ kind: 'rate_limit', requestId: 'b' }, 'g', now), now)).toBe(true)
197
+ })
198
+
199
+ it('no requestId, same kind: <window → 1, >window → 2', () => {
200
+ const now = 5 * ERROR_COLLAPSE_WINDOW_MS
201
+ const p = { kind: 'rate_limit' as const }
202
+ expect(gate.claim(gate.keyFor(p, 'g', now), now)).toBe(true)
203
+ // Within window (same bucket) → suppressed.
204
+ const within = now + ERROR_COLLAPSE_WINDOW_MS - 1
205
+ expect(gate.claim(gate.keyFor(p, 'g', within), within)).toBe(false)
206
+ // Past window (next bucket) → new claim.
207
+ const past = now + ERROR_COLLAPSE_WINDOW_MS + 1
208
+ expect(gate.claim(gate.keyFor(p, 'g', past), past)).toBe(true)
209
+ })
210
+ })
211
+
212
+ // ─── auto-retry silence ──────────────────────────────────────────────────────
213
+ //
214
+ // UNIT test of the module's own auto-retry-silence CONTRACT: given a retryState
215
+ // with attempt<max, a transient error is suppressed. This is NOT exercised by
216
+ // the production gateway wiring (it calls `parseLlmError(detail)` without a
217
+ // retryState, and session-tail already drops in-flight transients upstream — see
218
+ // the note in decideErrorSurface). These cases pin the module guarantee for any
219
+ // caller that DOES thread retryState; they do not imply the gateway does.
220
+
221
+ describe('decideErrorSurface — auto-retry silence (module contract, retryState-driven)', () => {
222
+ beforeEach(() => errorPresenceGate.reset())
223
+
224
+ it('a transient 529 mid-retry (attempt<max) is silenced on all surfaces', () => {
225
+ const retry: LlmErrorRetryState = { retryAttempt: 1, maxRetries: 5 }
226
+ const p = parseLlmError('overloaded_error 529 overloaded', retry)
227
+ expect(p.autoRetrying).toBe(true)
228
+ expect(p.terminal).toBe(false)
229
+ expect(decideErrorSurface(p, 'g', { claim: true })).toBe('suppress')
230
+ })
231
+
232
+ it('the same 529 at attempt>=max renders exactly one card', () => {
233
+ const retry: LlmErrorRetryState = { retryAttempt: 5, maxRetries: 5 }
234
+ const p = parseLlmError('overloaded_error 529 overloaded', retry)
235
+ expect(p.autoRetrying).toBe(false)
236
+ expect(p.terminal).toBe(true)
237
+ expect(decideErrorSurface(p, 'g', { claim: true })).toBe('render')
238
+ // A second surface for the same terminal error collapses.
239
+ expect(decideErrorSurface(p, 'g', { claim: true })).toBe('suppress')
240
+ })
241
+ })
242
+
243
+ // ─── actionable never suppressed ─────────────────────────────────────────────
244
+
245
+ describe('decideErrorSurface — actionable kinds never suppressed', () => {
246
+ beforeEach(() => errorPresenceGate.reset())
247
+
248
+ it('auth + quota_wall always render even inside an active collapse window', () => {
249
+ const auth = parseLlmError('authentication_error: credentials expired')
250
+ const quota = parseLlmError("You've hit your limit · resets 5pm")
251
+ expect(isActionableKind(auth.kind)).toBe(true)
252
+ expect(isActionableKind(quota.kind)).toBe(true)
253
+ // Pre-claim the same-kind window key to simulate a busy collapse window.
254
+ const now = 2_000_000
255
+ errorPresenceGate.claim(errorPresenceGate.keyFor(auth, 'g', now), now)
256
+ errorPresenceGate.claim(errorPresenceGate.keyFor(quota, 'g', now), now)
257
+ expect(decideErrorSurface(auth, 'g', { claim: true, now })).toBe('render')
258
+ expect(decideErrorSurface(quota, 'g', { claim: true, now })).toBe('render')
259
+ })
260
+ })
261
+
262
+ // ─── golden fan-out ──────────────────────────────────────────────────────────
263
+
264
+ describe('golden fan-out — one 429 line → exactly one JSON-free user message', () => {
265
+ beforeEach(() => errorPresenceGate.reset())
266
+
267
+ // The v2.1.x synthetic-assistant usage/rate-limit shape: the raw error text
268
+ // lives inside message.content[].text, with the byte-blob attached.
269
+ const line = JSON.stringify({
270
+ type: 'assistant',
271
+ message: {
272
+ role: 'assistant',
273
+ content: [
274
+ {
275
+ type: 'text',
276
+ text: `Server is temporarily limiting requests (not your usage limit) · resets 4:52pm (Australia/Melbourne) · ${RAW_BYTES}`,
277
+ },
278
+ ],
279
+ },
280
+ error: 'rate_limit',
281
+ isApiErrorMessage: true,
282
+ apiErrorStatus: 429,
283
+ })
284
+
285
+ it('starves the reply + done-card surfaces and renders one operator card', () => {
286
+ const userMessages: string[] = []
287
+
288
+ // Surfaces #1 (done-card recap) and #3 (reply passthrough) both derive from
289
+ // the projected transcript `text` events. Post-fix: NO text/thinking event
290
+ // is projected for an isApiErrorMessage line, so neither surface can relay
291
+ // the raw bytes as the turn's answer.
292
+ const projected = projectTranscriptLine(line)
293
+ for (const ev of projected) {
294
+ if (ev.kind === 'text') userMessages.push(`reply:${ev.text}`)
295
+ }
296
+ expect(projected.some((e) => e.kind === 'text')).toBe(false)
297
+
298
+ // Surface #2 (operator card): the SAME line is independently classified and
299
+ // rendered as the ONE humanized card. This mirrors the production gateway
300
+ // call exactly — `parseLlmError(detail)` with NO retryState (session-tail
301
+ // has already dropped any in-flight transient before forwarding, so the
302
+ // event that reaches the gateway is terminal by construction).
303
+ const detected = detectErrorInTranscriptLine(line)
304
+ expect(detected).not.toBeNull()
305
+ expect(detected!.terminal).toBe(true)
306
+ const parsed = parseLlmError(detected!.detail)
307
+ if (decideErrorSurface(parsed, 'gymbro', { claim: true }) === 'render') {
308
+ userMessages.push(renderLlmError(parsed, 'gymbro', 'Australia/Melbourne').text)
309
+ }
310
+
311
+ // EXACTLY ONE user-facing message, and it carries no raw bytes.
312
+ expect(userMessages).toHaveLength(1)
313
+ assertNoRawBytes(userMessages[0])
314
+ })
315
+ })
316
+
317
+ // ─── operator-card sanitize (surface #2 regression) ──────────────────────────
318
+
319
+ describe('renderOperatorEvent — never relays raw error bytes', () => {
320
+ function ev(kind: OperatorEvent['kind'], detail: string): OperatorEvent {
321
+ return { kind, agent: 'g', detail, suggestedActions: [], firstSeenAt: new Date() }
322
+ }
323
+
324
+ it('strips raw bytes from a rate-limited detail', () => {
325
+ const { text } = renderOperatorEvent(ev('rate-limited', `Rate limited · ${RAW_BYTES}`))
326
+ assertNoRawBytes(text)
327
+ expect(text).toContain('Rate limited')
328
+ })
329
+
330
+ it('strips raw bytes from an unknown-4xx detail', () => {
331
+ const { text } = renderOperatorEvent(ev('unknown-4xx', `API Error: 400 bad · ${RAW_BYTES}`))
332
+ assertNoRawBytes(text)
333
+ })
334
+
335
+ it('leaves a clean detail intact', () => {
336
+ const { text } = renderOperatorEvent(ev('credentials-expired', 'token expired 2026-04-01'))
337
+ expect(text).toContain('token expired 2026-04-01')
338
+ })
339
+ })
340
+
341
+ // ─── request_id survives the 1000-char bridge truncation (MEDIUM) ─────────────
342
+
343
+ describe('truncateDetailPreservingRequestId — exact dedup key survives truncation', () => {
344
+ beforeEach(() => errorPresenceGate.reset())
345
+
346
+ // A realistic long Anthropic rate-limit body whose request_id sits PAST char
347
+ // 1000 (the bridge's OPERATOR_EVENT_DETAIL_MAX). `filler` pads the human part.
348
+ function longRateLimitLine(requestId: string): string {
349
+ const filler = 'Server is temporarily limiting requests (not your usage limit). '.repeat(30)
350
+ return `${filler} b'{"type":"error","error":{"type":"rate_limit_error","message":"rate limited"},"request_id":"${requestId}"}'`
351
+ }
352
+
353
+ it('preserves the request_id that a naive slice(0,1000) would drop', () => {
354
+ const line = longRateLimitLine('req_deadbeef01')
355
+ expect(line.length).toBeGreaterThan(1000)
356
+ // A naive slice loses the trailing id...
357
+ expect(extractRequestId(line.slice(0, 1000))).toBeUndefined()
358
+ // ...but the preserving truncation keeps it, within budget.
359
+ const truncated = truncateDetailPreservingRequestId(line, 1000)
360
+ expect(truncated.length).toBeLessThanOrEqual(1000)
361
+ expect(extractRequestId(truncated)).toBe('req_deadbeef01')
362
+ })
363
+
364
+ it('two DISTINCT request_ids past char 1000 render TWO messages (exact key still fires)', () => {
365
+ const now = 9_000_000
366
+ const messages: string[] = []
367
+ for (const rid of ['req_aaaaaaaa', 'req_bbbbbbbb']) {
368
+ // Simulate the bridge hop: truncate-preserving, THEN parse gateway-side.
369
+ const detail = truncateDetailPreservingRequestId(longRateLimitLine(rid), 1000)
370
+ const parsed = parseLlmError(detail)
371
+ expect(parsed.requestId).toBe(rid) // exact key preserved, not coarse fallback
372
+ if (decideErrorSurface(parsed, 'gymbro', { claim: true, now }) === 'render') {
373
+ messages.push(renderLlmError(parsed, 'gymbro', 'UTC', new Date(now)).text)
374
+ }
375
+ }
376
+ // Pre-fix (id truncated away) both fall to the SAME `${kind}:${agent}:${bucket}`
377
+ // key and collapse to ONE. With the id preserved, the exact keys differ → TWO.
378
+ expect(messages).toHaveLength(2)
379
+ })
380
+ })
@@ -16,6 +16,9 @@
16
16
  import { describe, it, expect, beforeAll } from "vitest";
17
17
  import {
18
18
  parseModelCommand,
19
+ planModelCommand,
20
+ isModelCommandBusy,
21
+ modelCommandReceiptLine,
19
22
  handleModelCommand,
20
23
  isValidModelArg,
21
24
  isSrModel,
@@ -203,7 +206,7 @@ describe("handleModelCommand — set", () => {
203
206
  const reply = await handleModelCommand({ kind: "set", model: "opus" }, deps);
204
207
  expect(calls).toEqual([{ agent: "klanker", command: "/model opus" }]);
205
208
  expect(reply.text).toContain("<pre>⏺ Set model to sonnet</pre>");
206
- expect(reply.text).toContain("persists across restarts, deploys, and crashes");
209
+ expect(reply.text).toContain("lasts until the agent’s next restart");
207
210
  expect(reply.html).toBe(true);
208
211
  // A verified confirmation records the live model so /status stays honest
209
212
  // (bug 1: the typed path never recorded the switch before).
@@ -1374,3 +1377,83 @@ describe("isOfflineTrustedModelToken (#3042 blocker 2a)", () => {
1374
1377
  expect(isOfflineTrustedModelToken(alias.toUpperCase())).toBe(true);
1375
1378
  });
1376
1379
  });
1380
+
1381
+ // ---------------------------------------------------------------------------
1382
+ // #3177 — a typed /model must NEVER be swallowed without trace when sent
1383
+ // mid-turn. The routing decision folds BOTH busy signals, and every parsed
1384
+ // shape maps to a visible action (menu / queue / apply — never "do nothing").
1385
+ // ---------------------------------------------------------------------------
1386
+ describe("#3177 typed /model never silently swallowed mid-turn", () => {
1387
+ describe("isModelCommandBusy folds both busy signals", () => {
1388
+ it("is busy when the turn atom is set", () => {
1389
+ expect(isModelCommandBusy({ currentTurnActive: true, turnInFlight: false })).toBe(true);
1390
+ });
1391
+
1392
+ it("is busy when only the authoritative delivery-machine/approval gate is set", () => {
1393
+ // THE SWALLOW REGRESSION: the pre-fix handler gated on `currentTurn !==
1394
+ // null` ALONE. A session busy by the delivery-machine / pending-approval
1395
+ // gate while the turn atom is cleared (the recovered-late / premature-
1396
+ // turn-end window) read as idle, so the switch injected into a busy pane
1397
+ // and was swallowed as literal text. Folding turnInFlight closes it.
1398
+ expect(isModelCommandBusy({ currentTurnActive: false, turnInFlight: true })).toBe(true);
1399
+ });
1400
+
1401
+ it("is idle only when BOTH signals are clear", () => {
1402
+ expect(isModelCommandBusy({ currentTurnActive: false, turnInFlight: false })).toBe(false);
1403
+ });
1404
+ });
1405
+
1406
+ describe("planModelCommand routes every shape to a visible action", () => {
1407
+ const busyByAtom = { currentTurnActive: true, turnInFlight: false, menuEnabled: true };
1408
+ const busyByGate = { currentTurnActive: false, turnInFlight: true, menuEnabled: true };
1409
+ const idle = { currentTurnActive: false, turnInFlight: false, menuEnabled: true };
1410
+
1411
+ it("QUEUES a set while busy by the turn atom (ack, not silent inject)", () => {
1412
+ const d = planModelCommand({ kind: "set", model: "opus" }, busyByAtom);
1413
+ expect(d).toEqual({ kind: "queue", target: "opus" });
1414
+ });
1415
+
1416
+ it("QUEUES a set while busy by the delivery-machine/approval gate ONLY — the swallow case", () => {
1417
+ // Sabotage-verify: revert the fix (gate on currentTurnActive alone) and
1418
+ // this flips to { kind: 'apply' } → the direct-inject swallow returns.
1419
+ const d = planModelCommand({ kind: "set", model: "opus" }, busyByGate);
1420
+ expect(d).toEqual({ kind: "queue", target: "opus" });
1421
+ });
1422
+
1423
+ it("expands sr-* aliases into the queued target token", () => {
1424
+ const d = planModelCommand({ kind: "set", model: "flash" }, busyByAtom);
1425
+ expect(d).toEqual({ kind: "queue", target: "sr-gemini-2.5-flash" });
1426
+ });
1427
+
1428
+ it("APPLIES a set immediately when idle by both signals", () => {
1429
+ const d = planModelCommand({ kind: "set", model: "opus" }, idle);
1430
+ expect(d).toEqual({ kind: "apply", parsed: { kind: "set", model: "opus" } });
1431
+ });
1432
+
1433
+ it("renders the MENU for bare /model when the picker is enabled (even mid-turn)", () => {
1434
+ expect(planModelCommand({ kind: "show" }, busyByGate)).toEqual({ kind: "menu" });
1435
+ });
1436
+
1437
+ it("APPLIES the text show path when the picker is disabled", () => {
1438
+ const d = planModelCommand({ kind: "show" }, { ...idle, menuEnabled: false });
1439
+ expect(d).toEqual({ kind: "apply", parsed: { kind: "show" } });
1440
+ });
1441
+
1442
+ it("APPLIES help so a bad arg still gets an explicit reply", () => {
1443
+ const parsed = { kind: "help", reason: "not a valid model name: !!" } as const;
1444
+ expect(planModelCommand(parsed, busyByGate)).toEqual({ kind: "apply", parsed });
1445
+ });
1446
+ });
1447
+
1448
+ describe("modelCommandReceiptLine — the durable, greppable entry trace", () => {
1449
+ it("stamps agent, kind, arg, and busy for a typed set", () => {
1450
+ const line = modelCommandReceiptLine("finn", { kind: "set", model: "opus" }, true);
1451
+ expect(line).toBe("telegram gateway: gw /model received agent=finn kind=set arg=opus busy=true");
1452
+ });
1453
+
1454
+ it("marks the show + help forms with placeholder args", () => {
1455
+ expect(modelCommandReceiptLine("finn", { kind: "show" }, false)).toContain("kind=show arg=(show)");
1456
+ expect(modelCommandReceiptLine("finn", { kind: "help" }, false)).toContain("kind=help arg=(help)");
1457
+ });
1458
+ });
1459
+ });
@@ -156,25 +156,188 @@ describe('formatPermissionCardBody', () => {
156
156
  expect(body).toContain('why: _listing temp files_')
157
157
  })
158
158
 
159
- test('shows "not provided" when no caller reason is present (never the description)', () => {
159
+ // #3167: no caller reason → an honest synthesized `context:` line built
160
+ // from the tool's salient input, NEVER a bare "why: not provided" and never
161
+ // the static schema description. The distinct `context:` label keeps the
162
+ // agent's omission of a rationale visible.
163
+ test('synthesizes a context line when no caller reason is present (never "not provided" / the description)', () => {
160
164
  const body = formatPermissionCardBody({
161
165
  toolName: 'Bash',
162
166
  inputPreview: JSON.stringify({ command: 'ls /tmp' }),
163
167
  description: 'Run a shell command on the host.',
164
168
  agentName: 'gymbro',
165
169
  })
166
- expect(body).toContain('why: _not provided_')
170
+ expect(body).not.toContain('not provided')
167
171
  expect(body).not.toContain('Run a shell command')
172
+ expect(body).toContain('context: _command: ls /tmp_')
168
173
  })
169
174
 
170
- test('shows "not provided" when caller reason is whitespace only', () => {
175
+ test('synthesizes context when the caller reason is whitespace only', () => {
171
176
  const body = formatPermissionCardBody({
172
177
  toolName: 'Bash',
173
178
  inputPreview: JSON.stringify({ command: 'ls /tmp', reason: ' \n ' }),
174
179
  description: 'Run a shell command.',
175
180
  agentName: 'gymbro',
176
181
  })
177
- expect(body).toContain('why: _not provided_')
182
+ expect(body).not.toContain('not provided')
183
+ expect(body).toContain('context: _command: ls /tmp_')
184
+ })
185
+
186
+ // #3167 root case: the `reply` tool (and react/edit_message/…) carries NO
187
+ // `reason` argument, so its cards used to render a contentless
188
+ // "🔐 Clerk wants to reply / why: not provided". Now the reply text is
189
+ // synthesized onto a `context:` line so the operator has something to judge.
190
+ test('reply (no reason arg) synthesizes the reply text as context, not "not provided"', () => {
191
+ const body = formatPermissionCardBody({
192
+ toolName: 'mcp__switchroom-telegram__reply',
193
+ inputPreview: JSON.stringify({
194
+ chat_id: '12345',
195
+ text: "On it — pulling yesterday's GitHub activity now.",
196
+ format: 'html',
197
+ disable_notification: true,
198
+ }),
199
+ description: 'Reply on Telegram.',
200
+ agentName: 'clerk',
201
+ })
202
+ const lines = body.split('\n')
203
+ expect(lines[0]).toBe('🔐 **Clerk** wants to reply')
204
+ expect(body).not.toContain('not provided')
205
+ // Salient field surfaced (truncated); id/routing/formatting noise stripped.
206
+ expect(body).toContain('context: _text: On it — pulling')
207
+ expect(body).toMatch(/context: _text: On it — pulling[^_]*…_/)
208
+ expect(body).not.toContain('chat_id')
209
+ expect(body).not.toContain('disable_notification')
210
+ expect(body).not.toContain('format')
211
+ })
212
+
213
+ test('a reason ON a reply card still renders as why:, not context: (#3167)', () => {
214
+ const body = formatPermissionCardBody({
215
+ toolName: 'mcp__switchroom-telegram__reply',
216
+ inputPreview: JSON.stringify({
217
+ chat_id: '12345',
218
+ text: 'done',
219
+ reason: 'answering the operator’s status question',
220
+ }),
221
+ description: 'Reply on Telegram.',
222
+ agentName: 'clerk',
223
+ })
224
+ expect(body).toContain('why: _answering the operator’s status question_')
225
+ expect(body).not.toContain('context:')
226
+ })
227
+
228
+ test('synthesized context redacts secrets in the salient input (#3167)', () => {
229
+ const fakeToken = 'sk-ant-' + 'api03-' + 'B'.repeat(48)
230
+ const body = formatPermissionCardBody({
231
+ toolName: 'mcp__switchroom-telegram__reply',
232
+ inputPreview: JSON.stringify({ chat_id: '1', text: `here is the key: ${fakeToken}` }),
233
+ description: 'Reply on Telegram.',
234
+ agentName: 'clerk',
235
+ })
236
+ expect(body).toContain('context:')
237
+ expect(body).not.toContain(fakeToken)
238
+ })
239
+
240
+ test('falls back to the natural action when the input exposes nothing salient (#3167)', () => {
241
+ const body = formatPermissionCardBody({
242
+ toolName: 'ExitPlanMode',
243
+ inputPreview: undefined,
244
+ description: 'Exit plan mode.',
245
+ agentName: 'clerk',
246
+ })
247
+ expect(body).not.toContain('not provided')
248
+ expect(body).toContain('context: _exit plan mode_')
249
+ })
250
+
251
+ // #3167 review (HIGH/MEDIUM): the salient-value redactor must catch secrets
252
+ // the bare-value path missed. redact() excludes generic_high_entropy and the
253
+ // contextual kv detectors need a `key=value` shape — so a prefixless token
254
+ // under a credential-shaped key leaked verbatim. These FAIL pre-fix.
255
+ test('hard-masks a prefixless high-entropy value under a `token` key (#3167)', () => {
256
+ // No sk-/ghp-/JWT prefix — shape detection alone cannot catch it; only the
257
+ // key-name signal ("token") does.
258
+ const bare = 'Zx8Kq2Lm9Rn4Tv6Wb1Yc3Hd5Jf7Ug0Pe'
259
+ const body = formatPermissionCardBody({
260
+ toolName: 'mcp__acme__do',
261
+ inputPreview: JSON.stringify({ token: bare, note: 'ping' }),
262
+ description: 'do a thing',
263
+ agentName: 'clerk',
264
+ })
265
+ expect(body).not.toContain(bare)
266
+ // `token` value is hard-masked; the benign `note` still surfaces.
267
+ expect(body).toContain('token:')
268
+ expect(body).toContain('REDACTED')
269
+ expect(body).toContain('note: ping')
270
+ })
271
+
272
+ test('hard-masks a 40-hex secret under a `secret`-shaped key (#3167)', () => {
273
+ const hex = 'a3f1'.repeat(10) // 40 hex chars, no prefix
274
+ const body = formatPermissionCardBody({
275
+ toolName: 'mcp__acme__do',
276
+ inputPreview: JSON.stringify({ webhook_secret: hex }),
277
+ description: 'do a thing',
278
+ agentName: 'clerk',
279
+ })
280
+ expect(body).not.toContain(hex)
281
+ expect(body).toContain('REDACTED')
282
+ })
283
+
284
+ test('masks a non-http DSN credential embedded in free text (redactUrls misses it) (#3167)', () => {
285
+ // Under a benign key (`text`) so the hard-mask key path does NOT fire —
286
+ // this exercises the NON_HTTP_DSN_RE scheme coverage specifically.
287
+ const dsn = 'postgres://user:S3cretPass99@db.host:5432/app'
288
+ const body = formatPermissionCardBody({
289
+ toolName: 'mcp__switchroom-telegram__reply',
290
+ inputPreview: JSON.stringify({ chat_id: '1', text: `connect via ${dsn} then run` }),
291
+ description: 'Reply on Telegram.',
292
+ agentName: 'clerk',
293
+ })
294
+ expect(body).not.toContain('S3cretPass99')
295
+ expect(body).not.toContain(dsn)
296
+ expect(body).toContain('REDACTED')
297
+ })
298
+
299
+ test('hard-masks a DATABASE_URL under a url-shaped key (#3167)', () => {
300
+ const dsn = 'postgres://admin:hunter2pass@10.0.0.5/prod'
301
+ const body = formatPermissionCardBody({
302
+ toolName: 'mcp__acme__migrate',
303
+ inputPreview: JSON.stringify({ database_url: dsn }),
304
+ description: 'run a migration',
305
+ agentName: 'clerk',
306
+ })
307
+ expect(body).not.toContain('hunter2pass')
308
+ expect(body).not.toContain(dsn)
309
+ expect(body).toContain('REDACTED')
310
+ })
311
+
312
+ // #3167 review (LOW-2): acceptance asks for "tool + summarized input +
313
+ // originating turn" — surface a compact origin-turn reference when present.
314
+ test('appends a compact originating-turn reference when origin_turn_id is present (#3167)', () => {
315
+ const body = formatPermissionCardBody({
316
+ toolName: 'mcp__switchroom-telegram__reply',
317
+ inputPreview: JSON.stringify({ chat_id: '1', text: 'hi', origin_turn_id: 'turn-abcdef123456' }),
318
+ description: 'Reply on Telegram.',
319
+ agentName: 'clerk',
320
+ })
321
+ expect(body).toContain('· turn …ef123456')
322
+ // The raw routing id is NOT dumped as its own kv pair.
323
+ expect(body).not.toContain('origin_turn_id:')
324
+ })
325
+
326
+ // #3167 review (LOW-3): raw commands/text now reach the context line, so a
327
+ // markdown metachar in the value must be escaped or it can spoof/break the
328
+ // card's own `_italic_` / `code` formatting.
329
+ test('escapes markdown metachars in the synthesized context value (#3167)', () => {
330
+ const body = formatPermissionCardBody({
331
+ toolName: 'Bash',
332
+ inputPreview: JSON.stringify({ command: 'echo _a_ *b* `c`' }),
333
+ description: 'run a shell command',
334
+ agentName: 'clerk',
335
+ })
336
+ // The metachars are backslash-escaped so they can't open emphasis/code.
337
+ expect(body).toContain('\\_a\\_')
338
+ expect(body).toContain('\\*b\\*')
339
+ expect(body).toContain('\\`c\\`')
340
+ expect(body).not.toContain('echo _a_ *b* `c`')
178
341
  })
179
342
 
180
343
  test('drops the agent prefix when agentName is null (early-boot edge)', () => {