switchroom 0.18.15 → 0.18.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/agent-scheduler/index.js +16 -0
  2. package/dist/auth-broker/index.js +445 -10
  3. package/dist/cli/notion-write-pretool.mjs +16 -0
  4. package/dist/cli/switchroom.js +654 -479
  5. package/dist/host-control/main.js +20 -1
  6. package/dist/vault/approvals/kernel-server.js +16 -0
  7. package/dist/vault/broker/server.js +16 -0
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/bridge/bridge.ts +7 -1
  11. package/telegram-plugin/dist/bridge/bridge.js +26 -1
  12. package/telegram-plugin/dist/gateway/gateway.js +1758 -661
  13. package/telegram-plugin/dist/server.js +26 -1
  14. package/telegram-plugin/draft-stream.ts +78 -3
  15. package/telegram-plugin/fleet-fallback-resume.ts +26 -3
  16. package/telegram-plugin/gateway/approval-hold.ts +49 -0
  17. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +64 -22
  18. package/telegram-plugin/gateway/effort-command.ts +9 -7
  19. package/telegram-plugin/gateway/gateway.ts +627 -291
  20. package/telegram-plugin/gateway/linear-activity.ts +20 -4
  21. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  22. package/telegram-plugin/gateway/model-command.ts +96 -18
  23. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  24. package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
  25. package/telegram-plugin/gateway/session-model-file.ts +141 -172
  26. package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
  27. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
  28. package/telegram-plugin/litellm-local-notice.ts +189 -0
  29. package/telegram-plugin/llm-error-present.ts +436 -0
  30. package/telegram-plugin/operator-events.ts +7 -1
  31. package/telegram-plugin/permission-title.ts +172 -10
  32. package/telegram-plugin/premium-recovery.ts +101 -0
  33. package/telegram-plugin/quota-watch.ts +16 -4
  34. package/telegram-plugin/raw-error-scrub.ts +73 -0
  35. package/telegram-plugin/retry-api-call.ts +8 -2
  36. package/telegram-plugin/runtime-metrics.ts +16 -0
  37. package/telegram-plugin/send-gate-degraded.test.ts +161 -8
  38. package/telegram-plugin/send-gate-observability.test.ts +140 -0
  39. package/telegram-plugin/send-gate-observability.ts +65 -20
  40. package/telegram-plugin/send-gate.test.ts +143 -1
  41. package/telegram-plugin/send-gate.ts +246 -23
  42. package/telegram-plugin/session-tail.ts +16 -0
  43. package/telegram-plugin/shared/local-time.ts +69 -0
  44. package/telegram-plugin/stream-controller.ts +143 -20
  45. package/telegram-plugin/stream-reply-handler.ts +12 -2
  46. package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
  47. package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
  48. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  49. package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
  50. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  51. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  52. package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
  53. package/telegram-plugin/tests/flood-windows-persistence.test.ts +5 -4
  54. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  55. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  56. package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
  57. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  58. package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
  59. package/telegram-plugin/tests/model-command.test.ts +84 -1
  60. package/telegram-plugin/tests/permission-title.test.ts +167 -4
  61. package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
  62. package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
  63. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  64. package/telegram-plugin/tests/reaction-gate-routing.test.ts +8 -3
  65. package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
  66. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  67. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  68. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  69. package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
  70. package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
  71. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
  72. package/telegram-plugin/tests/worker-activity-feed.test.ts +212 -2
  73. package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
  74. package/telegram-plugin/tier-downgrade.ts +198 -0
  75. package/telegram-plugin/tool-activity-summary.ts +99 -0
  76. package/telegram-plugin/worker-activity-feed.ts +543 -368
@@ -0,0 +1,521 @@
1
+ /**
2
+ * Integration tests: `createStreamController` driven through a REAL
3
+ * `createSendGate` (#3110).
4
+ *
5
+ * Production wires stream-controller's `retry` to the gateway's
6
+ * `robustApiCall`, which is the send gate wrapped over the retry policy —
7
+ * so what the gate does to the opts a call site passes IS the production
8
+ * behavior. Before #3110 the controller passed only `{ threadId, chat_id }`,
9
+ * the gate saw ordinary sends, and the draft stream's own 400 ms DM throttle
10
+ * drove same-message editMessageText straight past the >=1.5s per-message
11
+ * edit floor — the #1 documented flood-ban trigger (part3-design §4;
12
+ * production ~6h ban 2026-07-12).
13
+ *
14
+ * These tests pin the wiring end to end:
15
+ * - draft edits inside the floor never reach the API; intermediates
16
+ * collapse; the latest snapshot lands floor-paced
17
+ * - the floor is per-message (stream A does not delay stream B)
18
+ * - the gate's no-op skip drops a repeat payload for the same message
19
+ * - an open flood window sheds draft edits with ZERO API calls, and the
20
+ * stream recovers with full state after the window closes
21
+ * - a shed draft is NOT recorded as delivered — a later flush of the
22
+ * SAME text (the completed answer) still lands
23
+ * - the finalize flush is `critical`: never shed; waits out a short
24
+ * window; fails fast (structured, logged) on a long one
25
+ * - regression pin: the controller passes messageId / editPayload /
26
+ * priorityClass through the retry policy on every edit
27
+ *
28
+ * Two clocks, deliberately separate: vitest fake timers drive draft-stream's
29
+ * internal `setTimeout` throttle (only `setTimeout`/`clearTimeout` are faked
30
+ * so the gate's microtask pump below stays real); the gate runs on the same
31
+ * injectable fake `Clock` used by send-gate.test.ts.
32
+ */
33
+ import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'
34
+ import { createStreamController, type RetryPolicy } from '../stream-controller.js'
35
+ import { createSendGate, type Clock, type SendGateConfig } from '../send-gate.js'
36
+ import { isFloodWaitActiveError } from '../retry-api-call.js'
37
+ import { renderOutboundChunks } from '../render/rich-render.js'
38
+ import { createMockBot, installBotResetHook } from './bot-api.harness.js'
39
+
40
+ /** Deterministic fake clock — same contract as send-gate.test.ts's. */
41
+ class FakeClock implements Clock {
42
+ private cur = 0
43
+ private seq = 0
44
+ private timers: { at: number; id: number; resolve: () => void }[] = []
45
+
46
+ now(): number {
47
+ return this.cur
48
+ }
49
+
50
+ sleep(ms: number): Promise<void> {
51
+ return new Promise<void>((resolve) => {
52
+ this.timers.push({ at: this.cur + ms, id: this.seq++, resolve })
53
+ })
54
+ }
55
+
56
+ async advance(ms: number): Promise<void> {
57
+ const target = this.cur + ms
58
+ for (;;) {
59
+ const due = this.timers
60
+ .filter((t) => t.at <= target)
61
+ .sort((a, b) => a.at - b.at || a.id - b.id)
62
+ if (due.length === 0) break
63
+ const t = due[0]
64
+ this.timers = this.timers.filter((x) => x !== t)
65
+ this.cur = t.at
66
+ t.resolve()
67
+ await flush()
68
+ }
69
+ this.cur = target
70
+ }
71
+ }
72
+
73
+ /** Flush pending microtasks via a REAL macrotask hop (setImmediate is not faked). */
74
+ function flush(): Promise<void> {
75
+ return new Promise<void>((r) => setImmediate(r))
76
+ }
77
+
78
+ /** A real send gate on a fake clock, exposed as a stream-controller RetryPolicy. */
79
+ function makeGatedRetry(cfg: Partial<SendGateConfig> = {}) {
80
+ const clock = new FakeClock()
81
+ const gate = createSendGate({ enabled: true, clock, jitter: () => 0, ...cfg })
82
+ const retry: RetryPolicy = (fn, opts) => gate.gate(fn, opts)
83
+ return { clock, gate, retry }
84
+ }
85
+
86
+ describe('stream-controller × send gate (#3110)', () => {
87
+ const bot = createMockBot()
88
+ installBotResetHook(bot)
89
+
90
+ // Fake ONLY draft-stream's throttle timers. The gate's fake clock is
91
+ // driven explicitly; its microtask pump (setImmediate) must stay real.
92
+ beforeEach(() => vi.useFakeTimers({ toFake: ['setTimeout', 'clearTimeout'] }))
93
+ afterEach(() => vi.useRealTimers())
94
+
95
+ /** Release draft-stream's local throttle, then settle microtasks. */
96
+ async function tick(throttleMs = 250): Promise<void> {
97
+ vi.advanceTimersByTime(throttleMs)
98
+ await flush()
99
+ }
100
+
101
+ const editBodies = () =>
102
+ bot.api.editMessageText.mock.calls.map(([, , text]) =>
103
+ typeof text === 'object' && text != null ? text.markdown : text,
104
+ )
105
+
106
+ it('draft edits inside the floor never hit the API; intermediates collapse; latest lands floor-paced', async () => {
107
+ const { clock, retry } = makeGatedRetry({ editFloorMs: 1500 })
108
+ // Record the FAKE-clock time of every editMessageText that reaches the API.
109
+ const editTimes: number[] = []
110
+ bot.api.editMessageText.mockImplementation(async () => {
111
+ editTimes.push(clock.now())
112
+ return true as const
113
+ })
114
+ const stream = createStreamController({ bot, chatId: '1', throttleMs: 250, retry })
115
+
116
+ // Anchor send, then the first edit — no prior edit on the message, so the
117
+ // gate admits it immediately.
118
+ void stream.update('draft 1')
119
+ await flush()
120
+ expect(bot.api.sendRichMessage).toHaveBeenCalledTimes(1)
121
+ void stream.update('draft 2')
122
+ await tick()
123
+ expect(editBodies()).toEqual(['draft 2'])
124
+
125
+ // Three rapid snapshots inside the 1.5s floor: d3 reaches the gate (edit
126
+ // floor holds it); d4 and d5 arrive while that edit is in flight, so
127
+ // draft-stream's own last-write-wins collapses d4 — it must NEVER reach
128
+ // the API.
129
+ void stream.update('draft 3')
130
+ await tick()
131
+ void stream.update('draft 4')
132
+ void stream.update('draft 5')
133
+ await flush()
134
+ // Still inside the floor: nothing new hit the API.
135
+ expect(editBodies()).toEqual(['draft 2'])
136
+
137
+ await clock.advance(1500) // floor clears → d3 lands; d5 flushes into the gate
138
+ await clock.advance(1500) // next floor clears → d5 lands
139
+ expect(editBodies()).toEqual(['draft 2', 'draft 3', 'draft 5'])
140
+ // Wire spacing is the gate floor, not the 250 ms local throttle.
141
+ expect(editTimes).toEqual([0, 1500, 3000])
142
+ })
143
+
144
+ it('the edit floor is per-message: stream B is not delayed by stream A holding its floor', async () => {
145
+ const { clock, retry } = makeGatedRetry({ editFloorMs: 1500 })
146
+ const editTimes = new Map<number, number[]>()
147
+ bot.api.editMessageText.mockImplementation(async (_chat, id) => {
148
+ const arr = editTimes.get(id) ?? []
149
+ arr.push(clock.now())
150
+ editTimes.set(id, arr)
151
+ return true as const
152
+ })
153
+ // Two chats → independent per-chat buckets; ONE shared gate (as in the
154
+ // gateway, where every surface transits the same robustApiCall).
155
+ const a = createStreamController({ bot, chatId: '100', throttleMs: 250, retry })
156
+ const b = createStreamController({ bot, chatId: '200', throttleMs: 250, retry })
157
+
158
+ void a.update('a1')
159
+ void b.update('b1')
160
+ await flush()
161
+ void a.update('a2')
162
+ void b.update('b2')
163
+ await tick()
164
+ // First edit per message admits immediately on each message's own floor.
165
+ const idA = a.getMessageId() as number
166
+ const idB = b.getMessageId() as number
167
+ expect(editTimes.get(idA)).toEqual([0])
168
+ expect(editTimes.get(idB)).toEqual([0])
169
+
170
+ // Refill the global bucket a little so the cosmetic shed check (token
171
+ // pressure) doesn't fire — this test isolates the FLOOR.
172
+ await clock.advance(400)
173
+ void a.update('a3')
174
+ void b.update('b3')
175
+ await tick()
176
+ expect(editTimes.get(idA)).toEqual([0]) // held by A's floor…
177
+ expect(editTimes.get(idB)).toEqual([0]) // …and B by B's own, not A's.
178
+
179
+ await clock.advance(1100) // 400 + 1100 = 1500 → BOTH floors clear together
180
+ expect(editTimes.get(idA)).toEqual([0, 1500])
181
+ expect(editTimes.get(idB)).toEqual([0, 1500])
182
+ })
183
+
184
+ it('gate no-op skip: a repeated payload is dropped BENIGNLY — no API call, treated as delivered (F4)', async () => {
185
+ const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
186
+ // Two controllers re-attached to the SAME anchor message (the #626
187
+ // initialMessageId path — e.g. a stream re-created after a restart).
188
+ const first = createStreamController({
189
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
190
+ })
191
+ void first.update('**status: done**')
192
+ await flush()
193
+ expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
194
+
195
+ await clock.advance(1500) // clear the floor — isolate the no-op skip
196
+ const onEdit = vi.fn()
197
+ const logs: string[] = []
198
+ const second = createStreamController({
199
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777, onEdit,
200
+ log: (m) => logs.push(m),
201
+ })
202
+ void second.update('**status: done**')
203
+ await flush()
204
+
205
+ // Identical rendered payload for the same chat+message → dropped by the
206
+ // gate before the API…
207
+ expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
208
+ expect(gate.stats().global.dropped).toBe(1)
209
+ // …and the drop is BENIGN (the payload IS on screen): the controller
210
+ // reports it as delivered — onEdit fires, no shed marker, no failure log
211
+ // (F4: pre-sentinel this test passed via the wrong mechanism — the
212
+ // drop's `undefined` was misread as a shed and onEdit stayed silent).
213
+ expect(onEdit).toHaveBeenCalledTimes(1)
214
+ expect(logs.filter((m) => m.includes('shed') || m.includes('edit failed'))).toEqual([])
215
+
216
+ // Because the drop was recorded as delivered, a THIRD identical update
217
+ // dedupes inside draft-stream itself — it never even reaches the gate.
218
+ void second.update('**status: done**')
219
+ await tick() // release the local throttle so the dedupe check runs
220
+ expect(gate.stats().global.dropped).toBe(1) // unchanged — gate never consulted
221
+ expect(bot.api.editMessageText).toHaveBeenCalledTimes(1)
222
+ })
223
+
224
+ it('open flood window: draft edits shed with ZERO API calls; full state lands after it closes', async () => {
225
+ const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
226
+ const stream = createStreamController({
227
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
228
+ })
229
+
230
+ gate.openFloodWindow('global', clock.now() + 30_000)
231
+
232
+ void stream.update('draft a')
233
+ await flush()
234
+ void stream.update('draft b')
235
+ await tick()
236
+ // Both drafts shed as cosmetic — nothing reached the API.
237
+ expect(bot.api.editMessageText).not.toHaveBeenCalled()
238
+ expect(gate.stats().global.shed).toBe(2)
239
+
240
+ await clock.advance(30_000) // window closes
241
+ void stream.update('draft c — full state')
242
+ await tick()
243
+ expect(editBodies()).toEqual(['draft c — full state'])
244
+ })
245
+
246
+ it('a shed draft is NOT recorded as delivered: a later finalize of the SAME text still lands', async () => {
247
+ const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
248
+ const stream = createStreamController({
249
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
250
+ })
251
+
252
+ gate.openFloodWindow('global', clock.now() + 30_000)
253
+ void stream.update('the completed answer')
254
+ await flush()
255
+ expect(bot.api.editMessageText).not.toHaveBeenCalled()
256
+ expect(gate.stats().global.shed).toBe(1)
257
+
258
+ await clock.advance(30_000)
259
+ // stream_reply done=true with the same text → finalize(text). If the shed
260
+ // draft had been recorded as on-screen, draft-stream's dedupe would skip
261
+ // this flush and the completed answer would never render.
262
+ await stream.finalize('the completed answer')
263
+ expect(editBodies()).toEqual(['the completed answer'])
264
+ })
265
+
266
+ it('finalize is critical: not shed by an open short window — waits it out, then lands', async () => {
267
+ const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
268
+ const editTimes: number[] = []
269
+ bot.api.editMessageText.mockImplementation(async () => {
270
+ editTimes.push(clock.now())
271
+ return true as const
272
+ })
273
+ const stream = createStreamController({
274
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
275
+ })
276
+
277
+ gate.openFloodWindow('global', clock.now() + 30_000) // short: <= 60s fail-fast ceiling
278
+ void stream.update('draft while banned')
279
+ await flush()
280
+ expect(bot.api.editMessageText).not.toHaveBeenCalled() // draft shed
281
+
282
+ const fin = stream.finalize('the answer')
283
+ await flush()
284
+ // Critical: NOT shed — queued against the window, still zero API calls.
285
+ expect(bot.api.editMessageText).not.toHaveBeenCalled()
286
+
287
+ await clock.advance(30_000)
288
+ await fin
289
+ expect(editBodies()).toEqual(['the answer'])
290
+ expect(editTimes).toEqual([30_000])
291
+ expect(gate.stats().global.shed).toBe(1) // only the draft
292
+ })
293
+
294
+ it('finalize under a LONG window fails fast (structured FLOOD_WAIT_ACTIVE, logged) — no API call, no hang', async () => {
295
+ const { clock, gate, retry } = makeGatedRetry({ editFloorMs: 1500 })
296
+ const logs: string[] = []
297
+ const stream = createStreamController({
298
+ bot, chatId: '1', throttleMs: 250, retry, initialMessageId: 7777,
299
+ log: (m) => logs.push(m),
300
+ })
301
+
302
+ gate.openFloodWindow('global', clock.now() + 21_397_000) // the 2026-07-12 ban: ~5.9h
303
+ const fin = stream.finalize('the answer')
304
+ await flush()
305
+ await fin // resolves — draft-stream surfaces the failure via log, not a hang
306
+
307
+ expect(bot.api.editMessageText).not.toHaveBeenCalled()
308
+ expect(gate.stats().global.failedFast).toBe(1)
309
+ expect(logs.some((m) => m.includes('FLOOD_WAIT_ACTIVE'))).toBe(true)
310
+ // Sanity: the gate's structured error shape is what the reply path keys on.
311
+ try {
312
+ await retry(async () => true, { chat_id: '1', priorityClass: 'critical' })
313
+ expect.unreachable('critical send during a long window must fail fast')
314
+ } catch (err) {
315
+ expect(isFloodWaitActiveError(err)).toBe(true)
316
+ }
317
+ })
318
+
319
+ it('REGRESSION PIN: every edit passes messageId / editPayload / priorityClass to the retry policy', async () => {
320
+ // Spy retry with NO gate — pins exactly what the controller hands to
321
+ // robustApiCall (the #3110 bypass was these fields being absent).
322
+ const seen: Array<Record<string, unknown> | undefined> = []
323
+ const retry: RetryPolicy = (fn, opts) => {
324
+ seen.push(opts as Record<string, unknown> | undefined)
325
+ return fn()
326
+ }
327
+ const stream = createStreamController({ bot, chatId: '9', threadId: 3, throttleMs: 250, retry })
328
+
329
+ void stream.update('draft')
330
+ await flush()
331
+ // The anchor SEND stays untagged: the gate admits untagged non-edit
332
+ // sends as critical, and a shed send would break message-id capture.
333
+ expect(seen[0]).toEqual({ threadId: 3, chat_id: '9' })
334
+
335
+ void stream.update('draft v2')
336
+ await tick()
337
+ const id = stream.getMessageId() as number
338
+ expect(seen[1]).toEqual({
339
+ threadId: 3,
340
+ chat_id: '9',
341
+ messageId: id,
342
+ editPayload: { markdown: 'draft v2' },
343
+ priorityClass: 'cosmetic',
344
+ })
345
+
346
+ await stream.finalize('the final answer')
347
+ expect(seen[2]).toEqual({
348
+ threadId: 3,
349
+ chat_id: '9',
350
+ messageId: id,
351
+ editPayload: { markdown: 'the final answer' },
352
+ priorityClass: 'critical',
353
+ })
354
+ })
355
+
356
+ it('REGRESSION PIN: the plain parse-entities fallback edit carries the gate fields too', async () => {
357
+ const seen: Array<Record<string, unknown> | undefined> = []
358
+ const retry: RetryPolicy = (fn, opts) => {
359
+ seen.push(opts as Record<string, unknown> | undefined)
360
+ return fn()
361
+ }
362
+ const stream = createStreamController({ bot, chatId: '9', throttleMs: 250, retry })
363
+ void stream.update('draft')
364
+ await flush()
365
+ const id = stream.getMessageId() as number
366
+
367
+ // Rich edit 400s with a parse-entities rejection → same-id plain fallback.
368
+ const { GrammyError } = await import('grammy')
369
+ bot.api.editMessageText.mockRejectedValueOnce(
370
+ new GrammyError(
371
+ "Bad Request: can't parse entities",
372
+ { ok: false, error_code: 400, description: "Bad Request: can't parse entities" },
373
+ 'editMessageText',
374
+ {},
375
+ ),
376
+ )
377
+ void stream.update('draft *broken')
378
+ await tick()
379
+
380
+ expect(bot.api.editMessageText).toHaveBeenCalledTimes(2)
381
+ // The fallback is an EDIT of the same message: gate fields present, and
382
+ // the payload is the PLAIN body (hashes differently from the rich form,
383
+ // so the gate never no-op-drops the recovery edit).
384
+ expect(seen[2]).toEqual({
385
+ threadId: undefined,
386
+ chat_id: '9',
387
+ messageId: id,
388
+ editPayload: 'draft *broken',
389
+ priorityClass: 'cosmetic',
390
+ })
391
+ })
392
+
393
+ /**
394
+ * The reviewer's F1 production-faithful probe: a multi-piece body (the
395
+ * escaped-markdown expansion splits it across several wire-cap pieces)
396
+ * where only the TAIL grows across flushes. The anchor piece stops
397
+ * changing, so on a perfectly healthy chat its edit resolves `undefined`
398
+ * every flush — first via robustApiCall's swallowed "message is not
399
+ * modified" 400, then via the gate's no-op drop. Pre-sentinel, that
400
+ * `undefined` was misread as a shed and thrown BEFORE the tail loop: no
401
+ * tail edit ever landed (screen frozen at "tail v1") and a false "shed"
402
+ * line hit stderr every flush.
403
+ */
404
+ it('multi-piece body: a benign anchor outcome never starves the tail edits (F1)', async () => {
405
+ const { clock, gate, retry } = makeGatedRetry({
406
+ editFloorMs: 1500,
407
+ // Generous buckets: this test isolates the benign-undefined path, not
408
+ // token pressure.
409
+ globalPerSec: 1000, globalBurst: 100, perChatPerSec: 1000, perChatBurst: 100,
410
+ })
411
+ const base = ('a_b_c_d_e ').repeat(3000)
412
+ const b1 = `${base}tail v1`
413
+ const b2 = `${base}tail v2`
414
+ const b3 = `${base}tail v3`
415
+ // Sanity-pin the repro geometry: several pieces, prefix-stable chunking,
416
+ // only the LAST piece changes as the tail grows.
417
+ const p1 = renderOutboundChunks(b1)
418
+ const p2 = renderOutboundChunks(b2)
419
+ expect(p1.length).toBeGreaterThan(1)
420
+ expect(p2.length).toBe(p1.length)
421
+ for (let i = 0; i < p1.length - 1; i++) expect(p2[i].text).toBe(p1[i].text)
422
+ expect(p2[p1.length - 1].text).not.toBe(p1[p1.length - 1].text)
423
+
424
+ // Faithful edit mock: repeating a message's current text 400s with
425
+ // "message is not modified", which the production retry policy swallows
426
+ // to `undefined` — simulate the post-swallow result directly.
427
+ const currentText = new Map<number, string>()
428
+ bot.api.sendMessage.mockImplementation(async (_c, text) => {
429
+ const id = bot.nextMessageId++
430
+ currentText.set(id, text)
431
+ return { message_id: id }
432
+ })
433
+ bot.api.editMessageText.mockImplementation(async (_c, id, text) => {
434
+ const body = typeof text === 'object' && text != null ? text.markdown : text
435
+ if (currentText.get(id) === body) return undefined // swallowed not-modified
436
+ currentText.set(id, body)
437
+ return true as const
438
+ })
439
+
440
+ const logs: string[] = []
441
+ const stream = createStreamController({
442
+ bot, chatId: '1', throttleMs: 250, retry,
443
+ log: (m) => logs.push(m), warn: (m) => logs.push(m),
444
+ })
445
+
446
+ void stream.update(b1)
447
+ await flush()
448
+ const anchorId = stream.getMessageId() as number
449
+ const lastTailId = anchorId + p1.length - 1
450
+ expect(bot.api.sendMessage).toHaveBeenCalledTimes(p1.length)
451
+ expect(currentText.get(lastTailId)).toContain('tail v1')
452
+
453
+ // Flush 2: anchor unchanged → its edit resolves undefined via the
454
+ // not-modified swallow. The tail edit MUST still land.
455
+ void stream.update(b2)
456
+ await tick()
457
+ expect(currentText.get(lastTailId)).toContain('tail v2')
458
+
459
+ // Flush 3: anchor unchanged again → this time the GATE no-op-drops it
460
+ // (its hash is recorded from flush 2). The tail edit MUST still land.
461
+ await clock.advance(1600) // clear the tail message's own edit floor
462
+ void stream.update(b3)
463
+ await tick()
464
+ expect(currentText.get(lastTailId)).toContain('tail v3')
465
+ expect(gate.stats().global.dropped).toBeGreaterThanOrEqual(1)
466
+
467
+ // The whole run was healthy: no shed marker, no failure lines.
468
+ expect(logs.filter((m) => m.includes('shed') || m.includes('FAILED') || m.includes('edit failed'))).toEqual([])
469
+
470
+ // And the flushes were recorded as delivered: an identical finalize
471
+ // dedupes inside draft-stream — zero further API calls.
472
+ const editCalls = bot.api.editMessageText.mock.calls.length
473
+ await stream.finalize(b3)
474
+ expect(bot.api.editMessageText.mock.calls.length).toBe(editCalls)
475
+ })
476
+
477
+ it('a shed TAIL is not recorded as delivered: argument-less finalize() re-flushes and lands it (F2+F3)', async () => {
478
+ const { clock, gate, retry } = makeGatedRetry({
479
+ editFloorMs: 1500,
480
+ globalPerSec: 1000, globalBurst: 100, perChatPerSec: 1000, perChatBurst: 100,
481
+ })
482
+ const base = ('a_b_c_d_e ').repeat(3000)
483
+ const b1 = `${base}tail v1`
484
+ const b2 = `${base}tail v2 — the completed answer`
485
+ const pieceCount = renderOutboundChunks(b1).length
486
+ expect(pieceCount).toBeGreaterThan(1)
487
+
488
+ const logs: string[] = []
489
+ const stream = createStreamController({
490
+ bot, chatId: '1', throttleMs: 250, retry, log: (m) => logs.push(m),
491
+ })
492
+ void stream.update(b1)
493
+ await flush()
494
+ const anchorId = stream.getMessageId() as number
495
+ const lastTailId = anchorId + pieceCount - 1
496
+
497
+ // Flood window scoped to the LAST tail message only (H1 msg-edit scope):
498
+ // its edit sheds; the anchor and other pieces are unaffected.
499
+ gate.openFloodWindow(`msg-edit:1:${lastTailId}`, clock.now() + 30_000)
500
+
501
+ void stream.update(b2)
502
+ await tick()
503
+ // The changed piece is the suppressed tail → shed, zero edits landed on
504
+ // it; the flush is reported shed and the snapshot preserved (F2), NOT
505
+ // recorded as delivered (F3).
506
+ const tailEdits = () =>
507
+ bot.api.editMessageText.mock.calls.filter(([, id]) => id === lastTailId)
508
+ expect(tailEdits()).toHaveLength(0)
509
+ expect(gate.stats().global.shed).toBe(1)
510
+ expect(logs.some((m) => m.includes('shed by send gate'))).toBe(true)
511
+
512
+ // Window closes; the gateway-style ARGUMENT-LESS finalize (the
513
+ // disconnect-flush / turn-end cleanup path) must re-deliver the shed
514
+ // snapshot — pre-F2 the content was silently lost here.
515
+ await clock.advance(31_000)
516
+ await stream.finalize()
517
+ expect(tailEdits()).toHaveLength(1)
518
+ const [, , tailBody] = tailEdits()[0]
519
+ expect(String(tailBody)).toContain('tail v2')
520
+ })
521
+ })
@@ -207,6 +207,50 @@ describe('handleStreamReply', () => {
207
207
  expect(state.activeDraftStreams.size).toBe(0)
208
208
  })
209
209
 
210
+ it('done=true routes the final text through finalize(text) — the answer edit is CRITICAL for the send gate, drafts cosmetic (#3110)', async () => {
211
+ const state = makeState()
212
+ // Spy retry: pins exactly what reaches robustApiCall (= the send gate).
213
+ const seen: Array<Record<string, unknown> | undefined> = []
214
+ const retry = (<T,>(fn: () => Promise<T>, opts?: unknown) => {
215
+ seen.push(opts as Record<string, unknown> | undefined)
216
+ return fn()
217
+ }) as StreamReplyDeps['retry']
218
+ const deps = makeDeps(bot, { retry })
219
+
220
+ // Turn start → anchor send.
221
+ let pending = handleStreamReply({ chat_id: '1', text: 'thinking…' }, state, deps)
222
+ await microtaskFlush()
223
+ await pending
224
+
225
+ // Intermediate snapshot → a sheddable (cosmetic) draft edit.
226
+ pending = handleStreamReply({ chat_id: '1', text: 'half the answer' }, state, deps)
227
+ await microtaskFlush()
228
+ vi.advanceTimersByTime(600) // release the local throttle
229
+ await microtaskFlush()
230
+ await pending
231
+
232
+ // done=true → the completed answer flushes via finalize(text): the edit
233
+ // runs with the stream already final and is tagged critical — never shed
234
+ // as a draft under flood pressure.
235
+ pending = handleStreamReply(
236
+ { chat_id: '1', text: 'the full answer', done: true },
237
+ state,
238
+ deps,
239
+ )
240
+ await microtaskFlush()
241
+ const result = await pending
242
+
243
+ expect(result.status).toBe('finalized')
244
+ expect(richEditMarkdown(bot, 1)).toBe('the full answer')
245
+ expect(seen[0]).toEqual({ threadId: undefined, chat_id: '1' }) // anchor send: untagged
246
+ expect(seen[1]).toMatchObject({ messageId: 500, priorityClass: 'cosmetic' })
247
+ expect(seen[2]).toMatchObject({
248
+ messageId: 500,
249
+ priorityClass: 'critical',
250
+ editPayload: { markdown: 'the full answer' },
251
+ })
252
+ })
253
+
210
254
  it('done=true on a named lane does NOT fire terminal 👍', async () => {
211
255
  const state = makeState()
212
256
  const deps = makeDeps(bot)