switchroom 0.18.15 → 0.18.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/agent-scheduler/index.js +16 -0
  2. package/dist/auth-broker/index.js +445 -10
  3. package/dist/cli/notion-write-pretool.mjs +16 -0
  4. package/dist/cli/switchroom.js +654 -479
  5. package/dist/host-control/main.js +20 -1
  6. package/dist/vault/approvals/kernel-server.js +16 -0
  7. package/dist/vault/broker/server.js +16 -0
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/bridge/bridge.ts +7 -1
  11. package/telegram-plugin/dist/bridge/bridge.js +26 -1
  12. package/telegram-plugin/dist/gateway/gateway.js +1758 -661
  13. package/telegram-plugin/dist/server.js +26 -1
  14. package/telegram-plugin/draft-stream.ts +78 -3
  15. package/telegram-plugin/fleet-fallback-resume.ts +26 -3
  16. package/telegram-plugin/gateway/approval-hold.ts +49 -0
  17. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +64 -22
  18. package/telegram-plugin/gateway/effort-command.ts +9 -7
  19. package/telegram-plugin/gateway/gateway.ts +627 -291
  20. package/telegram-plugin/gateway/linear-activity.ts +20 -4
  21. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  22. package/telegram-plugin/gateway/model-command.ts +96 -18
  23. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  24. package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
  25. package/telegram-plugin/gateway/session-model-file.ts +141 -172
  26. package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
  27. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
  28. package/telegram-plugin/litellm-local-notice.ts +189 -0
  29. package/telegram-plugin/llm-error-present.ts +436 -0
  30. package/telegram-plugin/operator-events.ts +7 -1
  31. package/telegram-plugin/permission-title.ts +172 -10
  32. package/telegram-plugin/premium-recovery.ts +101 -0
  33. package/telegram-plugin/quota-watch.ts +16 -4
  34. package/telegram-plugin/raw-error-scrub.ts +73 -0
  35. package/telegram-plugin/retry-api-call.ts +8 -2
  36. package/telegram-plugin/runtime-metrics.ts +16 -0
  37. package/telegram-plugin/send-gate-degraded.test.ts +161 -8
  38. package/telegram-plugin/send-gate-observability.test.ts +140 -0
  39. package/telegram-plugin/send-gate-observability.ts +65 -20
  40. package/telegram-plugin/send-gate.test.ts +143 -1
  41. package/telegram-plugin/send-gate.ts +246 -23
  42. package/telegram-plugin/session-tail.ts +16 -0
  43. package/telegram-plugin/shared/local-time.ts +69 -0
  44. package/telegram-plugin/stream-controller.ts +143 -20
  45. package/telegram-plugin/stream-reply-handler.ts +12 -2
  46. package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
  47. package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
  48. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  49. package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
  50. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  51. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  52. package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
  53. package/telegram-plugin/tests/flood-windows-persistence.test.ts +5 -4
  54. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  55. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  56. package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
  57. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  58. package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
  59. package/telegram-plugin/tests/model-command.test.ts +84 -1
  60. package/telegram-plugin/tests/permission-title.test.ts +167 -4
  61. package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
  62. package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
  63. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  64. package/telegram-plugin/tests/reaction-gate-routing.test.ts +8 -3
  65. package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
  66. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  67. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  68. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  69. package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
  70. package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
  71. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
  72. package/telegram-plugin/tests/worker-activity-feed.test.ts +212 -2
  73. package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
  74. package/telegram-plugin/tier-downgrade.ts +198 -0
  75. package/telegram-plugin/tool-activity-summary.ts +99 -0
  76. package/telegram-plugin/worker-activity-feed.ts +543 -368
@@ -19,9 +19,14 @@
19
19
  * entire server.ts top-level initialization.
20
20
  */
21
21
 
22
- import { createDraftStream, type DraftStreamHandle } from './draft-stream.js'
22
+ import {
23
+ createDraftStream,
24
+ makeDraftEditShedError,
25
+ type DraftStreamHandle,
26
+ } from './draft-stream.js'
23
27
  import { richMessage, isParseEntitiesError } from './rich-send.js'
24
28
  import { renderOutboundChunks } from './render/rich-render.js'
29
+ import { isSendGateShed, type SendGateOpts } from './send-gate.js'
25
30
 
26
31
  /**
27
32
  * Minimal bot.api surface the controller needs. Real callers pass grammy's
@@ -77,9 +82,21 @@ export interface StreamSendOpts {
77
82
  disable_notification?: boolean
78
83
  }
79
84
 
85
+ /**
86
+ * Options a stream-controller call passes to its retry wrapper. A structural
87
+ * superset of `SendGateOpts` (send-gate.ts) plus the retry policy's own
88
+ * `threadId`, and assignable to `RetryCallOpts` (retry-api-call.ts) — so the
89
+ * production wrapper (`robustApiCall` = send gate over `createRetryApiCall`)
90
+ * receives `messageId` / `editPayload` / `priorityClass` and the gate's
91
+ * per-message edit floor, last-write-wins coalescing, no-op skip and
92
+ * cosmetic shedding govern the draft/answer stream (#3110; part3-design §4
93
+ * names rapid same-message editMessageText as the #1 flood-ban trigger).
94
+ */
95
+ export type RetryPolicyOpts = SendGateOpts & { threadId?: number }
96
+
80
97
  export type RetryPolicy = <T>(
81
98
  fn: () => Promise<T>,
82
- opts?: { threadId?: number; chat_id?: string },
99
+ opts?: RetryPolicyOpts,
83
100
  ) => Promise<T>
84
101
 
85
102
  export interface StreamControllerConfig {
@@ -251,6 +268,70 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
251
268
  return bot.api.editMessageText(chatId, id, richMessage(piece.text), opts)
252
269
  }
253
270
 
271
+ // ---- Send-gate wiring for the edit path (#3110) -------------------------
272
+ //
273
+ // Every edit call passes `messageId` / `editPayload` / `priorityClass`
274
+ // through the retry wrapper so the send gate's per-message edit floor
275
+ // (>=1.5s), last-write-wins coalescing, and no-op skip govern the draft
276
+ // stream — previously these edits carried only `{ threadId, chat_id }`,
277
+ // so the gate treated them as ordinary sends and the stream's own 400 ms
278
+ // DM throttle drove same-message editMessageText well under the floor
279
+ // (the #1 documented flood-ban trigger, part3-design §4; production ban
280
+ // 2026-07-12 on #3110). The local per-surface throttle stays as a cheap
281
+ // pre-filter; the gate is the authority.
282
+ //
283
+ // Priority classes: intermediate draft edits are `cosmetic` (part3-design
284
+ // §2 lists "stream updates" there) — shed under pressure / an open flood
285
+ // window; the next flush carries full state. The FINALIZE flush (the edit
286
+ // that renders the completed answer) is `critical`, mirroring the reply
287
+ // path's preview-finalize convention (gateway.ts editPreview): never shed,
288
+ // waits out a short window, fails fast with a structured FLOOD_WAIT_ACTIVE
289
+ // on a long one. SENDS stay untagged (the gate admits untagged non-edit
290
+ // sends as `critical`) — a shed send would resolve the gate's shed
291
+ // sentinel instead of `{ message_id }` and break message-id capture, and
292
+ // the anchor/tail sends ARE the answer surface.
293
+ //
294
+ // `handleRef.isFinal()` is true from the moment finalize() is entered
295
+ // (draft-stream sets `final` before its last flush), so the closures below
296
+ // classify exactly the finalize flush — and anything after it — as
297
+ // critical. The ref is assigned right after createDraftStream returns,
298
+ // before any closure can run (closures only fire from update/finalize).
299
+ //
300
+ // KNOWN MISSED BENEFIT (review F5, deliberate): the gate's last-write-wins
301
+ // coalescing never engages for THIS surface, because draft-stream
302
+ // serializes its flushes — it awaits each edit before issuing the next, so
303
+ // at most one edit per message is ever inside the gate. Consequence: a
304
+ // stale draft sleeping on the gate's floor still lands (one API call the
305
+ // coalescer would have replaced) before the newer snapshot, and a finalize
306
+ // issued mid-floor can trail by up to ~2x editFloorMs (floor wait for the
307
+ // stale draft, then floor wait for the final). Correctness is unaffected —
308
+ // the latest state always lands, floor-paced — and the gate coalescing
309
+ // remains live protection for CONCURRENT writers to one message (e.g. a
310
+ // re-attached #626 controller racing its predecessor).
311
+ let handleRef: DraftStreamHandle | null = null
312
+ const editGateOpts = (id: number, payload: unknown): RetryPolicyOpts => ({
313
+ threadId,
314
+ chat_id: chatId,
315
+ messageId: id,
316
+ editPayload: payload,
317
+ priorityClass: handleRef?.isFinal() === true ? 'critical' : 'cosmetic',
318
+ })
319
+ // The rendered payload the gate hashes for the no-op skip / coalescing —
320
+ // exactly what goes over the wire (rich wrapper included), so a plain
321
+ // fallback of the same text never hashes equal to its rich form.
322
+ const piecePayload = (piece: { text: string; rich: boolean }): unknown =>
323
+ piece.rich ? richMessage(piece.text) : piece.text
324
+ // Shed detection (#3110 review F1): keyed EXACTLY off the gate's
325
+ // SEND_GATE_SHED sentinel — never off `undefined`, which is overloaded
326
+ // (gate no-op drop; robustApiCall's swallowed benign 400s like "message is
327
+ // not modified"). Those benign cases mean the payload is ALREADY on screen
328
+ // and are treated as delivered, exactly as before this wiring existed. A
329
+ // true shed means the edit did NOT land: draft-stream must not record the
330
+ // snapshot as on-screen (its dedupe would skip a later flush of the same
331
+ // text — the completed answer would never render), so the edit closure
332
+ // throws the marker error draft-stream recognizes and recovers from
333
+ // (`makeDraftEditShedError` → snapshot preserved for finalize, review F2).
334
+
254
335
  // Overflow-tail bookkeeping, shared across the send + edit closures for the
255
336
  // whole stream lifetime. A body large enough to split into several
256
337
  // wire-cap pieces anchors on piece[0] (edited in place by draft-stream) and
@@ -269,12 +350,25 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
269
350
  // text) thereafter. A non-parse failure is logged as a partial-delivery
270
351
  // warning and swallowed so the remaining tail pieces still get a chance to
271
352
  // land — never a silent drop, never an abort of pieces K..N (concern C1).
272
- const upsertTail = async (ti: number, piece: { text: string; rich: boolean }): Promise<void> => {
353
+ //
354
+ // Returns TRUE when this piece's edit was SHED by the send gate (did not
355
+ // land): the caller must then report the whole flush as shed so
356
+ // draft-stream does not record the full body as delivered while a tail is
357
+ // stale (review F3). The piece's own tailLastText stays stale too, so the
358
+ // recovery flush re-attempts exactly the shed piece.
359
+ const upsertTail = async (
360
+ ti: number,
361
+ piece: { text: string; rich: boolean },
362
+ ): Promise<boolean> => {
273
363
  const existingId = tailIds[ti]
274
364
  if (existingId != null) {
275
- if (tailLastText[ti] === piece.text) return // unchanged — skip the API call
365
+ if (tailLastText[ti] === piece.text) return false // unchanged — skip the API call
276
366
  try {
277
- await retry(() => editPiece(existingId, piece, baseOpts), { threadId, chat_id: chatId })
367
+ const res = await retry(
368
+ () => editPiece(existingId, piece, baseOpts),
369
+ editGateOpts(existingId, piecePayload(piece)),
370
+ )
371
+ if (isSendGateShed(res)) return true // shed — stale; retried by the recovery flush
278
372
  tailLastText[ti] = piece.text
279
373
  onEdit?.(existingId, piece.text.length)
280
374
  } catch (err) {
@@ -282,10 +376,11 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
282
376
  warn?.(
283
377
  `stream-controller: tail-piece #${ti + 1} edit parse-entities rejected — retrying same id=${existingId} as plain text (${err instanceof Error ? err.message : String(err)})`,
284
378
  )
285
- await retry(
379
+ const res = await retry(
286
380
  () => bot.api.editMessageText(chatId, existingId, piece.text, baseOpts),
287
- { threadId, chat_id: chatId },
381
+ editGateOpts(existingId, piece.text),
288
382
  )
383
+ if (isSendGateShed(res)) return true // shed — stale; retried by the recovery flush
289
384
  tailLastText[ti] = piece.text
290
385
  onEdit?.(existingId, piece.text.length)
291
386
  } else {
@@ -296,7 +391,7 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
296
391
  )
297
392
  }
298
393
  }
299
- return
394
+ return false
300
395
  }
301
396
  // First emission of this tail piece → a fresh follow-up message.
302
397
  try {
@@ -324,9 +419,12 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
324
419
  )
325
420
  }
326
421
  }
422
+ // First-emission SENDS are untagged (critical) — the gate never sheds
423
+ // them, so this path can only land or fail (handled above).
424
+ return false
327
425
  }
328
426
 
329
- return createDraftStream(
427
+ const handle = createDraftStream(
330
428
  async (text) => {
331
429
  // Render → 1+ cap-respecting pieces. The FIRST piece's message_id anchors
332
430
  // the stream (later edits target it); any overflow pieces are parked as
@@ -378,14 +476,26 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
378
476
  async (id, text) => {
379
477
  const pieces = renderPieces(text)
380
478
  const head = pieces[0]
479
+ // Whether any piece of THIS flush was shed by the gate (did not land).
480
+ // Decided at the very end — AFTER the tail loop — so a benign anchor
481
+ // outcome (or even a shed anchor) never starves the tail pieces of
482
+ // their own upsert attempt (review F1: an anchor whose payload stopped
483
+ // changing resolves benignly every flush while only the tail grows).
484
+ let anchorShed = false
381
485
  // Edit the anchor message in place with the FIRST piece.
382
486
  try {
383
- await retry(
384
- () => editPiece(id, head, baseOpts),
385
- { threadId, chat_id: chatId },
386
- )
387
- // C2: report the actual head-piece length, not the full body length.
388
- onEdit?.(id, head.text.length)
487
+ const res = await retry(() => editPiece(id, head, baseOpts), editGateOpts(id, piecePayload(head)))
488
+ if (isSendGateShed(res)) {
489
+ // Shed by the gate (cosmetic under pressure / an open flood
490
+ // window) — did NOT land. Benign `undefined` resolutions (gate
491
+ // no-op drop, robustApiCall's swallowed "message is not modified")
492
+ // deliberately do NOT take this branch: the payload is already on
493
+ // screen and the flush proceeds as delivered.
494
+ anchorShed = true
495
+ } else {
496
+ // C2: report the actual head-piece length, not the full body length.
497
+ onEdit?.(id, head.text.length)
498
+ }
389
499
  } catch (err) {
390
500
  if (!literalText && head.rich && isParseEntitiesError(err)) {
391
501
  // Edit rejected because the markdown couldn't be parsed — DO NOT
@@ -400,11 +510,12 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
400
510
  // contract). For a single-piece stream (common case) this is the
401
511
  // whole body; a rare oversize split edits the head piece's body.
402
512
  const fallbackBody = pieces.length === 1 ? text : head.text
403
- await retry(
513
+ const res = await retry(
404
514
  () => bot.api.editMessageText(chatId, id, fallbackBody, baseOpts),
405
- { threadId, chat_id: chatId },
515
+ editGateOpts(id, fallbackBody),
406
516
  )
407
- onEdit?.(id, head.text.length)
517
+ if (isSendGateShed(res)) anchorShed = true
518
+ else onEdit?.(id, head.text.length)
408
519
  } else {
409
520
  throw err
410
521
  }
@@ -414,10 +525,20 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
414
525
  // that only just came into existence) — we never re-send tails already
415
526
  // emitted on a prior flush. This is the fix for the duplicate-flood
416
527
  // blocker: the previous code re-sent pieces[1..n] as brand-new messages
417
- // on each throttled edit tick.
528
+ // on each throttled edit tick. Runs BEFORE the shed decision below so a
529
+ // shed (or benign) anchor never starves the tails (review F1).
530
+ let anyTailShed = false
418
531
  for (let pi = 1; pi < pieces.length; pi++) {
419
- await upsertTail(pi - 1, pieces[pi])
532
+ if (await upsertTail(pi - 1, pieces[pi])) anyTailShed = true
420
533
  }
534
+ // Review F3: ANY shed piece — anchor or tail — means this flush did not
535
+ // fully land. Throw the marker error so draft-stream does not record
536
+ // the body as delivered (its dedupe would freeze the shed piece
537
+ // forever) and instead preserves the snapshot for the finalize
538
+ // re-flush. Pieces that DID land are unaffected on that re-flush: the
539
+ // gate's no-op skip drops their identical payloads before the API, and
540
+ // landed tails short-circuit on tailLastText.
541
+ if (anchorShed || anyTailShed) throw makeDraftEditShedError(id)
421
542
  },
422
543
  {
423
544
  ...(throttleMs != null ? { throttleMs } : {}),
@@ -429,4 +550,6 @@ export function createStreamController(cfg: StreamControllerConfig): DraftStream
429
550
  chatId,
430
551
  },
431
552
  )
553
+ handleRef = handle
554
+ return handle
432
555
  }
@@ -516,10 +516,20 @@ export async function handleStreamReply(
516
516
  state.activeDraftStreams.set(sKey, stream)
517
517
  }
518
518
 
519
- await stream.update(effectiveText)
519
+ if (!done) {
520
+ // Intermediate snapshot — an ordinary throttled draft update.
521
+ await stream.update(effectiveText)
522
+ }
520
523
 
521
524
  if (done) {
522
- await stream.finalize()
525
+ // #3110: route the FINAL text through finalize(text) so the flush that
526
+ // renders the completed answer runs with the stream already final —
527
+ // stream-controller then classifies that edit `critical` for the send
528
+ // gate (never shed; fails fast with a structured FLOOD_WAIT_ACTIVE on a
529
+ // long flood window) while intermediate draft edits stay `cosmetic`
530
+ // (sheddable). The previous update()-then-finalize() pair flushed the
531
+ // final text inside update(), i.e. as an ordinary sheddable draft edit.
532
+ await stream.finalize(effectiveText)
523
533
  state.activeDraftStreams.delete(sKey)
524
534
  // #1713: stream_reply done=true is a NON-EVENT for the status
525
535
  // reaction. The reaction reflects current turn activity, not
@@ -42,6 +42,7 @@ import {
42
42
  safeActionForRecord,
43
43
  holdReasonFor,
44
44
  heldRetryBackoffMs,
45
+ applyDeliveredHoldReset,
45
46
  type UndeliverableMark,
46
47
  type BlockedApprovalStore,
47
48
  } from '../gateway/approval-hold.js'
@@ -280,12 +281,11 @@ export function createHarness(opts: { cap?: number; targets?: string[] } = {}):
280
281
  const live = pending.get(requestId)
281
282
  if (live && sent) {
282
283
  live.cards.push({ chatId, messageId: sent.message_id })
283
- if (live.undeliverable != null) {
284
- live.undeliverable = null
285
- live.redeliveryFailures = 0
286
- // PR 3: the operator's decision window starts NOW — until this
287
- // moment they had nothing to answer.
288
- live.startedAt = clock.now()
284
+ // Drive the SAME reset the gateway drives (#3128) — no private copy.
285
+ // Deleting `startedAt = now` from `applyDeliveredHoldReset` turns the
286
+ // `(d2)` full-window outcome test RED, because there is one shared
287
+ // implementation, not a mirror the harness silently compensates with.
288
+ if (applyDeliveredHoldReset(live, clock.now())) {
289
289
  reconcile()
290
290
  }
291
291
  }
@@ -289,8 +289,16 @@ describe('gateway wiring — the leash', () => {
289
289
  expect(sweep).not.toContain('if (now - v.startedAt > ttl)')
290
290
  })
291
291
 
292
- it('successful (re)delivery resets startedAt', () => {
293
- expect(GATEWAY_SRC).toContain('live.startedAt = Date.now()')
292
+ it('successful (re)delivery resets startedAt through the SHARED reset (#3128)', () => {
293
+ // This pin is no longer the safety net for the reset — the behavioural `(d2)`
294
+ // test above is, and it now drives the same `applyDeliveredHoldReset` the
295
+ // gateway does, so deleting `startedAt = now` from that one shared function
296
+ // turns `(d2)` RED. This only guards the WIRING: that the gateway routes
297
+ // delivery through the shared reset and can't regrow a private inline copy
298
+ // that drifts from what the harness exercises.
299
+ expect(GATEWAY_SRC).toContain('applyDeliveredHoldReset(live, Date.now())')
300
+ // …and must NOT carry its own inline copy of the reset.
301
+ expect(GATEWAY_SRC).not.toContain('live.startedAt = Date.now()')
294
302
  })
295
303
 
296
304
  it('the missed-approvals re-offer no longer drops a card that never landed', () => {
@@ -84,7 +84,11 @@ export function createMockBot(startMessageId = 500): MockBot {
84
84
  const api: MockBotApi = {
85
85
  sendMessage: vi.fn(async () => ({ message_id: state.nextMessageId++ })),
86
86
  sendRichMessage: vi.fn(async () => ({ message_id: state.nextMessageId++ })),
87
- editMessageText: vi.fn(async () => undefined),
87
+ // Faithful to grammy: editMessageText resolves `Message | true`, NEVER
88
+ // undefined. An `undefined` from the production retry stack means the
89
+ // send gate shed/skipped the call (#3110) — stream-controller treats it
90
+ // as not-landed — so the mock default must not be undefined.
91
+ editMessageText: vi.fn(async () => true as const),
88
92
  deleteMessage: vi.fn(async () => true as const),
89
93
  setMessageReaction: vi.fn(async () => true as const),
90
94
  editMessageReplyMarkup: vi.fn(async () => undefined),
@@ -122,7 +126,8 @@ export function installBotResetHook(bot: MockBot): void {
122
126
  bot.api.sendRichMessage.mockImplementation(async () => ({
123
127
  message_id: bot.nextMessageId++,
124
128
  }))
125
- bot.api.editMessageText.mockImplementation(async () => undefined)
129
+ // Faithful to grammy: `Message | true`, never undefined (see above).
130
+ bot.api.editMessageText.mockImplementation(async () => true as const)
126
131
  bot.api.deleteMessage.mockImplementation(async () => true as const)
127
132
  bot.api.setMessageReaction.mockImplementation(async () => true as const)
128
133
  bot.api.editMessageReplyMarkup.mockImplementation(async () => undefined)
@@ -99,6 +99,9 @@ function makeWatchdog(overrides: Partial<BridgeDeadWatchdogOpts> = {}) {
99
99
  const state = { sessionAlive: true, shuttingDown: false, escalateResult: true }
100
100
  const wd = createBridgeDeadWatchdog({
101
101
  graceMs: 90_000,
102
+ // The gateway serves AGENT — only AGENT's own bridge drives the
103
+ // watchdog (#3086). Overridable per-test.
104
+ selfAgentName: AGENT,
102
105
  isSessionAlive: () => state.sessionAlive,
103
106
  isShuttingDown: () => state.shuttingDown,
104
107
  escalate: (reason) => {
@@ -386,6 +389,64 @@ describe('cron / anonymous identity gating', () => {
386
389
  })
387
390
  })
388
391
 
392
+ // ─── #3086: secondary/relay identity gating ──────────────────────────────────
393
+ //
394
+ // Repro of the klanker incident: a short-lived `overlord-relay` IPC client
395
+ // registers and disconnects against klanker's gateway socket while klanker's
396
+ // OWN bridge is registered and actively replying. Pre-fix, isRealBridgeIdentity
397
+ // treated the relay (named, non-cron) as the real bridge, so its disconnect
398
+ // flipped bridgeRegistered=false, re-armed the grace window, and 90s later the
399
+ // watchdog SIGTERM'd a healthy container — killing the in-flight claude session.
400
+ describe('#3086 secondary/relay identity gating', () => {
401
+ const RELAY = 'overlord-relay' // a DIFFERENT agent name than AGENT
402
+
403
+ it('a relay disconnecting while the primary bridge is alive does NOT re-arm or escalate', () => {
404
+ const { wd, harness, escalations } = makeWatchdog() // selfAgentName === AGENT
405
+ wd.arm()
406
+ wd.noteBridgeRegistered(AGENT) // klanker's own bridge — healthy
407
+ expect(harness.liveCount()).toBe(0) // grace timer stood down
408
+ // The relay connects then disconnects (flood-ban workaround churn).
409
+ wd.noteBridgeRegistered(RELAY)
410
+ wd.noteBridgeDisconnected(RELAY)
411
+ // Must NOT re-arm: the primary bridge never went away.
412
+ expect(harness.liveCount()).toBe(0)
413
+ expect(escalations).toEqual([])
414
+ expect(wd.hasEscalated()).toBe(false)
415
+ })
416
+
417
+ it('a relay register does NOT satisfy the watchdog (only the primary bridge does)', () => {
418
+ const { wd, harness, escalations } = makeWatchdog()
419
+ wd.arm()
420
+ wd.noteBridgeRegistered(RELAY) // relay is not THIS gateway's bridge
421
+ expect(harness.liveCount()).toBe(1) // still armed — primary still missing
422
+ harness.fireLatest()
423
+ // Primary bridge genuinely absent → the real failure still escalates.
424
+ expect(escalations).toEqual([BRIDGE_DEAD_RESTART_REASON])
425
+ })
426
+
427
+ it('a genuine PRIMARY bridge death still escalates (fix does not over-suppress)', () => {
428
+ const { wd, harness, escalations } = makeWatchdog()
429
+ wd.arm()
430
+ wd.noteBridgeRegistered(AGENT) // primary registers
431
+ expect(harness.liveCount()).toBe(0)
432
+ wd.noteBridgeDisconnected(AGENT) // primary dies for good, never reconnects
433
+ expect(harness.liveCount()).toBe(1) // re-armed
434
+ harness.fireLatest()
435
+ expect(escalations).toEqual([BRIDGE_DEAD_RESTART_REASON])
436
+ })
437
+
438
+ it('fallback: with no selfAgentName, any named non-cron client still counts (pre-#3086)', () => {
439
+ const { wd, harness, escalations } = makeWatchdog({ selfAgentName: '' })
440
+ wd.arm()
441
+ wd.noteBridgeRegistered(RELAY) // no identity to match against → treated as real
442
+ expect(harness.liveCount()).toBe(0) // satisfied the watchdog
443
+ wd.noteBridgeDisconnected(RELAY)
444
+ expect(harness.liveCount()).toBe(1) // re-armed
445
+ harness.fireLatest()
446
+ expect(escalations).toEqual([BRIDGE_DEAD_RESTART_REASON])
447
+ })
448
+ })
449
+
389
450
  // ─── Crash-log tail ──────────────────────────────────────────────────────────
390
451
 
391
452
  describe('readFreshCrashLogTail', () => {
@@ -1,5 +1,5 @@
1
1
  import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'
2
- import { createDraftStream } from '../draft-stream.js'
2
+ import { createDraftStream, makeDraftEditShedError } from '../draft-stream.js'
3
3
 
4
4
  interface MockTelegram {
5
5
  send: (text: string) => Promise<number>
@@ -137,6 +137,115 @@ describe('createDraftStream', () => {
137
137
  expect(stream.isFinal()).toBe(true)
138
138
  })
139
139
 
140
+ it('finalize(finalText) flushes the supplied snapshot with the stream already final (#3110)', async () => {
141
+ const m = makeMock()
142
+ // Capture what isFinal() reads AT EDIT TIME — the transport layer
143
+ // (stream-controller) classifies the send-gate priority from exactly
144
+ // this signal, so the final snapshot MUST flush with final=true.
145
+ const finalAtEdit: boolean[] = []
146
+ const stream = createDraftStream(
147
+ m.send,
148
+ async (id, text) => {
149
+ finalAtEdit.push(stream.isFinal())
150
+ await m.edit(id, text)
151
+ },
152
+ { throttleMs: 1000 },
153
+ )
154
+
155
+ void stream.update('initial')
156
+ await microtaskFlush()
157
+ expect(m.sendCalls.length).toBe(1)
158
+
159
+ // A stale draft is pending; finalize(text) supersedes it (last-write-wins).
160
+ void stream.update('stale draft')
161
+ await microtaskFlush()
162
+ await stream.finalize('the completed answer')
163
+
164
+ expect(m.editCalls.length).toBe(1)
165
+ expect(m.editCalls[0].text).toBe('the completed answer')
166
+ expect(finalAtEdit).toEqual([true])
167
+ expect(stream.isFinal()).toBe(true)
168
+ })
169
+
170
+ it('a shed flush preserves the snapshot; argument-less finalize() re-delivers it (#3110 F2)', async () => {
171
+ const m = makeMock()
172
+ let shedNext = true
173
+ const stream = createDraftStream(
174
+ m.send,
175
+ async (id, text) => {
176
+ if (shedNext) {
177
+ shedNext = false
178
+ throw makeDraftEditShedError(id)
179
+ }
180
+ await m.edit(id, text)
181
+ },
182
+ { throttleMs: 1000 },
183
+ )
184
+
185
+ void stream.update('v1')
186
+ await microtaskFlush()
187
+ expect(m.sendCalls.length).toBe(1)
188
+
189
+ void stream.update('v2 — shed by the gate')
190
+ vi.advanceTimersByTime(1000)
191
+ await microtaskFlush()
192
+ // The edit was shed: nothing landed, and the snapshot must NOT be
193
+ // recorded as sent.
194
+ expect(m.editCalls.length).toBe(0)
195
+
196
+ // The gateway's cleanup paths finalize with NO argument — the shed
197
+ // snapshot must be re-flushed as the stream's final state, not lost.
198
+ await stream.finalize()
199
+ expect(m.editCalls.length).toBe(1)
200
+ expect(m.editCalls[0].text).toBe('v2 — shed by the gate')
201
+ })
202
+
203
+ it('a newer landed flush supersedes an earlier shed snapshot (no stale resurrect)', async () => {
204
+ const m = makeMock()
205
+ let shedNext = true
206
+ const stream = createDraftStream(
207
+ m.send,
208
+ async (id, text) => {
209
+ if (shedNext) {
210
+ shedNext = false
211
+ throw makeDraftEditShedError(id)
212
+ }
213
+ await m.edit(id, text)
214
+ },
215
+ { throttleMs: 1000 },
216
+ )
217
+
218
+ void stream.update('v1')
219
+ await microtaskFlush()
220
+ void stream.update('v2 — shed')
221
+ vi.advanceTimersByTime(1000)
222
+ await microtaskFlush()
223
+ expect(m.editCalls.length).toBe(0)
224
+
225
+ // A NEWER snapshot lands normally — the shed one is now stale.
226
+ void stream.update('v3 — landed')
227
+ vi.advanceTimersByTime(1000)
228
+ await microtaskFlush()
229
+ expect(m.editCalls.map((c) => c.text)).toEqual(['v3 — landed'])
230
+
231
+ // finalize() must NOT resurrect the superseded shed snapshot.
232
+ await stream.finalize()
233
+ expect(m.editCalls.map((c) => c.text)).toEqual(['v3 — landed'])
234
+ })
235
+
236
+ it('finalize(finalText) still dedupes against text that actually landed', async () => {
237
+ const m = makeMock()
238
+ const stream = createDraftStream(m.send, m.edit, { throttleMs: 1000 })
239
+
240
+ void stream.update('the answer')
241
+ await microtaskFlush()
242
+ expect(m.sendCalls.length).toBe(1)
243
+
244
+ // Same text again as the final snapshot → already on screen, no edit.
245
+ await stream.finalize('the answer')
246
+ expect(m.editCalls.length).toBe(0)
247
+ })
248
+
140
249
  it('updates after finalize are silently dropped', async () => {
141
250
  const m = makeMock()
142
251
  const stream = createDraftStream(m.send, m.edit, { throttleMs: 1000 })
@@ -107,7 +107,7 @@ describe("effort-command: handler", () => {
107
107
  const { deps } = makeDeps({ getConfiguredEffort: () => "medium" });
108
108
  const r = await handleEffortCommand({ kind: "show" }, deps);
109
109
  expect(r.text).toContain("medium");
110
- expect(r.text).toMatch(/persists across restarts and deploys/);
110
+ expect(r.text).toMatch(/lasts until the agent’s next restart/);
111
111
  });
112
112
 
113
113
  it("show falls back to low when effort is unreadable", async () => {
@@ -121,7 +121,7 @@ describe("effort-command: handler", () => {
121
121
  const r = await handleEffortCommand({ kind: "set", level: "high" }, deps);
122
122
  expect(calls).toEqual([{ agent: "carrie", level: "high" }]);
123
123
  expect(r.text).toContain("Set effort level to high");
124
- expect(r.text).toMatch(/persists across restarts and deploys/);
124
+ expect(r.text).toMatch(/lasts until the agent’s next restart/);
125
125
  });
126
126
 
127
127
  it("set notes the re-read cost when a confirmation was needed", async () => {
@@ -249,9 +249,9 @@ describe("effort-command: /effort default (#3039)", () => {
249
249
  expect(marked?.text).toBe("✅ max");
250
250
  });
251
251
 
252
- it("help text advertises /effort default and the sticky contract", async () => {
252
+ it("help text advertises /effort default and the session-only contract", async () => {
253
253
  const r = await handleEffortCommand({ kind: "help" }, makeDeps().deps);
254
254
  expect(r.text).toContain("/effort default");
255
- expect(r.text).toContain("persists across restarts and deploys");
255
+ expect(r.text).toContain("lasts until the agent’s next restart");
256
256
  });
257
257
  });
@@ -122,6 +122,45 @@ describe('createFleetFallbackResumeGate — staleness guard', () => {
122
122
  })
123
123
  })
124
124
 
125
+ describe('createFleetFallbackResumeGate — peek / arm seams (tier-downgrade LOW-3 + LOW-9a)', () => {
126
+ it('peek() evaluates the verdict WITHOUT arming the single-flight latch', () => {
127
+ const clk = fakeClock()
128
+ const gate = createFleetFallbackResumeGate({ nowFn: clk.now })
129
+ // Peeking 'resume' must not record an arm time — a following peek/decide is
130
+ // still 'resume'. This is what lets the gateway do a fallible carrier write
131
+ // between the peek and the arm without leaving a phantom armed latch (LOW-3).
132
+ expect(gate.peek(clk.now())).toBe('resume')
133
+ expect(gate.inspect().lastResumedAtMs).toBe(Number.NEGATIVE_INFINITY)
134
+ expect(gate.peek(clk.now())).toBe('resume')
135
+ // The real restart never fires (write threw) → nothing armed → a genuine
136
+ // later swap still resumes.
137
+ expect(gate.decide(clk.now())).toBe('resume')
138
+ })
139
+
140
+ it('arm() commits the single-flight window so a later peek reports skip-inflight (LOW-9a)', () => {
141
+ const clk = fakeClock()
142
+ const gate = createFleetFallbackResumeGate({ nowFn: clk.now, singleFlightMs: 60_000 })
143
+ // Turn 1: peek 'resume' → do the write → arm.
144
+ expect(gate.peek(clk.now())).toBe('resume')
145
+ gate.arm()
146
+ // Turn 2 (concurrent, same process, 1s later): a resume restart is already
147
+ // armed, so peek reports skip-inflight → the gateway suppresses the give-up.
148
+ clk.advance(1_000)
149
+ expect(gate.peek(clk.now())).toBe('skip-inflight')
150
+ })
151
+
152
+ it('peek(skip-inflight/skip-stale) never arms; only arm() records the window', () => {
153
+ const clk = fakeClock()
154
+ const gate = createFleetFallbackResumeGate({ nowFn: clk.now })
155
+ // A stale peek must not arm (mirrors decide()'s stale behaviour).
156
+ expect(gate.peek(clk.now() - (DEFAULT_RESUME_MAX_AGE_MS + 1))).toBe('skip-stale')
157
+ expect(gate.inspect().lastResumedAtMs).toBe(Number.NEGATIVE_INFINITY)
158
+ // decide() remains equivalent to peek+arm-on-resume.
159
+ expect(gate.decide(clk.now())).toBe('resume')
160
+ expect(gate.peek(clk.now())).toBe('skip-inflight')
161
+ })
162
+ })
163
+
125
164
  describe('createFleetFallbackResumeGate — reset / inspect seams', () => {
126
165
  it('reset() clears the single-flight arm', () => {
127
166
  const clk = fakeClock()
@@ -14,7 +14,7 @@ import {
14
14
  FLOOD_STATE_MODE,
15
15
  FLOOD_WINDOWS_CORRUPT_SUPPRESS_MS,
16
16
  } from '../flood-circuit-breaker.js'
17
- import { createSendGate, type Clock } from '../send-gate.js'
17
+ import { createSendGate, SEND_GATE_SHED, type Clock } from '../send-gate.js'
18
18
 
19
19
  /**
20
20
  * #3084 PR 2 — restart-proof SCOPED flood windows (part3-design §7). Verifies
@@ -103,9 +103,10 @@ describe('#3084 scoped flood-window persistence', () => {
103
103
  ),
104
104
  ).rejects.toBe(floodErr)
105
105
 
106
- // The window is on disk.
106
+ // The window is on disk. #3111: a chat-bound 429 persists ONLY the chat
107
+ // scope — no coincident `global` window that would suppress unrelated chats.
107
108
  const persisted = readFloodWindows(winPath, 0)
108
- expect(persisted.map((w) => w.scopeKey).sort()).toEqual(['chat:7', 'global'])
109
+ expect(persisted.map((w) => w.scopeKey).sort()).toEqual(['chat:7'])
109
110
 
110
111
  // Gate 2 (simulated restart): built BEFORE any outbound call from the
111
112
  // persisted windows. A cosmetic send on chat:7 must still shed.
@@ -119,7 +120,7 @@ describe('#3084 scoped flood-window persistence', () => {
119
120
  chat_id: '7',
120
121
  priorityClass: 'cosmetic',
121
122
  })
122
- expect(res).toBeUndefined()
123
+ expect(res).toBe(SEND_GATE_SHED) // shed sentinel (#3110 F1)
123
124
  expect(gate2.stats().global.shed).toBe(1)
124
125
 
125
126
  // And a critical into the still-long window fails fast — restart did not