switchroom 0.18.10 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/dist/agent-scheduler/index.js +29 -5
  2. package/dist/auth-broker/index.js +53 -13
  3. package/dist/cli/hindsight-mental-model-pretool.mjs +39 -0
  4. package/dist/cli/notion-write-pretool.mjs +29 -5
  5. package/dist/cli/switchroom.js +2636 -1369
  6. package/dist/cli/ui/apple-touch-icon.png +0 -0
  7. package/dist/cli/ui/favicon-32.png +0 -0
  8. package/dist/cli/ui/favicon.ico +0 -0
  9. package/dist/cli/ui/index.html +163 -17
  10. package/dist/host-control/main.js +1248 -342
  11. package/dist/vault/approvals/kernel-server.js +54 -13
  12. package/dist/vault/broker/server.js +163 -114
  13. package/package.json +3 -4
  14. package/profiles/_base/start.sh.hbs +65 -0
  15. package/profiles/_shared/vault-protocol.md.hbs +3 -1
  16. package/profiles/coding/CLAUDE.md.hbs +1 -1
  17. package/profiles/default/CLAUDE.md.hbs +2 -2
  18. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  19. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  20. package/telegram-plugin/bridge/bridge.ts +37 -0
  21. package/telegram-plugin/bridge/inbound-dedup.ts +101 -0
  22. package/telegram-plugin/dist/bridge/bridge.js +73 -1
  23. package/telegram-plugin/dist/gateway/gateway.js +3603 -1007
  24. package/telegram-plugin/dist/server.js +74 -2
  25. package/telegram-plugin/flood-circuit-breaker.ts +493 -21
  26. package/telegram-plugin/gateway/approval-hold.ts +583 -0
  27. package/telegram-plugin/gateway/auth-command.ts +92 -2
  28. package/telegram-plugin/gateway/auth-loopback-relay.ts +670 -0
  29. package/telegram-plugin/gateway/boot-card.ts +12 -5
  30. package/telegram-plugin/gateway/callback-query-handlers.ts +76 -1
  31. package/telegram-plugin/gateway/config-approval-handler.ts +6 -1
  32. package/telegram-plugin/gateway/disconnect-flush.ts +19 -0
  33. package/telegram-plugin/gateway/dm-pin-sweep.test.ts +251 -0
  34. package/telegram-plugin/gateway/dm-pin-sweep.ts +178 -0
  35. package/telegram-plugin/gateway/gateway.ts +1482 -165
  36. package/telegram-plugin/gateway/hostd-dispatch.ts +23 -0
  37. package/telegram-plugin/gateway/idle-clear.ts +90 -6
  38. package/telegram-plugin/gateway/inbound-delivery-machine-shadow.ts +26 -5
  39. package/telegram-plugin/gateway/inject-handler.ts +8 -0
  40. package/telegram-plugin/gateway/ipc-protocol.ts +46 -3
  41. package/telegram-plugin/gateway/ipc-server.ts +43 -0
  42. package/telegram-plugin/gateway/mental-model-propose-resolve.ts +145 -37
  43. package/telegram-plugin/gateway/model-command.ts +9 -3
  44. package/telegram-plugin/gateway/pending-session-command.ts +13 -1
  45. package/telegram-plugin/gateway/permission-ttl-sweep.ts +66 -0
  46. package/telegram-plugin/gateway/pre-approval-check.ts +74 -0
  47. package/telegram-plugin/gateway/queued-card-store.ts +217 -0
  48. package/telegram-plugin/gateway/session-model-file.ts +26 -1
  49. package/telegram-plugin/gateway/turn-end-gate-backstop.ts +59 -0
  50. package/telegram-plugin/gateway/turn-end-gate.ts +95 -0
  51. package/telegram-plugin/gateway/turn-typing-loop.ts +10 -2
  52. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +13 -0
  53. package/telegram-plugin/hooks/dispatch-claim-scan.mjs +259 -0
  54. package/telegram-plugin/hooks/dispatch-claim-stop.mjs +129 -0
  55. package/telegram-plugin/hooks/hooks.json +9 -0
  56. package/telegram-plugin/inline-keyboard-callbacks.ts +209 -2
  57. package/telegram-plugin/operator-events.ts +23 -0
  58. package/telegram-plugin/package.json +0 -1
  59. package/telegram-plugin/permission-rule.ts +1 -0
  60. package/telegram-plugin/permission-title.ts +1 -0
  61. package/telegram-plugin/retry-api-call.ts +212 -2
  62. package/telegram-plugin/send-gate-degraded.test.ts +443 -0
  63. package/telegram-plugin/send-gate-observability.test.ts +470 -0
  64. package/telegram-plugin/send-gate-observability.ts +355 -0
  65. package/telegram-plugin/send-gate.test.ts +698 -0
  66. package/telegram-plugin/send-gate.ts +982 -0
  67. package/telegram-plugin/shared/bot-runtime.ts +17 -5
  68. package/telegram-plugin/shared/gw-trace-gate.ts +105 -0
  69. package/telegram-plugin/status-pin-driver.ts +52 -7
  70. package/telegram-plugin/status-pin.ts +81 -0
  71. package/telegram-plugin/subagent-watcher.ts +102 -2
  72. package/telegram-plugin/tests/activity-card-wiring.test.ts +18 -5
  73. package/telegram-plugin/tests/approval-hold-harness.ts +425 -0
  74. package/telegram-plugin/tests/approval-hold-outcome.test.ts +296 -0
  75. package/telegram-plugin/tests/approval-hold-record.test.ts +531 -0
  76. package/telegram-plugin/tests/approval-hold-redeliver.test.ts +602 -0
  77. package/telegram-plugin/tests/auth-loopback-relay.test.ts +533 -0
  78. package/telegram-plugin/tests/boot-card-flood-suppress.test.ts +53 -7
  79. package/telegram-plugin/tests/busy-key-reaper.test.ts +1 -0
  80. package/telegram-plugin/tests/dispatch-claim-scan.test.ts +250 -0
  81. package/telegram-plugin/tests/flood-breaker-blindness.test.ts +213 -0
  82. package/telegram-plugin/tests/flood-windows-persistence.test.ts +224 -0
  83. package/telegram-plugin/tests/gateway-boot-marker-clear.test.ts +3 -3
  84. package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +29 -1
  85. package/telegram-plugin/tests/gateway-loopback-paste-redact.test.ts +66 -0
  86. package/telegram-plugin/tests/gw-trace-gate.test.ts +105 -0
  87. package/telegram-plugin/tests/idle-clear.test.ts +233 -3
  88. package/telegram-plugin/tests/inbound-dedup.test.ts +93 -0
  89. package/telegram-plugin/tests/inline-keyboard-callbacks.test.ts +284 -0
  90. package/telegram-plugin/tests/ipc-server-check-pre-approved.test.ts +194 -0
  91. package/telegram-plugin/tests/mental-model-propose-resolve.test.ts +123 -0
  92. package/telegram-plugin/tests/missed-approvals-wiring.test.ts +1 -1
  93. package/telegram-plugin/tests/model-command.test.ts +14 -0
  94. package/telegram-plugin/tests/pending-session-command.test.ts +21 -0
  95. package/telegram-plugin/tests/permission-card-routing.test.ts +30 -5
  96. package/telegram-plugin/tests/permission-no-repeat-wiring.test.ts +8 -7
  97. package/telegram-plugin/tests/permission-rearm-wiring.test.ts +1 -1
  98. package/telegram-plugin/tests/pre-approval-check.test.ts +148 -0
  99. package/telegram-plugin/tests/queued-card-store.test.ts +232 -0
  100. package/telegram-plugin/tests/reaction-flush-turn-gated.test.ts +100 -0
  101. package/telegram-plugin/tests/retry-api-call.test.ts +398 -0
  102. package/telegram-plugin/tests/session-model-file.test.ts +50 -0
  103. package/telegram-plugin/tests/status-pin-boot-recovery.test.ts +35 -14
  104. package/telegram-plugin/tests/status-pin.test.ts +275 -1
  105. package/telegram-plugin/tests/subagent-watcher-deferral-log-ratelimit.test.ts +316 -0
  106. package/telegram-plugin/tests/turn-end-gate-backstop.test.ts +92 -0
  107. package/telegram-plugin/tests/turn-end-gate.test.ts +137 -0
  108. package/telegram-plugin/tests/typing-emitter.test.ts +586 -0
  109. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +20 -0
  110. package/telegram-plugin/typing-emitter.ts +224 -0
  111. package/telegram-plugin/uat/scenarios/jtbd-feel-like-a-colleague-dm.test.ts +136 -0
  112. package/telegram-plugin/welcome-text.ts +42 -0
  113. package/vendor/hindsight-memory/scripts/drain_pending.py +22 -6
  114. package/vendor/hindsight-memory/scripts/lib/client.py +12 -5
  115. package/vendor/hindsight-memory/scripts/lib/directives.py +38 -3
  116. package/vendor/hindsight-memory/scripts/lib/pending.py +36 -9
  117. package/vendor/hindsight-memory/scripts/session_end.py +14 -3
  118. package/vendor/hindsight-memory/scripts/session_start.py +21 -0
  119. package/vendor/hindsight-memory/scripts/tests/test_directives.py +38 -0
  120. package/vendor/hindsight-memory/tests/test_drain_pending.py +68 -0
  121. package/vendor/hindsight-memory/tests/test_pending.py +44 -0
  122. package/vendor/hindsight-memory/tests/test_session_end_pending.py +38 -0
  123. package/vendor/hindsight-memory/tests/test_session_start_drain.py +155 -0
  124. package/telegram-plugin/channel-envelope-safety.test.ts +0 -56
  125. package/telegram-plugin/channel-envelope-safety.ts +0 -56
@@ -112,7 +112,7 @@ import {
112
112
  import type { WebhookGatewayRecord } from '../../src/web/webhook-gateway-record.js'
113
113
  import { reconcilePin, type PinBotApi } from '../status-pin-driver.js'
114
114
  import type { PinState, DesiredPin } from '../status-pin.js'
115
- import { decidePinAction } from '../status-pin.js'
115
+ import { decidePinAction, PinRightsCache } from '../status-pin.js'
116
116
  import { formatTurnLifecycle, detectStatusSurfaceDegraded } from './status-surface-log.js'
117
117
  import { parseSourceMessageId } from './source-message-id.js'
118
118
  import {
@@ -155,6 +155,7 @@ import {
155
155
  computeBootSweepStripTargets,
156
156
  distinctRequestIds,
157
157
  } from './permission-rearm.js'
158
+ import { sweepPermissionTtl } from './permission-ttl-sweep.js'
158
159
  import { createMissedApprovalsStore, type MissedApproval } from './missed-approvals-store.js'
159
160
  import {
160
161
  createAlwaysAllowPersistQueue,
@@ -168,6 +169,15 @@ import {
168
169
  buildMissedApprovalRetryInbound,
169
170
  } from './missed-approvals-card.js'
170
171
  import { pickRecoveredPermissionOrigin } from './permission-card-origin.js'
172
+ import {
173
+ createBlockedApprovalStore,
174
+ safeActionForRecord,
175
+ selectOldestHeld,
176
+ selectHeldForRedelivery,
177
+ holdReasonFor,
178
+ heldRetryBackoffMs,
179
+ type UndeliverableMark,
180
+ } from './approval-hold.js'
171
181
  import { isTelegramReplyTool, isTelegramSurfaceTool } from '../tool-names.js'
172
182
  import { appendActivityLabel, clipNarrative, renderActivityFeedWithNested, formatStepSuffix, type SessionActivityHeader } from '../tool-activity-summary.js'
173
183
  import { formatModelLabel } from '../model-label.js'
@@ -177,6 +187,7 @@ import { REPLY_TOOLS, isDraftOfReply } from '../narrative-dedup.js'
177
187
  import { toolLabel } from '../tool-labels.js'
178
188
  import { createTypingWrapper } from '../typing-wrap.js'
179
189
  import { createTurnTypingLoop } from './turn-typing-loop.js'
190
+ import { createTypingEmitter, TYPING_REFRESH_MS } from '../typing-emitter.js'
180
191
  import { type DraftStreamHandle } from '../draft-stream.js'
181
192
  import { handlePtyPartialPure, type PtyHandlerState } from '../pty-partial-handler.js'
182
193
  import { handleStreamReply } from '../stream-reply-handler.js'
@@ -186,10 +197,23 @@ import {
186
197
  createSwallowingRetryApiCall,
187
198
  retryWithThreadFallback,
188
199
  isPhotoDimensionRejectError,
200
+ isFloodWaitActiveError,
189
201
  } from '../retry-api-call.js'
202
+ import { createSendGate, sendGateEnabledFromEnv } from '../send-gate.js'
203
+ import { createStatsLogger, createFloodWindowObserver } from '../send-gate-observability.js'
190
204
  import { classifyPhotoFile, rerouteResultSuffix } from '../photo-precheck.js'
191
205
  import { installTgPostLogger, withTgPostTags } from '../shared/bot-runtime.js'
192
- import { floodStatePath, makeFloodWaitRecorder } from '../flood-circuit-breaker.js'
206
+ import {
207
+ floodStatePath,
208
+ floodWindowsPath,
209
+ makeFloodWaitRecorder,
210
+ makeFloodWaitProbe,
211
+ makeFloodWindowRecorder,
212
+ loadInitialFloodWindows,
213
+ readFloodWindows,
214
+ markFloodWindowAlerted,
215
+ suppressNonEssentialSendMs,
216
+ } from '../flood-circuit-breaker.js'
193
217
  import { buildAttachmentPath, assertInsideInbox } from '../attachment-path.js'
194
218
  import { logStreamingEvent } from '../streaming-metrics.js'
195
219
  import * as signalTracker from '../turn-signal-tracker.js'
@@ -253,6 +277,14 @@ import {
253
277
  cancelAccountAuthSession,
254
278
  cleanScratchDir as cleanAuthAddScratchDir,
255
279
  } from './auth-add-flow.js'
280
+ import {
281
+ pendingLoopbackFlows,
282
+ startLoopbackFlow,
283
+ submitLoopbackRedirect,
284
+ cancelLoopbackFlow,
285
+ shouldConsumeLoopbackPaste,
286
+ trackFlowExit,
287
+ } from './auth-loopback-relay.js'
256
288
  import {
257
289
  initHistory, recordInbound, recordOutbound, recordEdit,
258
290
  deleteFromHistory, query as queryHistory, getLatestInboundMessageId,
@@ -312,6 +344,8 @@ import {
312
344
  extractAgentButtonMeta,
313
345
  keyboardIsSingleUse,
314
346
  finalizeCallback,
347
+ resolveTapAnnotation,
348
+ applyTapAnnotationEdit,
315
349
  type AgentButtonMeta,
316
350
  } from '../inline-keyboard-callbacks.js'
317
351
  import {
@@ -340,6 +374,8 @@ import {
340
374
  decideTurnFlush,
341
375
  isTurnFlushSafetyEnabled,
342
376
  } from '../turn-flush-safety.js'
377
+ // #1667 — pure decision core for the turn_end answer-delivery gate (#1664).
378
+ import { decideTurnEndGate } from './turn-end-gate.js'
343
379
  // #1122 PR3: turn-flush-prose-recovery removed with the progress card.
344
380
  import { resolveAgentDirFromEnv } from '../agent-dir.js'
345
381
  import {
@@ -408,6 +444,7 @@ import {
408
444
  readSessionModelFileRaw,
409
445
  restoreSessionModelFileRaw,
410
446
  clearSessionModelFile,
447
+ clearSessionModelBootAttempts,
411
448
  readConfiguredDefaultModel,
412
449
  writeRelaunchModelIntent,
413
450
  clearRelaunchModelIntent,
@@ -453,7 +490,7 @@ import { resolveOutboundTopic as resolveOutboundTopicHelper, topicForRecipient,
453
490
  import { readTurnUsages } from '../../src/agents/perf.js'
454
491
  import { buildContextOccupancy, writeContextOccupancySnapshot } from './context-occupancy.js'
455
492
  import { decideProactiveCompact, initialCompactState, type CompactState } from './proactive-compact.js'
456
- import { decideIdleClear, idleDurationToMs, DEFAULT_IDLE_CLEAR_MS } from './idle-clear.js'
493
+ import { decideIdleClear, classifyIdleEvent, idleDurationToMs, DEFAULT_IDLE_CLEAR_MS } from './idle-clear.js'
457
494
  import { nextCompactNotify, idleCompactNotifyState, type CompactNotifyState } from './compact-notify.js'
458
495
  import {
459
496
  tryHostdDispatch,
@@ -463,12 +500,19 @@ import {
463
500
  pollHostdStatus,
464
501
  hostdGetStatusOnce,
465
502
  warnLegacySpawnIfHostdDisabled,
503
+ withOperatorAttestation,
466
504
  _resetHostdEnabledCache,
467
505
  } from './hostd-dispatch.js'
468
506
  import { formatUpdateStatusLine } from './update-status-line.js'
469
507
  import type { HostdRequest, HostdResponse } from '../../src/host-control/protocol.js'
470
508
  import type { AgentAudit } from '../welcome-text.js'
471
509
  import { shouldSweepChatAtBoot } from './boot-sweep-filter.js'
510
+ import {
511
+ createDmPinSweeper,
512
+ collectDmChatIdsFromStores,
513
+ unexpiredStoreRepinIds,
514
+ type DmPinSweeper,
515
+ } from './dm-pin-sweep.js'
472
516
  import { startWebhookIngestServer } from './webhook-ingest-server.js'
473
517
  import { recordWebhookEvent } from '../../src/web/webhook-gateway-record.js'
474
518
 
@@ -495,6 +539,7 @@ import {
495
539
  type TrackedStatusPin,
496
540
  } from './status-pin-store.js'
497
541
  import {
542
+ loadActivityCards,
498
543
  writeActivityCardRecord,
499
544
  clearActivityCardRecord,
500
545
  runActivityCardBootReaper,
@@ -502,6 +547,13 @@ import {
502
547
  restartOrphanCardFinalizeText,
503
548
  type ActivityCardStoreFsSeam,
504
549
  } from './activity-card-store.js'
550
+ import {
551
+ loadQueuedCards,
552
+ writeQueuedCardRecord,
553
+ clearQueuedCardRecord,
554
+ runQueuedCardBootReaper,
555
+ type QueuedCardStoreFsSeam,
556
+ } from './queued-card-store.js'
505
557
  import {
506
558
  decideWorkerPinReaps,
507
559
  WORKER_PIN_TTL_MS_DEFAULT,
@@ -511,6 +563,7 @@ import { shouldSuppressRepresent } from './represent-guard.js'
511
563
  import { shouldDeferEscalationForBridge } from './escalation-bridge-gate.js'
512
564
  import { createInboundSpool } from './inbound-spool.js'
513
565
  import { purgeStaleTurnsForChat } from './turn-state-purge.js'
566
+ import { withTurnEndGateBackstop } from './turn-end-gate-backstop.js'
514
567
  import { decideInboundDelivery, reserveInboundDelivery } from './inbound-delivery-gate.js'
515
568
  import { mayDrainBufferedInbound, shouldArmNoReplyDrain } from './serialize-drain-gate.js'
516
569
  import { decideFeedReopen } from './feed-reopen-gate.js'
@@ -589,6 +642,7 @@ import type {
589
642
  SendOutboundMessage,
590
643
  QuotaWallDetectedMessage,
591
644
  QueryPendingPermissionMessage,
645
+ CheckPreApprovedMessage,
592
646
  PostSkillProposalMessage,
593
647
  PermissionEvent,
594
648
  RolloutStatusPostMessage,
@@ -664,6 +718,7 @@ import {
664
718
  import { createScopedGrantStore } from './scoped-grant-store.js'
665
719
  import { grantRestartDecision, type GrantRestartDecision } from './grant-restart.js'
666
720
  import { synthesizeAllowRuleDiff, extractAddedAllowRule } from '../permission-diff.js'
721
+ import { isDiffPreApproved } from './pre-approval-check.js'
667
722
  import {
668
723
  readClaudeJsonOverage,
669
724
  evaluateCreditState,
@@ -790,6 +845,69 @@ process.on('beforeExit', () => {
790
845
  // ─── Env + state dir ──────────────────────────────────────────────────────
791
846
  const STATE_DIR = process.env.TELEGRAM_STATE_DIR ?? join(homedir(), '.claude', 'channels', 'telegram')
792
847
  const permCardStore = createPermissionCardStore(STATE_DIR)
848
+ // #3084 follow-up — the blocked-approval surface. A SHARED top-level directory
849
+ // (compose binds host `~/.switchroom/blocked-approvals` here), not per-agent
850
+ // state: switchroom-web reads every agent's record from one place, and the
851
+ // records are 0644 so its uid-1000 process can actually read them. The agent's
852
+ // own telegram state is 0600 and is NOT web-readable.
853
+ // If the mount is missing (a container predating the volume), the write lands
854
+ // inside the container and the hold still works — only the off-Telegram
855
+ // SURFACE is lost. The record is best-effort by construction; it must never be
856
+ // able to fail the hold back into an auto-deny.
857
+ const BLOCKED_APPROVALS_DIR =
858
+ process.env.SWITCHROOM_BLOCKED_APPROVALS_DIR ?? '/state/blocked-approvals'
859
+ // The canonical "which agent am I" identity (profiles/_base/start.sh.hbs:430).
860
+ const AGENT_NAME = process.env.SWITCHROOM_AGENT_NAME ?? 'agent'
861
+ const blockedApprovalStore = createBlockedApprovalStore(
862
+ BLOCKED_APPROVALS_DIR,
863
+ AGENT_NAME,
864
+ // Fallback: the agent's OWN state dir, which the scaffold chowns to the agent
865
+ // uid, so a write there always succeeds. Guarantees the record can never be
866
+ // silently lost when the shared dir isn't writable by this agent's uid.
867
+ process.env.SWITCHROOM_AGENT_STATE_DIR ?? '/state/agent',
868
+ )
869
+
870
+ /**
871
+ * Rewrite the blocked-approval surface from the LIVE pending state.
872
+ *
873
+ * A reconcile, not a delete — deliberately. The store holds one record per
874
+ * agent, but an agent can hold SEVERAL permissions at once: Claude Code issues
875
+ * parallel tool calls, each hitting `canUseTool`, and
876
+ * `cancelPendingPermissionsForHalt` iterates multiple pendings. A naive
877
+ * `clear()` on any single resolution would delete the file while ANOTHER
878
+ * request was still held, blinding the operator to a real, live block.
879
+ *
880
+ * So derive the record from whatever is still held. Oldest block wins — it has
881
+ * been waiting longest. Nothing held → clear.
882
+ *
883
+ * Also called at BOOT. `pendingPermissions` is in-memory and empty on a fresh
884
+ * process, so this clears any record orphaned by a restart mid-hold — otherwise
885
+ * the dashboard would show a permanently blocked agent forever. Nothing can
886
+ * legitimately be held across a restart: the bridge re-sends unresolved requests
887
+ * on IPC reconnect (`bridge/permission-ledger.ts`), which re-raises them and
888
+ * re-marks them if the channel is still shut.
889
+ */
890
+ function reconcileBlockedApprovals(): void {
891
+ // Selection (oldest hold wins; null ⇒ nothing held) is the pure, unit-tested
892
+ // `selectOldestHeld` in approval-hold.ts — keep policy out of gateway.ts.
893
+ const oldest = selectOldestHeld(pendingPermissions)
894
+ if (oldest == null) {
895
+ blockedApprovalStore.clear()
896
+ return
897
+ }
898
+ blockedApprovalStore.write({
899
+ agent: AGENT_NAME,
900
+ requestId: oldest.requestId,
901
+ toolName: oldest.pend.tool_name,
902
+ // NOT bare naturalAction(): for Bash/Glob/Grep/WebFetch/mcp__* it
903
+ // interpolates the RAW tool input, and this file is world-readable.
904
+ action: safeActionForRecord(naturalAction, oldest.pend.tool_name, oldest.pend.input_preview),
905
+ blockedSince: oldest.pend.startedAt,
906
+ undeliverableSince: oldest.mark.since,
907
+ retryableAt: oldest.mark.retryableAt,
908
+ reason: oldest.mark.reason,
909
+ })
910
+ }
793
911
  // Durable store for the four AGENT-INITIATED approval-card families
794
912
  // (vault_request_access / vault_request_save / request_secret /
795
913
  // mental_model_propose). Persists card METADATA only — never any secret value
@@ -1347,6 +1465,17 @@ type Access = {
1347
1465
  short_name?: string
1348
1466
  author_name?: string
1349
1467
  }
1468
+ /** Auto-confirm "✅ You chose: X" annotation on inline_keyboard taps
1469
+ * (#789). When enabled, tapping an agent-emitted single-use button
1470
+ * annotates the source message body so the chat surface is
1471
+ * self-documenting. Off by default; only honored when parseMode is the
1472
+ * default 'html'. Projected from
1473
+ * channels.telegram.button_choice_confirmation by scaffold. */
1474
+ button_choice_confirmation?: {
1475
+ enabled?: boolean
1476
+ format?: string
1477
+ timezone?: 'gateway' | 'utc'
1478
+ }
1350
1479
  }
1351
1480
 
1352
1481
  function defaultAccess(): Access {
@@ -1431,6 +1560,8 @@ function readAccessFile(): Access {
1431
1560
  voice_in: parsed.voice_in,
1432
1561
  voice_out: parsed.voice_out,
1433
1562
  telegraph: parsed.telegraph,
1563
+ // #789: button-choice-confirmation config projected by scaffold.
1564
+ button_choice_confirmation: parsed.button_choice_confirmation,
1434
1565
  }
1435
1566
  } catch (err) {
1436
1567
  if ((err as NodeJS.ErrnoException).code === 'ENOENT') return defaultAccess()
@@ -2586,6 +2717,17 @@ function turnInFlightForGate(): boolean {
2586
2717
  // Keeping the gate closed for as long as there is an outstanding approval
2587
2718
  // card prevents that race. The gate reopens when the card is tapped or times
2588
2719
  // out (pendingPermissions.delete in finalizeCallback / TTL sweep).
2720
+ //
2721
+ // #3084 follow-up — ONE case now has no timeout valve: an approval HELD
2722
+ // because its card could not be DELIVERED (a Telegram flood ban) never
2723
+ // expires, by design. The leash may not fabricate a verdict for an ask no
2724
+ // human ever saw, so the entry — and therefore this gate — persists until a
2725
+ // human answers. An agent can consequently buffer normal inbound for the
2726
+ // duration of a channel outage. That is the intended trade: a buffered
2727
+ // message is recoverable, a silently auto-denied approval is not. The escape
2728
+ // hatches are unaffected — /allow, /deny and /pending are grammy command
2729
+ // handlers and do not sit behind this gate — and the block is surfaced
2730
+ // off-Telegram in blocked-approvals/<agent>.json.
2589
2731
  const hasPendingApproval = pendingPermissions.size > 0
2590
2732
  if (!isDeliveryCutoverEnabled()) return claudeBusyKeys.size > 0 || hasPendingApproval
2591
2733
  // Machine is authoritative. Run the log-only drift canary (#2794): the
@@ -3659,6 +3801,9 @@ function postQueuedStatus(chatId: string, bufferedThread: number, inFlightThread
3659
3801
  return
3660
3802
  }
3661
3803
  queuedStatusMsgIds.set(key, { chatId, threadId: bufferedThread, messageId })
3804
+ // #3002: write through to the durable store so a gateway restart before the
3805
+ // in-process reap can still delete this orphaned card on boot.
3806
+ persistQueuedCard(key, chatId, bufferedThread, messageId)
3662
3807
  })()
3663
3808
  }
3664
3809
 
@@ -3695,6 +3840,9 @@ function reapQueuedStatus(chatId: string, thread: number | undefined): void {
3695
3840
  const entry = queuedStatusMsgIds.get(key)
3696
3841
  if (entry == null) return
3697
3842
  queuedStatusMsgIds.delete(key)
3843
+ // #3002: drop the durable row too, scoped to this exact message id (reap-race
3844
+ // guard) so a fresh live card under the same key keeps its own protection.
3845
+ clearQueuedCard(key, entry.messageId)
3698
3846
  void swallowingApiCall(
3699
3847
  () => bot.api.deleteMessage(chatId, entry.messageId),
3700
3848
  { chat_id: chatId, verb: 'queued-status.reap', ...(entry.threadId != null ? { threadId: entry.threadId } : {}) },
@@ -3807,6 +3955,8 @@ function postBusyAck(chatId: string, threadId: number | undefined, text: string)
3807
3955
  return
3808
3956
  }
3809
3957
  queuedStatusMsgIds.set(key, { chatId, threadId: threadId ?? null, messageId })
3958
+ // #3002: write through to the durable store (see postQueuedStatus).
3959
+ persistQueuedCard(key, chatId, threadId ?? null, messageId)
3810
3960
  })()
3811
3961
  }
3812
3962
 
@@ -3981,6 +4131,7 @@ function cancelPendingPermissionsForHalt(origin: string): void {
3981
4131
  void stripCancelledPermissionCards(details.card_text, details.cards)
3982
4132
  pendingPermissions.delete(requestId)
3983
4133
  permCardStore.remove(requestId)
4134
+ reconcileBlockedApprovals()
3984
4135
  process.stderr.write(
3985
4136
  `telegram gateway: halt-now cancelled pending permission origin=${origin} ` +
3986
4137
  `request=${requestId} tool=${details.tool_name}\n`,
@@ -4805,10 +4956,18 @@ function maybeProactiveCompact(): void {
4805
4956
  // ─── Idle auto-clear ──────────────────────────────────────────────────────
4806
4957
  // Wall-clock idle → /clear (idle-clear.ts). Independent of proactive-compact
4807
4958
  // (occupancy-driven at the turn-end gate): a fully-idle agent never ends a
4808
- // turn, so this runs on its own interval. Any activity (inbound / turn start /
4809
- // cron fire) resets the timer via markIdleActivity(); fires once per idle
4810
- // period; never mid-turn (turnInFlightForGate, the same gate compaction uses).
4959
+ // turn, so this runs on its own interval. "Idle" means NOTHING HAS HAPPENED
4960
+ // since the last thing happened — inbound, cron fire, or ANY claude session
4961
+ // event (turn start, tool call, tool result, text, sub-agent event, turn end)
4962
+ // resets the timer via markIdleActivity(); a turn ending additionally stamps
4963
+ // markIdleTurnEnd(). Fires once per idle period; never mid-turn
4964
+ // (turnInFlightForGate, the same gate compaction uses).
4965
+ //
4966
+ // It is emphatically NOT "no turn has *started* recently" — that reading wiped
4967
+ // overlord's 3h of working context on 2026-07-11 for the crime of being busy.
4968
+ // See the idle-clear.ts header.
4811
4969
  let lastIdleActivityAt = Date.now();
4970
+ let lastIdleTurnEndAt: number | null = null;
4812
4971
  let idleAutoCleared = false;
4813
4972
  let idleClearDispatching = false;
4814
4973
 
@@ -4818,6 +4977,15 @@ function markIdleActivity(): void {
4818
4977
  idleAutoCleared = false;
4819
4978
  }
4820
4979
 
4980
+ /**
4981
+ * Stamp "a turn just ended". The idle window is measured from
4982
+ * max(lastActivityAt, lastTurnEndedAt), so a turn that ran LONGER than the
4983
+ * window can't be cleared on the first tick after `turnInFlight` goes false.
4984
+ */
4985
+ function markIdleTurnEnd(): void {
4986
+ lastIdleTurnEndAt = Date.now();
4987
+ }
4988
+
4821
4989
  /** Idle window in ms: env override → per-agent config → 3h default. 0 disables. */
4822
4990
  function resolveIdleClearMs(): number {
4823
4991
  const env = process.env.SWITCHROOM_IDLE_CLEAR_MS;
@@ -4849,6 +5017,7 @@ function maybeIdleClear(): void {
4849
5017
  const decision = decideIdleClear(
4850
5018
  {
4851
5019
  lastActivityAt: lastIdleActivityAt,
5020
+ lastTurnEndedAt: lastIdleTurnEndAt,
4852
5021
  idleClearMs,
4853
5022
  alreadyCleared: idleAutoCleared,
4854
5023
  turnInFlight: turnInFlightForGate(),
@@ -5275,6 +5444,181 @@ const PHOTO_EXTS = new Set(['.jpg', '.jpeg', '.png', '.gif', '.webp'])
5275
5444
  // The length/newline text splitter moved to outbound-send-path.ts (#2996) as
5276
5445
  // `chunkText`; the sole gateway caller now goes through `computeReplyChunks`.
5277
5446
 
5447
+ // ─── Robust API call wrapper ──────────────────────────────────────────────
5448
+ // Extracted to telegram-plugin/retry-api-call.ts so it's unit-testable in
5449
+ // isolation; the gateway just composes the pure policy with its own logger.
5450
+ // #2923: the shared flood-wait marker. Every observed 429 retry_after window
5451
+ // is persisted here via onFloodWait, and both boot-card callsites consult the
5452
+ // SAME file to suppress a restart card while a per-bot flood ban is open (so a
5453
+ // restart doesn't post into the window and extend the ban). Falls back to a
5454
+ // no-op recorder when TELEGRAM_STATE_DIR is unset (dev/one-shot contexts).
5455
+ // STATE_DIR always resolves (env or a ~/.claude fallback), so this is live.
5456
+ // Declared ABOVE the typing indicator because the typing sends now route
5457
+ // through the same policy (#3084) — see nonEssentialApiCall below.
5458
+ const FLOOD_STATE_PATH = floodStatePath(STATE_DIR)
5459
+ // #3084 PR 2/3 — sibling file for SCOPED flood windows (global / chat / group /
5460
+ // msg-edit). Kept separate from the single-object flood-wait.json so #3094's
5461
+ // makeFloodWaitProbe schema is untouched (part3-design §7).
5462
+ const FLOOD_WINDOWS_PATH = floodWindowsPath(STATE_DIR)
5463
+ const recordFloodWindow = makeFloodWindowRecorder(FLOOD_WINDOWS_PATH)
5464
+
5465
+ // #3084 PR 2/3 — deterministic outbound send gate (token buckets + per-message
5466
+ // edit floor + no-op-edit skip + PRIORITY SHEDDING + DEGRADED MODE). Wrapped
5467
+ // HERE at the robustApiCall layer so every Bot API call routed through the
5468
+ // standard retry policy also transits one scheduler (no call site can bypass
5469
+ // it). Feature-flagged, default OFF: when SWITCHROOM_TELEGRAM_SEND_GATE !== '1'
5470
+ // the gate is a pure passthrough and the retry policy behaves exactly as
5471
+ // before. Composes with #3094's pre-call flood gate and #3097's non-essential
5472
+ // drop policy — those decide whether a call happens at all; this paces the
5473
+ // calls that do.
5474
+ //
5475
+ // §7 restart-proof flood state: initialWindows are loaded from disk BEFORE the
5476
+ // gate (and thus before ANY outbound call, boot cards included), so a boot mid-
5477
+ // ban never resends into an open window. onWindowOpen write-throughs every
5478
+ // runtime-opened window to FLOOD_WINDOWS_PATH; bootRamp starts the global
5479
+ // bucket at half capacity for 10s to absorb the boot-card burst.
5480
+ const sendGate = createSendGate({
5481
+ enabled: sendGateEnabledFromEnv(),
5482
+ initialWindows: loadInitialFloodWindows(FLOOD_STATE_PATH, FLOOD_WINDOWS_PATH, Date.now()),
5483
+ bootRamp: {},
5484
+ onWindowOpen: (scopeKey, untilTs) => recordFloodWindow(scopeKey, untilTs),
5485
+ })
5486
+
5487
+ // Hoisted so the held-card re-delivery sweep (#3084 follow-up) asks the SAME
5488
+ // probe `robustApiCall` does, off the same on-disk window. The sweep's check and
5489
+ // the retry policy's own pre-call short-circuit are then two independent reads
5490
+ // of one source of truth, not two notions of "is the channel open".
5491
+ const probeFloodWaitRemainingMs = makeFloodWaitProbe(FLOOD_STATE_PATH)
5492
+ const rawRobustApiCall = createRetryApiCall({
5493
+ log: (line) => process.stderr.write(line),
5494
+ onFloodWait: (retryAfterSec) => {
5495
+ // #2923/#3094 — persist the single-object global window (probe reads this).
5496
+ makeFloodWaitRecorder(FLOOD_STATE_PATH)(retryAfterSec)
5497
+ // #3084 PR 2 — also open a GLOBAL send-gate window so cosmetic traffic sheds
5498
+ // for the ban's duration even on a SHORT (slept-and-retried) 429 that never
5499
+ // throws FLOOD_WAIT_ACTIVE. Scope-precise windows are opened by the gate's
5500
+ // own FLOOD_WAIT_ACTIVE catch (which has the call's opts).
5501
+ try {
5502
+ sendGate.openFloodWindow('global', Date.now() + Math.max(0, retryAfterSec) * 1000)
5503
+ } catch {
5504
+ /* best-effort — never let the window hook break the retry path */
5505
+ }
5506
+ },
5507
+ // #3094: while a LONG per-bot ban is open, don't issue the call at all.
5508
+ // retryApiCall no longer sleeps a multi-hour retry_after (it throws
5509
+ // FLOOD_WAIT_ACTIVE instead), and the card surfaces re-drive on a 5-6s
5510
+ // heartbeat — without this gate that turns the fix into an amplifier that
5511
+ // fires thousands of requests into the open window and extends the ban.
5512
+ floodWaitRemainingMs: probeFloodWaitRemainingMs,
5513
+ })
5514
+
5515
+ const robustApiCall = <T>(
5516
+ fn: () => Promise<T>,
5517
+ opts?: Parameters<typeof rawRobustApiCall<T>>[1],
5518
+ ): Promise<T> => sendGate.gate(() => rawRobustApiCall(fn, opts), opts)
5519
+
5520
+ // Fire-and-forget wrapper for outbound surfaces that previously had
5521
+ // `.catch(() => {})` directly on `bot.api.*` calls. Resolves to undefined
5522
+ // (instead of crashing the gateway) on THREAD_NOT_FOUND, give-up, 403,
5523
+ // and any non-benign error — logs a one-liner so the failure isn't
5524
+ // completely silent. See #1075.
5525
+ const swallowingApiCall = createSwallowingRetryApiCall(
5526
+ robustApiCall,
5527
+ (line) => process.stderr.write(line),
5528
+ )
5529
+
5530
+ /**
5531
+ * The wrapper for NON-ESSENTIAL sends that must NEVER retry (#3084).
5532
+ *
5533
+ * A typing indicator is disposable: if it fails, the correct answer is to drop
5534
+ * it, because a retry spends more of the per-bot flood budget the REPLIES need
5535
+ * — retrying is what feeds a ban. But it must not be SILENT either: the typing
5536
+ * loop was the single largest emitter (55% of outbound volume) and it called
5537
+ * `bot.api` raw with `.catch(() => {})`, so the #2923 flood circuit breaker —
5538
+ * which only learns about 429s through `createRetryApiCall`'s `onFloodWait`
5539
+ * hook — was blind to 429s from the biggest source of them.
5540
+ *
5541
+ * `maxRetries: 1` + a no-op sleep gives exactly one attempt at the API, whatever
5542
+ * the outcome: a 429 still runs `onFloodWait` (recording the window to the
5543
+ * breaker), and then the call ends — a SHORT ban falls out of the loop as
5544
+ * `max retries exceeded`, a LONG one throws #3094's `FLOOD_WAIT_ACTIVE`. Either
5545
+ * way the ping is dropped, never slept and never retried. That is the whole
5546
+ * point: a retried non-essential send is what feeds a ban.
5547
+ *
5548
+ * Composes cleanly on top of #3094 (`bbe14471`): that PR bounded the in-process
5549
+ * flood sleep and added a pre-call gate for LONG open windows. This wrapper
5550
+ * pre-dates neither mechanism nor fights them — it sits under the emitter's own
5551
+ * flood gate, which short-circuits earlier still (a typing ping during a ban
5552
+ * never even reaches the retry layer). Defense in depth, in that order.
5553
+ */
5554
+ // No `log` hook: retryApiCall's flood line reads "waiting Ns", which would be a
5555
+ // lie here — we never wait, we drop. And the error the caller finally sees names
5556
+ // neither the code nor the retry_after (retryApiCall replaces the original on
5557
+ // give-up) — and retry_after is the single number an operator needs during a
5558
+ // ban. So the honest 429 line is written HERE, in the onFloodWait hook, where
5559
+ // the real value is still in hand.
5560
+ const recordTypingFloodWait = makeFloodWaitRecorder(FLOOD_STATE_PATH)
5561
+ const nonEssentialApiCall = createRetryApiCall({
5562
+ maxRetries: 1,
5563
+ sleep: async () => {},
5564
+ onFloodWait: (retryAfterSec) => {
5565
+ recordTypingFloodWait(retryAfterSec)
5566
+ process.stderr.write(
5567
+ `telegram gateway: 429 flood-wait on a NON-ESSENTIAL send (retry_after=${retryAfterSec}s) — ` +
5568
+ `recorded to the flood breaker; the send is DROPPED, not retried (#3084)\n`,
5569
+ )
5570
+ },
5571
+ })
5572
+
5573
+ // #3084 PR 3/3 — observability + operator flood alerts (part3-design §6).
5574
+ // A low-frequency, change-gated stats line + flood-window open/close snapshots
5575
+ // to the gateway-supervisor log, and ONE operator alert per prolonged flood
5576
+ // window. All feature-flagged: when the send gate is OFF, `stats().enabled` is
5577
+ // false and both `tick()`s are pure no-ops (the interval is not even scheduled).
5578
+ const SEND_GATE_OBSERVE_INTERVAL_MS = 15_000
5579
+ const sendGateStatsLogger = createStatsLogger({
5580
+ stats: () => sendGate.stats(),
5581
+ log: (line) => process.stderr.write(line),
5582
+ clock: { now: () => Date.now(), sleep: (ms) => new Promise((r) => setTimeout(r, ms)) },
5583
+ })
5584
+ const floodWindowObserver = createFloodWindowObserver({
5585
+ clock: { now: () => Date.now(), sleep: (ms) => new Promise((r) => setTimeout(r, ms)) },
5586
+ log: (line) => process.stderr.write(line),
5587
+ stats: () => sendGate.stats(),
5588
+ readWindows: (now) => readFloodWindows(FLOOD_WINDOWS_PATH, now),
5589
+ markAlerted: (scopeKey, alertedAt) =>
5590
+ markFloodWindowAlerted(FLOOD_WINDOWS_PATH, scopeKey, alertedAt, Date.now()),
5591
+ operatorChatId: () => loadAccess().allowFrom[0], // resolved per-tick (allowFrom can change)
5592
+ sendAlert: async (text) => {
5593
+ const operator = loadAccess().allowFrom[0]
5594
+ if (operator === undefined) {
5595
+ process.stderr.write(
5596
+ `telegram gateway: send-gate flood alert not sent — no operator chat (allowFrom empty)\n`,
5597
+ )
5598
+ return
5599
+ }
5600
+ await robustApiCall(
5601
+ // allow-raw-bot-api: operator flood alert, routed through robustApiCall.
5602
+ () => bot.api.sendRichMessage(operator, richMessage(text), {}),
5603
+ { chat_id: String(operator), verb: 'send-gate-flood-alert', priorityClass: 'critical' },
5604
+ )
5605
+ },
5606
+ })
5607
+ if (sendGateEnabledFromEnv()) {
5608
+ const observeTimer = setInterval(() => {
5609
+ try {
5610
+ sendGateStatsLogger.tick()
5611
+ } catch {
5612
+ /* observability must never crash the gateway */
5613
+ }
5614
+ void floodWindowObserver.tick().catch(() => {
5615
+ /* best-effort — a failed observer tick must not surface */
5616
+ })
5617
+ }, SEND_GATE_OBSERVE_INTERVAL_MS)
5618
+ // Don't keep the event loop alive for observability alone.
5619
+ observeTimer.unref?.()
5620
+ }
5621
+
5278
5622
  // ─── Typing indicator ─────────────────────────────────────────────────────
5279
5623
  // All four state maps re-keyed from `chat_id` to `chatKey(chat, thread)`
5280
5624
  // in PR3 of the supergroup-mode rollout. In supergroup mode one agent
@@ -5311,33 +5655,94 @@ const CHAT_ACTION_WHITELIST = new Set([
5311
5655
  ] as const)
5312
5656
  type ChatAction = typeof CHAT_ACTION_WHITELIST extends Set<infer T> ? T : never
5313
5657
 
5314
- function startTypingLoop(
5315
- chat_id: string,
5316
- thread_id: number | null = null,
5317
- action: ChatAction = 'typing',
5318
- ): void {
5319
- stopTypingLoop(chat_id, thread_id)
5320
- const key = chatKey(chat_id, thread_id) as string
5321
- const sendOpts = thread_id != null ? { message_thread_id: thread_id } : undefined
5322
- const send = () => {
5323
- bot.api.sendChatAction(chat_id, action, sendOpts).then(
5658
+ /**
5659
+ * The ONE seam every chat action in this gateway goes through (#3084).
5660
+ *
5661
+ * Both typing loops (tool-use `typingIntervals` below and the turn-level
5662
+ * `turnTypingLoop`) and the one-shot inbound pings share this emitter, so the
5663
+ * per-chat-key floor holds ACROSS them: at most one `sendChatAction` per chat
5664
+ * key per ~4 s window, no matter how many times a loop is restarted. That is
5665
+ * what decouples the typing rate from the agent's TOOL-CALL rate — the
5666
+ * regression that earned a 4.6-hour flood ban. A cold start (no ping inside
5667
+ * the floor) still fires instantly, so "typing…" lands the moment a turn
5668
+ * begins; only redundant restarts are dropped — and a dropped tick arms a
5669
+ * coalesced catch-up, so the indicator never goes dark for longer than the
5670
+ * floor. While a flood window is open, typing — non-essential by definition —
5671
+ * is not emitted at all.
5672
+ *
5673
+ * (`sendChatAction` is outside `check-bot-api-wrapping`'s pattern by design —
5674
+ * chat actions take no `message_thread_id`, so they were never in the
5675
+ * THREAD_NOT_FOUND blast radius the guard polices. It's routed through the
5676
+ * retry module anyway, because that is where the flood breaker's `onFloodWait`
5677
+ * hook lives and typing is the largest 429 source there is.)
5678
+ */
5679
+ const typingEmitter = createTypingEmitter({
5680
+ chatKey: (chat_id, thread_id) => chatKey(chat_id, thread_id) as string,
5681
+ isSuppressed: () => suppressNonEssentialSendMs(FLOOD_STATE_PATH, Date.now()) > 0,
5682
+ send: (chat_id, thread_id, action) => {
5683
+ const sendOpts = thread_id != null ? { message_thread_id: thread_id } : undefined
5684
+ void nonEssentialApiCall(
5685
+ () => bot.api.sendChatAction(chat_id, action as ChatAction, sendOpts),
5686
+ { chat_id, verb: 'sendChatAction' },
5687
+ ).then(
5324
5688
  () => { typingBackoffMs = 0 },
5325
5689
  (err) => {
5326
5690
  const msg = err instanceof Error ? err.message : String(err)
5327
5691
  if (msg.includes('401') || msg.includes('Unauthorized')) {
5692
+ const key = chatKey(chat_id, thread_id) as string
5328
5693
  typingBackoffMs = Math.min(Math.max(typingBackoffMs * 2 || 1000, 1000), TYPING_BACKOFF_MAX)
5329
5694
  stopTypingLoop(chat_id, thread_id)
5330
5695
  const retry = setTimeout(() => {
5331
5696
  typingRetryTimers.delete(key)
5332
- startTypingLoop(chat_id, thread_id, action)
5697
+ startTypingLoop(chat_id, thread_id, action as ChatAction)
5333
5698
  }, typingBackoffMs)
5334
5699
  typingRetryTimers.set(key, retry)
5700
+ return
5335
5701
  }
5702
+ // A flood-wait already logged its retry_after (and hit the breaker) in
5703
+ // onFloodWait above. What lands here is either retryApiCall's opaque
5704
+ // give-up error (short ban) or #3094's FLOOD_WAIT_ACTIVE marker (long
5705
+ // ban / pre-call gate) — neither names the retry_after, so don't
5706
+ // double-log a strictly less informative line. Both are DROPS: a
5707
+ // typing ping is never retried, and the marker is caught here (this is
5708
+ // an onRejected handler), so it can't surface as an unhandled rejection
5709
+ // out of a fire-and-forget ping.
5710
+ if (isFloodWaitActiveError(err) || msg.includes('max retries exceeded')) return
5711
+ // Everything else is DROPPED, never retried — but logged, so the
5712
+ // largest outbound emitter is no longer silent (#3084).
5713
+ process.stderr.write(
5714
+ `telegram gateway: sendChatAction dropped (non-essential, not retried): ${msg}\n`,
5715
+ )
5336
5716
  },
5337
5717
  )
5338
- }
5718
+ },
5719
+ })
5720
+
5721
+ /** Fire one chat action for (chat, thread), subject to the shared floor. */
5722
+ function emitChatAction(
5723
+ chat_id: string,
5724
+ thread_id: number | null = null,
5725
+ action: ChatAction = 'typing',
5726
+ ): void {
5727
+ typingEmitter.emit(chat_id, thread_id, action)
5728
+ }
5729
+
5730
+ function startTypingLoop(
5731
+ chat_id: string,
5732
+ thread_id: number | null = null,
5733
+ action: ChatAction = 'typing',
5734
+ ): void {
5735
+ stopTypingLoop(chat_id, thread_id)
5736
+ const key = chatKey(chat_id, thread_id) as string
5737
+ // The immediate fire is FLOOR-GATED (it was not, and that is #3084): the
5738
+ // tool-use wrapper restarts this loop on every tool call, so an unguarded
5739
+ // immediate send made the ping rate equal the tool-call rate and the 4 s
5740
+ // interval decorative. The emitter drops the restart's ping when this key
5741
+ // already pinged inside the window, and lets it straight through on a cold
5742
+ // start — so a turn still lights up "typing…" instantly.
5743
+ const send = () => emitChatAction(chat_id, thread_id, action)
5339
5744
  send()
5340
- typingIntervals.set(key, setInterval(send, 4000))
5745
+ typingIntervals.set(key, setInterval(send, TYPING_REFRESH_MS))
5341
5746
  }
5342
5747
 
5343
5748
  function stopTypingLoop(chat_id: string, thread_id: number | null = null): void {
@@ -5357,14 +5762,19 @@ function stopTypingLoop(chat_id: string, thread_id: number | null = null): void
5357
5762
  // kill it and the chat would go dark for the rest of the turn — the exact
5358
5763
  // black-box gap this closes. The dedicated map (private to the factory) makes
5359
5764
  // the turn loop structurally immune to those stops: only the canonical turn-end
5360
- // stop clears it. The redundant `typing` pings while a reply is mid-flight are
5361
- // harmless — same action, and sendChatAction is cheap.
5765
+ // stop clears it.
5766
+ //
5767
+ // The interval map stays separate — but the SENDS do not (#3084). This loop and
5768
+ // `typingIntervals` target the SAME chat key, and the old comment here claimed
5769
+ // the redundant pings were "harmless — same action, and sendChatAction is
5770
+ // cheap". They are not cheap: they spend the per-bot flood budget the replies
5771
+ // need, and two independently-restarting loops on one key is exactly how the
5772
+ // rate compounded. Both loops now emit through `typingEmitter`, so the
5773
+ // per-chat-key floor is SHARED and neither loop can out-shout the other.
5362
5774
  const turnTypingLoop = createTurnTypingLoop({
5363
- sendChatAction: (chat_id, thread_id) => {
5364
- const sendOpts = thread_id != null ? { message_thread_id: thread_id } : undefined
5365
- void bot.api.sendChatAction(chat_id, 'typing', sendOpts).catch(() => {})
5366
- },
5775
+ sendChatAction: (chat_id, thread_id) => emitChatAction(chat_id, thread_id, 'typing'),
5367
5776
  chatKey: (chat_id, thread_id) => chatKey(chat_id, thread_id) as string,
5777
+ refreshMs: TYPING_REFRESH_MS,
5368
5778
  })
5369
5779
 
5370
5780
  function startTurnTypingLoop(chat_id: string, thread_id: number | null = null): void {
@@ -5373,6 +5783,11 @@ function startTurnTypingLoop(chat_id: string, thread_id: number | null = null):
5373
5783
 
5374
5784
  function stopTurnTypingLoop(chat_id: string, thread_id: number | null = null): void {
5375
5785
  turnTypingLoop.stop(chat_id, thread_id)
5786
+ // Canonical turn-end: cancel any catch-up the floor armed, so a tick dropped
5787
+ // in the last seconds of the turn can't resurrect "typing…" after the reply
5788
+ // has landed. Deliberately NOT done in `stopTypingLoop` — the tool loop stops
5789
+ // on every tool result, which is exactly the churn the catch-up covers.
5790
+ typingEmitter.cancelPending(chat_id, thread_id)
5376
5791
  }
5377
5792
 
5378
5793
  const typingWrapper = createTypingWrapper({
@@ -5381,31 +5796,6 @@ const typingWrapper = createTypingWrapper({
5381
5796
  isSurfaceTool: isTelegramSurfaceTool,
5382
5797
  })
5383
5798
 
5384
- // ─── Robust API call wrapper ──────────────────────────────────────────────
5385
- // Extracted to telegram-plugin/retry-api-call.ts so it's unit-testable in
5386
- // isolation; the gateway just composes the pure policy with its own logger.
5387
- // #2923: the shared flood-wait marker. Every observed 429 retry_after window
5388
- // is persisted here via onFloodWait, and both boot-card callsites consult the
5389
- // SAME file to suppress a restart card while a per-bot flood ban is open (so a
5390
- // restart doesn't post into the window and extend the ban). Falls back to a
5391
- // no-op recorder when TELEGRAM_STATE_DIR is unset (dev/one-shot contexts).
5392
- // STATE_DIR always resolves (env or a ~/.claude fallback), so this is live.
5393
- const FLOOD_STATE_PATH = floodStatePath(STATE_DIR)
5394
- const robustApiCall = createRetryApiCall({
5395
- log: (line) => process.stderr.write(line),
5396
- onFloodWait: makeFloodWaitRecorder(FLOOD_STATE_PATH),
5397
- })
5398
-
5399
- // Fire-and-forget wrapper for outbound surfaces that previously had
5400
- // `.catch(() => {})` directly on `bot.api.*` calls. Resolves to undefined
5401
- // (instead of crashing the gateway) on THREAD_NOT_FOUND, give-up, 403,
5402
- // and any non-benign error — logs a one-liner so the failure isn't
5403
- // completely silent. See #1075.
5404
- const swallowingApiCall = createSwallowingRetryApiCall(
5405
- robustApiCall,
5406
- (line) => process.stderr.write(line),
5407
- )
5408
-
5409
5799
  /**
5410
5800
  * Adapter factory for `startBootCard`'s `BotApiForBootCard` interface.
5411
5801
  *
@@ -5440,7 +5830,8 @@ function wrapBootCardApi(
5440
5830
  richMessage(text),
5441
5831
  sendOpts as Parameters<typeof lockedBot.api.sendRichMessage>[2],
5442
5832
  ),
5443
- opts(cid),
5833
+ // #3084 PR 2: boot/config card CREATION is USEFUL (queue with TTL).
5834
+ { ...opts(cid), priorityClass: 'useful' },
5444
5835
  )
5445
5836
  return sent as { message_id: number }
5446
5837
  },
@@ -5453,7 +5844,9 @@ function wrapBootCardApi(
5453
5844
  richMessage(text),
5454
5845
  editOpts as Parameters<typeof lockedBot.api.editMessageText>[3],
5455
5846
  ),
5456
- opts(cid),
5847
+ // A boot-card EDIT is COSMETIC — shed under pressure; pass
5848
+ // messageId/editPayload so the per-message floor + no-op skip engage.
5849
+ { ...opts(cid), priorityClass: 'cosmetic', messageId: mid, editPayload: text },
5457
5850
  ) as Promise<unknown>,
5458
5851
  // Strict edit for the boot-card edit-in-place probe: distinguishes
5459
5852
  // "message gone" (→ 'gone', caller sends fresh) from a landed/identical
@@ -5505,7 +5898,8 @@ function wrapIssuesCardApi(
5505
5898
  richMessage(text),
5506
5899
  sendOpts as Parameters<typeof lockedBot.api.sendRichMessage>[2],
5507
5900
  ),
5508
- opts(cid),
5901
+ // #3084 PR 2: issues-card creation is USEFUL.
5902
+ { ...opts(cid), priorityClass: 'useful' },
5509
5903
  )
5510
5904
  return sent as { message_id: number }
5511
5905
  },
@@ -5518,7 +5912,8 @@ function wrapIssuesCardApi(
5518
5912
  richMessage(text),
5519
5913
  editOpts as Parameters<typeof lockedBot.api.editMessageText>[3],
5520
5914
  ),
5521
- opts(cid),
5915
+ // An issues-card EDIT is COSMETIC.
5916
+ { ...opts(cid), priorityClass: 'cosmetic', messageId: mid, editPayload: text },
5522
5917
  ) as Promise<unknown>,
5523
5918
  deleteMessage: (cid, mid) =>
5524
5919
  robustApiCall(() => lockedBot.api.deleteMessage(cid, mid), opts(cid)) as Promise<unknown>,
@@ -5552,7 +5947,11 @@ const STATUS_QUERY_RE = /^\s*status\??\s*$/i
5552
5947
 
5553
5948
  // ─── Permission handling ──────────────────────────────────────────────────
5554
5949
  const PERMISSION_REPLY_RE = /^\s*(y|yes|n|no)\s+([a-km-z]{5})\s*$/i
5555
- const pendingPermissions = new Map<string, { tool_name: string; description: string; input_preview: string; startedAt: number; card_text: string; cards: { chatId: string; messageId: number; threadId?: number | null }[] }>()
5950
+ // `undeliverable` (#3084 follow-up): set when the card send failed against a
5951
+ // known-open Telegram flood window. While it is set the entry is HELD — the TTL
5952
+ // sweep skips it (never auto-deny an ask no human ever saw) and the reaper
5953
+ // re-posts the card once the window closes. Cleared on successful delivery.
5954
+ const pendingPermissions = new Map<string, { tool_name: string; description: string; input_preview: string; startedAt: number; card_text: string; cards: { chatId: string; messageId: number; threadId?: number | null }[]; undeliverable?: UndeliverableMark | null; redeliveryFailures?: number }>()
5556
5955
  // PERMISSION_TTL_MS / ttlForTool / the timed-out card builder now live in
5557
5956
  // ./permission-timeout.ts (pure + unit-testable). hostd gated verbs get a
5558
5957
  // 30-min window; everything else keeps the 10-min default.
@@ -5770,6 +6169,23 @@ function sweepStaleMentalModelCorrelations(now = Date.now()): void {
5770
6169
  pendingMentalModelCorrelations.sweep(now)
5771
6170
  }
5772
6171
 
6172
+ // #2975 Stage 2 — read-only pre-approval predicate. hostd asks (over the
6173
+ // approval-gateway socket, via `check_pre_approved`) whether an EXACT
6174
+ // (agent, diff) pair is already operator-consented so it can skip the
6175
+ // config_propose_edit rate limit for that persist. The forge-resistant,
6176
+ // read-only matching lives in the pure `isDiffPreApproved` (pre-approval-
6177
+ // check.ts) so its contract — true only for a byte-exact registered pair,
6178
+ // NEVER a mutation — is unit-testable without importing this module. Here we
6179
+ // only bind it to the live correlation stores + helpers.
6180
+ function isDiffPreApprovedLive(agentName: string, unifiedDiff: string): boolean {
6181
+ return isDiffPreApproved(agentName, unifiedDiff, {
6182
+ alwaysAllow: pendingAlwaysAllowCorrelations,
6183
+ mentalModel: pendingMentalModelCorrelations,
6184
+ extractAddedAllowRule,
6185
+ mentalModelCorrelationKey,
6186
+ })
6187
+ }
6188
+
5773
6189
  // Scoped-approval store: the 30-min window that backs the "✅ Allow" tap for
5774
6190
  // narrow non-destructive scopes (not a separate button — it IS what Allow
5775
6191
  // means for those). Operator-tapped, gateway-side ONLY (never pushed to the
@@ -6525,9 +6941,260 @@ function restorePendingApprovalCards(): number {
6525
6941
  return restored
6526
6942
  }
6527
6943
 
6944
+ /**
6945
+ * Post the Approve/Deny card for `requestId` to every permission-card target.
6946
+ *
6947
+ * Extracted from `onPermissionRequest` (#3084 follow-up) so it has TWO callers:
6948
+ * the initial delivery, and the reaper's held-card re-delivery once a flood
6949
+ * window closes. Both paths must behave identically — same routing, same
6950
+ * thread-fallback, same landed-card bookkeeping, same hold-on-flood-wait — so
6951
+ * there is exactly one implementation of "post this card".
6952
+ *
6953
+ * Everything it needs is rebuilt from the pending entry, so a re-delivery an
6954
+ * hour later renders the same card the operator would have seen at T+0.
6955
+ */
6956
+ function postPermissionCard(
6957
+ requestId: string,
6958
+ pend: NonNullable<ReturnType<typeof pendingPermissions.get>>,
6959
+ ): void {
6960
+ // Register the in-flight flag HERE, not at the call sites, so BOTH callers are
6961
+ // covered. The sweep used to add it itself, but the ipcServer initial-delivery
6962
+ // call site did not — so a reaper tick landing between window-close and a slow
6963
+ // in-flight INITIAL send settling could post a duplicate card. One registration
6964
+ // point, one release point (the `.finally` below / the sync-throw catch).
6965
+ heldCardsInFlight.add(requestId)
6966
+
6967
+ // One increment per ATTEMPT, not per failing target. `targets` is N-wide and
6968
+ // every failing target runs the same `.catch` — incrementing there ratcheted
6969
+ // the backoff ladder N× faster than the documented 1m/2m/4m…30m. Compute the
6970
+ // attempt number once, up front; every failing target stamps the SAME value.
6971
+ const attemptFailures = (pend.redeliveryFailures ?? 0) + 1
6972
+
6973
+ try {
6974
+ const text = pend.card_text
6975
+ const showAlways =
6976
+ resolveScopedAllowChoices(pend.tool_name, pend.input_preview) != null
6977
+ const keyboard = buildPermissionActionRow(requestId, showAlways)
6978
+ // Route the card to the SAME place the post-verdict resume message lands
6979
+ // (resolvePermissionCardTargets): the ORIGINATING chat+topic when there's an
6980
+ // active turn — so a supergroup agent's card appears IN the topic the
6981
+ // operator asked from (marko's "CRM (Brevo)"), not the operator DM — else the
6982
+ // configured operator DMs, thread-stripped. The old code iterated `allowFrom`
6983
+ // unconditionally, so a supergroup card could only ever reach operator DMs
6984
+ // (the topic chat id is never in `allowFrom`) (marko, 2026-06-03).
6985
+ const targets = resolvePermissionCardTargets()
6986
+
6987
+ // ONE settle across ALL targets, not one per target (reviewer F1).
6988
+ //
6989
+ // The in-flight flag (guard iii) exists so the reaper can't re-post a card whose
6990
+ // previous post hasn't settled. `resolvePermissionCardTargets()` returns N
6991
+ // targets — it ends in `allowFrom.map(...)` — and on the RE-delivery path N>1 is
6992
+ // the COMMON case: re-delivery after a multi-hour ban is by construction past the
6993
+ // 30-min origin-recovery window, so it falls through to the operator-DM fan-out.
6994
+ // Clearing the flag per-target released it the moment the FIRST target settled
6995
+ // while others were still in flight, and the next tick re-posted to all of them —
6996
+ // a double delivery. Settle once, when every target is done.
6997
+ const sends = targets.map(({ chatId, threadId }) => {
6998
+ // The rich-markdown path pairs with formatPermissionCardBody (#1790) so its
6999
+ // bold/italic render. retryWithThreadFallback: if the topic was
7000
+ // deleted/recreated (stale thread id → 400 "message thread not found"),
7001
+ // re-send thread-less into the main chat so the card still ARRIVES rather
7002
+ // than vanishing.
7003
+ // allow-raw-bot-api: wrapped in retryWithThreadFallback (retry policy); topic-aware send
7004
+ return retryWithThreadFallback<{ message_id: number; message_thread_id?: number }>(
7005
+ robustApiCall,
7006
+ (tid) =>
7007
+ bot.api.sendRichMessage(chatId, richMessage(text), {
7008
+ reply_markup: keyboard,
7009
+ ...(tid != null ? { message_thread_id: tid } : {}),
7010
+ }),
7011
+ { threadId, chat_id: chatId, verb: 'permission_request' },
7012
+ ).then(sent => {
7013
+ // Record the live card's (chat, message) so the reaper can strip its inline
7014
+ // keyboard — a stale Approve button left tappable dispatches a verdict for a
7015
+ // dead request_id (Bug 2). The entry may already be gone (operator tapped
7016
+ // before this resolved); guard the lookup.
7017
+ const live = pendingPermissions.get(requestId)
7018
+ if (live && sent && typeof sent.message_id === 'number') {
7019
+ // #2787: record where the card ACTUALLY LANDED so the confirm sweep scopes
7020
+ // its re-delivery suspension to the real chat/topic. `sent.message_thread_id`
7021
+ // reflects reality in all three cases: topic success → tid, main-chat /
7022
+ // fallback → undefined.
7023
+ const landedThreadId = sent.message_thread_id ?? undefined
7024
+ live.cards.push({ chatId, messageId: sent.message_id, threadId: landedThreadId })
7025
+ // The card LANDED — the operator can see and tap it, so the block is over.
7026
+ // Drop the hold mark and reconcile the off-Telegram surface. (PR 3 also
7027
+ // resets `startedAt` here: the TTL measures how long the operator had to
7028
+ // answer, and until this moment they had nothing to answer.)
7029
+ if (live.undeliverable != null) {
7030
+ live.undeliverable = null
7031
+ live.redeliveryFailures = 0
7032
+ // RESET THE TTL CLOCK. Load-bearing, not cosmetic. `startedAt` is when
7033
+ // the agent asked; the TTL measures how long the operator had to
7034
+ // answer. Until this instant they had NOTHING to answer — the card did
7035
+ // not exist in any chat. Without the reset, a card held through a 4.6h
7036
+ // ban lands already-expired against a 60-min TTL and the very next
7037
+ // reaper tick auto-denies it: we would have MOVED the silent denial,
7038
+ // not removed it.
7039
+ live.startedAt = Date.now()
7040
+ reconcileBlockedApprovals()
7041
+ process.stderr.write(
7042
+ `telegram gateway: permission-card RE-DELIVERED request=${requestId} ` +
7043
+ `tool=${live.tool_name} chat=${chatId} — the operator can answer now; ` +
7044
+ `TTL clock restarted (they had zero seconds while the channel was shut)\n`,
7045
+ )
7046
+ }
7047
+ permCardStore.add({
7048
+ requestId,
7049
+ chatId,
7050
+ messageId: sent.message_id,
7051
+ startedAt: live.startedAt,
7052
+ toolName: live.tool_name,
7053
+ cardText: live.card_text,
7054
+ })
7055
+ }
7056
+ }).catch(e => {
7057
+ process.stderr.write(`telegram gateway: permission_request send to ${chatId} failed: ${e}\n`)
7058
+ const live = pendingPermissions.get(requestId)
7059
+ if (live == null) return // operator already resolved it — nothing to hold
7060
+
7061
+ // A card already landed on ANOTHER target, so the ask is NOT undeliverable —
7062
+ // the operator can see and tap it (reviewer F2). Marking it held here would be
7063
+ // a PERMANENT wedge: `selectHeldForRedelivery` skips entries that have cards,
7064
+ // so the mark could never be cleared, and PR 3's TTL freeze would then keep
7065
+ // the entry alive forever with the surface stuck on "blocked".
7066
+ if (live.cards.length > 0) return
7067
+
7068
+ // The card did NOT land anywhere. Before this the error was logged and
7069
+ // dropped: the entry sat with `cards: []`, the operator saw nothing, and 60
7070
+ // minutes later the TTL sweep AUTO-DENIED an approval no human had ever been
7071
+ // shown. Now we hold it — but only for a TRANSIENT cause. `holdReasonFor`
7072
+ // decides; a permanent 400 (a card that will never format) still falls through
7073
+ // to the TTL, which PR 3 makes safe by giving the no-card timeout a
7074
+ // missed-approvals fallback.
7075
+ const reason = holdReasonFor(e)
7076
+ if (reason == null) return
7077
+
7078
+ // Bound the RATE, never the RETRY. A held entry is re-selected every tick, so a
7079
+ // target that keeps failing would otherwise re-send every 60s forever — the
7080
+ // amplifier this series exists to avoid. Back off instead (1m, 2m, 4m … 30m).
7081
+ // We never STOP retrying: a terminal give-up would strand the card even after
7082
+ // the channel recovered, and because turnInFlightForGate() holds the inbound
7083
+ // gate while a permission is pending, that would silently buffer the operator's
7084
+ // messages forever. Escalate the surface, never the verdict — and never abandon
7085
+ // the card.
7086
+ const failures = attemptFailures
7087
+ live.redeliveryFailures = failures
7088
+ const now = Date.now()
7089
+ // A flood-wait carries Telegram's own window end; everything else backs off.
7090
+ const retryableAt = isFloodWaitActiveError(e) ? e.untilTs : now + heldRetryBackoffMs(failures)
7091
+ // `since` is set ONCE — it is when the block began, not when we last retried.
7092
+ live.undeliverable = { since: live.undeliverable?.since ?? now, retryableAt, reason }
7093
+ // Reconcile the shared surface SYNCHRONOUSLY, here in the failure handler — not
7094
+ // on the 60s reaper tick. The operator is locked out of Telegram; this record is
7095
+ // the only way they learn an agent is blocked, so it must be live in under a
7096
+ // second, not up to a minute.
7097
+ reconcileBlockedApprovals()
7098
+ process.stderr.write(
7099
+ `telegram gateway: permission-card HELD (undeliverable) request=${requestId} ` +
7100
+ `tool=${live.tool_name} reason=${reason} attempt=${failures} ` +
7101
+ `retryable_at=${new Date(retryableAt).toISOString()}` +
7102
+ ` — approval is held, NOT denied; re-delivering when the channel returns ` +
7103
+ `(backoff ${Math.round(heldRetryBackoffMs(failures) / 60000)}m)\n`,
7104
+ )
7105
+ })
7106
+ })
7107
+
7108
+ // Settle ONCE, after every target. `allSettled` never rejects, so the flag is
7109
+ // always released — including when `targets` is empty (no active turn AND no
7110
+ // configured operator DM), in which case `sends` is [] and this resolves at once.
7111
+ void Promise.allSettled(sends).finally(() => {
7112
+ heldCardsInFlight.delete(requestId)
7113
+ })
7114
+ } catch (e) {
7115
+ // A SYNCHRONOUS throw (resolveScopedAllowChoices / buildPermissionActionRow /
7116
+ // resolvePermissionCardTargets) never reaches the `.finally` above — without
7117
+ // this catch the in-flight flag would leak and the request would never be
7118
+ // re-selected by the sweep. Release the flag, then rethrow so the caller
7119
+ // (the sweep's own try/catch, or the ipc handler) sees the failure.
7120
+ heldCardsInFlight.delete(requestId)
7121
+ throw e
7122
+ }
7123
+ }
7124
+
7125
+ /**
7126
+ * Request ids whose card post is mid-flight — INITIAL delivery or re-delivery.
7127
+ * Guard (iii) against double-delivery: the 60s reaper must not re-post a card
7128
+ * whose previous post hasn't settled yet (a slow send would otherwise be
7129
+ * re-issued every tick). Registered at the top of postPermissionCard (so both
7130
+ * callers are covered) and cleared in its `.finally` — or its sync-throw catch.
7131
+ */
7132
+ const heldCardsInFlight = new Set<string>()
7133
+
7134
+ /**
7135
+ * Re-deliver held permission cards once the flood window closes (#3084
7136
+ * follow-up, PR 2).
7137
+ *
7138
+ * The whole point of holding rather than auto-denying is that the ask comes
7139
+ * BACK. This is the half that brings it back.
7140
+ *
7141
+ * `selectHeldForRedelivery` owns the guards (window closed, held, no landed
7142
+ * card, not in flight, per-tick cap). `robustApiCall`'s own pre-call probe is a
7143
+ * further, independent short-circuit downstream — if the window re-opens
7144
+ * between selection and send, the send refuses itself and the entry is simply
7145
+ * re-marked with the longer window.
7146
+ *
7147
+ * The per-tick cap is the ban-safety property: a backlog of held cards must not
7148
+ * BURST the instant the window closes, because that burst is the one realistic
7149
+ * way this design could re-earn the ban it exists to survive. Deferred ids are
7150
+ * LOGGED, never silently dropped — they go out on the next tick.
7151
+ */
7152
+ function sweepHeldPermissionCards(): void {
7153
+ const remaining = probeFloodWaitRemainingMs()
7154
+ const { send, deferred } = selectHeldForRedelivery(pendingPermissions.entries(), {
7155
+ floodRemainingMs: remaining,
7156
+ inFlight: heldCardsInFlight,
7157
+ now: Date.now(),
7158
+ })
7159
+ if (send.length === 0) return
7160
+
7161
+ process.stderr.write(
7162
+ `telegram gateway: flood window closed — re-delivering ${send.length} held ` +
7163
+ `permission card(s)` +
7164
+ (deferred.length > 0
7165
+ ? `; ${deferred.length} deferred to the next tick by the per-tick cap ` +
7166
+ `(${deferred.join(', ')}) — deferred, NOT dropped`
7167
+ : '') + '\n',
7168
+ )
7169
+ for (const requestId of send) {
7170
+ const pend = pendingPermissions.get(requestId)
7171
+ if (pend == null) continue
7172
+ try {
7173
+ // postPermissionCard registers the in-flight flag itself (so the ipc
7174
+ // initial-delivery caller is covered too) and releases it once every
7175
+ // target settles — or on a sync throw, in its own catch.
7176
+ postPermissionCard(requestId, pend)
7177
+ } catch (e) {
7178
+ // A synchronous throw must not escape the setInterval callback (it would
7179
+ // kill the whole reaper tick, TTL sweep included). postPermissionCard has
7180
+ // already released the flag; delete again defensively so the entry is
7181
+ // re-selectable on the next tick, and log rather than rethrow.
7182
+ heldCardsInFlight.delete(requestId)
7183
+ process.stderr.write(
7184
+ `telegram gateway: held-card re-delivery for ${requestId} threw ` +
7185
+ `synchronously: ${e} — flag released, will retry on the next tick\n`,
7186
+ )
7187
+ }
7188
+ }
7189
+ }
7190
+
6528
7191
  // 60-second sweep drops anything past its documented TTL.
6529
7192
  const pendingStateReaper = setInterval(() => {
6530
7193
  const now = Date.now()
7194
+ // #3084 follow-up — bring held cards BACK the moment the channel reopens.
7195
+ // Runs before the TTL sweep below so a card that can be re-delivered on this
7196
+ // tick is re-delivered, not considered for expiry.
7197
+ sweepHeldPermissionCards()
6531
7198
  // OAuth-code state grouped first (pinned by secret-detect-oauth-code.test.ts).
6532
7199
  pendingReauthFlows.sweep(now)
6533
7200
  for (const [k, v] of pendingAuthAddFlows) {
@@ -6539,6 +7206,21 @@ const pendingStateReaper = setInterval(() => {
6539
7206
  for (const [k, v] of awaitingAuthCodeAt) {
6540
7207
  if (now - v > AUTH_CODE_CONTEXT_TTL_MS) awaitingAuthCodeAt.delete(k)
6541
7208
  }
7209
+ // Loopback OAuth relay flows (issue #2582) — same TTL. Kill the waiting
7210
+ // CLI child so an abandoned consent never lingers with a bound listener.
7211
+ // Placed AFTER the OAuth-code cluster above, which secret-detect-oauth-
7212
+ // code.test.ts pins as contiguous within the first 800 chars of the
7213
+ // reaper (same precedent as the Microsoft connect sweep below). A flow
7214
+ // with a submit in flight is skipped (PR #3100 review finding 3): a paste
7215
+ // near the TTL boundary must not have its child killed mid-registration —
7216
+ // the submit path owns cleanup once `submitting` is set.
7217
+ for (const [k, v] of pendingLoopbackFlows) {
7218
+ if (v.submitting) continue
7219
+ if (now - v.startedAt > REAUTH_INTERCEPT_TTL_MS) {
7220
+ cancelLoopbackFlow(v)
7221
+ pendingLoopbackFlows.delete(k)
7222
+ }
7223
+ }
6542
7224
  // Microsoft connect flows self-expire at the device code's own expiry
6543
7225
  // (~15 min) — sweep past that + grace so an abandoned card doesn't pin
6544
7226
  // its key. Setting cancelled makes any still-running poll bail. Placed
@@ -6561,11 +7243,19 @@ const pendingStateReaper = setInterval(() => {
6561
7243
  if (now >= v.expiresAt) pendingAuthRmFlows.delete(k)
6562
7244
  }
6563
7245
  pendingVaultOps.sweep(now)
6564
- for (const [k, v] of pendingPermissions) {
6565
- // hostd gated fleet-mutation verbs get a longer (30-min) human-scale
6566
- // decision window than the 10-min default (Bug 2 fix #2).
6567
- const ttl = ttlForTool(v.tool_name)
6568
- if (now - v.startedAt > ttl) {
7246
+ // THE LEASH. The sweep — the guard AND the auto-deny it gates — now lives in
7247
+ // permission-ttl-sweep.ts, and the outcome test's harness drives that exact
7248
+ // function. It used to be an inline loop here while the harness kept a PRIVATE
7249
+ // copy of the guard, so deleting the real check left every behavioural assertion
7250
+ // GREEN and only a source-text grep noticed. A test that cannot fail is not a
7251
+ // test, and this is the test for `no-self-escalation`. One implementation, two
7252
+ // callers: there is no longer a gateway-side line whose deletion restores the
7253
+ // auto-deny without removing the sweep entirely.
7254
+ sweepPermissionTtl({
7255
+ entries: pendingPermissions,
7256
+ now,
7257
+ ttlForTool,
7258
+ onExpire: (k, v, ttl) => {
6569
7259
  // Don't just drop it: the claude turn is suspended INSIDE the MCP
6570
7260
  // permission call waiting for a verdict. A silent delete left it
6571
7261
  // wedged forever when the operator never tapped — permanent
@@ -6612,8 +7302,19 @@ const pendingStateReaper = setInterval(() => {
6612
7302
  // returns. Anchor to the card's own origin surface (where the operator
6613
7303
  // would have tapped), so the digest lands in the same topic — not a
6614
7304
  // fanned-out DM. Skip if the card was never posted anywhere.
7305
+ // The `cards[0]` hole (#3084 follow-up): this used to skip silently when the
7306
+ // card had never landed (`cards: []` → `origin === undefined`). That is
7307
+ // exactly the undeliverable case — so the request was auto-denied AND erased
7308
+ // from the only record that would have brought it back. The safety net had a
7309
+ // hole shaped exactly like the accident.
7310
+ //
7311
+ // Held entries no longer reach this code at all (the TTL freeze above skips
7312
+ // them), so this is belt-and-braces: ANY future path that times out a card
7313
+ // which never landed — including a PERMANENT 400, which holdReasonFor()
7314
+ // deliberately does not hold — still lands in the digest. Fall back to where
7315
+ // the card WOULD have gone.
6615
7316
  if (MISSED_APPROVAL_REOFFER_ENABLED) {
6616
- const origin = v.cards[0]
7317
+ const origin = v.cards[0] ?? resolvePermissionCardTargets()[0]
6617
7318
  if (origin != null) {
6618
7319
  missedApprovalsStore.add({
6619
7320
  requestId: k,
@@ -6632,8 +7333,9 @@ const pendingStateReaper = setInterval(() => {
6632
7333
  )
6633
7334
  pendingPermissions.delete(k)
6634
7335
  permCardStore.remove(k)
6635
- }
6636
- }
7336
+ reconcileBlockedApprovals()
7337
+ },
7338
+ })
6637
7339
  // Drop no-repeat suppression entries past the safety-cap window (the primary
6638
7340
  // bound is the operator-activity reset; this just keeps the map from growing).
6639
7341
  for (const [sig, at] of permissionTimeoutSignatures) {
@@ -7194,6 +7896,12 @@ const statusPinChatIds = new Map<string, string>()
7194
7896
  // re-pin of the same key keeps the original timestamp). Feeds the TTL gate of
7195
7897
  // the mid-session `wk:` pin reaper (#3001); cleared alongside the state.
7196
7898
  const statusPinPinnedAt = new Map<string, number>()
7899
+ // Rights-aware negative cache (#3024): chats where an auto status-pin attempt
7900
+ // failed with the permanent "not enough rights to manage pinned messages" 400.
7901
+ // Per-process only — a restart clears it so a later-granted pin right re-enables
7902
+ // auto-pin. The explicit `pin_message` MCP tool deliberately does NOT consult
7903
+ // this cache (it always attempts and surfaces the error to the agent).
7904
+ const statusPinRightsCache = new PinRightsCache()
7197
7905
 
7198
7906
  // Durable snapshot of the pin claim set on the persistent per-agent volume
7199
7907
  // (STATE_DIR = /state/agent/telegram in prod). Closes the crash hole: the
@@ -7228,6 +7936,48 @@ const activityCardStoreFs: ActivityCardStoreFsSeam = {
7228
7936
  }
7229
7937
  const activityCardPersistEnabled = !STATIC
7230
7938
 
7939
+ // Durable handle for the component-5 queued-status placeholder + #2995 busy-ack
7940
+ // card (#3002). Both track their sent message id only in the in-memory
7941
+ // `queuedStatusMsgIds` Map, so a restart between posting a card and its
7942
+ // promote/reap strands a permanent stale "⏳ Queued…" line. Persisted on POST,
7943
+ // cleared on in-process reap; a boot-time reaper (wired alongside
7944
+ // `activityCardBootReaper`, same startup-mutex ordering constraint) DELETES any
7945
+ // leftover card and clears the store — a "Queued" claim is always wrong after a
7946
+ // restart, so deletion is the honest terminal (reap-on-boot only, no
7947
+ // promote-across-restart). STATIC mode skips disk — same gate as the sibling
7948
+ // stores.
7949
+ const QUEUED_CARD_STORE_PATH = join(STATE_DIR, 'queued-cards-pending.json')
7950
+ const queuedCardStoreFs: QueuedCardStoreFsSeam = {
7951
+ readFileSync: (p: string) => readFileSync(p, 'utf8'),
7952
+ writeFileSync: (p: string, d: string) => writeFileSync(p, d),
7953
+ renameSync: (a: string, b: string) => renameSync(a, b),
7954
+ existsSync: (p: string) => existsSync(p),
7955
+ }
7956
+ const queuedCardPersistEnabled = !STATIC
7957
+
7958
+ // Write-through helpers (#3002) — mirror the in-memory `queuedStatusMsgIds`
7959
+ // set/delete onto the durable store. Best-effort + gated: no-op when the store
7960
+ // isn't usable (STATIC = no durable volume). Called from postQueuedStatus /
7961
+ // postBusyAck (post) and reapQueuedStatus (delete).
7962
+ function persistQueuedCard(
7963
+ key: string,
7964
+ chatId: string,
7965
+ threadId: number | null,
7966
+ messageId: number,
7967
+ ): void {
7968
+ if (!queuedCardPersistEnabled) return
7969
+ writeQueuedCardRecord(QUEUED_CARD_STORE_PATH, queuedCardStoreFs, {
7970
+ key,
7971
+ chatId,
7972
+ threadId,
7973
+ messageId,
7974
+ })
7975
+ }
7976
+ function clearQueuedCard(key: string, messageId?: number): void {
7977
+ if (!queuedCardPersistEnabled) return
7978
+ clearQueuedCardRecord(QUEUED_CARD_STORE_PATH, queuedCardStoreFs, key, messageId)
7979
+ }
7980
+
7231
7981
  // Slot-banner pin persistence (#421 crash-recovery). The slot banner is pinned
7232
7982
  // in the owner chat when the agent is on a non-default OAuth slot. Rather than a
7233
7983
  // parallel store + second boot hook, its pin is persisted in the SAME
@@ -7395,6 +8145,40 @@ async function activityCardBootReaper(): Promise<void> {
7395
8145
  )
7396
8146
  }
7397
8147
  }
8148
+
8149
+ /**
8150
+ * Boot-time reaper for orphaned queued-status / busy-ack cards (#3002). Thin
8151
+ * gateway wrapper over the pure `runQueuedCardBootReaper` — binds the live fs
8152
+ * seam, a robust Telegram delete, and the logger. Deletes any card persisted by
8153
+ * a prior (crashed/restarted) session and clears the store: a "⏳ Queued…" /
8154
+ * "On it" claim is always wrong after a restart, so deletion is the honest
8155
+ * terminal (reap-on-boot only, no promote-across-restart).
8156
+ *
8157
+ * MUST run ONLY after this gateway wins the startup mutex — same shared-file
8158
+ * ordering constraint as `activityCardBootReaper` / `statusPinBootCleanup`.
8159
+ */
8160
+ async function queuedCardBootReaper(): Promise<void> {
8161
+ if (!queuedCardPersistEnabled) return
8162
+ const { deleted, total } = await runQueuedCardBootReaper({
8163
+ path: QUEUED_CARD_STORE_PATH,
8164
+ fs: queuedCardStoreFs,
8165
+ deleteCard: (record) =>
8166
+ robustApiCall(
8167
+ () => lockedBot.api.deleteMessage(record.chatId, record.messageId),
8168
+ {
8169
+ chat_id: record.chatId,
8170
+ ...(record.threadId != null ? { threadId: record.threadId } : {}),
8171
+ verb: 'queued-card.boot-reap-delete',
8172
+ },
8173
+ ),
8174
+ })
8175
+ if (total > 0) {
8176
+ process.stderr.write(
8177
+ `telegram gateway: queued-card: deleted ${deleted}/${total} ` +
8178
+ `orphaned queued/busy-ack card(s) from a prior session\n`,
8179
+ )
8180
+ }
8181
+ }
7398
8182
  // ─── Mid-session stale-card reaper (#2918) ──────────────────────────────────
7399
8183
  // The boot reapers (markOrphanedWithTimeoutClassification + activityCardBoot-
7400
8184
  // Reaper) run ONCE at startup. A turn whose owning SDK subprocess dies
@@ -7651,6 +8435,16 @@ async function reconcileStatusPinInner(
7651
8435
  chatId,
7652
8436
  prevState: prev,
7653
8437
  desired,
8438
+ rightsCache: statusPinRightsCache,
8439
+ onPinRightsDisabled: (chat) => {
8440
+ // Logged ONCE per chat per process (#3024). Every subsequent auto-pin
8441
+ // attempt in this chat is skipped silently until a restart clears the
8442
+ // cache — replacing the 41x/48h `status-pin pin failed` spam marko saw.
8443
+ process.stderr.write(
8444
+ `telegram gateway: status-pin disabled for chat ${chat}: bot lacks ` +
8445
+ `pin rights; grant 'Pin messages' admin right to re-enable after restart\n`,
8446
+ )
8447
+ },
7654
8448
  onError: (phase, err) => {
7655
8449
  const msg = err instanceof Error ? err.message : String(err)
7656
8450
  process.stderr.write(
@@ -7752,6 +8546,105 @@ async function unpinAllStatusPins(): Promise<void> {
7752
8546
  }
7753
8547
  }
7754
8548
 
8549
+ // ─── DM stale-pin sweep (#3026) ─────────────────────────────────────────────
8550
+ // In a user DM the Bot API skips the boot getChat() probe (positive chat IDs
8551
+ // return `400 chat not found` until the user messages) AND getChat() exposes
8552
+ // only the NEWEST pin — DMs STACK pins and there is no list-pins method, so
8553
+ // older orphan pins are invisible to the probe-based sweep forever. The durable
8554
+ // fix: once per DM chat per boot, `unpinAllChatMessages` (safe in a DM — every
8555
+ // pin is bot-authored) then re-pin the live tracked cards. Groups keep the
8556
+ // probe path (unpin-all there would nuke human pins).
8557
+ //
8558
+ // Eligibility mirrors statusPinBootCleanup's mutex gate: set true ONLY after
8559
+ // this gateway wins the startup lock, so a losing double-boot never clears the
8560
+ // live holder's pins.
8561
+ let dmPinSweepEligible = false
8562
+ const dmPinSweeper: DmPinSweeper = createDmPinSweeper({
8563
+ unpinAll: (chatId) =>
8564
+ robustApiCall(() => lockedBot.api.unpinAllChatMessages(chatId), {
8565
+ chat_id: chatId,
8566
+ verb: 'dm-pin-sweep.unpin-all',
8567
+ }),
8568
+ pinSilent: (chatId, messageId) =>
8569
+ robustApiCall(
8570
+ () =>
8571
+ lockedBot.api.pinChatMessage(chatId, messageId, {
8572
+ disable_notification: true,
8573
+ }),
8574
+ { chat_id: chatId, verb: 'dm-pin-sweep.repin' },
8575
+ ),
8576
+ // Pins that must survive the unpin-all: live in-memory status-pin claims
8577
+ // for this chat (fg:/wk:/tool:/banner: — non-empty for a first-inbound
8578
+ // sweep landing mid-turn) UNIONED with the deliberately-retained store
8579
+ // rows — unexpired `tool:` pins (#3001) survive statusPinBootCleanup by
8580
+ // design and must survive this sweep too. Read LIVE from the store so
8581
+ // both the boot sweep (in-memory maps still empty then) and a later
8582
+ // first-inbound sweep see them. Best-effort: a store read failure
8583
+ // degrades to in-memory-only. The sweeper dedupes.
8584
+ liveTrackedMessageIds: (chatId) => {
8585
+ const ids: number[] = []
8586
+ for (const [key, st] of statusPinState.entries()) {
8587
+ if (statusPinChatIds.get(key) === chatId) ids.push(st.messageId)
8588
+ }
8589
+ if (statusPinPersistEnabled || bannerPinPersistEnabled || toolPinPersistEnabled) {
8590
+ try {
8591
+ ids.push(
8592
+ ...unexpiredStoreRepinIds(
8593
+ loadStatusPins(STATUS_PIN_STORE_PATH, statusPinStoreFs),
8594
+ chatId,
8595
+ Date.now(),
8596
+ ),
8597
+ )
8598
+ } catch (err) {
8599
+ process.stderr.write(
8600
+ `telegram gateway: dm-pin-sweep: store repin scan failed ` +
8601
+ `(chat=${chatId}): ${(err as Error).message}\n`,
8602
+ )
8603
+ }
8604
+ }
8605
+ return ids
8606
+ },
8607
+ eligible: () => dmPinSweepEligible,
8608
+ log: (line) => process.stderr.write(line),
8609
+ })
8610
+
8611
+ /**
8612
+ * Boot-time pin cleanup + DM stale-pin sweep, sequenced under the startup
8613
+ * mutex. Collects the DM chat IDs with a prior-session pin record BEFORE the
8614
+ * store reapers empty the stores, runs the three existing boot reapers, marks
8615
+ * the DM sweep eligible (this gateway now owns the shared state), then
8616
+ * unpin-alls each recorded DM chat. Fire-and-forget from the caller — never
8617
+ * blocks boot, never rejects unhandled.
8618
+ */
8619
+ async function runBootPinCleanupAndDmSweep(): Promise<void> {
8620
+ let dmChatIds: string[] = []
8621
+ try {
8622
+ dmChatIds = collectDmChatIdsFromStores({
8623
+ statusPins:
8624
+ statusPinPersistEnabled || bannerPinPersistEnabled || toolPinPersistEnabled
8625
+ ? loadStatusPins(STATUS_PIN_STORE_PATH, statusPinStoreFs)
8626
+ : [],
8627
+ activityCards: activityCardPersistEnabled
8628
+ ? loadActivityCards(ACTIVITY_CARD_STORE_PATH, activityCardStoreFs)
8629
+ : [],
8630
+ queuedCards: queuedCardPersistEnabled
8631
+ ? loadQueuedCards(QUEUED_CARD_STORE_PATH, queuedCardStoreFs)
8632
+ : [],
8633
+ })
8634
+ } catch (err) {
8635
+ process.stderr.write(
8636
+ `telegram gateway: dm-pin-sweep: store scan failed: ${(err as Error).message}\n`,
8637
+ )
8638
+ }
8639
+ await statusPinBootCleanup()
8640
+ await activityCardBootReaper()
8641
+ await queuedCardBootReaper()
8642
+ // This gateway now owns the shared per-agent pin state — enable the DM
8643
+ // unpin-all path (both the boot sweep below and lazy first-inbound sweeps).
8644
+ dmPinSweepEligible = true
8645
+ for (const id of dmChatIds) await dmPinSweeper.sweep(id)
8646
+ }
8647
+
7755
8648
  // Activity feed. The gateway streams a live "what it's doing" tool-activity
7756
8649
  // feed for every turn. The PreToolUse sidecar emits a `tool_label` per tool
7757
8650
  // call (flush-independent, so it stays real-time on fast/clustered-tool
@@ -7904,8 +8797,9 @@ function ensureIssuesCard(chatId: string, threadId: number | undefined): void {
7904
8797
  // pins from a prior (dead) session. Gated here (not at import time) so a
7905
8798
  // LOSING double-boot never unpins the live holder's legitimate pins.
7906
8799
  // Fire-and-forget: cleanup is best-effort and must not block boot.
7907
- void statusPinBootCleanup()
7908
- void activityCardBootReaper()
8800
+ // #3026: sequenced so the DM stale-pin sweep runs after the reapers and
8801
+ // only once this gateway owns the shared state.
8802
+ void runBootPinCleanupAndDmSweep()
7909
8803
  } catch (err) {
7910
8804
  process.stderr.write(
7911
8805
  `telegram gateway: boot.lock_acquire_failed err=${(err as Error).message} agent=${SWITCHROOM_AGENT_NAME}\n`,
@@ -7921,8 +8815,8 @@ function ensureIssuesCard(chatId: string, threadId: number | undefined): void {
7921
8815
  // probe + 409-retry loop is still the liveness guard on this path. A
7922
8816
  // successful writePidFile here means no live holder was detected, so
7923
8817
  // running orphan cleanup is consistent with the pre-mutex behaviour.
7924
- void statusPinBootCleanup()
7925
- void activityCardBootReaper()
8818
+ // #3026: same sequenced cleanup + DM stale-pin sweep as the mutex path.
8819
+ void runBootPinCleanupAndDmSweep()
7926
8820
  } catch (writeErr) {
7927
8821
  process.stderr.write(`telegram gateway: writePidFile failed: ${writeErr}\n`)
7928
8822
  }
@@ -8644,6 +9538,13 @@ pendingProgress.startTimer({
8644
9538
  chat_id: ctx.chatId,
8645
9539
  verb: 'pending-progress-edit',
8646
9540
  ...(ctx.threadId != null ? { threadId: ctx.threadId } : {}),
9541
+ // #3084 PR 2 / L1: the progress-card edit is the #1 documented ban
9542
+ // trigger (repeated editMessageText on one message). Tag COSMETIC and
9543
+ // pass messageId/editPayload so the gate's per-message floor + no-op
9544
+ // skip + coalescing engage, and the edit sheds while a window is open.
9545
+ priorityClass: 'cosmetic',
9546
+ messageId: ctx.messageId,
9547
+ editPayload: ctx.literalText ? ctx.newText : richMessage(ctx.newText),
8647
9548
  },
8648
9549
  )
8649
9550
  },
@@ -9256,6 +10157,16 @@ const ipcServer: IpcServer = createIpcServer({
9256
10157
  // this call is safe wherever it sits relative to the cron early-return
9257
10158
  // above (#3038 review finding 5).
9258
10159
  bridgeDeadWatchdog.noteBridgeRegistered(client.agentName)
10160
+ // #3043 item 2: a REAL bridge registering is proof the boot came all the
10161
+ // way up healthy — clear start.sh's crashloop boot-attempts counter so only
10162
+ // boots that genuinely fail BEFORE the bridge registers accumulate toward
10163
+ // the 3-strike override clear. Without this, three quick operator
10164
+ // hand-bounces of a healthy agent (each <150s apart) spuriously wipe a
10165
+ // working model override. Best-effort; no-op when the file is absent.
10166
+ if (client.agentName != null) {
10167
+ const smBootDir = resolveAgentDirFromEnv()
10168
+ if (smBootDir != null) clearSessionModelBootAttempts(smBootDir)
10169
+ }
9259
10170
  client.send({ type: 'status', status: 'agent_connected' })
9260
10171
 
9261
10172
  // Phase 2b PR 3a — bridgeUp cutover. The state machine's `bridgeUp`
@@ -9480,6 +10391,16 @@ const ipcServer: IpcServer = createIpcServer({
9480
10391
  // scripts/check-plugin-references.mjs (TS2722).
9481
10392
  progressDriver?.dispose?.({ preservePending: true })
9482
10393
  },
10394
+ // #2650: the bridge died mid-turn — its turn-long `typing…` loop never
10395
+ // hits the canonical turn-end stop, leaving a stale "typing…" until the
10396
+ // next turn. Sweep every live loop here (gated to registered-agent
10397
+ // disconnect inside flushOnAgentDisconnect).
10398
+ stopTurnTypingLoops: () => {
10399
+ turnTypingLoop.stopAll()
10400
+ // Shutdown drain: also cancel every armed catch-up so no typing ping
10401
+ // fires after the gateway has stopped.
10402
+ typingEmitter.reset()
10403
+ },
9483
10404
  // When dangling activeTurnStartedAt keys were swept (setDone raced
9484
10405
  // disconnect), the module-scope `currentTurn` may also point at the
9485
10406
  // dead bridge's turn. Null it so the next inbound starts a fresh
@@ -9702,7 +10623,8 @@ const ipcServer: IpcServer = createIpcServer({
9702
10623
  // `card_text` is retained so the TTL reaper can re-edit the SAME body
9703
10624
  // with a "timed out" footer while stripping the keyboard atomically
9704
10625
  // (Bug 2 fix #1) — the reaper has no grammy ctx to read the live text.
9705
- pendingPermissions.set(requestId, { tool_name: toolName, description, input_preview: inputPreview, startedAt: Date.now(), card_text: text, cards: [] })
10626
+ const pendEntry = { tool_name: toolName, description, input_preview: inputPreview, startedAt: Date.now(), card_text: text, cards: [] }
10627
+ pendingPermissions.set(requestId, pendEntry)
9706
10628
  // Compact action row: ❌ Deny · ✅ Allow · 🔁 Always… — the scope of an
9707
10629
  // "always" grant stays hidden until the operator taps "🔁 Always…",
9708
10630
  // which swaps the row for a scope choice (this file / any file ⚠️). The
@@ -9710,67 +10632,15 @@ const ipcServer: IpcServer = createIpcServer({
9710
10632
  // rule for this tool; unknown tools get the two-button row only. "Allow"
9711
10633
  // itself auto-grants a 30-min window for narrow non-destructive scopes
9712
10634
  // (decided in the allow handler), so there is no separate time-box button.
9713
- const showAlways = resolveScopedAllowChoices(toolName, inputPreview) != null
9714
- const keyboard = buildPermissionActionRow(requestId, showAlways)
9715
- // Route the card to the SAME place the post-verdict resume message
9716
- // lands (resolvePermissionCardTargets): the ORIGINATING chat+topic when
9717
- // there's an active turn — so a supergroup agent's card appears IN the
9718
- // topic the operator asked from (marko's "CRM (Brevo)"), not the
9719
- // operator DM — else the configured operator DMs, thread-stripped. The
9720
- // old code iterated `allowFrom` unconditionally, so a supergroup card
9721
- // could only ever reach operator DMs (the topic chat id is never in
9722
- // `allowFrom`) (marko, 2026-06-03).
10635
+ // Captured BEFORE the send: the status-reaction parking below needs the
10636
+ // turn that raised this card, and `currentTurn` can be re-pointed by a
10637
+ // concurrent inbound while the send is in flight.
9723
10638
  const activeTurn = currentTurn
9724
- const targets = resolvePermissionCardTargets()
9725
- for (const { chatId, threadId } of targets) {
9726
- // The rich-markdown path pairs with formatPermissionCardBody (#1790)
9727
- // so its bold/italic render. retryWithThreadFallback: if the topic was
9728
- // deleted/recreated (stale thread id → 400 "message thread not
9729
- // found"), re-send thread-less into the main chat so the card still
9730
- // ARRIVES rather than vanishing → 10-min TTL auto-deny → wedge.
9731
- // allow-raw-bot-api: wrapped in retryWithThreadFallback (retry policy); topic-aware send
9732
- void retryWithThreadFallback<{ message_id: number; message_thread_id?: number }>(
9733
- robustApiCall,
9734
- (tid) =>
9735
- bot.api.sendRichMessage(chatId, richMessage(text), {
9736
- reply_markup: keyboard,
9737
- ...(tid != null ? { message_thread_id: tid } : {}),
9738
- }),
9739
- { threadId, chat_id: chatId, verb: 'permission_request' },
9740
- ).then(sent => {
9741
- // Record the live card's (chat, message) so the TTL reaper can strip
9742
- // its inline keyboard on auto-deny — a stale Approve button left
9743
- // tappable dispatches a verdict for a dead request_id (Bug 2). The
9744
- // entry may already be gone (operator tapped before this resolved);
9745
- // guard the lookup.
9746
- const pend = pendingPermissions.get(requestId)
9747
- if (pend && sent && typeof sent.message_id === 'number') {
9748
- // #2787: record where the card ACTUALLY LANDED so the confirm sweep
9749
- // scopes its re-delivery suspension to the real chat/topic. When
9750
- // retryWithThreadFallback hit THREAD_NOT_FOUND (stale/renumbered
9751
- // topic) it re-sent thread-less into the main chat — the returned
9752
- // Message then carries no message_thread_id, so we must record the
9753
- // landed topic (undefined → suspend by bare chatId), NOT the stale
9754
- // requested `threadId`. Keying suspension on the stale topic would
9755
- // leave the main-chat card unsuspended (a re-deliver could clobber
9756
- // the live card) while needlessly suspending a topic that holds
9757
- // nothing. `sent.message_thread_id` reflects reality in all three
9758
- // cases: topic success → tid, main-chat / fallback → undefined.
9759
- const landedThreadId = sent.message_thread_id ?? undefined
9760
- pend.cards.push({ chatId, messageId: sent.message_id, threadId: landedThreadId })
9761
- permCardStore.add({
9762
- requestId,
9763
- chatId,
9764
- messageId: sent.message_id,
9765
- startedAt: pend.startedAt,
9766
- toolName: pend.tool_name,
9767
- cardText: pend.card_text,
9768
- })
9769
- }
9770
- }).catch(e => {
9771
- process.stderr.write(`telegram gateway: permission_request send to ${chatId} failed: ${e}\n`)
9772
- })
9773
- }
10639
+ // #3084 follow-up — the send now lives in postPermissionCard() so the
10640
+ // reaper can RE-DRIVE it when a flood window closes. This is the first
10641
+ // delivery attempt; if it fails against an open ban the entry is marked
10642
+ // undeliverable and held, never auto-denied.
10643
+ postPermissionCard(requestId, pendEntry)
9774
10644
  // Park the turn's status reaction on 🙏 (awaiting your tap) and
9775
10645
  // suspend the stall watchdog — a turn blocked on the operator is not
9776
10646
  // stalled, so it must not degrade to 🥱/😨 while the card sits
@@ -10481,6 +11351,36 @@ const ipcServer: IpcServer = createIpcServer({
10481
11351
  }
10482
11352
  },
10483
11353
 
11354
+ // #2975 Stage 2 — read-only pre-approval query from hostd. Answer whether the
11355
+ // EXACT (agent, diff) pair is already operator-consented so hostd can skip the
11356
+ // config_propose_edit rate limit for that persist. NEVER mutates state: it
11357
+ // only peeks the correlation maps by byte-exact match (isDiffPreApproved).
11358
+ // Fail-closed by construction — any agent mismatch or missing correlation
11359
+ // answers `preApproved: false`.
11360
+ onCheckPreApproved(client: IpcClient, msg: CheckPreApprovedMessage) {
11361
+ let preApproved = false
11362
+ try {
11363
+ const self = process.env.SWITCHROOM_AGENT_NAME
11364
+ // This gateway serves exactly one agent; a query for a different agent
11365
+ // can never be pre-approved here.
11366
+ if (!self || msg.agentName === self) {
11367
+ preApproved = isDiffPreApprovedLive(msg.agentName, msg.unifiedDiff)
11368
+ }
11369
+ } catch (err) {
11370
+ process.stderr.write(
11371
+ `telegram gateway: check_pre_approved errored — ${(err as Error).message} (failing closed)\n`,
11372
+ )
11373
+ preApproved = false
11374
+ }
11375
+ try {
11376
+ client.send({ type: 'pre_approved_result', correlationId: msg.correlationId, preApproved })
11377
+ } catch (err) {
11378
+ process.stderr.write(
11379
+ `telegram gateway: check_pre_approved reply failed: ${(err as Error).message}\n`,
11380
+ )
11381
+ }
11382
+ },
11383
+
10484
11384
  // #2670 one-tap self-improvement — persist a skill-improvement proposal and
10485
11385
  // post its Approve/Dismiss card. The store transition + apply-injection on
10486
11386
  // Approve are owned by handleSkillProposalCallback (so a gateway restart
@@ -11963,13 +12863,15 @@ async function executeReply(args: Record<string, unknown>): Promise<{ content: A
11963
12863
  robustApiCall(
11964
12864
  // allow-raw-bot-api: injected chunk-loop adapter — sendRichMessage routed through robustApiCall; THREAD_NOT_FOUND handled by sendReplyChunks' fallback ladder
11965
12865
  () => lockedBot.api.sendRichMessage(chat_id, body as never, opts as never),
11966
- { threadId: tid, chat_id },
12866
+ // #3084 PR 2: the final reply is CRITICAL — never shed; degraded mode
12867
+ // fails fast (structured flood_wait) instead of blocking the MCP reply.
12868
+ { threadId: tid, chat_id, priorityClass: 'critical' },
11967
12869
  ),
11968
12870
  sendLiteral: (opts, txt, tid) =>
11969
12871
  robustApiCall(
11970
12872
  // allow-raw-bot-api: injected chunk-loop adapter — literal format:'text' send routed through robustApiCall; THREAD_NOT_FOUND handled by sendReplyChunks
11971
12873
  () => lockedBot.api.sendMessage(chat_id, txt, opts as never),
11972
- { threadId: tid, chat_id },
12874
+ { threadId: tid, chat_id, priorityClass: 'critical' },
11973
12875
  ),
11974
12876
  sendLiteralRaw: (opts, txt) =>
11975
12877
  // allow-raw-bot-api: literal last-resort fallback (plaintext parse-reject / length re-split); wrapping would re-enter the parse/length policy that just rejected the payload
@@ -11981,7 +12883,10 @@ async function executeReply(args: Record<string, unknown>): Promise<{ content: A
11981
12883
  robustApiCall(
11982
12884
  // allow-raw-bot-api: preview edit-in-place routed through robustApiCall; thread fallback handled by sendReplyChunks
11983
12885
  () => lockedBot.api.editMessageText(chat_id, mid, body as never, opts as never),
11984
- { threadId: tid, chat_id },
12886
+ // Finalizing the reply into the preview message — still CRITICAL (this
12887
+ // IS the answer). Pass messageId/editPayload so the gate's per-message
12888
+ // floor + no-op skip engage on the edit (part3-design §4/§5, PR1 L1).
12889
+ { threadId: tid, chat_id, priorityClass: 'critical', messageId: mid, editPayload: body },
11985
12890
  ),
11986
12891
  richMessage,
11987
12892
  logOutbound,
@@ -13846,6 +14751,12 @@ async function executePinMessage(args: Record<string, unknown>): Promise<unknown
13846
14751
  () => lockedBot.api.pinChatMessage(pinChatId, pinMsgId),
13847
14752
  { chat_id: pinChatId, verb: 'pin_message' },
13848
14753
  )
14754
+ // An explicit pin succeeded here, so the bot demonstrably HAS pin rights in
14755
+ // this chat now — clear any auto-pin negative-cache entry (#3024) so the auto
14756
+ // status-pin path resumes immediately rather than waiting for a restart. The
14757
+ // explicit tool itself never consults the cache; a failure above still
14758
+ // surfaces to the agent as a normal tool error (robustApiCall rethrows).
14759
+ statusPinRightsCache.clear(pinChatId)
13849
14760
  // #3001: register the tool pin in the shared status-pin store under a
13850
14761
  // `tool:` key so it is no longer fire-and-forget. Unlike work-scoped
13851
14762
  // fg:/wk: rows a tool pin has no "work finished" event, so a restart does
@@ -14962,6 +15873,21 @@ function handleSessionEvent(ev: SessionEvent): void {
14962
15873
  liveTurn.liveness.onStreamEvent(ev.kind, durationMs, Date.now())
14963
15874
  }
14964
15875
  }
15876
+ // Idle-clear clocks (#3084 follow-up). EVERY genuine session event is
15877
+ // activity — an agent that is thinking, calling a tool, streaming text or
15878
+ // driving a sub-agent is NOT idle, whether or not the gateway currently has a
15879
+ // turn open. Stamping only at turn START (the old behaviour) is what let a
15880
+ // 3-hour working stretch be scored as zero activity and `/clear`ed the moment
15881
+ // the window elapsed. A turn ending stamps the turn-end clock too, so a turn
15882
+ // that outran the window is not wiped the instant `turnInFlight` goes false.
15883
+ // This runs for the whole event stream, including every `sub_agent_*` kind —
15884
+ // background workers keep the timer warm exactly as long as they are working.
15885
+ {
15886
+ const durationMs = ev.kind === 'turn_end' ? ev.durationMs : undefined
15887
+ const signal = classifyIdleEvent(ev.kind, durationMs)
15888
+ if (signal.activity) markIdleActivity()
15889
+ if (signal.turnEnded) markIdleTurnEnd()
15890
+ }
14965
15891
  switch (ev.kind) {
14966
15892
  case 'enqueue': {
14967
15893
  // Drain any orphaned typing-wrap entries left over from a crashed
@@ -15100,7 +16026,9 @@ function handleSessionEvent(ev: SessionEvent): void {
15100
16026
  // per-topic `byKey[statusKey]` entry AND the most-recent mirror. The key is
15101
16027
  // the SAME statusKey the ctor's façade was constructed with just above.
15102
16028
  setCurrentTurn(next, statusKey(ev.chatId, enqThreadIdNum))
15103
- markIdleActivity() // any turn start (main session) is activity — re-arm idle clear
16029
+ // (turn start already stamped the idle clock at the top of
16030
+ // handleSessionEvent, along with every other session event — see the
16031
+ // idle-clear block there.)
15104
16032
  // Early-open the "Working…" liveness card at turn start so narration /
15105
16033
  // thinking emitted BEFORE the first tool surfaces within ~a second
15106
16034
  // instead of after the old 12 s threshold (the dead-air gap). Fires the
@@ -15584,6 +16512,14 @@ function handleSessionEvent(ev: SessionEvent): void {
15584
16512
  chat_id: chatId,
15585
16513
  verb: 'answer-stream.editMessageText',
15586
16514
  ...(tid != null ? { threadId: tid } : {}),
16515
+ // #3084 PR 2 / L1: answer-stream edits are COSMETIC — a
16516
+ // dropped stream tick costs nothing (the next carries full
16517
+ // text). messageId/editPayload engage the per-message floor +
16518
+ // coalescing + no-op skip so rapid stream edits don't storm
16519
+ // the same message (the top ban trigger).
16520
+ priorityClass: 'cosmetic',
16521
+ messageId,
16522
+ editPayload: richMessage(text),
15587
16523
  },
15588
16524
  )
15589
16525
  },
@@ -15777,6 +16713,24 @@ function handleSessionEvent(ev: SessionEvent): void {
15777
16713
  return
15778
16714
  }
15779
16715
  }
16716
+ // #2094 finding 1 — turn_end gate-wedge backstop. Capture the turn
16717
+ // BEFORE the body runs (the body re-reads currentTurn as `turn`, then
16718
+ // nulls it via endCurrentTurnAtomic on every clean branch). The guarded
16719
+ // finally in withTurnEndGateBackstop forces the canonical purge iff a
16720
+ // throw in a pre-purge op (redactOutboundText, progressDriver?.
16721
+ // takeOverCard, narrative dedup, answer-stream finalize, …) skipped
16722
+ // endCurrentTurnAtomic → purgeReactionTracking, which would otherwise
16723
+ // leave activeTurnStartedAt + claudeBusyKeys populated and wedge the
16724
+ // #1556 inbound gate closed. No-op on the happy path (key already gone).
16725
+ const turnEndBackstopTurn = currentTurn
16726
+ const turnEndBackstopKey =
16727
+ turnEndBackstopTurn != null
16728
+ ? statusKey(turnEndBackstopTurn.sessionChatId, turnEndBackstopTurn.sessionThreadId)
16729
+ : null
16730
+ withTurnEndGateBackstop(
16731
+ turnEndBackstopKey,
16732
+ turnEndBackstopTurn,
16733
+ () => {
15780
16734
  // Drain any still-pending tool dispatch typing entries — covers
15781
16735
  // transcript truncation or a Claude Code crash mid-tool.
15782
16736
  typingWrapper.drainAll()
@@ -16003,6 +16957,17 @@ function handleSessionEvent(ev: SessionEvent): void {
16003
16957
  capturedText: turn.capturedText,
16004
16958
  flushEnabled: TURN_FLUSH_SAFETY_ENABLED,
16005
16959
  })
16960
+ // #1667 — resolve the turn_end answer-delivery gate once, here, via the
16961
+ // pure decision core. The three dispositions below (silent-marker,
16962
+ // turn-flush, #1664 re-prompt) delegate to this so the gateway runs the
16963
+ // exact code the regression test exercises. `finalAnswerDelivered` is read
16964
+ // at its tail value: the answer-stream materialize branch above has
16965
+ // already run (and may have set it true); the flush branch, which also
16966
+ // sets it, is not yet entered and does not affect the gate's own outcome.
16967
+ const turnEndDecision = decideTurnEndGate({
16968
+ flushDecision,
16969
+ finalAnswerDelivered: turn.finalAnswerDelivered,
16970
+ })
16006
16971
  if (flushDecision.kind === 'skip' && flushDecision.reason !== 'reply-called') {
16007
16972
  process.stderr.write(
16008
16973
  `telegram gateway: turn-flush skipped — reason=${flushDecision.reason}\n`,
@@ -16041,7 +17006,7 @@ function handleSessionEvent(ev: SessionEvent): void {
16041
17006
  // 2. NOT send any reply message to the user.
16042
17007
  // 3. Unpin the progress card so no orphaned ⚙️ Working… lingers.
16043
17008
  // 4. Log at debug level and fall through to normal state cleanup.
16044
- if (flushDecision.kind === 'skip' && flushDecision.reason === 'silent-marker') {
17009
+ if (turnEndDecision === 'silent_end') {
16045
17010
  // Don't try to distinguish NO_REPLY vs HEARTBEAT_OK in the log line:
16046
17011
  // `isSilentFlushMarker` accepts trailing punctuation (e.g. "NO_REPLY.")
16047
17012
  // and case variants, so a strict equality check would print the wrong
@@ -16131,7 +17096,7 @@ function handleSessionEvent(ev: SessionEvent): void {
16131
17096
  return
16132
17097
  }
16133
17098
 
16134
- if (flushDecision.kind === 'flush') {
17099
+ if (turnEndDecision === 'flush' && flushDecision.kind === 'flush') {
16135
17100
  let capturedText = flushDecision.text
16136
17101
  // #2798 — turn-flush delivers the model's terminal prose when it
16137
17102
  // skipped reply/stream_reply, but historically bypassed the reply
@@ -16267,9 +17232,12 @@ function handleSessionEvent(ev: SessionEvent): void {
16267
17232
  // the reaction is only finalized by the `turn_end` IPC
16268
17233
  // handler — mid-turn delivery proofs (local history,
16269
17234
  // stream finalize callbacks, executeReply post-send) no
16270
- // longer transition the emoji. This branch just purges
16271
- // the per-turn reaction tracking entry and returns.
16272
- purgeReactionTracking(statusKey(backstopChatId, backstopThreadId))
17235
+ // longer transition the emoji. This branch just returns.
17236
+ // #2094 cosmetic: the per-turn reaction tracking was ALREADY
17237
+ // purged synchronously by endCurrentTurnAtomic (before this
17238
+ // async IIFE ran). The old redundant purgeReactionTracking
17239
+ // here re-fired on an already-cleared key WITHOUT `endingTurn`,
17240
+ // emitting an inconsistent shadow trace. Removed.
16273
17241
  return
16274
17242
  }
16275
17243
  } catch {}
@@ -16405,9 +17373,13 @@ function handleSessionEvent(ev: SessionEvent): void {
16405
17373
  // #1713: backstop send failed — finalize as error so the
16406
17374
  // turn ends cleanly with 😱 rather than leaving it open.
16407
17375
  if (backstopCtrl) backstopCtrl.finalize('error')
16408
- } finally {
16409
- purgeReactionTracking(statusKey(backstopChatId, backstopThreadId))
16410
17376
  }
17377
+ // #2094 cosmetic: the trailing `finally { purgeReactionTracking() }`
17378
+ // was removed. endCurrentTurnAtomic already ran the canonical purge
17379
+ // (with the authoritative `endingTurn`) synchronously before this
17380
+ // async IIFE started, so re-purging here only re-fired on an
17381
+ // already-cleared key without `endingTurn` — an inconsistent shadow
17382
+ // trace. The #2094 finding-1 backstop covers any pre-purge throw.
16411
17383
  })()
16412
17384
  return
16413
17385
  }
@@ -16496,7 +17468,10 @@ function handleSessionEvent(ev: SessionEvent): void {
16496
17468
  // HEARTBEAT_OK silent-marker turns return earlier and never reach
16497
17469
  // this path. The turn-flush 'flush' branch also returns earlier
16498
17470
  // (and sets finalAnswerDelivered=true defensively).
16499
- if (turn.finalAnswerDelivered === false) {
17471
+ // #1667 — this is the reply-called tail; `turnEndDecision === 'reprompt'`
17472
+ // is exactly `turn.finalAnswerDelivered === false` here (silent-marker
17473
+ // and flush both returned earlier), delegated to the pure gate core.
17474
+ if (turnEndDecision === 'reprompt') {
16500
17475
  // PR #2892 (deterministic-turn-liveness RFC Phase 2) hardening:
16501
17476
  // wire the represent-guard-style staleness
16502
17477
  // check (`recordSilentTurnEnd`'s `hasOutboundDeliveredSince` dep) so
@@ -16600,6 +17575,14 @@ function handleSessionEvent(ev: SessionEvent): void {
16600
17575
  // #549 fix — preamble flush already happened at the TOP of this
16601
17576
  // turn_end handler (before turn.answerStream is nulled). See
16602
17577
  // comment near line 3431.
17578
+ return
17579
+ }, // end withTurnEndGateBackstop body (#2094 finding 1)
17580
+ {
17581
+ hasActiveTurn: (k) => activeTurnStartedAt.has(k),
17582
+ purge: (k, endingTurn) => purgeReactionTracking(k, endingTurn),
17583
+ log: (m) => process.stderr.write(m + '\n'),
17584
+ },
17585
+ )
16603
17586
  return
16604
17587
  }
16605
17588
  }
@@ -17004,7 +17987,9 @@ function maybeEarlyAckReaction(ctx: Context, from: NonNullable<Context['from']>)
17004
17987
  // model text. No fake content — Telegram clients render this natively
17005
17988
  // and it auto-expires after ~5s if not refreshed (the answer-lane
17006
17989
  // first edit will land long before then under the new defaults).
17007
- void bot.api.sendChatAction(chatId, 'typing').catch(() => {})
17990
+ // Through the shared emitter (#3084): a cold chat still lights up
17991
+ // instantly; a chat already pinged inside the floor doesn't pay twice.
17992
+ emitChatAction(chatId, null, 'typing')
17008
17993
  }
17009
17994
 
17010
17995
  /**
@@ -17063,6 +18048,16 @@ async function handleInbound(
17063
18048
  // #2862 — operator is back; re-offer any approvals that timed out meanwhile.
17064
18049
  maybePostMissedApprovalDigest('operator inbound')
17065
18050
 
18051
+ // #3026 — first inbound from a DM after boot: clear any stale STACKED pins
18052
+ // the boot getChat() probe can't see (DMs skip the probe AND getChat only
18053
+ // exposes the newest pin). unpin-all once per DM chat per boot, then re-pin
18054
+ // live tracked cards. once-guarded (dedups with the boot sweep); no-op for
18055
+ // groups and until the startup mutex is won. Fire-and-forget.
18056
+ {
18057
+ const inboundChatId = ctx.chat?.id
18058
+ if (inboundChatId != null) void dmPinSweeper.sweep(String(inboundChatId))
18059
+ }
18060
+
17066
18061
  // Capture wall-clock receive time for inbound_ack metric (#203).
17067
18062
  // Must be after gate() so early-exit paths (drop/pair) don't skew the delta.
17068
18063
  //
@@ -17474,6 +18469,101 @@ async function handleInbound(
17474
18469
  pendingAuthAddFlows.delete(interceptKey)
17475
18470
  }
17476
18471
 
18472
+ // Loopback OAuth relay paste-back intercept (issue #2582) — sibling to
18473
+ // the /auth add intercept above. When a Google/Microsoft loopback relay is
18474
+ // pending for this chat and the operator pastes their `127.0.0.1:<port>`
18475
+ // redirect URL, validate `state` and hand the code to the waiting CLI
18476
+ // listener. The LLM never sees the code (same hygiene rationale as above).
18477
+ // Consume-gate is deliberately narrow (PR #3100 review finding 1): only a
18478
+ // message that actually parses as a loopback redirect — carrying a code,
18479
+ // or a provider `error` param — is consumed and deleted. Unrelated chatter
18480
+ // that merely mentions localhost/127.0.0.1 flows through untouched.
18481
+ const pendingLoop = pendingLoopbackFlows.get(interceptKey)
18482
+ if (pendingLoop && shouldConsumeLoopbackPaste(text)) {
18483
+ const elapsed = Date.now() - pendingLoop.startedAt
18484
+ if (elapsed < REAUTH_INTERCEPT_TTL_MS) {
18485
+ if (pendingLoop.submitting) {
18486
+ // A submit is already in flight (double-paste race). Don't call
18487
+ // submitLoopbackRedirect — it would answer a non-retryable "already
18488
+ // completed" and we'd delete the entry out from under the first
18489
+ // submit. Benign ack; still redact (the paste carries a live code).
18490
+ await switchroomReply(
18491
+ ctx,
18492
+ '_Still finishing the previous paste — one moment._',
18493
+ { html: true },
18494
+ )
18495
+ redactAuthCodeMessage(bot.api as never, chat_id, msgId ?? null, line => process.stderr.write(line))
18496
+ return
18497
+ }
18498
+ const result = await submitLoopbackRedirect(pendingLoop, text.trim())
18499
+ if (result.ok) {
18500
+ pendingLoopbackFlows.delete(interceptKey)
18501
+ await switchroomReply(
18502
+ ctx,
18503
+ `✓ ${pendingLoop.provider === 'google' ? 'Google' : 'Microsoft'} account ` +
18504
+ `\`${escapeHtmlForTg(pendingLoop.email)}\` registered with the auth-broker.`,
18505
+ { html: true },
18506
+ )
18507
+ // Redact the pasted redirect (carries the OAuth code) from history.
18508
+ redactAuthCodeMessage(bot.api as never, chat_id, msgId ?? null, line => process.stderr.write(line))
18509
+ return
18510
+ }
18511
+ if (result.retryable) {
18512
+ // Keep the flow pending so the operator can paste again.
18513
+ await switchroomReply(
18514
+ ctx,
18515
+ `**Paste not accepted:** ${escapeHtmlForTg(result.reason)}\n` +
18516
+ `Re-open the consent URL, approve, and paste the full ` +
18517
+ `\`127.0.0.1\` URL from your address bar. \`/auth ${pendingLoop.provider} cancel\` to abort.`,
18518
+ { html: true },
18519
+ )
18520
+ // Redact even a rejected paste — it may still carry a live code.
18521
+ redactAuthCodeMessage(bot.api as never, chat_id, msgId ?? null, line => process.stderr.write(line))
18522
+ return
18523
+ }
18524
+ // Non-retryable — the flow is spent. Kill the CLI child before
18525
+ // dropping the entry (re-review finding, PR #3100): on the attempts-
18526
+ // exhausted path the child is still alive with a bound 127.0.0.1
18527
+ // listener and nothing else would ever reap it. cancelLoopbackFlow is
18528
+ // idempotent — safe on the already-exited / timed-out paths too.
18529
+ cancelLoopbackFlow(pendingLoop)
18530
+ pendingLoopbackFlows.delete(interceptKey)
18531
+ await switchroomReply(
18532
+ ctx,
18533
+ `**/auth ${pendingLoop.provider} add failed:** ${escapeHtmlForTg(result.reason)}`,
18534
+ { html: true },
18535
+ )
18536
+ redactAuthCodeMessage(bot.api as never, chat_id, msgId ?? null, line => process.stderr.write(line))
18537
+ return
18538
+ }
18539
+ // Stale — the intercept window has closed. Kill the child and drop the
18540
+ // entry, then fall through to the fail-safe below. We deliberately do NOT
18541
+ // let the paste reach the agent: the pasted `code` may still be live
18542
+ // (security audit #3084, F3), so the fail-safe redacts it.
18543
+ cancelLoopbackFlow(pendingLoop)
18544
+ pendingLoopbackFlows.delete(interceptKey)
18545
+ }
18546
+
18547
+ // Fail-safe redaction (security audit #3084, F2/F3). A message that looks
18548
+ // like a loopback OAuth redirect/code — even one too malformed to parse
18549
+ // cleanly, or one that arrived just after the intercept TTL closed, or one
18550
+ // with no active flow at all — must NEVER reach the (prompt-injectable)
18551
+ // agent session or linger unredacted in chat while carrying a possibly-live
18552
+ // credential. shouldConsumeLoopbackPaste is narrow (requires a loopback host
18553
+ // reference AND a code/error param), so ordinary chatter mentioning
18554
+ // localhost flows through untouched. Redact and drop rather than forward.
18555
+ if (shouldConsumeLoopbackPaste(text)) {
18556
+ redactAuthCodeMessage(bot.api as never, chat_id, msgId ?? null, line => process.stderr.write(line))
18557
+ await switchroomReply(
18558
+ ctx,
18559
+ '_That looked like an OAuth redirect/code, so I removed it from chat and did not forward it. ' +
18560
+ 'If a Google/Microsoft account add is in progress, re-run the add command and paste the fresh ' +
18561
+ '`127.0.0.1` URL — the previous code may have expired._',
18562
+ { html: true },
18563
+ )
18564
+ return
18565
+ }
18566
+
17477
18567
  // Auth-code intercept
17478
18568
  const pendingReauth = pendingReauthFlows.get(interceptKey)
17479
18569
  if (pendingReauth && looksLikeAuthCode(text)) {
@@ -17707,11 +18797,8 @@ async function handleInbound(
17707
18797
  // Typing indicator in the ORIGINATING topic — on a supergroup-topic inbound,
17708
18798
  // an un-threaded sendChatAction shows "typing" in General, not the topic the
17709
18799
  // user is in. messageThreadId is the inbound's thread (undefined in a DM).
17710
- void bot.api.sendChatAction(
17711
- chat_id,
17712
- 'typing',
17713
- messageThreadId != null ? { message_thread_id: messageThreadId } : {},
17714
- ).catch(() => {})
18800
+ // Floor-gated through the shared emitter (#3084).
18801
+ emitChatAction(chat_id, messageThreadId ?? null, 'typing')
17715
18802
 
17716
18803
  // Parse explicit prefixes first. `/steer ` / `/s ` opts IN to steering;
17717
18804
  // `/queue ` / `/q ` are legacy aliases that opt in to the new default (queued).
@@ -18792,6 +19879,17 @@ function escapeHtmlForTg(text: string): string {
18792
19879
  return text.replace(/([\\`*_~=\[\]|])/g, '\\$1')
18793
19880
  }
18794
19881
 
19882
+ // #789 — button-choice-confirmation ("✅ You chose: X") annotation state.
19883
+ //
19884
+ // Two once-per-process warning dedupe sets keyed by agent slug: one for the
19885
+ // parseMode gate (annotation only supports the default 'html' parse mode),
19886
+ // one for the single_use mismatch (a re-tappable keyboard can never be
19887
+ // annotated, so a confirmation request on it silently no-ops). The decision
19888
+ // + rendered text are pure functions in inline-keyboard-callbacks.ts
19889
+ // (resolveTapAnnotation) so they can be unit-tested against the exact payload.
19890
+ const buttonConfirmParseModeWarned = new Set<string>()
19891
+ const buttonConfirmSingleUseWarned = new Set<string>()
19892
+
18795
19893
  // Wrap CLI/command output in a fenced code block (content is literal there).
18796
19894
  function preBlock(text: string): string {
18797
19895
  return '```\n' + text.replace(/```/g, '`​``') + '\n```'
@@ -20127,6 +21225,30 @@ async function buildAgentMetadata(agentName: string): Promise<AgentMetadata> {
20127
21225
  auth: authSummary,
20128
21226
  audit: buildAgentAudit(agentName),
20129
21227
  live: await buildLiveProbeRows(agentName),
21228
+ sendGate: buildSendGateStatus(),
21229
+ }
21230
+ }
21231
+
21232
+ /**
21233
+ * Build the `/status` send-gate block (#3084 PR 3, part3-design §6). Returns
21234
+ * `undefined` when the gate flag is OFF so `/status` renders exactly as it did
21235
+ * before the gate existed. Queued / shed totals come from the live counters;
21236
+ * open flood windows are read from the persisted sibling file (already pruned).
21237
+ */
21238
+ function buildSendGateStatus(): AgentMetadata['sendGate'] {
21239
+ const s = sendGate.stats()
21240
+ if (!s.enabled) return undefined
21241
+ const openWindows = readFloodWindows(FLOOD_WINDOWS_PATH, Date.now()).map((w) => ({
21242
+ scopeKey: w.scopeKey,
21243
+ untilTs: w.untilTs,
21244
+ }))
21245
+ return {
21246
+ queued: s.global.queued,
21247
+ shed: s.global.shed,
21248
+ expired: s.global.expired,
21249
+ failedFast: s.global.failedFast,
21250
+ dropped: s.global.dropped,
21251
+ openWindows,
20130
21252
  }
20131
21253
  }
20132
21254
 
@@ -21447,15 +22569,25 @@ bot.command('update', async ctx => {
21447
22569
  const skipImages = passthrough.includes('--skip-images')
21448
22570
  const rebuild = passthrough.includes('--rebuild')
21449
22571
  const updateRequestId = hostdRequestId('gw-update')
21450
- const hostdResp = await tryHostdDispatch(getMyAgentName(), {
21451
- v: 1,
21452
- op: 'update_apply',
21453
- request_id: updateRequestId,
21454
- args: {
21455
- ...(skipImages ? { skip_images: true } : {}),
21456
- ...(rebuild ? { rebuild: true } : {}),
21457
- },
21458
- })
22572
+ // #1841 — forward the cached operator passphrase as the 2nd factor when
22573
+ // hostd requires operator-attest on update_apply. No-op when the vault
22574
+ // is locked / no passphrase is cached (feature-off posture unchanged).
22575
+ const updatePassphrase = vaultPassphraseCache.get(chatId)?.passphrase
22576
+ const hostdResp = await tryHostdDispatch(
22577
+ getMyAgentName(),
22578
+ withOperatorAttestation(
22579
+ {
22580
+ v: 1,
22581
+ op: 'update_apply',
22582
+ request_id: updateRequestId,
22583
+ args: {
22584
+ ...(skipImages ? { skip_images: true } : {}),
22585
+ ...(rebuild ? { rebuild: true } : {}),
22586
+ },
22587
+ },
22588
+ updatePassphrase,
22589
+ ),
22590
+ )
21459
22591
  if (hostdResp === 'not-configured') {
21460
22592
  warnLegacySpawnIfHostdDisabled('update_apply')
21461
22593
  spawnSwitchroomDetached(
@@ -21705,6 +22837,7 @@ async function handlePermissionSlash(ctx: Context, behavior: 'allow' | 'deny'):
21705
22837
  })
21706
22838
  pendingPermissions.delete(request_id)
21707
22839
  permCardStore.remove(request_id)
22840
+ reconcileBlockedApprovals()
21708
22841
  process.stderr.write(
21709
22842
  `[telegram gateway] slash-${behavior} request_id=${request_id} tool=${details.tool_name} by=${senderId}\n`,
21710
22843
  )
@@ -23008,6 +24141,84 @@ bot.command("auth", async ctx => {
23008
24141
  return
23009
24142
  }
23010
24143
 
24144
+ // `/auth google add|cancel` and `/auth microsoft add|cancel` — the
24145
+ // Telegram-native OAuth loopback relay (issue #2582). Gateway-routed for the
24146
+ // same reason as `/auth add`: they drive a child-process listener lifecycle
24147
+ // the broker client can't model. Admin-gated identically.
24148
+ if (parsed.kind === 'provider-add' || parsed.kind === 'provider-cancel') {
24149
+ if (!isAuthAdmin({ isAdmin })) {
24150
+ await switchroomReply(
24151
+ ctx,
24152
+ `**Not authorized.** \`/auth ${parsed.provider}\` is admin-only.\n` +
24153
+ `Set \`admin: true\` on this agent in switchroom.yaml to unlock.`,
24154
+ { html: true },
24155
+ )
24156
+ return
24157
+ }
24158
+ const loopKey = chatKey(chatId, ctx.message?.message_thread_id ?? null) as string
24159
+ if (parsed.kind === 'provider-cancel') {
24160
+ const existing = pendingLoopbackFlows.get(loopKey)
24161
+ if (!existing) {
24162
+ await switchroomReply(ctx, `_No pending \`/auth ${parsed.provider} add\` flow in this chat._`, { html: true })
24163
+ return
24164
+ }
24165
+ cancelLoopbackFlow(existing)
24166
+ pendingLoopbackFlows.delete(loopKey)
24167
+ await switchroomReply(ctx, 'Cancelled.', { html: true })
24168
+ return
24169
+ }
24170
+ // parsed.kind === 'provider-add'
24171
+ if (pendingLoopbackFlows.has(loopKey)) {
24172
+ await switchroomReply(
24173
+ ctx,
24174
+ `_An \`/auth ${parsed.provider} add\` flow is already in progress for this chat. ` +
24175
+ `Finish the paste, or send \`/auth ${parsed.provider} cancel\` to abort._`,
24176
+ { html: true },
24177
+ )
24178
+ return
24179
+ }
24180
+ try {
24181
+ const { consentUrl, state, port, child } = await startLoopbackFlow(
24182
+ parsed.provider,
24183
+ parsed.email,
24184
+ { replace: parsed.replace, write: parsed.write, orgMode: parsed.orgMode },
24185
+ )
24186
+ const newFlow = {
24187
+ provider: parsed.provider,
24188
+ email: parsed.email,
24189
+ state,
24190
+ port,
24191
+ consentUrl,
24192
+ child,
24193
+ startedAt: Date.now(),
24194
+ submitting: false,
24195
+ attempts: 0,
24196
+ }
24197
+ // Record child exit on the flow so a pre-paste crash fails fast
24198
+ // instead of hanging out the completion timeout (PR #3100 finding 4).
24199
+ trackFlowExit(newFlow)
24200
+ pendingLoopbackFlows.set(loopKey, newFlow)
24201
+ const providerName = parsed.provider === 'google' ? 'Google' : 'Microsoft'
24202
+ await switchroomReply(
24203
+ ctx,
24204
+ `**Adding ${providerName} account** \`${escapeHtmlForTg(parsed.email)}\`\n\n` +
24205
+ `1. Open this URL on your phone and approve:\n${consentUrl}\n\n` +
24206
+ `2. The redirect to \`127.0.0.1\` will fail to load — that's expected.\n` +
24207
+ `3. Copy the **full URL** from your browser's address bar (it holds ` +
24208
+ `\`?code=...&state=...\`) and paste it back here.\n\n` +
24209
+ `Send \`/auth ${parsed.provider} cancel\` to abort.`,
24210
+ { html: true },
24211
+ )
24212
+ } catch (err) {
24213
+ await switchroomReply(
24214
+ ctx,
24215
+ `**/auth ${parsed.provider} add failed:** ${escapeHtmlForTg((err as Error)?.message ?? String(err))}`,
24216
+ { html: true },
24217
+ )
24218
+ }
24219
+ return
24220
+ }
24221
+
23011
24222
  const client = await getAuthBrokerClient(currentAgent)
23012
24223
  if (!client) {
23013
24224
  await switchroomReply(ctx, "**/auth unavailable:** auth-broker client is not loaded (post-RFC-H rewire in progress?).", { html: true })
@@ -23201,6 +24412,9 @@ const callbackQueryHandlers = createCallbackQueryHandlers({
23201
24412
  getAdminOnlyKeys: () => ADMIN_ONLY_KEYS,
23202
24413
  vaultKeyRegex: VAULT_KEY_REGEX,
23203
24414
  mentalModelProposeTtlMs: MENTAL_MODEL_PROPOSE_TTL_MS,
24415
+ // #2975 Stage 1 — loud-failure funnel for a rate-window retry that also
24416
+ // failed (cooldown + record + broadcast in one place).
24417
+ emitOperatorEvent: emitGatewayOperatorEvent,
23204
24418
  })
23205
24419
  const {
23206
24420
  handleVaultRecentDenialCallback,
@@ -24634,8 +25848,74 @@ bot.on('callback_query:data', async ctx => {
24634
25848
  // opts out via `single_use: false`. With no stashed meta (e.g.
24635
25849
  // gateway restarted between send and tap) the default fires too,
24636
25850
  // which is the desired UX.
24637
- const stripKeyboard = metaForMessage == null || keyboardIsSingleUse(metaForMessage)
24638
- if (stripKeyboard && cbMessageId != null) {
25851
+ const singleUse = metaForMessage == null || keyboardIsSingleUse(metaForMessage)
25852
+
25853
+ // Source message body text — photos/stickers/etc. carry no `text`.
25854
+ const cbMsg = ctx.callbackQuery?.message
25855
+ const sourceText = cbMsg && 'text' in cbMsg && typeof cbMsg.text === 'string'
25856
+ ? cbMsg.text
25857
+ : undefined
25858
+
25859
+ // #789: pure decision for the "✅ You chose: X" body annotation. Per-message
25860
+ // override (the tapped button's inline_keyboard_confirm) wins over the agent
25861
+ // default; fires only on single-use keyboards with body text + a label and
25862
+ // the default 'html' parse mode.
25863
+ const annotation = resolveTapAnnotation({
25864
+ ...(tapMeta?.inline_keyboard_confirm != null
25865
+ ? { perMessageOverride: tapMeta.inline_keyboard_confirm }
25866
+ : {}),
25867
+ singleUse,
25868
+ ...(access.button_choice_confirmation != null
25869
+ ? { config: access.button_choice_confirmation }
25870
+ : {}),
25871
+ parseMode: access.parseMode ?? 'html',
25872
+ ...(sourceText != null ? { sourceText } : {}),
25873
+ ...(buttonText != null ? { label: buttonText } : {}),
25874
+ // escapeLabel deliberately omitted → defaults to the real HTML-entity
25875
+ // escaper (escapeHtmlEntities). The GFM-markdown escaper
25876
+ // (escapeHtmlForTg, #2669) is WRONG here: it doesn't escape &/</> (a
25877
+ // label containing them would 400 the HTML editMessageText) and it
25878
+ // garbles text under HTML parse mode (`Do_it` → `Do\_it`).
25879
+ })
25880
+
25881
+ const warnKey = process.env.SWITCHROOM_AGENT_NAME ?? ''
25882
+ // Round-1 minor finding: a confirmation requested on a single_use:false
25883
+ // (re-tappable) keyboard can never fire. Warn once per agent-process.
25884
+ if (annotation.warnSingleUseMismatch && !buttonConfirmSingleUseWarned.has(warnKey)) {
25885
+ buttonConfirmSingleUseWarned.add(warnKey)
25886
+ process.stderr.write(
25887
+ `telegram gateway: button_choice_confirmation skipped — inline_keyboard_confirm requested on a single_use:false (re-tappable) keyboard; annotation only fires on single-use keyboards (#789)\n`,
25888
+ )
25889
+ }
25890
+ // Parse-mode gate (Blocker 1): annotation is emitted with HTML. Warn once
25891
+ // per agent-process when a non-html parseMode suppresses it.
25892
+ if (annotation.warnParseMode && !buttonConfirmParseModeWarned.has(warnKey)) {
25893
+ buttonConfirmParseModeWarned.add(warnKey)
25894
+ process.stderr.write(
25895
+ `telegram gateway: button_choice_confirmation skipped — agent parseMode=${access.parseMode ?? 'html'}, only 'html' supported (#789)\n`,
25896
+ )
25897
+ }
25898
+
25899
+ if (annotation.annotate && cbMessageId != null) {
25900
+ // Single API call (Blocker 2): editMessageText accepts reply_markup, so
25901
+ // we annotate the body AND strip the keyboard in one edit. If the edit
25902
+ // rejects (400 on an HTML edge case, over-long body, too-old message…),
25903
+ // applyTapAnnotationEdit falls back to a keyboard-only strip so
25904
+ // single-use protection still holds — the meta is deleted below either
25905
+ // way, so a swallowed failure must not leave the keyboard live.
25906
+ await applyTapAnnotationEdit(
25907
+ {
25908
+ editMessageText: (text, other) => ctx.editMessageText(text, other),
25909
+ editMessageReplyMarkup: other => ctx.editMessageReplyMarkup(other),
25910
+ },
25911
+ annotation.text as string,
25912
+ )
25913
+ if (metaForMessage != null) {
25914
+ agentButtonMeta.delete(`${cbChatId}:${cbMessageId}`)
25915
+ }
25916
+ } else if (singleUse && cbMessageId != null) {
25917
+ // No annotation (disabled, non-single-use, non-html parseMode, or no
25918
+ // body text) — preserve the historical keyboard-only strip.
24639
25919
  await ctx.editMessageReplyMarkup({
24640
25920
  reply_markup: { inline_keyboard: [] },
24641
25921
  }).catch(() => {})
@@ -24714,6 +25994,7 @@ bot.on('callback_query:data', async ctx => {
24714
25994
 
24715
25995
  pendingPermissions.delete(request_id)
24716
25996
  permCardStore.remove(request_id)
25997
+ reconcileBlockedApprovals()
24717
25998
 
24718
25999
  // (2) Dispatch the in-flight permission verdict IMMEDIATELY — before
24719
26000
  // any host round-trip — so the turn never blocks on persistence.
@@ -25009,6 +26290,7 @@ bot.on('callback_query:data', async ctx => {
25009
26290
  const grantAgent = selfAgentName()
25010
26291
  pendingPermissions.delete(request_id)
25011
26292
  permCardStore.remove(request_id)
26293
+ reconcileBlockedApprovals()
25012
26294
  if (timeBox && grantAgent) {
25013
26295
  recordScopedGrant(scopedGrants, grantAgent, timeBox.rule, Date.now(), scopedTtl)
25014
26296
  // Write-through so the window survives a gateway restart (#2863). Absolute
@@ -26049,16 +27331,22 @@ function flushReactionBatch(batch: ReactionBatch): void {
26049
27331
  text,
26050
27332
  meta,
26051
27333
  }
26052
- const delivered = ipcServer.sendToAgent(agentName, inbound)
26053
- if (delivered) markClaudeBusyForInbound(inbound)
27334
+ // #2094 finding 3 — route the reaction flush through the SAME #1556
27335
+ // decideInboundDelivery gate every other synthetic inbound uses (via
27336
+ // deliverResumeSyntheticOrBuffer), instead of a raw ipcServer.sendToAgent.
27337
+ // A reaction that lands WHILE a turn is in flight was previously fired
27338
+ // mid-turn — the bridge typed it into the CLI composer where it stranded
27339
+ // by the turn-completion race (the #1556 composer wedge). Now a mid-turn
27340
+ // reaction buffers-until-idle (the turn-complete hook + idle-drain timer
27341
+ // flush pendingInboundBuffer the instant claude goes idle, landing cleanly
27342
+ // as a fresh turn); an idle reaction delivers now, buffering only on a
27343
+ // genuine bridge-offline miss — the #1150 buffer-on-failure guarantee,
27344
+ // preserved inside the helper.
27345
+ const delivered = deliverResumeSyntheticOrBuffer(agentName, inbound)
26054
27346
  process.stderr.write(
26055
27347
  `telegram gateway: reactions.dispatch agent=${agentName} chat=${batch.chatId} ` +
26056
27348
  `count=${batch.reactions.length} batched=${batch.batched} delivered=${delivered}\n`,
26057
27349
  )
26058
- // #1150: buffer-on-failure for reaction-triggered wake-ups too.
26059
- if (!delivered) {
26060
- pendingInboundBuffer.push(agentName, inbound)
26061
- }
26062
27350
  }
26063
27351
 
26064
27352
  // ─── Inbound message_reaction handler ────────────────────────────────────
@@ -26506,6 +27794,10 @@ async function shutdown(signal: string): Promise<void> {
26506
27794
  // Now finish the cleanup the drain didn't touch.
26507
27795
  inboundCoalescer.reset()
26508
27796
  pendingReauthFlows.clear()
27797
+ // Kill any in-flight loopback relay CLI children so they don't outlive the
27798
+ // gateway with a bound 127.0.0.1 listener (issue #2582).
27799
+ for (const [, v] of pendingLoopbackFlows) cancelLoopbackFlow(v)
27800
+ pendingLoopbackFlows.clear()
26509
27801
  pendingVaultOps.clear()
26510
27802
  pendingPermissions.clear()
26511
27803
  permissionTimeoutSignatures.clear()
@@ -26949,6 +28241,22 @@ void (async () => {
26949
28241
  )
26950
28242
  }
26951
28243
 
28244
+ // #3084 follow-up — drop a blocked-approval record orphaned by a restart
28245
+ // mid-hold. `pendingPermissions` is in-memory and empty on a fresh
28246
+ // process, so this reconciles the surface to "nothing held". Without it,
28247
+ // a gateway restart during a flood ban would leave a record on disk that
28248
+ // nothing ever clears, and the dashboard would show a permanently
28249
+ // blocked agent. The ASK itself survives: the bridge re-sends unresolved
28250
+ // permission requests on IPC reconnect, which re-raises them (and
28251
+ // re-marks them held if the channel is still shut).
28252
+ try {
28253
+ reconcileBlockedApprovals()
28254
+ } catch (err) {
28255
+ process.stderr.write(
28256
+ `telegram gateway: blocked-approval boot reconcile failed: ${(err as Error).message}\n`,
28257
+ )
28258
+ }
28259
+
26952
28260
  // Boot-time pin sweep
26953
28261
  try {
26954
28262
  const bootAccess = loadAccess()
@@ -27382,7 +28690,8 @@ void (async () => {
27382
28690
  richMessage(text),
27383
28691
  sendOpts as Parameters<typeof lockedBot.api.sendRichMessage>[2],
27384
28692
  ),
27385
- { chat_id: cid, verb: 'worker-feed' },
28693
+ // #3084 PR 2: worker-feed card CREATION (handback) is USEFUL.
28694
+ { chat_id: cid, verb: 'worker-feed', priorityClass: 'useful' },
27386
28695
  )
27387
28696
  return sent as { message_id: number }
27388
28697
  },
@@ -27395,7 +28704,15 @@ void (async () => {
27395
28704
  richMessage(text),
27396
28705
  editOpts as Parameters<typeof lockedBot.api.editMessageText>[3],
27397
28706
  ),
27398
- { chat_id: cid, verb: 'worker-feed' },
28707
+ // Worker-feed EDITs are COSMETIC — pass messageId/editPayload
28708
+ // so the per-message floor + coalescing + no-op skip engage.
28709
+ {
28710
+ chat_id: cid,
28711
+ verb: 'worker-feed',
28712
+ priorityClass: 'cosmetic',
28713
+ messageId: mid,
28714
+ editPayload: text,
28715
+ },
27399
28716
  ),
27400
28717
  },
27401
28718
  log: (msg) => process.stderr.write(`telegram gateway: ${msg}\n`),