switchroom 0.18.11 → 0.18.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/dist/agent-scheduler/index.js +29 -5
  2. package/dist/auth-broker/index.js +53 -13
  3. package/dist/cli/hindsight-mental-model-pretool.mjs +39 -0
  4. package/dist/cli/notion-write-pretool.mjs +29 -5
  5. package/dist/cli/switchroom.js +2453 -1293
  6. package/dist/cli/ui/index.html +163 -17
  7. package/dist/host-control/main.js +504 -100
  8. package/dist/vault/approvals/kernel-server.js +53 -13
  9. package/dist/vault/broker/server.js +162 -114
  10. package/package.json +3 -4
  11. package/profiles/_base/start.sh.hbs +65 -0
  12. package/profiles/_shared/vault-protocol.md.hbs +3 -1
  13. package/profiles/coding/CLAUDE.md.hbs +1 -1
  14. package/profiles/default/CLAUDE.md.hbs +2 -2
  15. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  16. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  17. package/telegram-plugin/bridge/bridge.ts +37 -0
  18. package/telegram-plugin/bridge/inbound-dedup.ts +101 -0
  19. package/telegram-plugin/dist/bridge/bridge.js +73 -1
  20. package/telegram-plugin/dist/gateway/gateway.js +3602 -1007
  21. package/telegram-plugin/dist/server.js +74 -2
  22. package/telegram-plugin/flood-circuit-breaker.ts +493 -21
  23. package/telegram-plugin/gateway/approval-hold.ts +583 -0
  24. package/telegram-plugin/gateway/auth-command.ts +92 -2
  25. package/telegram-plugin/gateway/auth-loopback-relay.ts +670 -0
  26. package/telegram-plugin/gateway/boot-card.ts +12 -5
  27. package/telegram-plugin/gateway/callback-query-handlers.ts +76 -1
  28. package/telegram-plugin/gateway/config-approval-handler.ts +6 -1
  29. package/telegram-plugin/gateway/disconnect-flush.ts +19 -0
  30. package/telegram-plugin/gateway/dm-pin-sweep.test.ts +251 -0
  31. package/telegram-plugin/gateway/dm-pin-sweep.ts +178 -0
  32. package/telegram-plugin/gateway/gateway.ts +1482 -165
  33. package/telegram-plugin/gateway/hostd-dispatch.ts +23 -0
  34. package/telegram-plugin/gateway/idle-clear.ts +90 -6
  35. package/telegram-plugin/gateway/inbound-delivery-machine-shadow.ts +26 -5
  36. package/telegram-plugin/gateway/inject-handler.ts +8 -0
  37. package/telegram-plugin/gateway/ipc-protocol.ts +46 -3
  38. package/telegram-plugin/gateway/ipc-server.ts +43 -0
  39. package/telegram-plugin/gateway/mental-model-propose-resolve.ts +145 -37
  40. package/telegram-plugin/gateway/model-command.ts +9 -3
  41. package/telegram-plugin/gateway/pending-session-command.ts +13 -1
  42. package/telegram-plugin/gateway/permission-ttl-sweep.ts +66 -0
  43. package/telegram-plugin/gateway/pre-approval-check.ts +74 -0
  44. package/telegram-plugin/gateway/queued-card-store.ts +217 -0
  45. package/telegram-plugin/gateway/session-model-file.ts +26 -1
  46. package/telegram-plugin/gateway/turn-end-gate-backstop.ts +59 -0
  47. package/telegram-plugin/gateway/turn-end-gate.ts +95 -0
  48. package/telegram-plugin/gateway/turn-typing-loop.ts +10 -2
  49. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +13 -0
  50. package/telegram-plugin/hooks/dispatch-claim-scan.mjs +259 -0
  51. package/telegram-plugin/hooks/dispatch-claim-stop.mjs +129 -0
  52. package/telegram-plugin/hooks/hooks.json +9 -0
  53. package/telegram-plugin/inline-keyboard-callbacks.ts +209 -2
  54. package/telegram-plugin/operator-events.ts +23 -0
  55. package/telegram-plugin/package.json +0 -1
  56. package/telegram-plugin/permission-rule.ts +1 -0
  57. package/telegram-plugin/permission-title.ts +1 -0
  58. package/telegram-plugin/retry-api-call.ts +212 -2
  59. package/telegram-plugin/send-gate-degraded.test.ts +443 -0
  60. package/telegram-plugin/send-gate-observability.test.ts +470 -0
  61. package/telegram-plugin/send-gate-observability.ts +355 -0
  62. package/telegram-plugin/send-gate.test.ts +698 -0
  63. package/telegram-plugin/send-gate.ts +982 -0
  64. package/telegram-plugin/shared/bot-runtime.ts +17 -5
  65. package/telegram-plugin/shared/gw-trace-gate.ts +105 -0
  66. package/telegram-plugin/status-pin-driver.ts +52 -7
  67. package/telegram-plugin/status-pin.ts +81 -0
  68. package/telegram-plugin/subagent-watcher.ts +102 -2
  69. package/telegram-plugin/tests/activity-card-wiring.test.ts +18 -5
  70. package/telegram-plugin/tests/approval-hold-harness.ts +425 -0
  71. package/telegram-plugin/tests/approval-hold-outcome.test.ts +296 -0
  72. package/telegram-plugin/tests/approval-hold-record.test.ts +531 -0
  73. package/telegram-plugin/tests/approval-hold-redeliver.test.ts +602 -0
  74. package/telegram-plugin/tests/auth-loopback-relay.test.ts +533 -0
  75. package/telegram-plugin/tests/boot-card-flood-suppress.test.ts +53 -7
  76. package/telegram-plugin/tests/busy-key-reaper.test.ts +1 -0
  77. package/telegram-plugin/tests/dispatch-claim-scan.test.ts +250 -0
  78. package/telegram-plugin/tests/flood-breaker-blindness.test.ts +213 -0
  79. package/telegram-plugin/tests/flood-windows-persistence.test.ts +224 -0
  80. package/telegram-plugin/tests/gateway-boot-marker-clear.test.ts +3 -3
  81. package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +29 -1
  82. package/telegram-plugin/tests/gateway-loopback-paste-redact.test.ts +66 -0
  83. package/telegram-plugin/tests/gw-trace-gate.test.ts +105 -0
  84. package/telegram-plugin/tests/idle-clear.test.ts +233 -3
  85. package/telegram-plugin/tests/inbound-dedup.test.ts +93 -0
  86. package/telegram-plugin/tests/inline-keyboard-callbacks.test.ts +284 -0
  87. package/telegram-plugin/tests/ipc-server-check-pre-approved.test.ts +194 -0
  88. package/telegram-plugin/tests/mental-model-propose-resolve.test.ts +123 -0
  89. package/telegram-plugin/tests/missed-approvals-wiring.test.ts +1 -1
  90. package/telegram-plugin/tests/model-command.test.ts +14 -0
  91. package/telegram-plugin/tests/pending-session-command.test.ts +21 -0
  92. package/telegram-plugin/tests/permission-card-routing.test.ts +30 -5
  93. package/telegram-plugin/tests/permission-no-repeat-wiring.test.ts +8 -7
  94. package/telegram-plugin/tests/permission-rearm-wiring.test.ts +1 -1
  95. package/telegram-plugin/tests/pre-approval-check.test.ts +148 -0
  96. package/telegram-plugin/tests/queued-card-store.test.ts +232 -0
  97. package/telegram-plugin/tests/reaction-flush-turn-gated.test.ts +100 -0
  98. package/telegram-plugin/tests/retry-api-call.test.ts +398 -0
  99. package/telegram-plugin/tests/session-model-file.test.ts +50 -0
  100. package/telegram-plugin/tests/status-pin-boot-recovery.test.ts +35 -14
  101. package/telegram-plugin/tests/status-pin.test.ts +275 -1
  102. package/telegram-plugin/tests/subagent-watcher-deferral-log-ratelimit.test.ts +316 -0
  103. package/telegram-plugin/tests/turn-end-gate-backstop.test.ts +92 -0
  104. package/telegram-plugin/tests/turn-end-gate.test.ts +137 -0
  105. package/telegram-plugin/tests/typing-emitter.test.ts +586 -0
  106. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +20 -0
  107. package/telegram-plugin/typing-emitter.ts +224 -0
  108. package/telegram-plugin/uat/scenarios/jtbd-feel-like-a-colleague-dm.test.ts +136 -0
  109. package/telegram-plugin/welcome-text.ts +42 -0
  110. package/vendor/hindsight-memory/scripts/drain_pending.py +22 -6
  111. package/vendor/hindsight-memory/scripts/lib/client.py +12 -5
  112. package/vendor/hindsight-memory/scripts/lib/directives.py +38 -3
  113. package/vendor/hindsight-memory/scripts/lib/pending.py +36 -9
  114. package/vendor/hindsight-memory/scripts/session_end.py +14 -3
  115. package/vendor/hindsight-memory/scripts/session_start.py +21 -0
  116. package/vendor/hindsight-memory/scripts/tests/test_directives.py +38 -0
  117. package/vendor/hindsight-memory/tests/test_drain_pending.py +68 -0
  118. package/vendor/hindsight-memory/tests/test_pending.py +44 -0
  119. package/vendor/hindsight-memory/tests/test_session_end_pending.py +38 -0
  120. package/vendor/hindsight-memory/tests/test_session_start_drain.py +155 -0
  121. package/telegram-plugin/channel-envelope-safety.test.ts +0 -56
  122. package/telegram-plugin/channel-envelope-safety.ts +0 -56
@@ -0,0 +1,982 @@
1
+ /**
2
+ * Deterministic outbound send gate for the Telegram Bot API (#3084, PR 1/3).
3
+ *
4
+ * WHY
5
+ * ---
6
+ * Each agent's gateway drives MANY outbound surfaces (reply chunks, answer /
7
+ * draft stream edits, typing, worker-feed edits, reactions, cards). Each has
8
+ * its own local throttle, but nothing composes them into a global or per-chat
9
+ * ceiling — so during a busy turn the per-surface throttles ADD UP with no cap
10
+ * and trip a per-bot-token flood ban (429 retry_after ~hours). See
11
+ * `part2-audit.md` §3 and issue #3084.
12
+ *
13
+ * This module is the core control from `part3-design.md` §1: ONE token-bucket
14
+ * scheduler that every Bot API call passes through (wired at the robustApiCall
15
+ * layer so no call site can bypass it). It enforces:
16
+ *
17
+ * - Global bucket: 25/sec sustained, small burst (headroom under ~30/s).
18
+ * - Per-chat bucket: 1/sec sustained, burst 3.
19
+ * - Per-group bucket: 18/min sustained, small burst (headroom under 20),
20
+ * keyed on chat type.
21
+ * - Per-message edit: >=1.5s between edits of the same message_id, with
22
+ * LAST-WRITE-WINS coalescing (an edit queued while one
23
+ * is pending REPLACES the pending payload; it never
24
+ * queues behind it — the next edit carries full state)
25
+ * AND in-flight serialization (only one send per
26
+ * message runs at a time; edits that arrive while a
27
+ * send is mid-flight — including one sleeping on a
28
+ * 429 retry_after — coalesce behind it rather than
29
+ * firing a second overlapping edit).
30
+ *
31
+ * Nested buckets like PTB's AIORateLimiter; grammY's Bottleneck numbers are the
32
+ * sanity reference. Sustained rate is the refill rate (= budget); the burst
33
+ * capacity is a small headroom under Telegram's hard window ceiling, NOT a full
34
+ * extra window's worth of budget (that would double the effective rate — see
35
+ * the M2 finding on #3092).
36
+ *
37
+ * It also skips no-op edits: identical rendered payload for the same
38
+ * message_id is dropped before hitting the API (Telegram 400s "message is not
39
+ * modified" today, which still costs flood budget).
40
+ *
41
+ * RESTART-PROOF FLOOD STATE (part3-design §7, PR 2 hook)
42
+ * -----------------------------------------------------
43
+ * `createSendGate` accepts `initialWindows` (flood windows loaded from
44
+ * `flood-wait.json` on boot) and `bootRamp` (start the global bucket at a
45
+ * fraction of capacity for the first N ms to absorb boot-card bursts).
46
+ * `openFloodWindow(scopeKey, untilTs)` lets PR 2 re-open a window at runtime
47
+ * on a 429. A bucket under an open window admits NOTHING until `untilTs`,
48
+ * regardless of token fill — a token bucket alone cannot express a
49
+ * "blocked until" suppression, so the buckets consult a per-scope window too.
50
+ *
51
+ * SAFETY / ROLLOUT
52
+ * ----------------
53
+ * Feature-flagged and default-OFF (`SWITCHROOM_TELEGRAM_SEND_GATE=1` to enable,
54
+ * following the `SWITCHROOM_*=== '1'` gateway flag convention). When disabled,
55
+ * `gate()` is a pure passthrough to the wrapped call — zero behaviour change.
56
+ * This is PR 1 of 3; priority-class shedding + degraded mode (PR 2) and
57
+ * observability + operator alert (PR 3) build on the counters exposed here.
58
+ *
59
+ * DETERMINISM / TESTABILITY
60
+ * -------------------------
61
+ * All time comes from an injectable `Clock` (`now()` + `sleep()`); there is no
62
+ * inline `Date.now()` / `setTimeout()` in the scheduling logic, so a fake clock
63
+ * makes the buckets fully deterministic under test. Token consumption happens
64
+ * in a synchronous critical section (no `await` between reading the clock and
65
+ * consuming), so concurrent admissions on a single-threaded runtime never
66
+ * double-spend a token.
67
+ */
68
+
69
+ import { createHash } from 'node:crypto'
70
+ import {
71
+ makeFloodWaitActiveError,
72
+ isFloodWaitActiveError,
73
+ } from './retry-api-call.js'
74
+
75
+ /** Injectable time source. Default binds to real wall clock + setTimeout. */
76
+ export interface Clock {
77
+ /** Milliseconds since epoch (monotonic-enough for bucket refill math). */
78
+ now(): number
79
+ /** Resolve after `ms` milliseconds. */
80
+ sleep(ms: number): Promise<void>
81
+ }
82
+
83
+ export const systemClock: Clock = {
84
+ now: () => Date.now(),
85
+ sleep: (ms: number) => new Promise<void>((r) => setTimeout(r, ms)),
86
+ }
87
+
88
+ /** Chat type as reported by Telegram, used to key the per-group bucket. */
89
+ export type ChatType = 'private' | 'group' | 'supergroup' | 'channel'
90
+
91
+ /**
92
+ * Priority class governing shedding + degraded mode (part3-design §2/§3):
93
+ * - `critical` — final reply chunks, approval / vault cards, error notices.
94
+ * Never shed; queued unbounded. Degraded mode: waits a short window,
95
+ * fails fast with a structured `FLOOD_WAIT_ACTIVE` for a long one.
96
+ * - `useful` — progress-card creation, worker handbacks, checklists,
97
+ * boot/config cards. Queued with a TTL; dropped (counted) when stale.
98
+ * - `cosmetic` — typing, reactions, all card EDITS, stream updates,
99
+ * heartbeats. Shed immediately (counted) when no token is free OR any
100
+ * flood window covering the scope is open.
101
+ *
102
+ * UNTAGGED default (L2, review PR #3106): an untagged NON-EDIT send defaults to
103
+ * `critical` — NON-droppable (queued unbounded, fail-fast only on a long
104
+ * window), NOT `useful`. Rationale: most call sites are untagged, and a `useful`
105
+ * default silently TTL-drops an untagged-but-important send under pressure
106
+ * (a resolved `undefined` a caller reads as "sent"). Conservative posture:
107
+ * nothing is droppable unless a call site OPTS IN by tagging `useful`/`cosmetic`.
108
+ * (Untagged EDITS keep coalescing — the latest payload always wins, never
109
+ * dropped — so their default is non-droppable already.)
110
+ */
111
+ export type PriorityClass = 'critical' | 'useful' | 'cosmetic'
112
+
113
+ /**
114
+ * Priority class an UNTAGGED non-edit send is admitted as. `critical` =
115
+ * non-droppable (L2). See the `PriorityClass` doc above.
116
+ */
117
+ export const UNTAGGED_SEND_CLASS: PriorityClass = 'critical'
118
+
119
+ /**
120
+ * Extra metadata a call site can attach so the gate can key the right buckets.
121
+ * All fields optional — a call with none still passes the global bucket. These
122
+ * mirror (a superset of) `RetryCallOpts` so the gate can wrap `robustApiCall`
123
+ * transparently.
124
+ */
125
+ export interface SendGateOpts {
126
+ /** Destination chat id — keys the per-chat and (for groups) per-group bucket. */
127
+ chat_id?: string
128
+ /** Chat type — a group/supergroup additionally passes the per-group bucket. */
129
+ chatType?: ChatType
130
+ /** For edits: the target message id. Enables the edit floor + coalescing. */
131
+ messageId?: number
132
+ /**
133
+ * For edits: the rendered payload (any stable-stringifiable value). Used for
134
+ * the no-op skip (hash equal to last sent → dropped) and last-write-wins
135
+ * coalescing. Only meaningful together with `messageId`.
136
+ */
137
+ editPayload?: unknown
138
+ /** Informational label (e.g. "editMessageText"); surfaced in stats/logs. */
139
+ verb?: string
140
+ /**
141
+ * Priority class for shedding + degraded mode (part3-design §2/§3). Untagged
142
+ * NON-EDIT calls default to `critical` (non-droppable — see
143
+ * `UNTAGGED_SEND_CLASS`); untagged edits coalesce (also non-droppable).
144
+ */
145
+ priorityClass?: PriorityClass
146
+ }
147
+
148
+ /** Per-bucket counters, snapshotted by `stats()`. */
149
+ export interface BucketCounters {
150
+ /** Calls that executed the wrapped fn. */
151
+ sent: number
152
+ /** Calls that had to wait on at least one bucket before executing. */
153
+ queued: number
154
+ /** Edits whose payload replaced a still-pending edit for the same message. */
155
+ coalesced: number
156
+ /** Calls dropped without hitting the API (no-op edit skip). */
157
+ dropped: number
158
+ /**
159
+ * Cosmetic calls shed under pressure — no token free OR a flood window open
160
+ * (part3-design §2). A shed resolves as `undefined`; the next send carries
161
+ * full state.
162
+ */
163
+ shed: number
164
+ /** Useful calls dropped because they exceeded their queue TTL (part3-design §2). */
165
+ expired: number
166
+ /**
167
+ * Critical calls that failed fast with a structured `FLOOD_WAIT_ACTIVE`
168
+ * because the open window exceeded the fail-fast ceiling (part3-design §3).
169
+ */
170
+ failedFast: number
171
+ }
172
+
173
+ export interface SendGateStats {
174
+ enabled: boolean
175
+ global: BucketCounters
176
+ /** Number of live per-message edit states (watch for the H2 leak). */
177
+ messageStates: number
178
+ /** Current token fill of each live bucket (for observability, PR 3). */
179
+ fill: {
180
+ global: number
181
+ perChat: Record<string, number>
182
+ perGroup: Record<string, number>
183
+ }
184
+ }
185
+
186
+ /** A flood window: a scope suppressed until `untilTs` (part3-design §7). */
187
+ export interface FloodWindow {
188
+ /** `global` | `chat:<id>` | `group:<id>` | `msg-edit:<id>`. */
189
+ scopeKey: string
190
+ /** Epoch ms until which the scope admits nothing. */
191
+ untilTs: number
192
+ }
193
+
194
+ /**
195
+ * Boot ramp for the global bucket (part3-design §7): start at a fraction of
196
+ * capacity for `durationMs` after boot so a burst of boot/config cards can't
197
+ * immediately saturate the freshly-full bucket.
198
+ */
199
+ export interface BootRamp {
200
+ /** Fraction of the global burst capacity during the ramp (0..1). Default 0.5. */
201
+ fraction?: number
202
+ /** Ramp duration in ms from gate construction. Default 10_000. */
203
+ durationMs?: number
204
+ }
205
+
206
+ export interface SendGateConfig {
207
+ /** Master switch. When false, `gate()` is a straight passthrough. */
208
+ enabled: boolean
209
+ /** Injected clock; defaults to the real system clock. */
210
+ clock?: Clock
211
+ /** Global bucket sustained rate (tokens/sec). Default 25. */
212
+ globalPerSec?: number
213
+ /**
214
+ * Global burst capacity (headroom under Telegram's ~30/s). Default 4.
215
+ * Worst-case sliding-window admissions = capacity + rate·T, so a full 1s
216
+ * window admits at most `globalBurst + globalPerSec` = 4 + 25 = 29 < 30 —
217
+ * a real margin under the ceiling (#3092 M2 residual: burst 5 hit exactly 30).
218
+ */
219
+ globalBurst?: number
220
+ /** Per-chat sustained rate (tokens/sec). Default 1. */
221
+ perChatPerSec?: number
222
+ /** Per-chat burst capacity. Default 3. */
223
+ perChatBurst?: number
224
+ /** Per-group sustained rate (tokens/min). Default 18. */
225
+ perGroupPerMin?: number
226
+ /** Per-group burst capacity (headroom under 20/min). Default 2. */
227
+ perGroupBurst?: number
228
+ /** Minimum ms between edits of the same message_id. Default 1500. */
229
+ editFloorMs?: number
230
+ /**
231
+ * Flood windows to re-open at construction (part3-design §7). PR 2 loads
232
+ * these from `flood-wait.json` BEFORE the first outbound call so a restart
233
+ * during an active ban does not immediately resend into the flood.
234
+ */
235
+ initialWindows?: FloodWindow[]
236
+ /** Global-bucket boot ramp (part3-design §7). Omit to start full. */
237
+ bootRamp?: BootRamp
238
+ /**
239
+ * TTL (ms) after which an idle per-message edit state is evicted. Default
240
+ * 60_000. Prevents unbounded growth of the perMessage map (#3092 H2).
241
+ */
242
+ messageStateTtlMs?: number
243
+ /**
244
+ * Hard cap on live per-message edit states; oldest-idle are evicted (LRU)
245
+ * once exceeded. Default 5_000.
246
+ */
247
+ maxMessageStates?: number
248
+ /**
249
+ * TTL (ms) a `useful` call may wait in the admission queue before it is
250
+ * dropped as stale (part3-design §2). Default 120_000 (~2 min).
251
+ */
252
+ usefulTtlMs?: number
253
+ /**
254
+ * Ceiling (ms) on an open flood window under which a `critical` send WAITS
255
+ * (single in-flight, jitter); above it, the send fails fast with a structured
256
+ * `FLOOD_WAIT_ACTIVE` carrying `untilTs` so the MCP reply path surfaces a real
257
+ * error instead of a long opaque block (part3-design §3). Default 60_000.
258
+ */
259
+ criticalFailFastMs?: number
260
+ /**
261
+ * Max jitter (ms) added before a `critical` send probes the API during an
262
+ * open (short) window, so serialized criticals don't thunder at the exact
263
+ * window edge. Default 250.
264
+ */
265
+ criticalJitterMaxMs?: number
266
+ /**
267
+ * Injectable 0..1 source for the critical jitter (tests pass `() => 0` for
268
+ * determinism). Default `Math.random`.
269
+ */
270
+ jitter?: () => number
271
+ /**
272
+ * Called whenever `openFloodWindow` opens / extends a window at RUNTIME (on a
273
+ * 429). Write-through persistence hook (part3-design §7) — the gateway wires
274
+ * this to `flood-windows.json` so a window survives a restart. NOT called for
275
+ * `initialWindows` applied at construction (those already came from disk).
276
+ */
277
+ onWindowOpen?: (scopeKey: string, untilTs: number) => void
278
+ }
279
+
280
+ /**
281
+ * Classic token bucket with an optional per-scope suppression window and an
282
+ * optional boot ramp. `tokens` refills continuously at `refillPerMs` up to
283
+ * `capacity` (or `rampCapacity` while inside the ramp). `consume` and
284
+ * `msUntilAvailable` both refill lazily against the supplied `now`, so there is
285
+ * no timer per bucket. A suppression window (`suppressUntil`) blocks all
286
+ * admission until its `untilTs`, regardless of token fill — a token bucket
287
+ * alone cannot represent "blocked until T" (part3-design §7).
288
+ */
289
+ class TokenBucket {
290
+ private tokens: number
291
+ private lastRefillMs: number
292
+ private suppressedUntilMs = 0
293
+ private readonly rampCapacity: number
294
+ private readonly rampUntilMs: number
295
+
296
+ constructor(
297
+ readonly capacity: number,
298
+ private readonly refillPerMs: number,
299
+ now: number,
300
+ ramp?: { capacity: number; untilMs: number },
301
+ ) {
302
+ this.rampCapacity = ramp?.capacity ?? capacity
303
+ this.rampUntilMs = ramp?.untilMs ?? 0
304
+ this.tokens = this.capAt(now)
305
+ this.lastRefillMs = now
306
+ }
307
+
308
+ /** Effective capacity ceiling at `now` (lower during the boot ramp). */
309
+ private capAt(now: number): number {
310
+ return now < this.rampUntilMs ? this.rampCapacity : this.capacity
311
+ }
312
+
313
+ private refill(now: number): void {
314
+ if (now <= this.lastRefillMs) return
315
+ const elapsed = now - this.lastRefillMs
316
+ this.tokens = Math.min(this.capAt(now), this.tokens + elapsed * this.refillPerMs)
317
+ this.lastRefillMs = now
318
+ }
319
+
320
+ /** Open/extend a suppression window (part3-design §7). */
321
+ suppressUntil(untilTs: number): void {
322
+ if (untilTs > this.suppressedUntilMs) this.suppressedUntilMs = untilTs
323
+ }
324
+
325
+ /** Remaining ms of an open suppression window (0 when none). */
326
+ windowRemainingMs(now: number): number {
327
+ return this.suppressedUntilMs > now ? this.suppressedUntilMs - now : 0
328
+ }
329
+
330
+ /**
331
+ * Milliseconds until at least one token is available AND no suppression
332
+ * window is open (0 if admissible now).
333
+ */
334
+ msUntilAvailable(now: number): number {
335
+ const windowWait = this.suppressedUntilMs > now ? this.suppressedUntilMs - now : 0
336
+ this.refill(now)
337
+ const tokenWait = this.tokens >= 1 ? 0 : Math.ceil((1 - this.tokens) / this.refillPerMs)
338
+ return Math.max(windowWait, tokenWait)
339
+ }
340
+
341
+ /** Consume one token. Caller must have checked availability in the SAME tick. */
342
+ consume(now: number): void {
343
+ this.refill(now)
344
+ this.tokens -= 1
345
+ }
346
+
347
+ fill(now: number): number {
348
+ this.refill(now)
349
+ return this.tokens
350
+ }
351
+ }
352
+
353
+ interface PendingEdit {
354
+ hash: string
355
+ fn: () => Promise<unknown>
356
+ promise: Promise<unknown>
357
+ resolve: (v: unknown) => void
358
+ reject: (e: unknown) => void
359
+ }
360
+
361
+ interface MessageEditState {
362
+ /** Wall time of the last edit SEND START for this message. */
363
+ lastSentMs: number
364
+ /** Hash of the last payload SUCCESSFULLY sent (for the no-op skip). */
365
+ lastHash: string | undefined
366
+ /** At most one queued coalesced edit per message (last-write-wins). */
367
+ pending: PendingEdit | null
368
+ /** True while a driver is actively sending / waiting for this message. */
369
+ running: boolean
370
+ /** Per-message flood suppression window (part3-design §7). */
371
+ suppressedUntilMs: number
372
+ }
373
+
374
+ function hashPayload(payload: unknown): string {
375
+ let s: string
376
+ if (typeof payload === 'string') {
377
+ s = payload
378
+ } else {
379
+ const j = stableStringify(payload)
380
+ // `stableStringify(undefined)` (and any value that JSON.stringify drops)
381
+ // returns undefined; hash a fixed sentinel so createHash never throws
382
+ // (#3092 L2 — a caller may set editPayload: undefined alongside messageId).
383
+ s = j === undefined ? 'undefined' : j
384
+ }
385
+ return createHash('sha256').update(s).digest('hex')
386
+ }
387
+
388
+ /** Deterministic JSON stringify (sorted keys) so equal payloads hash equal. */
389
+ function stableStringify(value: unknown): string | undefined {
390
+ return JSON.stringify(value, (_k, v) => {
391
+ if (v && typeof v === 'object' && !Array.isArray(v)) {
392
+ return Object.keys(v as Record<string, unknown>)
393
+ .sort()
394
+ .reduce<Record<string, unknown>>((acc, k) => {
395
+ acc[k] = (v as Record<string, unknown>)[k]
396
+ return acc
397
+ }, {})
398
+ }
399
+ return v
400
+ })
401
+ }
402
+
403
+ export interface SendGate {
404
+ /**
405
+ * Run `fn` (a single Bot API call) through the gate. Returns whatever `fn`
406
+ * resolves to. For a dropped no-op edit, resolves to `undefined` without
407
+ * calling `fn`. For a coalesced edit, resolves when the coalesced send
408
+ * completes (with that send's result).
409
+ */
410
+ gate<T>(fn: () => Promise<T>, opts?: SendGateOpts): Promise<T>
411
+ /**
412
+ * Open/extend a flood-suppression window on a scope (part3-design §7). Used
413
+ * by PR 2 on a 429 and at boot from the persisted `flood-wait.json`.
414
+ */
415
+ openFloodWindow(scopeKey: string, untilTs: number): void
416
+ /** Snapshot of counters + current bucket fill. */
417
+ stats(): SendGateStats
418
+ }
419
+
420
+ const GROUP_TYPES = new Set<ChatType>(['group', 'supergroup'])
421
+
422
+ export function createSendGate(config: SendGateConfig): SendGate {
423
+ const enabled = config.enabled
424
+ const clock = config.clock ?? systemClock
425
+ const globalPerSec = config.globalPerSec ?? 25
426
+ const globalBurst = config.globalBurst ?? 4
427
+ const perChatPerSec = config.perChatPerSec ?? 1
428
+ const perChatBurst = config.perChatBurst ?? 3
429
+ const perGroupPerMin = config.perGroupPerMin ?? 18
430
+ const perGroupBurst = config.perGroupBurst ?? 2
431
+ const editFloorMs = config.editFloorMs ?? 1500
432
+ const messageStateTtlMs = config.messageStateTtlMs ?? 60_000
433
+ const maxMessageStates = config.maxMessageStates ?? 5_000
434
+ const usefulTtlMs = config.usefulTtlMs ?? 120_000
435
+ const criticalFailFastMs = config.criticalFailFastMs ?? 60_000
436
+ const criticalJitterMaxMs = config.criticalJitterMaxMs ?? 250
437
+ const jitter = config.jitter ?? Math.random
438
+ const onWindowOpen = config.onWindowOpen
439
+
440
+ const counters: BucketCounters = {
441
+ sent: 0,
442
+ queued: 0,
443
+ coalesced: 0,
444
+ dropped: 0,
445
+ shed: 0,
446
+ expired: 0,
447
+ failedFast: 0,
448
+ }
449
+
450
+ const bootStart = clock.now()
451
+ const globalRamp = config.bootRamp
452
+ ? {
453
+ capacity: Math.max(1, Math.floor(globalBurst * (config.bootRamp.fraction ?? 0.5))),
454
+ untilMs: bootStart + (config.bootRamp.durationMs ?? 10_000),
455
+ }
456
+ : undefined
457
+
458
+ const globalBucket = new TokenBucket(globalBurst, globalPerSec / 1000, bootStart, globalRamp)
459
+ const perChat = new Map<string, TokenBucket>()
460
+ const perGroup = new Map<string, TokenBucket>()
461
+ // H1 (review PR #3106): keyed by `${chat_id}:${messageId}`, NOT messageId
462
+ // alone. Telegram `message_id` is per-chat, not globally unique — two chats
463
+ // routinely both hold a low-numbered card (id 3, 100, …). Keying on the id
464
+ // alone collides cross-chat: chat B's edit overwrites chat A's pending edit
465
+ // (A's real update silently lost), and the edit floor / msg-edit suppression
466
+ // window of one chat sheds an unrelated card in another.
467
+ const perMessage = new Map<string, MessageEditState>()
468
+ let lastSweepMs = bootStart
469
+
470
+ function chatBucket(chatId: string): TokenBucket {
471
+ let b = perChat.get(chatId)
472
+ if (!b) {
473
+ b = new TokenBucket(perChatBurst, perChatPerSec / 1000, clock.now())
474
+ perChat.set(chatId, b)
475
+ }
476
+ return b
477
+ }
478
+
479
+ function groupBucket(chatId: string): TokenBucket {
480
+ let b = perGroup.get(chatId)
481
+ if (!b) {
482
+ b = new TokenBucket(perGroupBurst, perGroupPerMin / 60000, clock.now())
483
+ perGroup.set(chatId, b)
484
+ }
485
+ return b
486
+ }
487
+
488
+ /**
489
+ * Per-message state key: `${chat_id}:${messageId}` (H1). A missing chat_id
490
+ * degrades to `:${messageId}` — still unique per call site, and edits always
491
+ * carry a chat_id in production. This is ALSO the `msg-edit:` scope suffix, so
492
+ * the boot-reloaded scoped windows and the runtime edit floor agree.
493
+ */
494
+ function messageKey(chatId: string | undefined, messageId: number): string {
495
+ return `${chatId ?? ''}:${messageId}`
496
+ }
497
+
498
+ function messageState(key: string): MessageEditState {
499
+ let state = perMessage.get(key)
500
+ if (!state) {
501
+ state = {
502
+ lastSentMs: Number.NEGATIVE_INFINITY,
503
+ lastHash: undefined,
504
+ pending: null,
505
+ running: false,
506
+ suppressedUntilMs: 0,
507
+ }
508
+ perMessage.set(key, state)
509
+ }
510
+ return state
511
+ }
512
+
513
+ /**
514
+ * Apply any flood windows injected at construction (§7 boot load) WITHOUT
515
+ * re-persisting them (they already came from disk). Applied before the first
516
+ * outbound call so a boot mid-ban never resends into an open window.
517
+ */
518
+ for (const w of config.initialWindows ?? []) applyWindow(w.scopeKey, w.untilTs, false)
519
+
520
+ function bucketsFor(opts?: SendGateOpts): TokenBucket[] {
521
+ const buckets: TokenBucket[] = [globalBucket]
522
+ if (opts?.chat_id) {
523
+ buckets.push(chatBucket(opts.chat_id))
524
+ if (opts.chatType && GROUP_TYPES.has(opts.chatType)) {
525
+ buckets.push(groupBucket(opts.chat_id))
526
+ }
527
+ }
528
+ return buckets
529
+ }
530
+
531
+ /**
532
+ * Open/extend a flood window on a scope. `global` suppresses the global
533
+ * bucket; `chat:<id>` / `group:<id>` the per-chat / per-group buckets;
534
+ * `msg-edit:<id>` the per-message edit floor. Idempotent + monotonic
535
+ * (only ever extends). Persists write-through via `onWindowOpen` (part3-design
536
+ * §7) so the window survives a restart.
537
+ */
538
+ function openFloodWindow(scopeKey: string, untilTs: number): void {
539
+ // M3 (review PR #3106): flag-OFF is a PURE no-op. The gateway wires
540
+ // `onFloodWait` → `openFloodWindow('global', …)` unconditionally, so without
541
+ // this guard a 429 would mutate in-memory suppression AND write a new
542
+ // `flood-windows.json` even while the feature is disabled — a real fs side
543
+ // effect and behaviour change before the flag is ever enabled. When
544
+ // disabled the gate must open no window and write no file (the separate
545
+ // #2923 `flood-wait.json` recorder is unaffected — it is not wired here).
546
+ if (!enabled) return
547
+ applyWindow(scopeKey, untilTs, true)
548
+ }
549
+
550
+ /**
551
+ * Open all scope windows implied by a call's `opts` at `untilTs` — global,
552
+ * plus per-chat / per-group / per-message where keyed. Called when a wrapped
553
+ * send surfaces a `FLOOD_WAIT_ACTIVE` (a real 429 or a pre-call
554
+ * short-circuit) so the ban is remembered at the finest scope we know
555
+ * (part3-design §7). Global is always opened as the conservative floor.
556
+ */
557
+ function openScopedWindowsForOpts(opts: SendGateOpts | undefined, untilTs: number): void {
558
+ applyWindow('global', untilTs, true)
559
+ if (opts?.chat_id) {
560
+ applyWindow(`chat:${opts.chat_id}`, untilTs, true)
561
+ if (opts.chatType && GROUP_TYPES.has(opts.chatType)) {
562
+ applyWindow(`group:${opts.chat_id}`, untilTs, true)
563
+ }
564
+ }
565
+ if (opts?.messageId != null) {
566
+ // H1: msg-edit scope keyed by chat_id+messageId so a 429 on one chat's
567
+ // card never sheds another chat's same-id card.
568
+ applyWindow(`msg-edit:${messageKey(opts.chat_id, opts.messageId)}`, untilTs, true)
569
+ }
570
+ }
571
+
572
+ /** Core window application; `persist` gates the write-through hook. */
573
+ function applyWindow(scopeKey: string, untilTs: number, persist: boolean): void {
574
+ if (persist && onWindowOpen) {
575
+ try {
576
+ onWindowOpen(scopeKey, untilTs)
577
+ } catch {
578
+ /* best-effort — persistence must not break the send path */
579
+ }
580
+ }
581
+ if (scopeKey === 'global') {
582
+ globalBucket.suppressUntil(untilTs)
583
+ } else if (scopeKey.startsWith('chat:')) {
584
+ chatBucket(scopeKey.slice('chat:'.length)).suppressUntil(untilTs)
585
+ } else if (scopeKey.startsWith('group:')) {
586
+ groupBucket(scopeKey.slice('group:'.length)).suppressUntil(untilTs)
587
+ } else if (scopeKey.startsWith('msg-edit:')) {
588
+ // H1: the suffix is the composite `${chat_id}:${messageId}` state key.
589
+ const key = scopeKey.slice('msg-edit:'.length)
590
+ if (key) {
591
+ // N2: creating a per-message edit state here must go through the same
592
+ // eviction accounting (maybeEvict / LRU cap) as normal state creation,
593
+ // so opening many per-message flood windows can't grow perMessage past
594
+ // the cap.
595
+ const now = clock.now()
596
+ const existing = perMessage.get(key)
597
+ maybeEvict(now, existing === undefined)
598
+ const st = existing ?? messageState(key)
599
+ if (untilTs > st.suppressedUntilMs) st.suppressedUntilMs = untilTs
600
+ }
601
+ }
602
+ }
603
+
604
+ function isEvictable(s: MessageEditState, now: number): boolean {
605
+ return !s.pending && !s.running && s.suppressedUntilMs <= now
606
+ }
607
+
608
+ /**
609
+ * Bound the perMessage map (#3092 H2). Two mechanisms: a TTL sweep of idle
610
+ * states older than `messageStateTtlMs`, and a hard LRU cap. An entry with a
611
+ * pending/running send or an open suppression window is never evicted.
612
+ */
613
+ function maybeEvict(now: number, reserving: boolean): void {
614
+ // Leave room for the entry about to be created so the map never settles
615
+ // ABOVE the cap (evicting AFTER the insert would perpetually sit at cap+1).
616
+ const target = Math.max(0, maxMessageStates - (reserving ? 1 : 0))
617
+ if (perMessage.size > target) {
618
+ const evictable = [...perMessage.entries()]
619
+ .filter(([, s]) => isEvictable(s, now))
620
+ .sort((a, b) => a[1].lastSentMs - b[1].lastSentMs)
621
+ let over = perMessage.size - target
622
+ for (const [k] of evictable) {
623
+ if (over <= 0) break
624
+ perMessage.delete(k)
625
+ over--
626
+ }
627
+ }
628
+ if (now - lastSweepMs >= messageStateTtlMs) {
629
+ lastSweepMs = now
630
+ for (const [k, s] of perMessage) {
631
+ if (isEvictable(s, now) && now - s.lastSentMs > messageStateTtlMs) {
632
+ perMessage.delete(k)
633
+ }
634
+ }
635
+ }
636
+ }
637
+
638
+ /**
639
+ * Wait until every bucket has a token (and no scope window is open), then
640
+ * consume one from each. The check-and-consume block runs synchronously (no
641
+ * `await` after reading the clock), so concurrent admissions never
642
+ * double-spend. Loops because a competing admission may drain a bucket during
643
+ * our sleep. When `deadline` is set (a `useful` call's TTL), returns `false`
644
+ * without consuming if admission cannot happen before it — the caller drops
645
+ * the call as stale (part3-design §2).
646
+ */
647
+ async function admitLoop(buckets: TokenBucket[], deadline?: number): Promise<boolean> {
648
+ let counted = false
649
+ for (;;) {
650
+ const now = clock.now()
651
+ let wait = 0
652
+ for (const b of buckets) wait = Math.max(wait, b.msUntilAvailable(now))
653
+ if (wait <= 0) {
654
+ for (const b of buckets) b.consume(now)
655
+ return true
656
+ }
657
+ if (deadline !== undefined && now + wait > deadline) return false
658
+ if (!counted) {
659
+ counters.queued++
660
+ counted = true
661
+ }
662
+ await clock.sleep(deadline !== undefined ? Math.min(wait, deadline - now) : wait)
663
+ }
664
+ }
665
+
666
+ /** Back-compat unbounded admission (used by the per-message edit driver). */
667
+ function admit(buckets: TokenBucket[]): Promise<boolean> {
668
+ return admitLoop(buckets)
669
+ }
670
+
671
+ // Serializes `critical` sends that must probe the API during an open (short)
672
+ // flood window: only ONE in-flight at a time so a burst of criticals doesn't
673
+ // thunder at the window edge (part3-design §3, "single in-flight, jitter").
674
+ let criticalTail: Promise<void> = Promise.resolve()
675
+ function criticalSerialize<R>(job: () => Promise<R>): Promise<R> {
676
+ const prev = criticalTail
677
+ let release!: () => void
678
+ criticalTail = new Promise<void>((r) => {
679
+ release = r
680
+ })
681
+ return (async () => {
682
+ await prev
683
+ try {
684
+ return await job()
685
+ } finally {
686
+ release()
687
+ }
688
+ })()
689
+ }
690
+
691
+ type AdmitOutcome =
692
+ | { result: 'ok' }
693
+ | { result: 'shed' }
694
+ | { result: 'expired' }
695
+ | { result: 'failfast'; untilTs: number }
696
+
697
+ /**
698
+ * Priority-aware admission (part3-design §2/§3):
699
+ * - cosmetic — shed immediately when any bucket has a wait (no token OR a
700
+ * window open); otherwise consume and go.
701
+ * - useful — queue with a TTL; drop as stale past the deadline.
702
+ * - critical — never shed. If a window covering the scope exceeds the
703
+ * fail-fast ceiling, fail fast with `untilTs` (caller throws a structured
704
+ * FLOOD_WAIT_ACTIVE). A short window → wait single-in-flight with jitter.
705
+ * No window → queue unbounded.
706
+ */
707
+ async function admitPriority(
708
+ buckets: TokenBucket[],
709
+ priority: PriorityClass,
710
+ ): Promise<AdmitOutcome> {
711
+ const now = clock.now()
712
+ if (priority === 'cosmetic') {
713
+ let wait = 0
714
+ for (const b of buckets) wait = Math.max(wait, b.msUntilAvailable(now))
715
+ if (wait > 0) return { result: 'shed' }
716
+ for (const b of buckets) b.consume(now)
717
+ return { result: 'ok' }
718
+ }
719
+ if (priority === 'critical') {
720
+ let remaining = 0
721
+ for (const b of buckets) remaining = Math.max(remaining, b.windowRemainingMs(now))
722
+ if (remaining > criticalFailFastMs) {
723
+ return { result: 'failfast', untilTs: now + remaining }
724
+ }
725
+ if (remaining > 0) {
726
+ // Degraded but short window: serialize + jitter, then wait it out —
727
+ // re-checking the ceiling EACH loop (admitCriticalLoop) so a window
728
+ // extended past `criticalFailFastMs` while we wait converts to fail-fast
729
+ // instead of blocking the reply path unbounded (M2, review PR #3106).
730
+ return await criticalSerialize(async () => {
731
+ const j = Math.floor(jitter() * criticalJitterMaxMs)
732
+ if (j > 0) await clock.sleep(j)
733
+ return await admitCriticalLoop(buckets)
734
+ })
735
+ }
736
+ return await admitCriticalLoop(buckets)
737
+ }
738
+ // useful (default): TTL-bounded queue.
739
+ const admitted = await admitLoop(buckets, now + usefulTtlMs)
740
+ return admitted ? { result: 'ok' } : { result: 'expired' }
741
+ }
742
+
743
+ /**
744
+ * Unbounded critical admission that RE-EVALUATES the covering window on every
745
+ * iteration (M2, review PR #3106). If a competing 429 extends the window past
746
+ * the fail-fast ceiling while a critical waits out a short window, this stops
747
+ * blocking and returns `failfast` (the caller throws a structured
748
+ * `FLOOD_WAIT_ACTIVE`) rather than blocking the MCP reply path for the whole
749
+ * extended ban — exactly the "opaque long block" part3-design §3 forbids.
750
+ */
751
+ async function admitCriticalLoop(buckets: TokenBucket[]): Promise<AdmitOutcome> {
752
+ let counted = false
753
+ for (;;) {
754
+ const now = clock.now()
755
+ let remaining = 0
756
+ let wait = 0
757
+ for (const b of buckets) {
758
+ remaining = Math.max(remaining, b.windowRemainingMs(now))
759
+ wait = Math.max(wait, b.msUntilAvailable(now))
760
+ }
761
+ if (remaining > criticalFailFastMs) return { result: 'failfast', untilTs: now + remaining }
762
+ if (wait <= 0) {
763
+ for (const b of buckets) b.consume(now)
764
+ return { result: 'ok' }
765
+ }
766
+ if (!counted) {
767
+ counters.queued++
768
+ counted = true
769
+ }
770
+ await clock.sleep(wait)
771
+ }
772
+ }
773
+
774
+ /**
775
+ * The single serialized driver for one message. Only ONE driver runs per
776
+ * message at a time (`state.running`); every edit that arrives while it runs
777
+ * — whether it is sleeping on the floor, waiting on a bucket, or mid-flight
778
+ * on the network (e.g. a 429 retry_after sleep INSIDE `fn`) — coalesces into
779
+ * `state.pending` rather than firing a second overlapping send. This closes
780
+ * the same-tick double-send (M3) and the in-flight parallelism (M4).
781
+ */
782
+ async function drive(state: MessageEditState, opts: SendGateOpts): Promise<void> {
783
+ state.running = true
784
+ try {
785
+ while (state.pending) {
786
+ const now = clock.now()
787
+ const readyAt = Math.max(state.lastSentMs + editFloorMs, state.suppressedUntilMs)
788
+ const waitMs = readyAt - now
789
+ if (waitMs > 0) {
790
+ // Still inside the floor / an open window — sleep, then re-read
791
+ // state.pending (a newer edit may have replaced it in the meantime).
792
+ await clock.sleep(waitMs)
793
+ continue
794
+ }
795
+
796
+ // Floor cleared: take the current pending edit. A newer edit arriving
797
+ // from here on lands in a FRESH state.pending and is handled next loop.
798
+ const p = state.pending
799
+ state.pending = null
800
+
801
+ // Re-check the no-op skip against the last SUCCESSFULLY-sent payload —
802
+ // the latest coalesced payload may have reverted to what's on screen.
803
+ if (p.hash === state.lastHash) {
804
+ counters.dropped++
805
+ p.resolve(undefined)
806
+ continue
807
+ }
808
+
809
+ await admit(bucketsFor(opts))
810
+ // Reserve the send-start time BEFORE awaiting the network so the floor
811
+ // is measured from send start (matches the per-message serialization).
812
+ state.lastSentMs = clock.now()
813
+ try {
814
+ // N3 (liveness): this awaits `p.fn()` with no watchdog. A `fn` that
815
+ // NEVER settles would keep `state.running` true forever, making the
816
+ // state unevictable and hanging every coalesced caller. We rely on
817
+ // the fact that the only production caller is `robustApiCall`, whose
818
+ // own request/retry timeouts bound every send — so `p.fn()` is
819
+ // guaranteed to settle. No separate watchdog is needed here.
820
+ const res = await p.fn()
821
+ // M1: only record the payload as on-screen AFTER a successful send,
822
+ // so a FAILED edit can be retried with the same payload (not dropped
823
+ // as a phantom no-op).
824
+ state.lastHash = p.hash
825
+ counters.sent++
826
+ p.resolve(res)
827
+ } catch (err) {
828
+ // A 429 surfaced from the edit send opens the flood windows at this
829
+ // scope so later cosmetic edits shed and the window persists (§3/§7).
830
+ if (isFloodWaitActiveError(err)) {
831
+ openScopedWindowsForOpts(opts, err.untilTs)
832
+ }
833
+ // H1: reject THIS edit's own promise (the closure-local `p`), never
834
+ // state.pending — a distinct newer edit that arrived during the send
835
+ // owns state.pending and must survive to be sent on the next loop.
836
+ p.reject(err)
837
+ }
838
+ }
839
+ } finally {
840
+ state.running = false
841
+ }
842
+ }
843
+
844
+ function handleEdit<T>(fn: () => Promise<T>, opts: SendGateOpts): Promise<T> {
845
+ const messageId = opts.messageId as number
846
+ const key = messageKey(opts.chat_id, messageId)
847
+ const hash = hashPayload(opts.editPayload)
848
+ const now = clock.now()
849
+ const existing = perMessage.get(key)
850
+ maybeEvict(now, existing === undefined)
851
+ const state = existing ?? messageState(key)
852
+
853
+ // Cosmetic edits (part3-design §2) shed under pressure — a flood window
854
+ // covering the scope is open, or a bucket has no token free right now. A
855
+ // dropped edit costs nothing: the next edit carries the full state. Only an
856
+ // EXPLICITLY-cosmetic edit sheds; untagged edits keep PR 1's coalescing
857
+ // behaviour (default = useful), so this never changes an untagged caller.
858
+ if ((opts.priorityClass ?? 'useful') === 'cosmetic') {
859
+ let wait = 0
860
+ for (const b of bucketsFor(opts)) wait = Math.max(wait, b.msUntilAvailable(now))
861
+ const msgWait = state.suppressedUntilMs > now ? state.suppressedUntilMs - now : 0
862
+ if (wait > 0 || msgWait > 0) {
863
+ counters.shed++
864
+ return Promise.resolve(undefined as unknown as T)
865
+ }
866
+ }
867
+
868
+ // N1: the coalesce (last-write-wins) check runs BEFORE the no-op skip. When
869
+ // a distinct edit is already queued, the newest payload always replaces it —
870
+ // even if that payload reverts to the on-screen one (hash === lastHash).
871
+ // Otherwise a revert-to-on-screen edit would be dropped as a no-op while the
872
+ // stale queued edit still rendered, violating last-write-wins. When the
873
+ // driver dequeues, its own no-op re-check (`p.hash === state.lastHash`) drops
874
+ // a coalesced revert so nothing needless hits the API. All callers share the
875
+ // single pending promise, which resolves with the coalesced send's result.
876
+ if (state.pending) {
877
+ if (state.pending.hash !== hash) {
878
+ counters.coalesced++
879
+ state.pending.hash = hash
880
+ state.pending.fn = fn as () => Promise<unknown>
881
+ }
882
+ return state.pending.promise as Promise<T>
883
+ }
884
+
885
+ // No edit queued → a repeat of the last payload we actually sent is a plain
886
+ // no-op skip: drop it before the API.
887
+ if (hash === state.lastHash) {
888
+ counters.dropped++
889
+ return Promise.resolve(undefined as unknown as T)
890
+ }
891
+
892
+ // Otherwise create a fresh pending edit. If no driver is currently running
893
+ // for this message, start one; if one IS running (mid-flight send, floor
894
+ // sleep, or bucket wait), it will pick this up on its next loop — this is
895
+ // what serializes sends per message (M3/M4).
896
+ let resolve!: (v: unknown) => void
897
+ let reject!: (e: unknown) => void
898
+ const promise = new Promise<unknown>((res, rej) => {
899
+ resolve = res
900
+ reject = rej
901
+ })
902
+ const pending: PendingEdit = {
903
+ hash,
904
+ fn: fn as () => Promise<unknown>,
905
+ promise,
906
+ resolve,
907
+ reject,
908
+ }
909
+ state.pending = pending
910
+ if (!state.running) void drive(state, opts)
911
+ return promise as Promise<T>
912
+ }
913
+
914
+ async function gate<T>(fn: () => Promise<T>, opts?: SendGateOpts): Promise<T> {
915
+ // Flag OFF → pure passthrough, zero behaviour change.
916
+ if (!enabled) return fn()
917
+
918
+ // Edit path (floor + coalescing + no-op skip) only when we can key it.
919
+ if (opts && opts.messageId != null && 'editPayload' in opts) {
920
+ return handleEdit(fn, opts)
921
+ }
922
+
923
+ // Priority-aware admission (part3-design §2/§3). Untagged → critical
924
+ // (non-droppable — L2, review PR #3106); tag `useful`/`cosmetic` to opt into
925
+ // shedding.
926
+ const priority = opts?.priorityClass ?? UNTAGGED_SEND_CLASS
927
+ const outcome = await admitPriority(bucketsFor(opts), priority)
928
+ if (outcome.result === 'shed') {
929
+ counters.shed++
930
+ return undefined as unknown as T
931
+ }
932
+ if (outcome.result === 'expired') {
933
+ counters.expired++
934
+ return undefined as unknown as T
935
+ }
936
+ if (outcome.result === 'failfast') {
937
+ counters.failedFast++
938
+ // Compose #3094's structured error so the MCP reply path surfaces a real
939
+ // flood_wait (untilTs / retry_after) instead of an opaque long block.
940
+ const retryAfterSec = Math.ceil((outcome.untilTs - clock.now()) / 1000)
941
+ openScopedWindowsForOpts(opts, outcome.untilTs)
942
+ throw makeFloodWaitActiveError(retryAfterSec, outcome.untilTs, null)
943
+ }
944
+
945
+ try {
946
+ // N4: count `sent` only AFTER a successful send, mirroring the edit path.
947
+ const res = await fn()
948
+ counters.sent++
949
+ return res
950
+ } catch (err) {
951
+ // A 429 surfaced from a non-edit send opens the scope's flood windows so
952
+ // subsequent cosmetic traffic sheds and the window persists (§3/§7).
953
+ if (isFloodWaitActiveError(err)) openScopedWindowsForOpts(opts, err.untilTs)
954
+ throw err
955
+ }
956
+ }
957
+
958
+ function stats(): SendGateStats {
959
+ const now = clock.now()
960
+ const perChatFill: Record<string, number> = {}
961
+ for (const [k, b] of perChat) perChatFill[k] = b.fill(now)
962
+ const perGroupFill: Record<string, number> = {}
963
+ for (const [k, b] of perGroup) perGroupFill[k] = b.fill(now)
964
+ return {
965
+ enabled,
966
+ global: { ...counters },
967
+ messageStates: perMessage.size,
968
+ fill: {
969
+ global: globalBucket.fill(now),
970
+ perChat: perChatFill,
971
+ perGroup: perGroupFill,
972
+ },
973
+ }
974
+ }
975
+
976
+ return { gate, openFloodWindow, stats }
977
+ }
978
+
979
+ /** Read the feature flag using the standard `SWITCHROOM_*=== '1'` convention. */
980
+ export function sendGateEnabledFromEnv(env: NodeJS.ProcessEnv = process.env): boolean {
981
+ return env.SWITCHROOM_TELEGRAM_SEND_GATE === '1'
982
+ }