switchroom 0.18.14 → 0.18.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent-scheduler/index.js +3 -0
  2. package/dist/auth-broker/index.js +473 -49
  3. package/dist/cli/notion-write-pretool.mjs +3 -0
  4. package/dist/cli/switchroom.js +1200 -1067
  5. package/dist/host-control/main.js +56 -51
  6. package/dist/vault/approvals/kernel-server.js +19 -12
  7. package/dist/vault/broker/server.js +675 -668
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/dist/bridge/bridge.js +21 -0
  11. package/telegram-plugin/dist/gateway/gateway.js +531 -259
  12. package/telegram-plugin/dist/server.js +22 -1
  13. package/telegram-plugin/draft-stream.ts +78 -3
  14. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
  15. package/telegram-plugin/gateway/effort-command.ts +9 -7
  16. package/telegram-plugin/gateway/gateway.ts +310 -219
  17. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  18. package/telegram-plugin/gateway/model-command.ts +96 -18
  19. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  20. package/telegram-plugin/gateway/session-model-file.ts +38 -172
  21. package/telegram-plugin/litellm-local-notice.ts +189 -0
  22. package/telegram-plugin/model-unavailable.ts +214 -0
  23. package/telegram-plugin/quota-watch.ts +16 -4
  24. package/telegram-plugin/runtime-metrics.ts +47 -0
  25. package/telegram-plugin/send-gate-degraded.test.ts +9 -7
  26. package/telegram-plugin/send-gate.ts +34 -4
  27. package/telegram-plugin/session-tail.ts +14 -2
  28. package/telegram-plugin/stream-controller.ts +143 -20
  29. package/telegram-plugin/stream-reply-handler.ts +12 -2
  30. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  31. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  32. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  33. package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
  34. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  35. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  36. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  37. package/telegram-plugin/tests/model-command.test.ts +84 -1
  38. package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
  39. package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
  40. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  41. package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
  42. package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
  43. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  44. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  45. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  46. package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
  47. package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
  48. package/telegram-plugin/throttle-tier.ts +98 -1
  49. package/telegram-plugin/worker-activity-feed.ts +83 -8
@@ -0,0 +1,189 @@
1
+ /**
2
+ * litellm-local-notice.ts — debounced operator notice for LiteLLM-proxy-local
3
+ * 429s (pure module: state machine + text + config parsing, no IPC, no bot,
4
+ * no clock except the injected `now`).
5
+ *
6
+ * When an agent routes through the LiteLLM gateway and trips the proxy's OWN
7
+ * `tpm_limit`/`rpm_limit` limiter, `classify429Detail` (throttle-tier.ts)
8
+ * classifies the terminal 429 `litellm-local` and the gateway takes the calm
9
+ * path — no broker mark, no failover, no throttle tier (the request never
10
+ * reached Anthropic, so the condition says nothing about the account). But
11
+ * pre-notice the user-facing surface was the GENERIC "🚦 Rate limited" card,
12
+ * which reads like an Anthropic problem. This module owns the honest
13
+ * replacement:
14
+ *
15
+ * - ONE calm notice per agent per cooldown window (default 15 min,
16
+ * operator-tunable via channels.telegram.litellm_notice.window_ms in
17
+ * switchroom.yaml, projected into access.json by scaffold).
18
+ * - Further litellm-local 429s inside the window are counted SILENTLY;
19
+ * the first notice after the window expires carries "throttled N more
20
+ * times since the last notice".
21
+ * - A notice only ever fires on an actual throttle event — a quiet agent
22
+ * posts nothing (the state machine is evaluate-on-event, no timers).
23
+ *
24
+ * The copy must make clear this is the fleet token limiter (LiteLLM
25
+ * `tpm_limit`/`rpm_limit`), NOT an Anthropic account limit, and that the
26
+ * turn retries — no action needed.
27
+ *
28
+ * Side-effect sequencing lives in gateway/litellm-local-notice-wiring.ts.
29
+ * This module MUST NOT change classification, failover, or quota-ledger
30
+ * behavior — it only renders and debounces the notice.
31
+ */
32
+
33
+ import { escapeMarkdown } from './card-format.js'
34
+ import type { RateLimit429Classification } from './throttle-tier.js'
35
+ import type { RuntimeMetricEvent } from './runtime-metrics.js'
36
+
37
+ // ─── Cooldown window config ──────────────────────────────────────────────────
38
+
39
+ /**
40
+ * Default per-agent notice cooldown. 15 minutes: long enough that a sustained
41
+ * burst produces a handful of notices per hour at most, short enough that the
42
+ * operator still sees the limiter working while it's working.
43
+ */
44
+ export const LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT = 15 * 60_000
45
+
46
+ /**
47
+ * Resolve the cooldown window from the raw access-file value
48
+ * (`access.litellmNoticeWindowMs`, projected by scaffold from
49
+ * channels.telegram.litellm_notice.window_ms). Any non-finite, non-positive,
50
+ * or non-number value falls back to the default — an operator typo must
51
+ * never produce a zero-width window (notice storm) or a NaN comparison
52
+ * (notices never fire again).
53
+ */
54
+ export function parseLitellmNoticeWindowMs(raw: unknown): number {
55
+ if (typeof raw !== 'number') return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
56
+ if (!Number.isFinite(raw) || raw <= 0) return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
57
+ return raw
58
+ }
59
+
60
+ // ─── Per-agent cooldown state machine ────────────────────────────────────────
61
+
62
+ export interface LitellmLocalNoticeState {
63
+ /** agent → unix ms of the last notice sent for it. */
64
+ lastSentAtMsByAgent: Record<string, number>
65
+ /** agent → litellm-local 429s counted silently since the last notice. */
66
+ suppressedCountByAgent: Record<string, number>
67
+ }
68
+
69
+ export function initialLitellmLocalNoticeState(): LitellmLocalNoticeState {
70
+ return { lastSentAtMsByAgent: {}, suppressedCountByAgent: {} }
71
+ }
72
+
73
+ export interface LitellmLocalNoticeVerdict {
74
+ send: boolean
75
+ /** Silent throttle events since the previous notice (0 on the first). Only
76
+ * meaningful when `send` is true — the renderer adds the "N more times"
77
+ * line when > 0. */
78
+ suppressedSinceLastNotice: number
79
+ next: LitellmLocalNoticeState
80
+ }
81
+
82
+ /**
83
+ * Evaluate one litellm-local 429 for `agent` at `now`.
84
+ *
85
+ * - First event ever (or ≥ windowMs since the last notice) → send; the
86
+ * verdict carries the silently-counted events since the previous notice
87
+ * and the counter resets.
88
+ * - Inside the window → suppress; the counter increments.
89
+ *
90
+ * Boundary is INCLUSIVE on expiry (elapsed === windowMs sends), matching
91
+ * `evaluateThrottleNotice` in throttle-tier.ts. Pure: returns the next
92
+ * state, never mutates `prev`.
93
+ */
94
+ export function evaluateLitellmLocalNotice(
95
+ prev: LitellmLocalNoticeState,
96
+ agent: string,
97
+ now: number,
98
+ windowMs: number = LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT,
99
+ ): LitellmLocalNoticeVerdict {
100
+ const last = prev.lastSentAtMsByAgent[agent]
101
+ if (last == null || now - last >= windowMs) {
102
+ return {
103
+ send: true,
104
+ suppressedSinceLastNotice: prev.suppressedCountByAgent[agent] ?? 0,
105
+ next: {
106
+ lastSentAtMsByAgent: { ...prev.lastSentAtMsByAgent, [agent]: now },
107
+ suppressedCountByAgent: { ...prev.suppressedCountByAgent, [agent]: 0 },
108
+ },
109
+ }
110
+ }
111
+ return {
112
+ send: false,
113
+ suppressedSinceLastNotice: 0,
114
+ next: {
115
+ lastSentAtMsByAgent: prev.lastSentAtMsByAgent,
116
+ suppressedCountByAgent: {
117
+ ...prev.suppressedCountByAgent,
118
+ [agent]: (prev.suppressedCountByAgent[agent] ?? 0) + 1,
119
+ },
120
+ },
121
+ }
122
+ }
123
+
124
+ // ─── Notice rendering ────────────────────────────────────────────────────────
125
+
126
+ /**
127
+ * The ONE calm notice for a litellm-local 429. Markdown (not HTML) per the
128
+ * card-format.ts convention. Copy contract (operator spec):
129
+ * - names the fleet token limiter (LiteLLM `tpm_limit`/`rpm_limit`)
130
+ * - explicitly NOT an Anthropic account limit — nothing exhausted, no
131
+ * account touched
132
+ * - the turn retries automatically; no action needed
133
+ * - after a window with silent suppressions, says "throttled N more
134
+ * time(s) since the last notice"
135
+ */
136
+ export function renderLitellmLocalNotice(opts: {
137
+ agent: string
138
+ /** Silent litellm-local 429s since the previous notice (0 on the first). */
139
+ suppressedSinceLastNotice: number
140
+ }): string {
141
+ const agent = escapeMarkdown(opts.agent)
142
+ const n = opts.suppressedSinceLastNotice
143
+ const lines = [
144
+ `🚦 **Fleet token limiter engaged** — **${agent}** hit the local LiteLLM proxy cap (\`tpm_limit\`/\`rpm_limit\`).`,
145
+ `This is switchroom's own fleet limiter smoothing a burst, not an Anthropic account limit — nothing is exhausted and no account was touched.`,
146
+ ]
147
+ if (n > 0) {
148
+ lines.push(`Throttled ${n} more ${n === 1 ? 'time' : 'times'} since the last notice.`)
149
+ }
150
+ lines.push(`_The turn retries automatically — no action needed._`)
151
+ return lines.join('\n')
152
+ }
153
+
154
+ // ─── Metric builder ──────────────────────────────────────────────────────────
155
+
156
+ /**
157
+ * Build the `litellm_local_429_notice` runtime metric for one SENT notice.
158
+ * Distinct from `rate_limit_429_classified` (which fires on EVERY classified
159
+ * event): this one fires only when a notice actually posts, and carries how
160
+ * many events the window silently absorbed — the debounce's effectiveness in
161
+ * one number. Pure builder so the payload shape is unit-testable; the wiring
162
+ * emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
163
+ */
164
+ export function buildLitellmLocalNoticeMetric(opts: {
165
+ agent: string
166
+ suppressedCount: number
167
+ windowMs: number
168
+ }): Extract<RuntimeMetricEvent, { kind: 'litellm_local_429_notice' }> {
169
+ return {
170
+ kind: 'litellm_local_429_notice',
171
+ agent: opts.agent,
172
+ suppressed_count: opts.suppressedCount,
173
+ window_ms: opts.windowMs,
174
+ }
175
+ }
176
+
177
+ // ─── Classification guard ────────────────────────────────────────────────────
178
+
179
+ /**
180
+ * True only for the `litellm-local` classification. The wiring runs this
181
+ * guard so "a notice never fires for non-litellm-local classifications" is a
182
+ * deterministic mechanism, not caller discipline — account-scoped 429s keep
183
+ * the throttle tier, generic transients keep the calm rate-limited card.
184
+ */
185
+ export function isLitellmLocalNoticeEligible(
186
+ classification: RateLimit429Classification | null | undefined,
187
+ ): classification is 'litellm-local' {
188
+ return classification === 'litellm-local'
189
+ }
@@ -92,6 +92,203 @@ export function isTransientUpstreamSignal(text: string): boolean {
92
92
  return transientUpstreamSignals.some(s => lower.includes(s))
93
93
  }
94
94
 
95
+ // ─── LiteLLM-proxy-LOCAL 429 signals (canonical, single source of truth) ─────
96
+
97
+ /**
98
+ * Explicit markers of a 429 generated by the LiteLLM proxy's OWN rate
99
+ * limiters — a router `tpm_limit`/`rpm_limit` deployment cap, a virtual-key /
100
+ * team / user cap, or an all-deployments-cooling-down router state. The
101
+ * request never reached Anthropic, so the condition is PROXY-LOCAL: it says
102
+ * nothing about the Anthropic account's quota and must never mark the account
103
+ * throttled/exhausted or trigger fleet failover.
104
+ *
105
+ * Each wording key is grounded in LiteLLM's source (verified against
106
+ * BerriAI/litellm `main`, 2026-07), matching how
107
+ * `accountScopedThrottleSignals` (throttle-tier.ts) documents its keys:
108
+ *
109
+ * - "deployment over user-defined ratelimit" —
110
+ * `RouterErrors.user_defined_ratelimit_error` (litellm/types/router.py),
111
+ * the body prefix of every deployment tpm/rpm cap 429 raised by
112
+ * litellm/router_utils/pre_call_checks/model_rate_limit_check.py:
113
+ * "Deployment over user-defined ratelimit. tpm limit={n}. current
114
+ * usage={n}. id={id}, model_group={g}".
115
+ * - "model rate limit exceeded. tpm/rpm limit" — the `message` field of the
116
+ * same enforcement RateLimitError: "Model rate limit exceeded. TPM
117
+ * limit={n}, current usage={n}" (model_rate_limit_check.py).
118
+ * - "deployment over defined rpm limit" — the usage-based-routing-v2
119
+ * strategy wording "Deployment over defined rpm limit={n}. current
120
+ * usage={n}" (litellm/router_strategy/lowest_tpm_rpm_v2.py). There is NO
121
+ * tpm sibling in any known release — v2 TPM exhaustion surfaces as the
122
+ * `no_deployments_available` wording below (the deployment is filtered
123
+ * from the candidate set rather than erroring in the increment path).
124
+ * - "no deployments available for selected model" —
125
+ * `RouterErrors.no_deployments_available`, the RouterRateLimitError
126
+ * message when every deployment for the model group is cooling down:
127
+ * "No deployments available for selected model, Try again in {n}
128
+ * seconds…" (litellm/types/router.py).
129
+ * - "litellm rate limit handler" — the v1 parallel-request limiter's
130
+ * ProxyRateLimitError detail prefix: "LiteLLM Rate Limit Handler for
131
+ * rate limit type = {t}. Crossed TPM / RPM / Max Parallel Request
132
+ * Limit. current rpm: {n}, rpm limit: {n}…"
133
+ * (litellm/proxy/hooks/parallel_request_limiter.py).
134
+ * - "crossed tpm / rpm" — `CommonProxyErrors
135
+ * .max_parallel_request_limit_reached.value` = "Crossed TPM / RPM / Max
136
+ * Parallel Request Limit" (litellm/proxy/_types.py), interpolated into
137
+ * BOTH v1 detail shapes, so it also covers the zero-limit branch's
138
+ * "Max parallel request limit reached {additional_details}" message.
139
+ * - "max parallel request limit reached" — the standalone prefix
140
+ * `raise_rate_limit_error` builds when a key's limit is set to 0
141
+ * (parallel_request_limiter.py `error_message`).
142
+ *
143
+ * The proxy-side per-key/team/user limiter (parallel_request_limiter_v3.py
144
+ * `_handle_rate_limit_error`) is matched by CO-OCCURRENCE instead of a
145
+ * per-descriptor entry — see `isLitellmProxyLocal429`. Its detail shape:
146
+ * "Rate limit exceeded for {descriptor}: {value}. Limit type: {t}. Current
147
+ * limit: {n}, Remaining: {n}. Limit resets at: {ts}". Descriptor keys in
148
+ * source are numerous and growing (api_key / user / team / team_member /
149
+ * organization / end_user / agent / agent_session / model_per_key /
150
+ * model_per_team / model_per_organization / model_per_project / tag_per_key
151
+ * / mcp_per_key / mcp_per_team as of 2026-07), so enumerating them is a
152
+ * treadmill: the pair "rate limit exceeded for " + "limit type:" appears in
153
+ * every v3 body and in no Anthropic error. This is the shape a `tpm_limit`
154
+ * on the per-agent virtual keys (src/litellm/provision.ts) trips.
155
+ *
156
+ * Deliberately EXCLUDED: the bare exception-mapping prefix
157
+ * "litellm.RateLimitError:". LiteLLM wraps FORWARDED upstream 429s with the
158
+ * same prefix on the pass-through, so its presence is NOT evidence the limit
159
+ * was proxy-local — a genuine Anthropic account throttle traversing the
160
+ * proxy must keep its account-scoped classification (see
161
+ * `classify429Detail` in throttle-tier.ts for the tie-break).
162
+ */
163
+ export const litellmProxyLocal429Signals = [
164
+ 'deployment over user-defined ratelimit',
165
+ 'model rate limit exceeded. tpm limit',
166
+ 'model rate limit exceeded. rpm limit',
167
+ 'deployment over defined rpm limit',
168
+ 'no deployments available for selected model',
169
+ 'litellm rate limit handler',
170
+ 'crossed tpm / rpm',
171
+ 'max parallel request limit reached',
172
+ ]
173
+
174
+ /**
175
+ * The v3 proxy-limiter co-occurrence pair (see the provenance comment on
176
+ * `litellmProxyLocal429Signals`): both substrings appear in every
177
+ * parallel_request_limiter_v3 429 body regardless of descriptor key, and
178
+ * never in an Anthropic error. Exported so tests can pin the rule.
179
+ */
180
+ export const litellmV3LimiterSignalPair = ['rate limit exceeded for ', 'limit type:'] as const
181
+
182
+ /**
183
+ * True when `text` carries an EXPLICIT LiteLLM-proxy-local rate-limit marker:
184
+ * one of `litellmProxyLocal429Signals`, OR the v3 limiter co-occurrence pair
185
+ * (`litellmV3LimiterSignalPair` — descriptor-agnostic, so new v3 descriptor
186
+ * keys are covered without a list update). Never throws on weird input. Pure
187
+ * wording detection only — precedence against account-scoped wording is
188
+ * owned by `classify429Detail` (throttle-tier.ts).
189
+ */
190
+ export function isLitellmProxyLocal429(text: string): boolean {
191
+ if (typeof text !== 'string' || text.length === 0) return false
192
+ const sample = text.length > 16_384 ? text.slice(0, 16_384) : text
193
+ const lower = sample.toLowerCase()
194
+ if (litellmProxyLocal429Signals.some(s => lower.includes(s))) return true
195
+ return litellmV3LimiterSignalPair.every(s => lower.includes(s))
196
+ }
197
+
198
+ /**
199
+ * Best-effort extraction of the limit detail LiteLLM embeds in its
200
+ * proxy-local 429 bodies, for instrumentation (the `rate_limit_429_classified`
201
+ * runtime metric). All fields null when nothing parseable is found — never
202
+ * throws. Shapes covered (same provenance as `litellmProxyLocal429Signals`):
203
+ *
204
+ * - "tpm limit=8000. current usage=8241" (model_rate_limit_check body)
205
+ * - "TPM limit=8000, current usage=8241" (model_rate_limit_check message)
206
+ * - "Deployment over defined rpm limit=60. current usage=61" (v2 strategy)
207
+ * - "Limit type: tokens. Current limit: 8000 … Limit resets at:
208
+ * 2026-07-12 08:05:00 UTC" (parallel_request_limiter_v3)
209
+ * - "Try again in 27.5 seconds" (RouterRateLimitError cooldown)
210
+ *
211
+ * CAVEAT — the v3 "Limit resets at" timestamp is NOT reliably UTC despite
212
+ * the literal: litellm formats it with a naive `datetime.fromtimestamp(...)
213
+ * .strftime("%Y-%m-%d %H:%M:%S UTC")` (parallel_request_limiter_v3.py
214
+ * `_handle_rate_limit_error`), i.e. proxy-LOCAL wall-clock time with a
215
+ * hard-coded "UTC" suffix. We parse it as UTC (nothing better is possible
216
+ * from the body alone), so `resetAtMs` — and the metric fields derived from
217
+ * it — can be skewed by the proxy host's UTC offset when the proxy doesn't
218
+ * run on UTC. Treat v3-derived resets as approximate; don't chase phantom
219
+ * clock drift from these metrics. The rate-limit windows are ≤1 minute
220
+ * anyway, so the absolute timestamp is informational.
221
+ */
222
+ export function parseLitellmLimitDetail(
223
+ text: string,
224
+ parseTimeNow: Date = new Date(),
225
+ ): {
226
+ limitType: string | null
227
+ limit: number | null
228
+ currentUsage: number | null
229
+ resetAtMs: number | null
230
+ } {
231
+ const empty = { limitType: null, limit: null, currentUsage: null, resetAtMs: null }
232
+ if (typeof text !== 'string' || text.length === 0) return empty
233
+ const sample = text.length > 16_384 ? text.slice(0, 16_384) : text
234
+ const lower = sample.toLowerCase()
235
+
236
+ // "tpm limit=8000" / "rpm limit=60" (body), "TPM limit=8000" (message),
237
+ // and the v1 limiter's colon form "rpm limit: 60".
238
+ let limitType: string | null = null
239
+ let limit: number | null = null
240
+ const eqLimit = lower.match(/\b([tr]pm)[ _]limit[=:]\s*(\d+)/)
241
+ if (eqLimit) {
242
+ limitType = eqLimit[1]
243
+ limit = Number(eqLimit[2])
244
+ }
245
+ // v3 limiter: "Limit type: tokens. Current limit: 8000"
246
+ if (limitType == null) {
247
+ const v3Type = lower.match(/limit type:\s*(tokens|requests|max_parallel_requests)/)
248
+ if (v3Type) limitType = v3Type[1]
249
+ }
250
+ if (limit == null) {
251
+ const v3Limit = lower.match(/current limit:\s*(\d+)/)
252
+ if (v3Limit) limit = Number(v3Limit[1])
253
+ }
254
+
255
+ // "current usage=8241" (body/message, both `=` and `= ` never occur with
256
+ // separators other than the literal shown in source).
257
+ let currentUsage: number | null = null
258
+ const usage = lower.match(/current usage=(\d+)/)
259
+ if (usage) currentUsage = Number(usage[1])
260
+
261
+ // Reset hints LiteLLM emits that the Anthropic-shaped parseResetTime does
262
+ // not cover: "Limit resets at: 2026-07-12 08:05:00 UTC" (v3 limiter — note
263
+ // the space separator, not ISO 'T') and "Try again in 27.5 seconds"
264
+ // (RouterRateLimitError).
265
+ let resetAtMs: number | null = null
266
+ const resetsAt = sample.match(
267
+ /limit resets at:\s*(\d{4}-\d{2}-\d{2}) (\d{2}:\d{2}:\d{2}) UTC/i,
268
+ )
269
+ if (resetsAt) {
270
+ const d = new Date(`${resetsAt[1]}T${resetsAt[2]}Z`)
271
+ if (!Number.isNaN(d.getTime())) resetAtMs = d.getTime()
272
+ }
273
+ if (resetAtMs == null) {
274
+ const tryAgain = lower.match(/try again in\s+(\d+(?:\.\d+)?)\s*seconds/)
275
+ if (tryAgain) {
276
+ const secs = Number(tryAgain[1])
277
+ if (Number.isFinite(secs) && secs > 0 && secs < 7 * 24 * 3600) {
278
+ resetAtMs = parseTimeNow.getTime() + Math.round(secs * 1000)
279
+ }
280
+ }
281
+ }
282
+
283
+ return {
284
+ limitType,
285
+ limit: limit != null && Number.isFinite(limit) ? limit : null,
286
+ currentUsage:
287
+ currentUsage != null && Number.isFinite(currentUsage) ? currentUsage : null,
288
+ resetAtMs,
289
+ }
290
+ }
291
+
95
292
  // ─── Detection ───────────────────────────────────────────────────────────────
96
293
 
97
294
  /**
@@ -138,6 +335,23 @@ export function detectModelUnavailable(
138
335
  : { kind: 'overload', raw: stderr }
139
336
  }
140
337
 
338
+ // ── 0.5. LiteLLM-proxy-LOCAL 429 — never a quota signal ─────────────────
339
+ // A 429 raised by the LiteLLM proxy's own limiters (`tpm_limit`/`rpm_limit`
340
+ // caps, router cooldown) never reached Anthropic — nothing about the
341
+ // account is exhausted, so it must classify to the calm retryable kind
342
+ // BEFORE the quota substrings run (some LiteLLM bodies contain the word
343
+ // "limit", which a negation-blind quota match could seize on). Runs after
344
+ // step 0 deliberately: account-affirming wording, when present, can only
345
+ // have originated upstream (LiteLLM never emits it) and wins — see the
346
+ // tie-break note on `classify429Detail` (throttle-tier.ts). Uses the full
347
+ // matcher (list + v3 co-occurrence pair) so every descriptor is covered.
348
+ if (isLitellmProxyLocal429(sample)) {
349
+ const resetAt = parseResetTime(sample)
350
+ return resetAt !== undefined
351
+ ? { kind: 'overload', resetAt, raw: stderr }
352
+ : { kind: 'overload', raw: stderr }
353
+ }
354
+
141
355
  // ── 1. Quota / billing exhaustion ──────────────────────────────────────
142
356
  const quotaSignals = [
143
357
  'out of extra usage',
@@ -192,10 +192,14 @@ export interface FleetRollInfo {
192
192
  /**
193
193
  * Trigger attribution (#3031 PR 2): "soft-avoid" = proactive
194
194
  * serving-preference roll off an account APPROACHING its limits;
195
- * "hard-exhaustion" = the probe saw a genuine quota wall. Absent on
196
- * pre-PR-2 brokers — rendered as hard exhaustion.
195
+ * "hard-exhaustion" = the probe saw a genuine quota wall;
196
+ * "model-tier-wall" = the flagship-tier (7d_oi) canary saw the account
197
+ * walled on the premium tier (#3176). Absent on pre-PR-2 brokers —
198
+ * rendered as hard exhaustion.
197
199
  */
198
- reason?: "soft-avoid" | "hard-exhaustion";
200
+ reason?: "soft-avoid" | "hard-exhaustion" | "model-tier-wall";
201
+ /** #3176 — the binding tier bucket, set only when reason is model-tier-wall. */
202
+ bucket?: string;
199
203
  }
200
204
 
201
205
  export type FleetRollAnnounceDecision =
@@ -241,9 +245,17 @@ export function buildFleetRollMessage(roll: FleetRollInfo, now: number): string
241
245
  ? ` (resets ${formatRelative(new Date(roll.exhausted_until), new Date(now))})`
242
246
  : "";
243
247
  const softAvoid = roll.reason === "soft-avoid";
248
+ // #3176 — a model-tier wall binds on the flagship (premium) 7d_oi bucket, not
249
+ // the 5h/7d windows (which read healthy — the whole point of the bug), so it
250
+ // has no window/pct to cite. Name the flagship tier explicitly and reassure
251
+ // that opus/haiku on the walled account are unaffected (the mark is
252
+ // tier-scoped), rather than rendering a generic, misleading "quota window".
253
+ const modelTierWall = roll.reason === "model-tier-wall";
244
254
  const causeLine = softAvoid
245
255
  ? `Proactive switch — \`${codeSpanSafe(roll.from)}\` is approaching its limits (${winLabel}${pctPart})${resetPart}, so the fleet moved early instead of hitting the wall.`
246
- : `${winLabel}${pctPart} on \`${codeSpanSafe(roll.from)}\`${resetPart}.`;
256
+ : modelTierWall
257
+ ? `Flagship (premium) tier weekly limit reached on \`${codeSpanSafe(roll.from)}\`${resetPart} — opus/haiku on that account are unaffected.`
258
+ : `${winLabel}${pctPart} on \`${codeSpanSafe(roll.from)}\`${resetPart}.`;
247
259
  return [
248
260
  `🔁 **Switched fleet to \`${codeSpanSafe(roll.to)}\`**`,
249
261
  ``,
@@ -164,6 +164,53 @@ export type RuntimeMetricEvent =
164
164
  agent: string
165
165
  prompt_key: string
166
166
  }
167
+ /**
168
+ * Every terminal rate-limit-family operator event (429 burst / 529
169
+ * overload), classified by origin BEFORE the gateway acts on it — fires
170
+ * even when the user-facing card is cooldown-suppressed, so the count is
171
+ * honest. `classification` says where the limit lives (see
172
+ * `RateLimit429Classification` in throttle-tier.ts): `account-scoped` =
173
+ * Anthropic throttled the account (throttle tier ran), `litellm-local` =
174
+ * the LiteLLM proxy's own tpm/rpm/router limiter tripped (calm path, no
175
+ * account attribution), `generic-transient` = other server-side 429/529
176
+ * wording. `action` is what the gateway DECIDED (throttle / failover /
177
+ * calm), emitted PRE-execution: the actual failover fire is dedup-gated
178
+ * downstream (fleetFallbackGate), so N long-reset 429s inside one dedup
179
+ * window emit N `action: 'failover'` metrics for ONE real roll — count
180
+ * decisions here, count rolls via the broker/fallback announcements.
181
+ * Correlating `account-scoped` fires against fleet token throughput is
182
+ * the operator's evidence base for setting LiteLLM `tpm_limit` caps; the
183
+ * `litellm-local` count then shows those caps actually absorbing load.
184
+ * Limit/reset fields are best-effort parses of the error body (null when
185
+ * absent).
186
+ */
187
+ | {
188
+ kind: 'rate_limit_429_classified'
189
+ agent: string
190
+ classification: 'account-scoped' | 'litellm-local' | 'generic-transient'
191
+ action: 'throttle' | 'failover' | 'calm'
192
+ reset_at_ms: number | null
193
+ reset_in_ms: number | null
194
+ limit_type: string | null
195
+ limit: number | null
196
+ current_usage: number | null
197
+ }
198
+ /**
199
+ * A litellm-local throttle NOTICE actually posted (litellm-local-notice.ts
200
+ * — the debounced "fleet token limiter engaged" message). Distinct from
201
+ * `rate_limit_429_classified`, which fires on EVERY classified 429: this
202
+ * fires once per sent notice, and `suppressed_count` is how many
203
+ * litellm-local 429s the cooldown window silently absorbed since the
204
+ * previous notice (0 on the first) — the debounce's effectiveness in one
205
+ * number. `window_ms` is the resolved cooldown window (default 15 min;
206
+ * channels.telegram.litellm_notice.window_ms).
207
+ */
208
+ | {
209
+ kind: 'litellm_local_429_notice'
210
+ agent: string
211
+ suppressed_count: number
212
+ window_ms: number
213
+ }
167
214
 
168
215
  /**
169
216
  * The JSONL sink lives under the runtime state dir so it's per-agent
@@ -1,5 +1,5 @@
1
1
  import { describe, expect, it } from 'vitest'
2
- import { createSendGate, type Clock } from './send-gate.js'
2
+ import { createSendGate, SEND_GATE_SHED, type Clock } from './send-gate.js'
3
3
  import { isFloodWaitActiveError } from './retry-api-call.js'
4
4
 
5
5
  /**
@@ -73,8 +73,10 @@ describe('send-gate PR2: cosmetic shedding', () => {
73
73
 
74
74
  const res = await gate.gate(fn('cosmetic'), { priorityClass: 'cosmetic' })
75
75
 
76
- // Shed resolves undefined; fn never ran; the shed counter moved.
77
- expect(res).toBeUndefined()
76
+ // Shed resolves the distinguishable sentinel (#3110 F1 — never a bare
77
+ // undefined, which is reserved for benign no-op drops); fn never ran;
78
+ // the shed counter moved.
79
+ expect(res).toBe(SEND_GATE_SHED)
78
80
  expect(calls).toHaveLength(0)
79
81
  expect(gate.stats().global.shed).toBe(1)
80
82
  expect(gate.stats().global.sent).toBe(0)
@@ -95,7 +97,7 @@ describe('send-gate PR2: cosmetic shedding', () => {
95
97
  const shed = await gate.gate(fn('b'), { chat_id: '5', priorityClass: 'cosmetic' })
96
98
 
97
99
  expect(calls.map((c) => c.label)).toEqual(['a'])
98
- expect(shed).toBeUndefined()
100
+ expect(shed).toBe(SEND_GATE_SHED)
99
101
  expect(gate.stats().global.shed).toBe(1)
100
102
  })
101
103
 
@@ -128,7 +130,7 @@ describe('send-gate PR2: cosmetic shedding', () => {
128
130
  priorityClass: 'cosmetic',
129
131
  })
130
132
 
131
- expect(res).toBeUndefined()
133
+ expect(res).toBe(SEND_GATE_SHED)
132
134
  expect(calls).toHaveLength(0)
133
135
  expect(gate.stats().global.shed).toBe(1)
134
136
  })
@@ -366,7 +368,7 @@ describe('send-gate PR2: H1 cross-chat message_id isolation', () => {
366
368
  priorityClass: 'cosmetic',
367
369
  })
368
370
 
369
- expect(rA).toBeUndefined() // A shed
371
+ expect(rA).toBe(SEND_GATE_SHED) // A shed
370
372
  expect(rB).toBe('B-edit') // B sent
371
373
  expect(calls.map((c) => c.label)).toEqual(['B-edit'])
372
374
  expect(gate.stats().global.shed).toBe(1)
@@ -568,7 +570,7 @@ describe('send-gate PR2: opening a window from a 429, and persistence hook', ()
568
570
  expect(scopes).toEqual(['chat:7', 'global', 'group:7', 'msg-edit:7:3'])
569
571
  // After the window opens, a later cosmetic on the same chat sheds.
570
572
  const shed = await gate.gate(async () => 'x', { chat_id: '7', priorityClass: 'cosmetic' })
571
- expect(shed).toBeUndefined()
573
+ expect(shed).toBe(SEND_GATE_SHED)
572
574
  expect(gate.stats().global.shed).toBe(1)
573
575
  })
574
576
  })
@@ -118,6 +118,32 @@ export type PriorityClass = 'critical' | 'useful' | 'cosmetic'
118
118
  */
119
119
  export const UNTAGGED_SEND_CLASS: PriorityClass = 'critical'
120
120
 
121
+ /**
122
+ * Distinguishable resolution value for a SHED call (#3110 review F1).
123
+ *
124
+ * `undefined` was overloaded three ways at the robustApiCall seam: a gate
125
+ * SHED (cosmetic call dropped under pressure — did NOT land), a gate no-op
126
+ * drop (identical payload already on screen — benign), and the retry
127
+ * policy's swallowed benign 400s ("message is not modified" — also benign).
128
+ * A caller that needs to know "did my edit land?" (the draft stream's
129
+ * shed-honesty handling in stream-controller.ts) could not tell these
130
+ * apart, and treating every `undefined` as a shed froze multi-piece
131
+ * streams on perfectly healthy chats.
132
+ *
133
+ * A shed now resolves THIS sentinel instead. `Symbol.for` keys it in the
134
+ * global symbol registry so duplicated module instances (server + gateway
135
+ * builds) agree on identity. The no-op drop and coalesced-revert drop keep
136
+ * resolving `undefined` — for those the payload IS on screen. Callers that
137
+ * ignore the result (reactions, typing, fire-and-forget card edits) are
138
+ * unaffected either way.
139
+ */
140
+ export const SEND_GATE_SHED: unique symbol = Symbol.for('switchroom.send-gate.shed')
141
+
142
+ /** True when a gate result is the {@link SEND_GATE_SHED} sentinel. */
143
+ export function isSendGateShed(value: unknown): value is typeof SEND_GATE_SHED {
144
+ return value === SEND_GATE_SHED
145
+ }
146
+
121
147
  /**
122
148
  * Extra metadata a call site can attach so the gate can key the right buckets.
123
149
  * All fields optional — a call with none still passes the global bucket. These
@@ -159,8 +185,9 @@ export interface BucketCounters {
159
185
  dropped: number
160
186
  /**
161
187
  * Cosmetic calls shed under pressure — no token free OR a flood window open
162
- * (part3-design §2). A shed resolves as `undefined`; the next send carries
163
- * full state.
188
+ * (part3-design §2). A shed resolves the SEND_GATE_SHED sentinel (F1 —
189
+ * distinguishable from a benign no-op drop's `undefined`); the next send
190
+ * carries full state.
164
191
  */
165
192
  shed: number
166
193
  /** Useful calls dropped because they exceeded their queue TTL (part3-design §2). */
@@ -913,7 +940,9 @@ export function createSendGate(config: SendGateConfig): SendGate {
913
940
  const msgWait = state.suppressedUntilMs > now ? state.suppressedUntilMs - now : 0
914
941
  if (wait > 0 || msgWait > 0) {
915
942
  counters.shed++
916
- return Promise.resolve(undefined as unknown as T)
943
+ // Distinguishable from the no-op drop below (undefined): a shed did
944
+ // NOT land, and edit-driving callers must be able to tell (F1).
945
+ return Promise.resolve(SEND_GATE_SHED as unknown as T)
917
946
  }
918
947
  }
919
948
 
@@ -991,7 +1020,8 @@ export function createSendGate(config: SendGateConfig): SendGate {
991
1020
  const outcome = await admitPriority(bucketsFor(opts), priority)
992
1021
  if (outcome.result === 'shed') {
993
1022
  counters.shed++
994
- return undefined as unknown as T
1023
+ // See SEND_GATE_SHED: distinguishable "did not land" resolution (F1).
1024
+ return SEND_GATE_SHED as unknown as T
995
1025
  }
996
1026
  if (outcome.result === 'expired') {
997
1027
  counters.expired++
@@ -41,7 +41,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
41
41
  return env.PROGRESS_CARD_MULTI_AGENT !== '0'
42
42
  }
43
43
  import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
44
- import { isTransientUpstreamSignal } from './model-unavailable.js'
44
+ import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
45
45
  import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
46
46
  import { isModelSentinel } from './model-label.js'
47
47
 
@@ -684,9 +684,21 @@ export function detectErrorInTranscriptLine(
684
684
  // model-unavailable.ts) — an ambiguous 429 that merely says "limit" stays
685
685
  // quota-exhausted, biasing toward surfacing a real wall. Other statuses fall
686
686
  // through to the shared classifier.
687
+ //
688
+ // A 429 carrying LiteLLM-proxy-LOCAL limiter wording ("Deployment over
689
+ // user-defined ratelimit", "Rate limit exceeded for api_key: …" — the
690
+ // canonical `litellmProxyLocal429Signals` list) is the proxy's own
691
+ // `tpm_limit`/`rpm_limit` cap tripping BEFORE the request reached
692
+ // Anthropic. Nothing about the account is exhausted — blanket-labeling it
693
+ // quota-exhausted would fire the model-unavailable card + mark-exhausted +
694
+ // fleet failover for a purely proxy-local condition. It is rate-limited
695
+ // (calm path). Precedence when BOTH wordings appear is owned by
696
+ // `classify429Detail` gateway-side; here both branches yield the same
697
+ // 'rate-limited' kind, so order is immaterial.
687
698
  const kind: OperatorEventKind =
688
699
  status === 429
689
- ? isTransientUpstreamSignal(`${text}\n${errStr}`)
700
+ ? isTransientUpstreamSignal(`${text}\n${errStr}`) ||
701
+ isLitellmProxyLocal429(`${text}\n${errStr}`)
690
702
  ? 'rate-limited'
691
703
  : 'quota-exhausted'
692
704
  : classifyClaudeError({ type: errStr, status, message: text })