switchroom 0.18.14 → 0.18.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -41,7 +41,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
41
41
  return env.PROGRESS_CARD_MULTI_AGENT !== '0'
42
42
  }
43
43
  import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
44
- import { isTransientUpstreamSignal } from './model-unavailable.js'
44
+ import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
45
45
  import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
46
46
  import { isModelSentinel } from './model-label.js'
47
47
 
@@ -684,9 +684,21 @@ export function detectErrorInTranscriptLine(
684
684
  // model-unavailable.ts) — an ambiguous 429 that merely says "limit" stays
685
685
  // quota-exhausted, biasing toward surfacing a real wall. Other statuses fall
686
686
  // through to the shared classifier.
687
+ //
688
+ // A 429 carrying LiteLLM-proxy-LOCAL limiter wording ("Deployment over
689
+ // user-defined ratelimit", "Rate limit exceeded for api_key: …" — the
690
+ // canonical `litellmProxyLocal429Signals` list) is the proxy's own
691
+ // `tpm_limit`/`rpm_limit` cap tripping BEFORE the request reached
692
+ // Anthropic. Nothing about the account is exhausted — blanket-labeling it
693
+ // quota-exhausted would fire the model-unavailable card + mark-exhausted +
694
+ // fleet failover for a purely proxy-local condition. It is rate-limited
695
+ // (calm path). Precedence when BOTH wordings appear is owned by
696
+ // `classify429Detail` gateway-side; here both branches yield the same
697
+ // 'rate-limited' kind, so order is immaterial.
687
698
  const kind: OperatorEventKind =
688
699
  status === 429
689
- ? isTransientUpstreamSignal(`${text}\n${errStr}`)
700
+ ? isTransientUpstreamSignal(`${text}\n${errStr}`) ||
701
+ isLitellmProxyLocal429(`${text}\n${errStr}`)
690
702
  ? 'rate-limited'
691
703
  : 'quota-exhausted'
692
704
  : classifyClaudeError({ type: errStr, status, message: text })
@@ -14,10 +14,52 @@ import { describe, it, expect } from 'vitest'
14
14
  import {
15
15
  detectModelUnavailable,
16
16
  formatModelUnavailableCard,
17
+ isLitellmProxyLocal429,
18
+ parseLitellmLimitDetail,
17
19
  resolveModelUnavailableFromOperatorEvent,
18
20
  type ModelUnavailableDetection,
19
21
  } from '../model-unavailable.js'
20
22
 
23
+ // Real LiteLLM proxy-local 429 bodies, verbatim shapes from BerriAI/litellm
24
+ // source (see the provenance comment on `litellmProxyLocal429Signals`).
25
+ const LITELLM_DEPLOYMENT_CAP_BODY =
26
+ 'Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241. ' +
27
+ 'id=abc123def, model_group=claude-fable-5'
28
+ const LITELLM_DEPLOYMENT_CAP_MESSAGE =
29
+ 'litellm.RateLimitError: Model rate limit exceeded. TPM limit=8000, current usage=8241'
30
+ const LITELLM_V2_STRATEGY_BODY =
31
+ "Deployment over defined rpm limit=60. current usage=61. id=abc123def, " +
32
+ "model_group=claude-fable-5. Get the model info by calling 'router.get_model_info(id)"
33
+ const LITELLM_ROUTER_COOLDOWN_BODY =
34
+ 'No deployments available for selected model, Try again in 27.5 seconds. ' +
35
+ "Passed model=claude-fable-5. pre-call-checks=False, cooldown_list=['abc123def']"
36
+ const LITELLM_V3_KEY_LIMIT_BODY =
37
+ 'Rate limit exceeded for api_key: hashed-key-1a2b3c. Limit type: tokens. ' +
38
+ 'Current limit: 8000, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC'
39
+ // v1 shape A — the ProxyRateLimitError detail (parallel_request_limiter.py
40
+ // ~line 124; the interpolated CommonProxyErrors.max_parallel_request_limit_
41
+ // reached.value is "Crossed TPM / RPM / Max Parallel Request Limit").
42
+ const LITELLM_V1_PARALLEL_BODY =
43
+ 'LiteLLM Rate Limit Handler for rate limit type = requests. ' +
44
+ 'Crossed TPM / RPM / Max Parallel Request Limit. ' +
45
+ 'current rpm: 61, rpm limit: 60, current tpm: 100, tpm limit: 8000, ' +
46
+ 'current max_parallel_requests: 1, max_parallel_requests: 10'
47
+ // v1 shape B — raise_rate_limit_error's zero-limit branch: the standalone
48
+ // "Max parallel request limit reached" prefix + additional_details
49
+ // (parallel_request_limiter.py ~lines 88 + 183).
50
+ const LITELLM_V1_ZERO_LIMIT_BODY =
51
+ 'Max parallel request limit reached Crossed TPM / RPM / Max Parallel ' +
52
+ 'Request Limit. Hit limit for tokens. Current limits: ' +
53
+ 'max_parallel_requests: 10, tpm_limit: 0, rpm_limit: 60'
54
+ // v3 limiter with a descriptor OUTSIDE any enumerated list — covered by the
55
+ // litellmV3LimiterSignalPair co-occurrence rule (v3 descriptor keys also
56
+ // include organization / team_member / model_per_team / agent / tag_per_key
57
+ // / mcp_per_key and keep growing, so enumeration is a treadmill).
58
+ const LITELLM_V3_TEAM_MODEL_BODY =
59
+ 'Rate limit exceeded for model_per_team: team-1a2b3c:claude-fable-5. ' +
60
+ 'Limit type: tokens. Current limit: 50000, Remaining: 0. ' +
61
+ 'Limit resets at: 2026-07-12 08:05:00 UTC'
62
+
21
63
  // ─── detectModelUnavailable ──────────────────────────────────────────────────
22
64
 
23
65
  describe('detectModelUnavailable — quota / billing strings', () => {
@@ -117,6 +159,151 @@ describe('detectModelUnavailable — transient upstream 429 vs account quota (#2
117
159
  })
118
160
  })
119
161
 
162
+ describe('detectModelUnavailable — LiteLLM-proxy-LOCAL 429s (never quota)', () => {
163
+ // A 429 raised by LiteLLM's OWN limiter never reached Anthropic — it must
164
+ // classify to the calm retryable kind, never quota_exhausted (which drives
165
+ // the scary card + mark-exhausted + fleet failover for an account that was
166
+ // never touched). Bodies are verbatim LiteLLM source shapes.
167
+ it.each([
168
+ ['deployment tpm cap (model_rate_limit_check body)', LITELLM_DEPLOYMENT_CAP_BODY],
169
+ ['deployment tpm cap (RateLimitError message)', LITELLM_DEPLOYMENT_CAP_MESSAGE],
170
+ ['usage-based-routing v2 rpm wording', LITELLM_V2_STRATEGY_BODY],
171
+ ['router cooldown (RouterRateLimitError)', LITELLM_ROUTER_COOLDOWN_BODY],
172
+ ['virtual-key tpm_limit (parallel_request_limiter_v3)', LITELLM_V3_KEY_LIMIT_BODY],
173
+ ['team model cap — non-enumerated v3 descriptor (co-occurrence rule)', LITELLM_V3_TEAM_MODEL_BODY],
174
+ ['v1 parallel-request limiter (Rate Limit Handler detail)', LITELLM_V1_PARALLEL_BODY],
175
+ ['v1 zero-limit branch (Max parallel request limit reached prefix)', LITELLM_V1_ZERO_LIMIT_BODY],
176
+ ])('classifies %s as overload, NOT quota_exhausted', (_name, body) => {
177
+ const d = detectModelUnavailable(body)
178
+ expect(d?.kind).toBe('overload')
179
+ })
180
+
181
+ it('a rate-limited operator event with a LiteLLM-local body resolves to NO model-unavailable card', () => {
182
+ // End-to-end seam the gateway uses: resolveModelUnavailableFromOperatorEvent
183
+ // returning null is what keeps the calm 🚦 card AND blocks the
184
+ // auto-fallback branch (fallback only fires when a detection resolves).
185
+ for (const body of [
186
+ LITELLM_DEPLOYMENT_CAP_BODY,
187
+ LITELLM_V3_KEY_LIMIT_BODY,
188
+ LITELLM_ROUTER_COOLDOWN_BODY,
189
+ ]) {
190
+ expect(
191
+ resolveModelUnavailableFromOperatorEvent({ kind: 'rate-limited', detail: body }),
192
+ ).toBeNull()
193
+ }
194
+ })
195
+ })
196
+
197
+ describe('isLitellmProxyLocal429 — signal matching', () => {
198
+ it('matches every canonical LiteLLM limiter wording', () => {
199
+ for (const body of [
200
+ LITELLM_DEPLOYMENT_CAP_BODY,
201
+ LITELLM_DEPLOYMENT_CAP_MESSAGE,
202
+ LITELLM_V2_STRATEGY_BODY,
203
+ LITELLM_ROUTER_COOLDOWN_BODY,
204
+ LITELLM_V3_KEY_LIMIT_BODY,
205
+ LITELLM_V1_PARALLEL_BODY,
206
+ LITELLM_V1_ZERO_LIMIT_BODY,
207
+ ]) {
208
+ expect(isLitellmProxyLocal429(body)).toBe(true)
209
+ }
210
+ })
211
+
212
+ it('matches ANY v3 descriptor via the co-occurrence pair, not an enumerated list', () => {
213
+ // model_per_team is not in litellmProxyLocal429Signals — only the
214
+ // "rate limit exceeded for " + "limit type:" pair catches it. Same for
215
+ // any descriptor litellm adds later.
216
+ expect(isLitellmProxyLocal429(LITELLM_V3_TEAM_MODEL_BODY)).toBe(true)
217
+ expect(
218
+ isLitellmProxyLocal429(
219
+ 'Rate limit exceeded for organization: org-1a2b3c. Limit type: requests. ' +
220
+ 'Current limit: 100, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC',
221
+ ),
222
+ ).toBe(true)
223
+ // HALF the pair is not enough — "rate limit exceeded for" prose without
224
+ // the v3 "Limit type:" field must not classify proxy-local.
225
+ expect(
226
+ isLitellmProxyLocal429('Rate limit exceeded for this account, try later'),
227
+ ).toBe(false)
228
+ expect(isLitellmProxyLocal429('Limit type: tokens')).toBe(false)
229
+ })
230
+
231
+ it('does NOT match Anthropic account/wall/server wordings', () => {
232
+ expect(
233
+ isLitellmProxyLocal429(
234
+ "This request would exceed your account's rate limit. Please try again later.",
235
+ ),
236
+ ).toBe(false)
237
+ expect(
238
+ isLitellmProxyLocal429('Server is temporarily limiting requests (not your usage limit)'),
239
+ ).toBe(false)
240
+ expect(isLitellmProxyLocal429("You've hit your limit · resets 8:50am")).toBe(false)
241
+ })
242
+
243
+ it('does NOT match the bare litellm.RateLimitError exception-mapping prefix', () => {
244
+ // The pass-through wraps FORWARDED upstream 429s with the same prefix —
245
+ // it is not evidence the limit was proxy-local.
246
+ expect(
247
+ isLitellmProxyLocal429(
248
+ "litellm.RateLimitError: RateLimitError: This request would exceed your account's rate limit.",
249
+ ),
250
+ ).toBe(false)
251
+ })
252
+
253
+ it('never throws on weird input', () => {
254
+ expect(isLitellmProxyLocal429('')).toBe(false)
255
+ expect(isLitellmProxyLocal429(undefined as unknown as string)).toBe(false)
256
+ expect(isLitellmProxyLocal429(42 as unknown as string)).toBe(false)
257
+ // Slice guard: signal past the 16KB sample is not scanned.
258
+ expect(isLitellmProxyLocal429('A'.repeat(100_000) + LITELLM_DEPLOYMENT_CAP_BODY)).toBe(false)
259
+ })
260
+ })
261
+
262
+ describe('parseLitellmLimitDetail — instrumentation extraction', () => {
263
+ const NOW = new Date(Date.UTC(2026, 6, 12, 8, 0, 0))
264
+
265
+ it('extracts tpm limit + current usage from the deployment-cap body', () => {
266
+ const d = parseLitellmLimitDetail(LITELLM_DEPLOYMENT_CAP_BODY, NOW)
267
+ expect(d.limitType).toBe('tpm')
268
+ expect(d.limit).toBe(8000)
269
+ expect(d.currentUsage).toBe(8241)
270
+ expect(d.resetAtMs).toBeNull()
271
+ })
272
+
273
+ it('extracts rpm limit from the v2 strategy body', () => {
274
+ const d = parseLitellmLimitDetail(LITELLM_V2_STRATEGY_BODY, NOW)
275
+ expect(d.limitType).toBe('rpm')
276
+ expect(d.limit).toBe(60)
277
+ expect(d.currentUsage).toBe(61)
278
+ })
279
+
280
+ it('extracts the v3 limiter limit type, limit, and "Limit resets at" UTC timestamp', () => {
281
+ const d = parseLitellmLimitDetail(LITELLM_V3_KEY_LIMIT_BODY, NOW)
282
+ expect(d.limitType).toBe('tokens')
283
+ expect(d.limit).toBe(8000)
284
+ expect(d.resetAtMs).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
285
+ })
286
+
287
+ it('extracts the router-cooldown "Try again in N seconds" relative reset', () => {
288
+ const d = parseLitellmLimitDetail(LITELLM_ROUTER_COOLDOWN_BODY, NOW)
289
+ expect(d.resetAtMs).toBe(NOW.getTime() + 27_500)
290
+ })
291
+
292
+ it('extracts the v1 limiter colon-form rpm limit', () => {
293
+ const d = parseLitellmLimitDetail(LITELLM_V1_PARALLEL_BODY, NOW)
294
+ expect(d.limitType).toBe('rpm')
295
+ expect(d.limit).toBe(60)
296
+ })
297
+
298
+ it('returns all-null on non-LiteLLM prose and never throws on weird input', () => {
299
+ const empty = { limitType: null, limit: null, currentUsage: null, resetAtMs: null }
300
+ expect(parseLitellmLimitDetail('Please try again later.', NOW)).toEqual(empty)
301
+ expect(parseLitellmLimitDetail('', NOW)).toEqual(empty)
302
+ expect(parseLitellmLimitDetail(undefined as unknown as string, NOW)).toEqual(empty)
303
+ expect(parseLitellmLimitDetail({} as unknown as string, NOW)).toEqual(empty)
304
+ })
305
+ })
306
+
120
307
  describe('detectModelUnavailable — network failures', () => {
121
308
  it('classifies ECONNREFUSED', () => {
122
309
  expect(detectModelUnavailable('connect ECONNREFUSED 1.2.3.4:443')?.kind).toBe('network')
@@ -198,6 +198,61 @@ describe('detectErrorInTranscriptLine — error detection', () => {
198
198
  expect(detection).toBeNull()
199
199
  })
200
200
 
201
+ // A 429 whose body carries LiteLLM-proxy-LOCAL limiter wording is the
202
+ // proxy's own tpm_limit/rpm_limit cap tripping BEFORE the request reached
203
+ // Anthropic. Nothing about the account is exhausted — it must take the
204
+ // calm rate-limited path (no model-unavailable card, no mark-exhausted,
205
+ // no fleet failover). Prerequisite for enabling litellm tpm caps on the
206
+ // fleet: without this, every cap trip would bench a healthy account.
207
+ it('classifies a LiteLLM-proxy-local 429 as rate-limited, NOT quota-exhausted', () => {
208
+ const litellmBodies = [
209
+ // Deployment tpm cap (model_rate_limit_check.py body, verbatim shape).
210
+ 'API Error: 429 Deployment over user-defined ratelimit. tpm limit=8000. ' +
211
+ 'current usage=8241. id=abc123def, model_group=claude-fable-5',
212
+ // Virtual-key tpm_limit (parallel_request_limiter_v3.py).
213
+ 'API Error: 429 Rate limit exceeded for api_key: hashed-key-1a2b3c. ' +
214
+ 'Limit type: tokens. Current limit: 8000, Remaining: 0. ' +
215
+ 'Limit resets at: 2026-07-12 08:05:00 UTC',
216
+ // Router cooldown (RouterRateLimitError).
217
+ 'API Error: 429 No deployments available for selected model, ' +
218
+ 'Try again in 27.5 seconds. Passed model=claude-fable-5.',
219
+ // Team-level model cap — a v3 descriptor OUTSIDE the enumerated signal
220
+ // list, covered only by the co-occurrence pair
221
+ // (litellmV3LimiterSignalPair). Pre-fix this missed every signal →
222
+ // quota-exhausted → scary card + failover.
223
+ 'API Error: 429 Rate limit exceeded for model_per_team: ' +
224
+ 'team-1a2b3c:claude-fable-5. Limit type: tokens. ' +
225
+ 'Current limit: 50000, Remaining: 0. ' +
226
+ 'Limit resets at: 2026-07-12 08:05:00 UTC',
227
+ ]
228
+ for (const text of litellmBodies) {
229
+ const line = JSON.stringify({
230
+ type: 'assistant',
231
+ message: {
232
+ role: 'assistant',
233
+ model: '<synthetic>',
234
+ content: [{ type: 'text', text }],
235
+ },
236
+ error: 'rate_limit',
237
+ isApiErrorMessage: true,
238
+ apiErrorStatus: 429,
239
+ })
240
+ const result = detectErrorInTranscriptLine(line)
241
+ expect(result).not.toBeNull()
242
+ expect(result!.kind).toBe('rate-limited')
243
+ expect(result!.transient).toBe(true)
244
+ // End-to-end: the resolver must NOT produce a model-unavailable card —
245
+ // null is what keeps the calm 🚦 card and blocks the auto-fallback
246
+ // branch in emitGatewayOperatorEvent.
247
+ expect(
248
+ resolveModelUnavailableFromOperatorEvent({
249
+ kind: result!.kind,
250
+ detail: result!.detail,
251
+ }),
252
+ ).toBeNull()
253
+ }
254
+ })
255
+
201
256
  // Guard against over-correcting: a GENUINE quota wall (no transient marker)
202
257
  // must STILL be quota-exhausted AND still resolve to a card.
203
258
  it('a genuine quota-wall 429 still produces the quota-exhausted card', () => {
@@ -91,6 +91,30 @@ describe('runtime-metrics — JSONL sink', () => {
91
91
  expect(typeof parsed.ts).toBe('number')
92
92
  })
93
93
 
94
+ it('rate_limit_429_classified carries classification + action + limit/reset detail', () => {
95
+ emitRuntimeMetric({
96
+ kind: 'rate_limit_429_classified',
97
+ agent: 'carrie',
98
+ classification: 'litellm-local',
99
+ action: 'calm',
100
+ reset_at_ms: 1_783_850_700_000,
101
+ reset_in_ms: 300_000,
102
+ limit_type: 'tpm',
103
+ limit: 8000,
104
+ current_usage: 8241,
105
+ })
106
+ const parsed = JSON.parse(readFileSync(metricsPath, 'utf-8').trim())
107
+ expect(parsed.kind).toBe('rate_limit_429_classified')
108
+ expect(parsed.agent).toBe('carrie')
109
+ expect(parsed.classification).toBe('litellm-local')
110
+ expect(parsed.action).toBe('calm')
111
+ expect(parsed.reset_in_ms).toBe(300_000)
112
+ expect(parsed.limit_type).toBe('tpm')
113
+ expect(parsed.limit).toBe(8000)
114
+ expect(parsed.current_usage).toBe(8241)
115
+ expect(typeof parsed.ts).toBe('number')
116
+ })
117
+
94
118
  it('appends — does not overwrite — across calls', () => {
95
119
  for (let i = 0; i < 5; i++) {
96
120
  emitRuntimeMetric({
@@ -12,6 +12,8 @@
12
12
 
13
13
  import { describe, it, expect } from 'vitest'
14
14
  import {
15
+ build429ClassifiedMetric,
16
+ classify429Detail,
15
17
  decideThrottleTier,
16
18
  evaluateThrottleNotice,
17
19
  isAccountScopedThrottle,
@@ -136,6 +138,180 @@ describe('decideThrottleTier — decision matrix', () => {
136
138
  })
137
139
  })
138
140
 
141
+ // Verbatim LiteLLM proxy-local 429 shapes (provenance on
142
+ // `litellmProxyLocal429Signals`, model-unavailable.ts).
143
+ const LITELLM_TPM_CAP =
144
+ 'Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241. ' +
145
+ 'id=abc123def, model_group=claude-fable-5'
146
+ const LITELLM_KEY_LIMIT =
147
+ 'Rate limit exceeded for api_key: hashed-key-1a2b3c. Limit type: tokens. ' +
148
+ 'Current limit: 8000, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC'
149
+ const LITELLM_ROUTER_COOLDOWN =
150
+ 'No deployments available for selected model, Try again in 27.5 seconds. ' +
151
+ "Passed model=claude-fable-5. pre-call-checks=False, cooldown_list=['abc123def']"
152
+ // v3 limiter, descriptor NOT in the enumerated signal list — matched by the
153
+ // litellmV3LimiterSignalPair co-occurrence rule (model-unavailable.ts).
154
+ const LITELLM_TEAM_MODEL_LIMIT =
155
+ 'Rate limit exceeded for model_per_team: team-1a2b3c:claude-fable-5. ' +
156
+ 'Limit type: tokens. Current limit: 50000, Remaining: 0. ' +
157
+ 'Limit resets at: 2026-07-12 08:05:00 UTC'
158
+
159
+ describe('decideThrottleTier — LiteLLM-proxy-local 429s never enter the tier', () => {
160
+ // OUTCOME pin: `none` is what keeps a proxy-local cap trip off the
161
+ // account-scoped machinery — no broker mark-throttled, no throttle-tier
162
+ // runner fire, no failover escalation. The gateway additionally gates on
163
+ // classify429Detail, but the decision module must agree.
164
+ it.each([
165
+ ['deployment tpm cap', LITELLM_TPM_CAP],
166
+ ['virtual-key tpm_limit', LITELLM_KEY_LIMIT],
167
+ ['router cooldown', LITELLM_ROUTER_COOLDOWN],
168
+ ['team model cap (non-enumerated v3 descriptor)', LITELLM_TEAM_MODEL_LIMIT],
169
+ ])('%s → none (calm path owns it)', (_name, detail) => {
170
+ expect(decideThrottleTier({ detail, now: NOW, thresholdMs: THRESHOLD })).toEqual({
171
+ action: 'none',
172
+ })
173
+ })
174
+ })
175
+
176
+ describe('classify429Detail — three-way origin classification', () => {
177
+ it('classifies LiteLLM limiter wordings as litellm-local', () => {
178
+ expect(classify429Detail(LITELLM_TPM_CAP)).toBe('litellm-local')
179
+ expect(classify429Detail(LITELLM_KEY_LIMIT)).toBe('litellm-local')
180
+ expect(classify429Detail(LITELLM_ROUTER_COOLDOWN)).toBe('litellm-local')
181
+ // Non-enumerated v3 descriptor — the co-occurrence pair, not a list
182
+ // entry, carries this one.
183
+ expect(classify429Detail(LITELLM_TEAM_MODEL_LIMIT)).toBe('litellm-local')
184
+ })
185
+
186
+ it('classifies Anthropic account-affirming wording as account-scoped', () => {
187
+ expect(classify429Detail(TRANSIENT('resets in 3m'))).toBe('account-scoped')
188
+ expect(
189
+ classify429Detail('would exceed your account’s rate limit'),
190
+ ).toBe('account-scoped')
191
+ })
192
+
193
+ it('classifies server-side / bare rate-limit wording as generic-transient', () => {
194
+ expect(
195
+ classify429Detail('Server is temporarily limiting requests (not your usage limit)'),
196
+ ).toBe('generic-transient')
197
+ expect(classify429Detail('overloaded_error 529')).toBe('generic-transient')
198
+ expect(classify429Detail('rate_limit_error: try again later')).toBe('generic-transient')
199
+ })
200
+
201
+ it('TIE-BREAK: account-affirming wording wins over LiteLLM wording when both appear', () => {
202
+ // The pass-through shape: LiteLLM wraps a FORWARDED upstream Anthropic
203
+ // account 429 (exception mapping / limiter prose in the same body).
204
+ // LiteLLM never emits the account wording itself, so its presence means
205
+ // Anthropic really throttled the account — the broker throttle mark and
206
+ // the throttle tier must still run.
207
+ const mixed =
208
+ `litellm.RateLimitError: ${LITELLM_TPM_CAP} — upstream said: ` +
209
+ "This request would exceed your account's rate limit. Please try again later."
210
+ expect(classify429Detail(mixed)).toBe('account-scoped')
211
+ // And the tier decision still engages (not 'none').
212
+ expect(
213
+ decideThrottleTier({ detail: mixed, now: NOW, thresholdMs: THRESHOLD }).action,
214
+ ).not.toBe('none')
215
+ })
216
+
217
+ it('a bare litellm.RateLimitError wrapper WITHOUT limiter wording stays generic-transient', () => {
218
+ // The exception-mapping prefix alone is not proxy-local evidence.
219
+ expect(
220
+ classify429Detail('litellm.RateLimitError: RateLimitError: 429 try again later'),
221
+ ).toBe('generic-transient')
222
+ })
223
+
224
+ it('never throws on weird input', () => {
225
+ expect(classify429Detail('')).toBe('generic-transient')
226
+ expect(classify429Detail(undefined as unknown as string)).toBe('generic-transient')
227
+ expect(classify429Detail(9000 as unknown as string)).toBe('generic-transient')
228
+ expect(classify429Detail('A'.repeat(200_000))).toBe('generic-transient')
229
+ })
230
+ })
231
+
232
+ describe('build429ClassifiedMetric — instrumentation payload', () => {
233
+ it('litellm-local deployment cap → limit fields + calm action', () => {
234
+ const m = build429ClassifiedMetric({
235
+ agent: 'carrie',
236
+ detail: LITELLM_TPM_CAP,
237
+ classification: 'litellm-local',
238
+ action: 'calm',
239
+ now: NOW,
240
+ })
241
+ expect(m).toEqual({
242
+ kind: 'rate_limit_429_classified',
243
+ agent: 'carrie',
244
+ classification: 'litellm-local',
245
+ action: 'calm',
246
+ reset_at_ms: null,
247
+ reset_in_ms: null,
248
+ limit_type: 'tpm',
249
+ limit: 8000,
250
+ current_usage: 8241,
251
+ })
252
+ })
253
+
254
+ it('litellm-local NON-enumerated v3 descriptor (model_per_team) → full metric payload', () => {
255
+ // Finding-pin: a team-level model cap must produce the metric (it used
256
+ // to miss every signal → quota-exhausted → no metric at all).
257
+ const m = build429ClassifiedMetric({
258
+ agent: 'carrie',
259
+ detail: LITELLM_TEAM_MODEL_LIMIT,
260
+ classification: 'litellm-local',
261
+ action: 'calm',
262
+ now: NOW,
263
+ })
264
+ expect(m.kind).toBe('rate_limit_429_classified')
265
+ expect(m.classification).toBe('litellm-local')
266
+ expect(m.limit_type).toBe('tokens')
267
+ expect(m.limit).toBe(50_000)
268
+ expect(m.reset_at_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
269
+ })
270
+
271
+ it('litellm-local v3 key limit → tokens limit type + parsed "Limit resets at" UTC reset', () => {
272
+ const m = build429ClassifiedMetric({
273
+ agent: 'carrie',
274
+ detail: LITELLM_KEY_LIMIT,
275
+ classification: 'litellm-local',
276
+ action: 'calm',
277
+ now: NOW,
278
+ })
279
+ expect(m.limit_type).toBe('tokens')
280
+ expect(m.limit).toBe(8000)
281
+ expect(m.reset_at_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
282
+ expect(m.reset_in_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0) - NOW)
283
+ })
284
+
285
+ it('account-scoped 429 → Anthropic-parsed reset, no limit fields', () => {
286
+ const m = build429ClassifiedMetric({
287
+ agent: 'carrie',
288
+ detail: TRANSIENT('resets in 3m'),
289
+ classification: 'account-scoped',
290
+ action: 'throttle',
291
+ now: NOW,
292
+ })
293
+ expect(m.classification).toBe('account-scoped')
294
+ expect(m.action).toBe('throttle')
295
+ expect(m.reset_at_ms).toBe(NOW + 3 * 60_000)
296
+ expect(m.reset_in_ms).toBe(3 * 60_000)
297
+ expect(m.limit_type).toBeNull()
298
+ expect(m.limit).toBeNull()
299
+ expect(m.current_usage).toBeNull()
300
+ })
301
+
302
+ it('never throws on weird detail; nulls out unparseable resets', () => {
303
+ const m = build429ClassifiedMetric({
304
+ agent: 'carrie',
305
+ detail: undefined as unknown as string,
306
+ classification: 'generic-transient',
307
+ action: 'calm',
308
+ now: NOW,
309
+ })
310
+ expect(m.reset_at_ms).toBeNull()
311
+ expect(m.reset_in_ms).toBeNull()
312
+ })
313
+ })
314
+
139
315
  describe('isAccountScopedThrottle — the tier gate', () => {
140
316
  it("matches account-affirming wording (both apostrophe variants + 'not your account')", () => {
141
317
  expect(isAccountScopedThrottle("would exceed your account's rate limit")).toBe(true)
@@ -33,7 +33,12 @@
33
33
 
34
34
  import { escapeMarkdown } from './card-format.js'
35
35
  import { formatResetRelative } from './quota-check.js'
36
- import { parseResetTime } from './model-unavailable.js'
36
+ import {
37
+ isLitellmProxyLocal429,
38
+ parseLitellmLimitDetail,
39
+ parseResetTime,
40
+ } from './model-unavailable.js'
41
+ import type { RuntimeMetricEvent } from './runtime-metrics.js'
37
42
 
38
43
  // ─── Account-scoped throttle wording ─────────────────────────────────────────
39
44
 
@@ -65,6 +70,98 @@ export function isAccountScopedThrottle(text: string): boolean {
65
70
  return accountScopedThrottleSignals.some((s) => lower.includes(s))
66
71
  }
67
72
 
73
+ // ─── Three-way 429 classification ────────────────────────────────────────────
74
+
75
+ /**
76
+ * Where a terminal 429-family failure originated:
77
+ * - `account-scoped` — Anthropic throttled THIS account ("would exceed
78
+ * your account's rate limit"). Eligible for the throttle tier below
79
+ * (broker mark-throttled / failover).
80
+ * - `litellm-local` — the LiteLLM proxy's OWN limiter tripped
81
+ * (`tpm_limit`/`rpm_limit` cap, router cooldown — see
82
+ * `litellmProxyLocal429Signals` in model-unavailable.ts). The request
83
+ * never reached Anthropic; account state must not be touched.
84
+ * - `generic-transient` — everything else in the rate-limit family
85
+ * (server-side 429/529 wording, bare `rate_limit_error`). Calm path.
86
+ */
87
+ export type RateLimit429Classification =
88
+ | 'account-scoped'
89
+ | 'litellm-local'
90
+ | 'generic-transient'
91
+
92
+ /**
93
+ * Classify a terminal `rate-limited` operator event's detail text.
94
+ *
95
+ * TIE-BREAK (both wordings present): account-scoped wins, and only on its
96
+ * EXPLICIT wording. Rationale: LiteLLM never emits the account-affirming
97
+ * strings itself, so when they co-occur with LiteLLM wording the detail is a
98
+ * genuine upstream Anthropic account 429 that traversed (and was wrapped by)
99
+ * the proxy — e.g. the pass-through's "litellm.RateLimitError: …would exceed
100
+ * your account's rate limit…" exception mapping. Classifying that as
101
+ * proxy-local would drop the broker throttle mark and walk every retry
102
+ * straight back into the same account throttle. The reverse risk is nil: a
103
+ * purely proxy-local 429 (limiter fired BEFORE any upstream call) cannot
104
+ * contain Anthropic's account wording.
105
+ *
106
+ * BOUND: operator events forwarded over IPC carry `detail` truncated to
107
+ * 1000 chars (bridge.ts `sendOperatorEvent`'s `.slice(0, 1000)`;
108
+ * `OPERATOR_EVENT_DETAIL_MAX` in gateway/ipc-server.ts), so on that path the
109
+ * tie-break sees only the first 1000 chars of the body. In practice both
110
+ * wordings sit well inside that window (real Anthropic and LiteLLM bodies
111
+ * are <400 chars), and a truncated-away account marker would merely
112
+ * downgrade account-scoped → litellm-local/generic-transient — the calm
113
+ * path, never a wrong account mark.
114
+ *
115
+ * Never throws on weird input (both matchers are total).
116
+ */
117
+ export function classify429Detail(text: string): RateLimit429Classification {
118
+ if (isAccountScopedThrottle(text)) return 'account-scoped'
119
+ if (isLitellmProxyLocal429(text)) return 'litellm-local'
120
+ return 'generic-transient'
121
+ }
122
+
123
+ /**
124
+ * Build the `rate_limit_429_classified` runtime metric for one terminal
125
+ * rate-limited operator event — the instrumentation that lets an operator
126
+ * correlate Anthropic ACCOUNT 429s with fleet TPM (the prerequisite for
127
+ * enabling LiteLLM `tpm_limit` caps; see docs/auth.md § LiteLLM-proxy-local
128
+ * 429s). Pure builder so the payload shape is unit-testable; the gateway
129
+ * emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
130
+ *
131
+ * `action` is what the gateway decided for this event:
132
+ * - `throttle` / `failover` — the account-scoped throttle tier's decision
133
+ * - `calm` — the existing calm rate-limited path (litellm-local and
134
+ * generic-transient always land here; no broker mark, no failover)
135
+ *
136
+ * Reset detail is best-effort: the Anthropic-shaped `parseResetTime` first,
137
+ * then LiteLLM's own shapes ("Limit resets at: … UTC" / "Try again in Ns")
138
+ * via `parseLitellmLimitDetail`. Limit fields parse only from LiteLLM
139
+ * wording (Anthropic bodies carry no numeric limit).
140
+ */
141
+ export function build429ClassifiedMetric(opts: {
142
+ agent: string
143
+ detail: string
144
+ classification: RateLimit429Classification
145
+ action: 'throttle' | 'failover' | 'calm'
146
+ now: number
147
+ }): Extract<RuntimeMetricEvent, { kind: 'rate_limit_429_classified' }> {
148
+ const detail = typeof opts.detail === 'string' ? opts.detail : ''
149
+ const litellm = parseLitellmLimitDetail(detail, new Date(opts.now))
150
+ const anthropicResetMs = parseResetTime(detail, new Date(opts.now))?.getTime() ?? null
151
+ const resetAtMs = anthropicResetMs ?? litellm.resetAtMs
152
+ return {
153
+ kind: 'rate_limit_429_classified',
154
+ agent: opts.agent,
155
+ classification: opts.classification,
156
+ action: opts.action,
157
+ reset_at_ms: resetAtMs,
158
+ reset_in_ms: resetAtMs != null ? Math.max(0, resetAtMs - opts.now) : null,
159
+ limit_type: litellm.limitType,
160
+ limit: litellm.limit,
161
+ current_usage: litellm.currentUsage,
162
+ }
163
+ }
164
+
68
165
  /**
69
166
  * Default retry-in-place ceiling: a transient 429 whose reset is within this
70
167
  * window waits on the account instead of failing the fleet over. Override