switchroom 0.18.14 → 0.18.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/auth-broker/index.js +41 -39
- package/dist/cli/switchroom.js +1151 -1067
- package/dist/host-control/main.js +53 -51
- package/dist/vault/approvals/kernel-server.js +16 -12
- package/dist/vault/broker/server.js +672 -668
- package/package.json +1 -1
- package/telegram-plugin/dist/bridge/bridge.js +21 -0
- package/telegram-plugin/dist/gateway/gateway.js +150 -5
- package/telegram-plugin/dist/server.js +22 -1
- package/telegram-plugin/gateway/gateway.ts +48 -2
- package/telegram-plugin/model-unavailable.ts +214 -0
- package/telegram-plugin/runtime-metrics.ts +31 -0
- package/telegram-plugin/session-tail.ts +14 -2
- package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
- package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
- package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
- package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
- package/telegram-plugin/throttle-tier.ts +98 -1
|
@@ -41,7 +41,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
|
|
|
41
41
|
return env.PROGRESS_CARD_MULTI_AGENT !== '0'
|
|
42
42
|
}
|
|
43
43
|
import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
|
|
44
|
-
import { isTransientUpstreamSignal } from './model-unavailable.js'
|
|
44
|
+
import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
|
|
45
45
|
import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
|
|
46
46
|
import { isModelSentinel } from './model-label.js'
|
|
47
47
|
|
|
@@ -684,9 +684,21 @@ export function detectErrorInTranscriptLine(
|
|
|
684
684
|
// model-unavailable.ts) — an ambiguous 429 that merely says "limit" stays
|
|
685
685
|
// quota-exhausted, biasing toward surfacing a real wall. Other statuses fall
|
|
686
686
|
// through to the shared classifier.
|
|
687
|
+
//
|
|
688
|
+
// A 429 carrying LiteLLM-proxy-LOCAL limiter wording ("Deployment over
|
|
689
|
+
// user-defined ratelimit", "Rate limit exceeded for api_key: …" — the
|
|
690
|
+
// canonical `litellmProxyLocal429Signals` list) is the proxy's own
|
|
691
|
+
// `tpm_limit`/`rpm_limit` cap tripping BEFORE the request reached
|
|
692
|
+
// Anthropic. Nothing about the account is exhausted — blanket-labeling it
|
|
693
|
+
// quota-exhausted would fire the model-unavailable card + mark-exhausted +
|
|
694
|
+
// fleet failover for a purely proxy-local condition. It is rate-limited
|
|
695
|
+
// (calm path). Precedence when BOTH wordings appear is owned by
|
|
696
|
+
// `classify429Detail` gateway-side; here both branches yield the same
|
|
697
|
+
// 'rate-limited' kind, so order is immaterial.
|
|
687
698
|
const kind: OperatorEventKind =
|
|
688
699
|
status === 429
|
|
689
|
-
? isTransientUpstreamSignal(`${text}\n${errStr}`)
|
|
700
|
+
? isTransientUpstreamSignal(`${text}\n${errStr}`) ||
|
|
701
|
+
isLitellmProxyLocal429(`${text}\n${errStr}`)
|
|
690
702
|
? 'rate-limited'
|
|
691
703
|
: 'quota-exhausted'
|
|
692
704
|
: classifyClaudeError({ type: errStr, status, message: text })
|
|
@@ -14,10 +14,52 @@ import { describe, it, expect } from 'vitest'
|
|
|
14
14
|
import {
|
|
15
15
|
detectModelUnavailable,
|
|
16
16
|
formatModelUnavailableCard,
|
|
17
|
+
isLitellmProxyLocal429,
|
|
18
|
+
parseLitellmLimitDetail,
|
|
17
19
|
resolveModelUnavailableFromOperatorEvent,
|
|
18
20
|
type ModelUnavailableDetection,
|
|
19
21
|
} from '../model-unavailable.js'
|
|
20
22
|
|
|
23
|
+
// Real LiteLLM proxy-local 429 bodies, verbatim shapes from BerriAI/litellm
|
|
24
|
+
// source (see the provenance comment on `litellmProxyLocal429Signals`).
|
|
25
|
+
const LITELLM_DEPLOYMENT_CAP_BODY =
|
|
26
|
+
'Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241. ' +
|
|
27
|
+
'id=abc123def, model_group=claude-fable-5'
|
|
28
|
+
const LITELLM_DEPLOYMENT_CAP_MESSAGE =
|
|
29
|
+
'litellm.RateLimitError: Model rate limit exceeded. TPM limit=8000, current usage=8241'
|
|
30
|
+
const LITELLM_V2_STRATEGY_BODY =
|
|
31
|
+
"Deployment over defined rpm limit=60. current usage=61. id=abc123def, " +
|
|
32
|
+
"model_group=claude-fable-5. Get the model info by calling 'router.get_model_info(id)"
|
|
33
|
+
const LITELLM_ROUTER_COOLDOWN_BODY =
|
|
34
|
+
'No deployments available for selected model, Try again in 27.5 seconds. ' +
|
|
35
|
+
"Passed model=claude-fable-5. pre-call-checks=False, cooldown_list=['abc123def']"
|
|
36
|
+
const LITELLM_V3_KEY_LIMIT_BODY =
|
|
37
|
+
'Rate limit exceeded for api_key: hashed-key-1a2b3c. Limit type: tokens. ' +
|
|
38
|
+
'Current limit: 8000, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC'
|
|
39
|
+
// v1 shape A — the ProxyRateLimitError detail (parallel_request_limiter.py
|
|
40
|
+
// ~line 124; the interpolated CommonProxyErrors.max_parallel_request_limit_
|
|
41
|
+
// reached.value is "Crossed TPM / RPM / Max Parallel Request Limit").
|
|
42
|
+
const LITELLM_V1_PARALLEL_BODY =
|
|
43
|
+
'LiteLLM Rate Limit Handler for rate limit type = requests. ' +
|
|
44
|
+
'Crossed TPM / RPM / Max Parallel Request Limit. ' +
|
|
45
|
+
'current rpm: 61, rpm limit: 60, current tpm: 100, tpm limit: 8000, ' +
|
|
46
|
+
'current max_parallel_requests: 1, max_parallel_requests: 10'
|
|
47
|
+
// v1 shape B — raise_rate_limit_error's zero-limit branch: the standalone
|
|
48
|
+
// "Max parallel request limit reached" prefix + additional_details
|
|
49
|
+
// (parallel_request_limiter.py ~lines 88 + 183).
|
|
50
|
+
const LITELLM_V1_ZERO_LIMIT_BODY =
|
|
51
|
+
'Max parallel request limit reached Crossed TPM / RPM / Max Parallel ' +
|
|
52
|
+
'Request Limit. Hit limit for tokens. Current limits: ' +
|
|
53
|
+
'max_parallel_requests: 10, tpm_limit: 0, rpm_limit: 60'
|
|
54
|
+
// v3 limiter with a descriptor OUTSIDE any enumerated list — covered by the
|
|
55
|
+
// litellmV3LimiterSignalPair co-occurrence rule (v3 descriptor keys also
|
|
56
|
+
// include organization / team_member / model_per_team / agent / tag_per_key
|
|
57
|
+
// / mcp_per_key and keep growing, so enumeration is a treadmill).
|
|
58
|
+
const LITELLM_V3_TEAM_MODEL_BODY =
|
|
59
|
+
'Rate limit exceeded for model_per_team: team-1a2b3c:claude-fable-5. ' +
|
|
60
|
+
'Limit type: tokens. Current limit: 50000, Remaining: 0. ' +
|
|
61
|
+
'Limit resets at: 2026-07-12 08:05:00 UTC'
|
|
62
|
+
|
|
21
63
|
// ─── detectModelUnavailable ──────────────────────────────────────────────────
|
|
22
64
|
|
|
23
65
|
describe('detectModelUnavailable — quota / billing strings', () => {
|
|
@@ -117,6 +159,151 @@ describe('detectModelUnavailable — transient upstream 429 vs account quota (#2
|
|
|
117
159
|
})
|
|
118
160
|
})
|
|
119
161
|
|
|
162
|
+
describe('detectModelUnavailable — LiteLLM-proxy-LOCAL 429s (never quota)', () => {
|
|
163
|
+
// A 429 raised by LiteLLM's OWN limiter never reached Anthropic — it must
|
|
164
|
+
// classify to the calm retryable kind, never quota_exhausted (which drives
|
|
165
|
+
// the scary card + mark-exhausted + fleet failover for an account that was
|
|
166
|
+
// never touched). Bodies are verbatim LiteLLM source shapes.
|
|
167
|
+
it.each([
|
|
168
|
+
['deployment tpm cap (model_rate_limit_check body)', LITELLM_DEPLOYMENT_CAP_BODY],
|
|
169
|
+
['deployment tpm cap (RateLimitError message)', LITELLM_DEPLOYMENT_CAP_MESSAGE],
|
|
170
|
+
['usage-based-routing v2 rpm wording', LITELLM_V2_STRATEGY_BODY],
|
|
171
|
+
['router cooldown (RouterRateLimitError)', LITELLM_ROUTER_COOLDOWN_BODY],
|
|
172
|
+
['virtual-key tpm_limit (parallel_request_limiter_v3)', LITELLM_V3_KEY_LIMIT_BODY],
|
|
173
|
+
['team model cap — non-enumerated v3 descriptor (co-occurrence rule)', LITELLM_V3_TEAM_MODEL_BODY],
|
|
174
|
+
['v1 parallel-request limiter (Rate Limit Handler detail)', LITELLM_V1_PARALLEL_BODY],
|
|
175
|
+
['v1 zero-limit branch (Max parallel request limit reached prefix)', LITELLM_V1_ZERO_LIMIT_BODY],
|
|
176
|
+
])('classifies %s as overload, NOT quota_exhausted', (_name, body) => {
|
|
177
|
+
const d = detectModelUnavailable(body)
|
|
178
|
+
expect(d?.kind).toBe('overload')
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
it('a rate-limited operator event with a LiteLLM-local body resolves to NO model-unavailable card', () => {
|
|
182
|
+
// End-to-end seam the gateway uses: resolveModelUnavailableFromOperatorEvent
|
|
183
|
+
// returning null is what keeps the calm 🚦 card AND blocks the
|
|
184
|
+
// auto-fallback branch (fallback only fires when a detection resolves).
|
|
185
|
+
for (const body of [
|
|
186
|
+
LITELLM_DEPLOYMENT_CAP_BODY,
|
|
187
|
+
LITELLM_V3_KEY_LIMIT_BODY,
|
|
188
|
+
LITELLM_ROUTER_COOLDOWN_BODY,
|
|
189
|
+
]) {
|
|
190
|
+
expect(
|
|
191
|
+
resolveModelUnavailableFromOperatorEvent({ kind: 'rate-limited', detail: body }),
|
|
192
|
+
).toBeNull()
|
|
193
|
+
}
|
|
194
|
+
})
|
|
195
|
+
})
|
|
196
|
+
|
|
197
|
+
describe('isLitellmProxyLocal429 — signal matching', () => {
|
|
198
|
+
it('matches every canonical LiteLLM limiter wording', () => {
|
|
199
|
+
for (const body of [
|
|
200
|
+
LITELLM_DEPLOYMENT_CAP_BODY,
|
|
201
|
+
LITELLM_DEPLOYMENT_CAP_MESSAGE,
|
|
202
|
+
LITELLM_V2_STRATEGY_BODY,
|
|
203
|
+
LITELLM_ROUTER_COOLDOWN_BODY,
|
|
204
|
+
LITELLM_V3_KEY_LIMIT_BODY,
|
|
205
|
+
LITELLM_V1_PARALLEL_BODY,
|
|
206
|
+
LITELLM_V1_ZERO_LIMIT_BODY,
|
|
207
|
+
]) {
|
|
208
|
+
expect(isLitellmProxyLocal429(body)).toBe(true)
|
|
209
|
+
}
|
|
210
|
+
})
|
|
211
|
+
|
|
212
|
+
it('matches ANY v3 descriptor via the co-occurrence pair, not an enumerated list', () => {
|
|
213
|
+
// model_per_team is not in litellmProxyLocal429Signals — only the
|
|
214
|
+
// "rate limit exceeded for " + "limit type:" pair catches it. Same for
|
|
215
|
+
// any descriptor litellm adds later.
|
|
216
|
+
expect(isLitellmProxyLocal429(LITELLM_V3_TEAM_MODEL_BODY)).toBe(true)
|
|
217
|
+
expect(
|
|
218
|
+
isLitellmProxyLocal429(
|
|
219
|
+
'Rate limit exceeded for organization: org-1a2b3c. Limit type: requests. ' +
|
|
220
|
+
'Current limit: 100, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC',
|
|
221
|
+
),
|
|
222
|
+
).toBe(true)
|
|
223
|
+
// HALF the pair is not enough — "rate limit exceeded for" prose without
|
|
224
|
+
// the v3 "Limit type:" field must not classify proxy-local.
|
|
225
|
+
expect(
|
|
226
|
+
isLitellmProxyLocal429('Rate limit exceeded for this account, try later'),
|
|
227
|
+
).toBe(false)
|
|
228
|
+
expect(isLitellmProxyLocal429('Limit type: tokens')).toBe(false)
|
|
229
|
+
})
|
|
230
|
+
|
|
231
|
+
it('does NOT match Anthropic account/wall/server wordings', () => {
|
|
232
|
+
expect(
|
|
233
|
+
isLitellmProxyLocal429(
|
|
234
|
+
"This request would exceed your account's rate limit. Please try again later.",
|
|
235
|
+
),
|
|
236
|
+
).toBe(false)
|
|
237
|
+
expect(
|
|
238
|
+
isLitellmProxyLocal429('Server is temporarily limiting requests (not your usage limit)'),
|
|
239
|
+
).toBe(false)
|
|
240
|
+
expect(isLitellmProxyLocal429("You've hit your limit · resets 8:50am")).toBe(false)
|
|
241
|
+
})
|
|
242
|
+
|
|
243
|
+
it('does NOT match the bare litellm.RateLimitError exception-mapping prefix', () => {
|
|
244
|
+
// The pass-through wraps FORWARDED upstream 429s with the same prefix —
|
|
245
|
+
// it is not evidence the limit was proxy-local.
|
|
246
|
+
expect(
|
|
247
|
+
isLitellmProxyLocal429(
|
|
248
|
+
"litellm.RateLimitError: RateLimitError: This request would exceed your account's rate limit.",
|
|
249
|
+
),
|
|
250
|
+
).toBe(false)
|
|
251
|
+
})
|
|
252
|
+
|
|
253
|
+
it('never throws on weird input', () => {
|
|
254
|
+
expect(isLitellmProxyLocal429('')).toBe(false)
|
|
255
|
+
expect(isLitellmProxyLocal429(undefined as unknown as string)).toBe(false)
|
|
256
|
+
expect(isLitellmProxyLocal429(42 as unknown as string)).toBe(false)
|
|
257
|
+
// Slice guard: signal past the 16KB sample is not scanned.
|
|
258
|
+
expect(isLitellmProxyLocal429('A'.repeat(100_000) + LITELLM_DEPLOYMENT_CAP_BODY)).toBe(false)
|
|
259
|
+
})
|
|
260
|
+
})
|
|
261
|
+
|
|
262
|
+
describe('parseLitellmLimitDetail — instrumentation extraction', () => {
|
|
263
|
+
const NOW = new Date(Date.UTC(2026, 6, 12, 8, 0, 0))
|
|
264
|
+
|
|
265
|
+
it('extracts tpm limit + current usage from the deployment-cap body', () => {
|
|
266
|
+
const d = parseLitellmLimitDetail(LITELLM_DEPLOYMENT_CAP_BODY, NOW)
|
|
267
|
+
expect(d.limitType).toBe('tpm')
|
|
268
|
+
expect(d.limit).toBe(8000)
|
|
269
|
+
expect(d.currentUsage).toBe(8241)
|
|
270
|
+
expect(d.resetAtMs).toBeNull()
|
|
271
|
+
})
|
|
272
|
+
|
|
273
|
+
it('extracts rpm limit from the v2 strategy body', () => {
|
|
274
|
+
const d = parseLitellmLimitDetail(LITELLM_V2_STRATEGY_BODY, NOW)
|
|
275
|
+
expect(d.limitType).toBe('rpm')
|
|
276
|
+
expect(d.limit).toBe(60)
|
|
277
|
+
expect(d.currentUsage).toBe(61)
|
|
278
|
+
})
|
|
279
|
+
|
|
280
|
+
it('extracts the v3 limiter limit type, limit, and "Limit resets at" UTC timestamp', () => {
|
|
281
|
+
const d = parseLitellmLimitDetail(LITELLM_V3_KEY_LIMIT_BODY, NOW)
|
|
282
|
+
expect(d.limitType).toBe('tokens')
|
|
283
|
+
expect(d.limit).toBe(8000)
|
|
284
|
+
expect(d.resetAtMs).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
|
|
285
|
+
})
|
|
286
|
+
|
|
287
|
+
it('extracts the router-cooldown "Try again in N seconds" relative reset', () => {
|
|
288
|
+
const d = parseLitellmLimitDetail(LITELLM_ROUTER_COOLDOWN_BODY, NOW)
|
|
289
|
+
expect(d.resetAtMs).toBe(NOW.getTime() + 27_500)
|
|
290
|
+
})
|
|
291
|
+
|
|
292
|
+
it('extracts the v1 limiter colon-form rpm limit', () => {
|
|
293
|
+
const d = parseLitellmLimitDetail(LITELLM_V1_PARALLEL_BODY, NOW)
|
|
294
|
+
expect(d.limitType).toBe('rpm')
|
|
295
|
+
expect(d.limit).toBe(60)
|
|
296
|
+
})
|
|
297
|
+
|
|
298
|
+
it('returns all-null on non-LiteLLM prose and never throws on weird input', () => {
|
|
299
|
+
const empty = { limitType: null, limit: null, currentUsage: null, resetAtMs: null }
|
|
300
|
+
expect(parseLitellmLimitDetail('Please try again later.', NOW)).toEqual(empty)
|
|
301
|
+
expect(parseLitellmLimitDetail('', NOW)).toEqual(empty)
|
|
302
|
+
expect(parseLitellmLimitDetail(undefined as unknown as string, NOW)).toEqual(empty)
|
|
303
|
+
expect(parseLitellmLimitDetail({} as unknown as string, NOW)).toEqual(empty)
|
|
304
|
+
})
|
|
305
|
+
})
|
|
306
|
+
|
|
120
307
|
describe('detectModelUnavailable — network failures', () => {
|
|
121
308
|
it('classifies ECONNREFUSED', () => {
|
|
122
309
|
expect(detectModelUnavailable('connect ECONNREFUSED 1.2.3.4:443')?.kind).toBe('network')
|
|
@@ -198,6 +198,61 @@ describe('detectErrorInTranscriptLine — error detection', () => {
|
|
|
198
198
|
expect(detection).toBeNull()
|
|
199
199
|
})
|
|
200
200
|
|
|
201
|
+
// A 429 whose body carries LiteLLM-proxy-LOCAL limiter wording is the
|
|
202
|
+
// proxy's own tpm_limit/rpm_limit cap tripping BEFORE the request reached
|
|
203
|
+
// Anthropic. Nothing about the account is exhausted — it must take the
|
|
204
|
+
// calm rate-limited path (no model-unavailable card, no mark-exhausted,
|
|
205
|
+
// no fleet failover). Prerequisite for enabling litellm tpm caps on the
|
|
206
|
+
// fleet: without this, every cap trip would bench a healthy account.
|
|
207
|
+
it('classifies a LiteLLM-proxy-local 429 as rate-limited, NOT quota-exhausted', () => {
|
|
208
|
+
const litellmBodies = [
|
|
209
|
+
// Deployment tpm cap (model_rate_limit_check.py body, verbatim shape).
|
|
210
|
+
'API Error: 429 Deployment over user-defined ratelimit. tpm limit=8000. ' +
|
|
211
|
+
'current usage=8241. id=abc123def, model_group=claude-fable-5',
|
|
212
|
+
// Virtual-key tpm_limit (parallel_request_limiter_v3.py).
|
|
213
|
+
'API Error: 429 Rate limit exceeded for api_key: hashed-key-1a2b3c. ' +
|
|
214
|
+
'Limit type: tokens. Current limit: 8000, Remaining: 0. ' +
|
|
215
|
+
'Limit resets at: 2026-07-12 08:05:00 UTC',
|
|
216
|
+
// Router cooldown (RouterRateLimitError).
|
|
217
|
+
'API Error: 429 No deployments available for selected model, ' +
|
|
218
|
+
'Try again in 27.5 seconds. Passed model=claude-fable-5.',
|
|
219
|
+
// Team-level model cap — a v3 descriptor OUTSIDE the enumerated signal
|
|
220
|
+
// list, covered only by the co-occurrence pair
|
|
221
|
+
// (litellmV3LimiterSignalPair). Pre-fix this missed every signal →
|
|
222
|
+
// quota-exhausted → scary card + failover.
|
|
223
|
+
'API Error: 429 Rate limit exceeded for model_per_team: ' +
|
|
224
|
+
'team-1a2b3c:claude-fable-5. Limit type: tokens. ' +
|
|
225
|
+
'Current limit: 50000, Remaining: 0. ' +
|
|
226
|
+
'Limit resets at: 2026-07-12 08:05:00 UTC',
|
|
227
|
+
]
|
|
228
|
+
for (const text of litellmBodies) {
|
|
229
|
+
const line = JSON.stringify({
|
|
230
|
+
type: 'assistant',
|
|
231
|
+
message: {
|
|
232
|
+
role: 'assistant',
|
|
233
|
+
model: '<synthetic>',
|
|
234
|
+
content: [{ type: 'text', text }],
|
|
235
|
+
},
|
|
236
|
+
error: 'rate_limit',
|
|
237
|
+
isApiErrorMessage: true,
|
|
238
|
+
apiErrorStatus: 429,
|
|
239
|
+
})
|
|
240
|
+
const result = detectErrorInTranscriptLine(line)
|
|
241
|
+
expect(result).not.toBeNull()
|
|
242
|
+
expect(result!.kind).toBe('rate-limited')
|
|
243
|
+
expect(result!.transient).toBe(true)
|
|
244
|
+
// End-to-end: the resolver must NOT produce a model-unavailable card —
|
|
245
|
+
// null is what keeps the calm 🚦 card and blocks the auto-fallback
|
|
246
|
+
// branch in emitGatewayOperatorEvent.
|
|
247
|
+
expect(
|
|
248
|
+
resolveModelUnavailableFromOperatorEvent({
|
|
249
|
+
kind: result!.kind,
|
|
250
|
+
detail: result!.detail,
|
|
251
|
+
}),
|
|
252
|
+
).toBeNull()
|
|
253
|
+
}
|
|
254
|
+
})
|
|
255
|
+
|
|
201
256
|
// Guard against over-correcting: a GENUINE quota wall (no transient marker)
|
|
202
257
|
// must STILL be quota-exhausted AND still resolve to a card.
|
|
203
258
|
it('a genuine quota-wall 429 still produces the quota-exhausted card', () => {
|
|
@@ -91,6 +91,30 @@ describe('runtime-metrics — JSONL sink', () => {
|
|
|
91
91
|
expect(typeof parsed.ts).toBe('number')
|
|
92
92
|
})
|
|
93
93
|
|
|
94
|
+
it('rate_limit_429_classified carries classification + action + limit/reset detail', () => {
|
|
95
|
+
emitRuntimeMetric({
|
|
96
|
+
kind: 'rate_limit_429_classified',
|
|
97
|
+
agent: 'carrie',
|
|
98
|
+
classification: 'litellm-local',
|
|
99
|
+
action: 'calm',
|
|
100
|
+
reset_at_ms: 1_783_850_700_000,
|
|
101
|
+
reset_in_ms: 300_000,
|
|
102
|
+
limit_type: 'tpm',
|
|
103
|
+
limit: 8000,
|
|
104
|
+
current_usage: 8241,
|
|
105
|
+
})
|
|
106
|
+
const parsed = JSON.parse(readFileSync(metricsPath, 'utf-8').trim())
|
|
107
|
+
expect(parsed.kind).toBe('rate_limit_429_classified')
|
|
108
|
+
expect(parsed.agent).toBe('carrie')
|
|
109
|
+
expect(parsed.classification).toBe('litellm-local')
|
|
110
|
+
expect(parsed.action).toBe('calm')
|
|
111
|
+
expect(parsed.reset_in_ms).toBe(300_000)
|
|
112
|
+
expect(parsed.limit_type).toBe('tpm')
|
|
113
|
+
expect(parsed.limit).toBe(8000)
|
|
114
|
+
expect(parsed.current_usage).toBe(8241)
|
|
115
|
+
expect(typeof parsed.ts).toBe('number')
|
|
116
|
+
})
|
|
117
|
+
|
|
94
118
|
it('appends — does not overwrite — across calls', () => {
|
|
95
119
|
for (let i = 0; i < 5; i++) {
|
|
96
120
|
emitRuntimeMetric({
|
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
|
|
13
13
|
import { describe, it, expect } from 'vitest'
|
|
14
14
|
import {
|
|
15
|
+
build429ClassifiedMetric,
|
|
16
|
+
classify429Detail,
|
|
15
17
|
decideThrottleTier,
|
|
16
18
|
evaluateThrottleNotice,
|
|
17
19
|
isAccountScopedThrottle,
|
|
@@ -136,6 +138,180 @@ describe('decideThrottleTier — decision matrix', () => {
|
|
|
136
138
|
})
|
|
137
139
|
})
|
|
138
140
|
|
|
141
|
+
// Verbatim LiteLLM proxy-local 429 shapes (provenance on
|
|
142
|
+
// `litellmProxyLocal429Signals`, model-unavailable.ts).
|
|
143
|
+
const LITELLM_TPM_CAP =
|
|
144
|
+
'Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241. ' +
|
|
145
|
+
'id=abc123def, model_group=claude-fable-5'
|
|
146
|
+
const LITELLM_KEY_LIMIT =
|
|
147
|
+
'Rate limit exceeded for api_key: hashed-key-1a2b3c. Limit type: tokens. ' +
|
|
148
|
+
'Current limit: 8000, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC'
|
|
149
|
+
const LITELLM_ROUTER_COOLDOWN =
|
|
150
|
+
'No deployments available for selected model, Try again in 27.5 seconds. ' +
|
|
151
|
+
"Passed model=claude-fable-5. pre-call-checks=False, cooldown_list=['abc123def']"
|
|
152
|
+
// v3 limiter, descriptor NOT in the enumerated signal list — matched by the
|
|
153
|
+
// litellmV3LimiterSignalPair co-occurrence rule (model-unavailable.ts).
|
|
154
|
+
const LITELLM_TEAM_MODEL_LIMIT =
|
|
155
|
+
'Rate limit exceeded for model_per_team: team-1a2b3c:claude-fable-5. ' +
|
|
156
|
+
'Limit type: tokens. Current limit: 50000, Remaining: 0. ' +
|
|
157
|
+
'Limit resets at: 2026-07-12 08:05:00 UTC'
|
|
158
|
+
|
|
159
|
+
describe('decideThrottleTier — LiteLLM-proxy-local 429s never enter the tier', () => {
|
|
160
|
+
// OUTCOME pin: `none` is what keeps a proxy-local cap trip off the
|
|
161
|
+
// account-scoped machinery — no broker mark-throttled, no throttle-tier
|
|
162
|
+
// runner fire, no failover escalation. The gateway additionally gates on
|
|
163
|
+
// classify429Detail, but the decision module must agree.
|
|
164
|
+
it.each([
|
|
165
|
+
['deployment tpm cap', LITELLM_TPM_CAP],
|
|
166
|
+
['virtual-key tpm_limit', LITELLM_KEY_LIMIT],
|
|
167
|
+
['router cooldown', LITELLM_ROUTER_COOLDOWN],
|
|
168
|
+
['team model cap (non-enumerated v3 descriptor)', LITELLM_TEAM_MODEL_LIMIT],
|
|
169
|
+
])('%s → none (calm path owns it)', (_name, detail) => {
|
|
170
|
+
expect(decideThrottleTier({ detail, now: NOW, thresholdMs: THRESHOLD })).toEqual({
|
|
171
|
+
action: 'none',
|
|
172
|
+
})
|
|
173
|
+
})
|
|
174
|
+
})
|
|
175
|
+
|
|
176
|
+
describe('classify429Detail — three-way origin classification', () => {
|
|
177
|
+
it('classifies LiteLLM limiter wordings as litellm-local', () => {
|
|
178
|
+
expect(classify429Detail(LITELLM_TPM_CAP)).toBe('litellm-local')
|
|
179
|
+
expect(classify429Detail(LITELLM_KEY_LIMIT)).toBe('litellm-local')
|
|
180
|
+
expect(classify429Detail(LITELLM_ROUTER_COOLDOWN)).toBe('litellm-local')
|
|
181
|
+
// Non-enumerated v3 descriptor — the co-occurrence pair, not a list
|
|
182
|
+
// entry, carries this one.
|
|
183
|
+
expect(classify429Detail(LITELLM_TEAM_MODEL_LIMIT)).toBe('litellm-local')
|
|
184
|
+
})
|
|
185
|
+
|
|
186
|
+
it('classifies Anthropic account-affirming wording as account-scoped', () => {
|
|
187
|
+
expect(classify429Detail(TRANSIENT('resets in 3m'))).toBe('account-scoped')
|
|
188
|
+
expect(
|
|
189
|
+
classify429Detail('would exceed your account’s rate limit'),
|
|
190
|
+
).toBe('account-scoped')
|
|
191
|
+
})
|
|
192
|
+
|
|
193
|
+
it('classifies server-side / bare rate-limit wording as generic-transient', () => {
|
|
194
|
+
expect(
|
|
195
|
+
classify429Detail('Server is temporarily limiting requests (not your usage limit)'),
|
|
196
|
+
).toBe('generic-transient')
|
|
197
|
+
expect(classify429Detail('overloaded_error 529')).toBe('generic-transient')
|
|
198
|
+
expect(classify429Detail('rate_limit_error: try again later')).toBe('generic-transient')
|
|
199
|
+
})
|
|
200
|
+
|
|
201
|
+
it('TIE-BREAK: account-affirming wording wins over LiteLLM wording when both appear', () => {
|
|
202
|
+
// The pass-through shape: LiteLLM wraps a FORWARDED upstream Anthropic
|
|
203
|
+
// account 429 (exception mapping / limiter prose in the same body).
|
|
204
|
+
// LiteLLM never emits the account wording itself, so its presence means
|
|
205
|
+
// Anthropic really throttled the account — the broker throttle mark and
|
|
206
|
+
// the throttle tier must still run.
|
|
207
|
+
const mixed =
|
|
208
|
+
`litellm.RateLimitError: ${LITELLM_TPM_CAP} — upstream said: ` +
|
|
209
|
+
"This request would exceed your account's rate limit. Please try again later."
|
|
210
|
+
expect(classify429Detail(mixed)).toBe('account-scoped')
|
|
211
|
+
// And the tier decision still engages (not 'none').
|
|
212
|
+
expect(
|
|
213
|
+
decideThrottleTier({ detail: mixed, now: NOW, thresholdMs: THRESHOLD }).action,
|
|
214
|
+
).not.toBe('none')
|
|
215
|
+
})
|
|
216
|
+
|
|
217
|
+
it('a bare litellm.RateLimitError wrapper WITHOUT limiter wording stays generic-transient', () => {
|
|
218
|
+
// The exception-mapping prefix alone is not proxy-local evidence.
|
|
219
|
+
expect(
|
|
220
|
+
classify429Detail('litellm.RateLimitError: RateLimitError: 429 try again later'),
|
|
221
|
+
).toBe('generic-transient')
|
|
222
|
+
})
|
|
223
|
+
|
|
224
|
+
it('never throws on weird input', () => {
|
|
225
|
+
expect(classify429Detail('')).toBe('generic-transient')
|
|
226
|
+
expect(classify429Detail(undefined as unknown as string)).toBe('generic-transient')
|
|
227
|
+
expect(classify429Detail(9000 as unknown as string)).toBe('generic-transient')
|
|
228
|
+
expect(classify429Detail('A'.repeat(200_000))).toBe('generic-transient')
|
|
229
|
+
})
|
|
230
|
+
})
|
|
231
|
+
|
|
232
|
+
describe('build429ClassifiedMetric — instrumentation payload', () => {
|
|
233
|
+
it('litellm-local deployment cap → limit fields + calm action', () => {
|
|
234
|
+
const m = build429ClassifiedMetric({
|
|
235
|
+
agent: 'carrie',
|
|
236
|
+
detail: LITELLM_TPM_CAP,
|
|
237
|
+
classification: 'litellm-local',
|
|
238
|
+
action: 'calm',
|
|
239
|
+
now: NOW,
|
|
240
|
+
})
|
|
241
|
+
expect(m).toEqual({
|
|
242
|
+
kind: 'rate_limit_429_classified',
|
|
243
|
+
agent: 'carrie',
|
|
244
|
+
classification: 'litellm-local',
|
|
245
|
+
action: 'calm',
|
|
246
|
+
reset_at_ms: null,
|
|
247
|
+
reset_in_ms: null,
|
|
248
|
+
limit_type: 'tpm',
|
|
249
|
+
limit: 8000,
|
|
250
|
+
current_usage: 8241,
|
|
251
|
+
})
|
|
252
|
+
})
|
|
253
|
+
|
|
254
|
+
it('litellm-local NON-enumerated v3 descriptor (model_per_team) → full metric payload', () => {
|
|
255
|
+
// Finding-pin: a team-level model cap must produce the metric (it used
|
|
256
|
+
// to miss every signal → quota-exhausted → no metric at all).
|
|
257
|
+
const m = build429ClassifiedMetric({
|
|
258
|
+
agent: 'carrie',
|
|
259
|
+
detail: LITELLM_TEAM_MODEL_LIMIT,
|
|
260
|
+
classification: 'litellm-local',
|
|
261
|
+
action: 'calm',
|
|
262
|
+
now: NOW,
|
|
263
|
+
})
|
|
264
|
+
expect(m.kind).toBe('rate_limit_429_classified')
|
|
265
|
+
expect(m.classification).toBe('litellm-local')
|
|
266
|
+
expect(m.limit_type).toBe('tokens')
|
|
267
|
+
expect(m.limit).toBe(50_000)
|
|
268
|
+
expect(m.reset_at_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
|
|
269
|
+
})
|
|
270
|
+
|
|
271
|
+
it('litellm-local v3 key limit → tokens limit type + parsed "Limit resets at" UTC reset', () => {
|
|
272
|
+
const m = build429ClassifiedMetric({
|
|
273
|
+
agent: 'carrie',
|
|
274
|
+
detail: LITELLM_KEY_LIMIT,
|
|
275
|
+
classification: 'litellm-local',
|
|
276
|
+
action: 'calm',
|
|
277
|
+
now: NOW,
|
|
278
|
+
})
|
|
279
|
+
expect(m.limit_type).toBe('tokens')
|
|
280
|
+
expect(m.limit).toBe(8000)
|
|
281
|
+
expect(m.reset_at_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
|
|
282
|
+
expect(m.reset_in_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0) - NOW)
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
it('account-scoped 429 → Anthropic-parsed reset, no limit fields', () => {
|
|
286
|
+
const m = build429ClassifiedMetric({
|
|
287
|
+
agent: 'carrie',
|
|
288
|
+
detail: TRANSIENT('resets in 3m'),
|
|
289
|
+
classification: 'account-scoped',
|
|
290
|
+
action: 'throttle',
|
|
291
|
+
now: NOW,
|
|
292
|
+
})
|
|
293
|
+
expect(m.classification).toBe('account-scoped')
|
|
294
|
+
expect(m.action).toBe('throttle')
|
|
295
|
+
expect(m.reset_at_ms).toBe(NOW + 3 * 60_000)
|
|
296
|
+
expect(m.reset_in_ms).toBe(3 * 60_000)
|
|
297
|
+
expect(m.limit_type).toBeNull()
|
|
298
|
+
expect(m.limit).toBeNull()
|
|
299
|
+
expect(m.current_usage).toBeNull()
|
|
300
|
+
})
|
|
301
|
+
|
|
302
|
+
it('never throws on weird detail; nulls out unparseable resets', () => {
|
|
303
|
+
const m = build429ClassifiedMetric({
|
|
304
|
+
agent: 'carrie',
|
|
305
|
+
detail: undefined as unknown as string,
|
|
306
|
+
classification: 'generic-transient',
|
|
307
|
+
action: 'calm',
|
|
308
|
+
now: NOW,
|
|
309
|
+
})
|
|
310
|
+
expect(m.reset_at_ms).toBeNull()
|
|
311
|
+
expect(m.reset_in_ms).toBeNull()
|
|
312
|
+
})
|
|
313
|
+
})
|
|
314
|
+
|
|
139
315
|
describe('isAccountScopedThrottle — the tier gate', () => {
|
|
140
316
|
it("matches account-affirming wording (both apostrophe variants + 'not your account')", () => {
|
|
141
317
|
expect(isAccountScopedThrottle("would exceed your account's rate limit")).toBe(true)
|
|
@@ -33,7 +33,12 @@
|
|
|
33
33
|
|
|
34
34
|
import { escapeMarkdown } from './card-format.js'
|
|
35
35
|
import { formatResetRelative } from './quota-check.js'
|
|
36
|
-
import {
|
|
36
|
+
import {
|
|
37
|
+
isLitellmProxyLocal429,
|
|
38
|
+
parseLitellmLimitDetail,
|
|
39
|
+
parseResetTime,
|
|
40
|
+
} from './model-unavailable.js'
|
|
41
|
+
import type { RuntimeMetricEvent } from './runtime-metrics.js'
|
|
37
42
|
|
|
38
43
|
// ─── Account-scoped throttle wording ─────────────────────────────────────────
|
|
39
44
|
|
|
@@ -65,6 +70,98 @@ export function isAccountScopedThrottle(text: string): boolean {
|
|
|
65
70
|
return accountScopedThrottleSignals.some((s) => lower.includes(s))
|
|
66
71
|
}
|
|
67
72
|
|
|
73
|
+
// ─── Three-way 429 classification ────────────────────────────────────────────
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Where a terminal 429-family failure originated:
|
|
77
|
+
* - `account-scoped` — Anthropic throttled THIS account ("would exceed
|
|
78
|
+
* your account's rate limit"). Eligible for the throttle tier below
|
|
79
|
+
* (broker mark-throttled / failover).
|
|
80
|
+
* - `litellm-local` — the LiteLLM proxy's OWN limiter tripped
|
|
81
|
+
* (`tpm_limit`/`rpm_limit` cap, router cooldown — see
|
|
82
|
+
* `litellmProxyLocal429Signals` in model-unavailable.ts). The request
|
|
83
|
+
* never reached Anthropic; account state must not be touched.
|
|
84
|
+
* - `generic-transient` — everything else in the rate-limit family
|
|
85
|
+
* (server-side 429/529 wording, bare `rate_limit_error`). Calm path.
|
|
86
|
+
*/
|
|
87
|
+
export type RateLimit429Classification =
|
|
88
|
+
| 'account-scoped'
|
|
89
|
+
| 'litellm-local'
|
|
90
|
+
| 'generic-transient'
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Classify a terminal `rate-limited` operator event's detail text.
|
|
94
|
+
*
|
|
95
|
+
* TIE-BREAK (both wordings present): account-scoped wins, and only on its
|
|
96
|
+
* EXPLICIT wording. Rationale: LiteLLM never emits the account-affirming
|
|
97
|
+
* strings itself, so when they co-occur with LiteLLM wording the detail is a
|
|
98
|
+
* genuine upstream Anthropic account 429 that traversed (and was wrapped by)
|
|
99
|
+
* the proxy — e.g. the pass-through's "litellm.RateLimitError: …would exceed
|
|
100
|
+
* your account's rate limit…" exception mapping. Classifying that as
|
|
101
|
+
* proxy-local would drop the broker throttle mark and walk every retry
|
|
102
|
+
* straight back into the same account throttle. The reverse risk is nil: a
|
|
103
|
+
* purely proxy-local 429 (limiter fired BEFORE any upstream call) cannot
|
|
104
|
+
* contain Anthropic's account wording.
|
|
105
|
+
*
|
|
106
|
+
* BOUND: operator events forwarded over IPC carry `detail` truncated to
|
|
107
|
+
* 1000 chars (bridge.ts `sendOperatorEvent`'s `.slice(0, 1000)`;
|
|
108
|
+
* `OPERATOR_EVENT_DETAIL_MAX` in gateway/ipc-server.ts), so on that path the
|
|
109
|
+
* tie-break sees only the first 1000 chars of the body. In practice both
|
|
110
|
+
* wordings sit well inside that window (real Anthropic and LiteLLM bodies
|
|
111
|
+
* are <400 chars), and a truncated-away account marker would merely
|
|
112
|
+
* downgrade account-scoped → litellm-local/generic-transient — the calm
|
|
113
|
+
* path, never a wrong account mark.
|
|
114
|
+
*
|
|
115
|
+
* Never throws on weird input (both matchers are total).
|
|
116
|
+
*/
|
|
117
|
+
export function classify429Detail(text: string): RateLimit429Classification {
|
|
118
|
+
if (isAccountScopedThrottle(text)) return 'account-scoped'
|
|
119
|
+
if (isLitellmProxyLocal429(text)) return 'litellm-local'
|
|
120
|
+
return 'generic-transient'
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Build the `rate_limit_429_classified` runtime metric for one terminal
|
|
125
|
+
* rate-limited operator event — the instrumentation that lets an operator
|
|
126
|
+
* correlate Anthropic ACCOUNT 429s with fleet TPM (the prerequisite for
|
|
127
|
+
* enabling LiteLLM `tpm_limit` caps; see docs/auth.md § LiteLLM-proxy-local
|
|
128
|
+
* 429s). Pure builder so the payload shape is unit-testable; the gateway
|
|
129
|
+
* emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
|
|
130
|
+
*
|
|
131
|
+
* `action` is what the gateway decided for this event:
|
|
132
|
+
* - `throttle` / `failover` — the account-scoped throttle tier's decision
|
|
133
|
+
* - `calm` — the existing calm rate-limited path (litellm-local and
|
|
134
|
+
* generic-transient always land here; no broker mark, no failover)
|
|
135
|
+
*
|
|
136
|
+
* Reset detail is best-effort: the Anthropic-shaped `parseResetTime` first,
|
|
137
|
+
* then LiteLLM's own shapes ("Limit resets at: … UTC" / "Try again in Ns")
|
|
138
|
+
* via `parseLitellmLimitDetail`. Limit fields parse only from LiteLLM
|
|
139
|
+
* wording (Anthropic bodies carry no numeric limit).
|
|
140
|
+
*/
|
|
141
|
+
export function build429ClassifiedMetric(opts: {
|
|
142
|
+
agent: string
|
|
143
|
+
detail: string
|
|
144
|
+
classification: RateLimit429Classification
|
|
145
|
+
action: 'throttle' | 'failover' | 'calm'
|
|
146
|
+
now: number
|
|
147
|
+
}): Extract<RuntimeMetricEvent, { kind: 'rate_limit_429_classified' }> {
|
|
148
|
+
const detail = typeof opts.detail === 'string' ? opts.detail : ''
|
|
149
|
+
const litellm = parseLitellmLimitDetail(detail, new Date(opts.now))
|
|
150
|
+
const anthropicResetMs = parseResetTime(detail, new Date(opts.now))?.getTime() ?? null
|
|
151
|
+
const resetAtMs = anthropicResetMs ?? litellm.resetAtMs
|
|
152
|
+
return {
|
|
153
|
+
kind: 'rate_limit_429_classified',
|
|
154
|
+
agent: opts.agent,
|
|
155
|
+
classification: opts.classification,
|
|
156
|
+
action: opts.action,
|
|
157
|
+
reset_at_ms: resetAtMs,
|
|
158
|
+
reset_in_ms: resetAtMs != null ? Math.max(0, resetAtMs - opts.now) : null,
|
|
159
|
+
limit_type: litellm.limitType,
|
|
160
|
+
limit: litellm.limit,
|
|
161
|
+
current_usage: litellm.currentUsage,
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
|
|
68
165
|
/**
|
|
69
166
|
* Default retry-in-place ceiling: a transient 429 whose reset is within this
|
|
70
167
|
* window waits on the account instead of failing the fleet over. Override
|