switchroom 0.18.14 → 0.18.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +3 -0
- package/dist/auth-broker/index.js +473 -49
- package/dist/cli/notion-write-pretool.mjs +3 -0
- package/dist/cli/switchroom.js +1200 -1067
- package/dist/host-control/main.js +56 -51
- package/dist/vault/approvals/kernel-server.js +19 -12
- package/dist/vault/broker/server.js +675 -668
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +81 -139
- package/telegram-plugin/dist/bridge/bridge.js +21 -0
- package/telegram-plugin/dist/gateway/gateway.js +531 -259
- package/telegram-plugin/dist/server.js +22 -1
- package/telegram-plugin/draft-stream.ts +78 -3
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
- package/telegram-plugin/gateway/effort-command.ts +9 -7
- package/telegram-plugin/gateway/gateway.ts +310 -219
- package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
- package/telegram-plugin/gateway/model-command.ts +96 -18
- package/telegram-plugin/gateway/pending-session-command.ts +10 -8
- package/telegram-plugin/gateway/session-model-file.ts +38 -172
- package/telegram-plugin/litellm-local-notice.ts +189 -0
- package/telegram-plugin/model-unavailable.ts +214 -0
- package/telegram-plugin/quota-watch.ts +16 -4
- package/telegram-plugin/runtime-metrics.ts +47 -0
- package/telegram-plugin/send-gate-degraded.test.ts +9 -7
- package/telegram-plugin/send-gate.ts +34 -4
- package/telegram-plugin/session-tail.ts +14 -2
- package/telegram-plugin/stream-controller.ts +143 -20
- package/telegram-plugin/stream-reply-handler.ts +12 -2
- package/telegram-plugin/tests/bot-api.harness.ts +7 -2
- package/telegram-plugin/tests/draft-stream.test.ts +110 -1
- package/telegram-plugin/tests/effort-command.test.ts +4 -4
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
- package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
- package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
- package/telegram-plugin/tests/model-command.test.ts +84 -1
- package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
- package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
- package/telegram-plugin/tests/quota-watch.test.ts +21 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
- package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
- package/telegram-plugin/tests/session-model-file.test.ts +7 -155
- package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
- package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
- package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
- package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
- package/telegram-plugin/throttle-tier.ts +98 -1
- package/telegram-plugin/worker-activity-feed.ts +83 -8
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* litellm-local-notice.ts — debounced operator notice for LiteLLM-proxy-local
|
|
3
|
+
* 429s (pure module: state machine + text + config parsing, no IPC, no bot,
|
|
4
|
+
* no clock except the injected `now`).
|
|
5
|
+
*
|
|
6
|
+
* When an agent routes through the LiteLLM gateway and trips the proxy's OWN
|
|
7
|
+
* `tpm_limit`/`rpm_limit` limiter, `classify429Detail` (throttle-tier.ts)
|
|
8
|
+
* classifies the terminal 429 `litellm-local` and the gateway takes the calm
|
|
9
|
+
* path — no broker mark, no failover, no throttle tier (the request never
|
|
10
|
+
* reached Anthropic, so the condition says nothing about the account). But
|
|
11
|
+
* pre-notice the user-facing surface was the GENERIC "🚦 Rate limited" card,
|
|
12
|
+
* which reads like an Anthropic problem. This module owns the honest
|
|
13
|
+
* replacement:
|
|
14
|
+
*
|
|
15
|
+
* - ONE calm notice per agent per cooldown window (default 15 min,
|
|
16
|
+
* operator-tunable via channels.telegram.litellm_notice.window_ms in
|
|
17
|
+
* switchroom.yaml, projected into access.json by scaffold).
|
|
18
|
+
* - Further litellm-local 429s inside the window are counted SILENTLY;
|
|
19
|
+
* the first notice after the window expires carries "throttled N more
|
|
20
|
+
* times since the last notice".
|
|
21
|
+
* - A notice only ever fires on an actual throttle event — a quiet agent
|
|
22
|
+
* posts nothing (the state machine is evaluate-on-event, no timers).
|
|
23
|
+
*
|
|
24
|
+
* The copy must make clear this is the fleet token limiter (LiteLLM
|
|
25
|
+
* `tpm_limit`/`rpm_limit`), NOT an Anthropic account limit, and that the
|
|
26
|
+
* turn retries — no action needed.
|
|
27
|
+
*
|
|
28
|
+
* Side-effect sequencing lives in gateway/litellm-local-notice-wiring.ts.
|
|
29
|
+
* This module MUST NOT change classification, failover, or quota-ledger
|
|
30
|
+
* behavior — it only renders and debounces the notice.
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
import { escapeMarkdown } from './card-format.js'
|
|
34
|
+
import type { RateLimit429Classification } from './throttle-tier.js'
|
|
35
|
+
import type { RuntimeMetricEvent } from './runtime-metrics.js'
|
|
36
|
+
|
|
37
|
+
// ─── Cooldown window config ──────────────────────────────────────────────────
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Default per-agent notice cooldown. 15 minutes: long enough that a sustained
|
|
41
|
+
* burst produces a handful of notices per hour at most, short enough that the
|
|
42
|
+
* operator still sees the limiter working while it's working.
|
|
43
|
+
*/
|
|
44
|
+
export const LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT = 15 * 60_000
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Resolve the cooldown window from the raw access-file value
|
|
48
|
+
* (`access.litellmNoticeWindowMs`, projected by scaffold from
|
|
49
|
+
* channels.telegram.litellm_notice.window_ms). Any non-finite, non-positive,
|
|
50
|
+
* or non-number value falls back to the default — an operator typo must
|
|
51
|
+
* never produce a zero-width window (notice storm) or a NaN comparison
|
|
52
|
+
* (notices never fire again).
|
|
53
|
+
*/
|
|
54
|
+
export function parseLitellmNoticeWindowMs(raw: unknown): number {
|
|
55
|
+
if (typeof raw !== 'number') return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
|
|
56
|
+
if (!Number.isFinite(raw) || raw <= 0) return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
|
|
57
|
+
return raw
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// ─── Per-agent cooldown state machine ────────────────────────────────────────
|
|
61
|
+
|
|
62
|
+
export interface LitellmLocalNoticeState {
|
|
63
|
+
/** agent → unix ms of the last notice sent for it. */
|
|
64
|
+
lastSentAtMsByAgent: Record<string, number>
|
|
65
|
+
/** agent → litellm-local 429s counted silently since the last notice. */
|
|
66
|
+
suppressedCountByAgent: Record<string, number>
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export function initialLitellmLocalNoticeState(): LitellmLocalNoticeState {
|
|
70
|
+
return { lastSentAtMsByAgent: {}, suppressedCountByAgent: {} }
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface LitellmLocalNoticeVerdict {
|
|
74
|
+
send: boolean
|
|
75
|
+
/** Silent throttle events since the previous notice (0 on the first). Only
|
|
76
|
+
* meaningful when `send` is true — the renderer adds the "N more times"
|
|
77
|
+
* line when > 0. */
|
|
78
|
+
suppressedSinceLastNotice: number
|
|
79
|
+
next: LitellmLocalNoticeState
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Evaluate one litellm-local 429 for `agent` at `now`.
|
|
84
|
+
*
|
|
85
|
+
* - First event ever (or ≥ windowMs since the last notice) → send; the
|
|
86
|
+
* verdict carries the silently-counted events since the previous notice
|
|
87
|
+
* and the counter resets.
|
|
88
|
+
* - Inside the window → suppress; the counter increments.
|
|
89
|
+
*
|
|
90
|
+
* Boundary is INCLUSIVE on expiry (elapsed === windowMs sends), matching
|
|
91
|
+
* `evaluateThrottleNotice` in throttle-tier.ts. Pure: returns the next
|
|
92
|
+
* state, never mutates `prev`.
|
|
93
|
+
*/
|
|
94
|
+
export function evaluateLitellmLocalNotice(
|
|
95
|
+
prev: LitellmLocalNoticeState,
|
|
96
|
+
agent: string,
|
|
97
|
+
now: number,
|
|
98
|
+
windowMs: number = LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT,
|
|
99
|
+
): LitellmLocalNoticeVerdict {
|
|
100
|
+
const last = prev.lastSentAtMsByAgent[agent]
|
|
101
|
+
if (last == null || now - last >= windowMs) {
|
|
102
|
+
return {
|
|
103
|
+
send: true,
|
|
104
|
+
suppressedSinceLastNotice: prev.suppressedCountByAgent[agent] ?? 0,
|
|
105
|
+
next: {
|
|
106
|
+
lastSentAtMsByAgent: { ...prev.lastSentAtMsByAgent, [agent]: now },
|
|
107
|
+
suppressedCountByAgent: { ...prev.suppressedCountByAgent, [agent]: 0 },
|
|
108
|
+
},
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return {
|
|
112
|
+
send: false,
|
|
113
|
+
suppressedSinceLastNotice: 0,
|
|
114
|
+
next: {
|
|
115
|
+
lastSentAtMsByAgent: prev.lastSentAtMsByAgent,
|
|
116
|
+
suppressedCountByAgent: {
|
|
117
|
+
...prev.suppressedCountByAgent,
|
|
118
|
+
[agent]: (prev.suppressedCountByAgent[agent] ?? 0) + 1,
|
|
119
|
+
},
|
|
120
|
+
},
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// ─── Notice rendering ────────────────────────────────────────────────────────
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* The ONE calm notice for a litellm-local 429. Markdown (not HTML) per the
|
|
128
|
+
* card-format.ts convention. Copy contract (operator spec):
|
|
129
|
+
* - names the fleet token limiter (LiteLLM `tpm_limit`/`rpm_limit`)
|
|
130
|
+
* - explicitly NOT an Anthropic account limit — nothing exhausted, no
|
|
131
|
+
* account touched
|
|
132
|
+
* - the turn retries automatically; no action needed
|
|
133
|
+
* - after a window with silent suppressions, says "throttled N more
|
|
134
|
+
* time(s) since the last notice"
|
|
135
|
+
*/
|
|
136
|
+
export function renderLitellmLocalNotice(opts: {
|
|
137
|
+
agent: string
|
|
138
|
+
/** Silent litellm-local 429s since the previous notice (0 on the first). */
|
|
139
|
+
suppressedSinceLastNotice: number
|
|
140
|
+
}): string {
|
|
141
|
+
const agent = escapeMarkdown(opts.agent)
|
|
142
|
+
const n = opts.suppressedSinceLastNotice
|
|
143
|
+
const lines = [
|
|
144
|
+
`🚦 **Fleet token limiter engaged** — **${agent}** hit the local LiteLLM proxy cap (\`tpm_limit\`/\`rpm_limit\`).`,
|
|
145
|
+
`This is switchroom's own fleet limiter smoothing a burst, not an Anthropic account limit — nothing is exhausted and no account was touched.`,
|
|
146
|
+
]
|
|
147
|
+
if (n > 0) {
|
|
148
|
+
lines.push(`Throttled ${n} more ${n === 1 ? 'time' : 'times'} since the last notice.`)
|
|
149
|
+
}
|
|
150
|
+
lines.push(`_The turn retries automatically — no action needed._`)
|
|
151
|
+
return lines.join('\n')
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// ─── Metric builder ──────────────────────────────────────────────────────────
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Build the `litellm_local_429_notice` runtime metric for one SENT notice.
|
|
158
|
+
* Distinct from `rate_limit_429_classified` (which fires on EVERY classified
|
|
159
|
+
* event): this one fires only when a notice actually posts, and carries how
|
|
160
|
+
* many events the window silently absorbed — the debounce's effectiveness in
|
|
161
|
+
* one number. Pure builder so the payload shape is unit-testable; the wiring
|
|
162
|
+
* emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
|
|
163
|
+
*/
|
|
164
|
+
export function buildLitellmLocalNoticeMetric(opts: {
|
|
165
|
+
agent: string
|
|
166
|
+
suppressedCount: number
|
|
167
|
+
windowMs: number
|
|
168
|
+
}): Extract<RuntimeMetricEvent, { kind: 'litellm_local_429_notice' }> {
|
|
169
|
+
return {
|
|
170
|
+
kind: 'litellm_local_429_notice',
|
|
171
|
+
agent: opts.agent,
|
|
172
|
+
suppressed_count: opts.suppressedCount,
|
|
173
|
+
window_ms: opts.windowMs,
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// ─── Classification guard ────────────────────────────────────────────────────
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* True only for the `litellm-local` classification. The wiring runs this
|
|
181
|
+
* guard so "a notice never fires for non-litellm-local classifications" is a
|
|
182
|
+
* deterministic mechanism, not caller discipline — account-scoped 429s keep
|
|
183
|
+
* the throttle tier, generic transients keep the calm rate-limited card.
|
|
184
|
+
*/
|
|
185
|
+
export function isLitellmLocalNoticeEligible(
|
|
186
|
+
classification: RateLimit429Classification | null | undefined,
|
|
187
|
+
): classification is 'litellm-local' {
|
|
188
|
+
return classification === 'litellm-local'
|
|
189
|
+
}
|
|
@@ -92,6 +92,203 @@ export function isTransientUpstreamSignal(text: string): boolean {
|
|
|
92
92
|
return transientUpstreamSignals.some(s => lower.includes(s))
|
|
93
93
|
}
|
|
94
94
|
|
|
95
|
+
// ─── LiteLLM-proxy-LOCAL 429 signals (canonical, single source of truth) ─────
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Explicit markers of a 429 generated by the LiteLLM proxy's OWN rate
|
|
99
|
+
* limiters — a router `tpm_limit`/`rpm_limit` deployment cap, a virtual-key /
|
|
100
|
+
* team / user cap, or an all-deployments-cooling-down router state. The
|
|
101
|
+
* request never reached Anthropic, so the condition is PROXY-LOCAL: it says
|
|
102
|
+
* nothing about the Anthropic account's quota and must never mark the account
|
|
103
|
+
* throttled/exhausted or trigger fleet failover.
|
|
104
|
+
*
|
|
105
|
+
* Each wording key is grounded in LiteLLM's source (verified against
|
|
106
|
+
* BerriAI/litellm `main`, 2026-07), matching how
|
|
107
|
+
* `accountScopedThrottleSignals` (throttle-tier.ts) documents its keys:
|
|
108
|
+
*
|
|
109
|
+
* - "deployment over user-defined ratelimit" —
|
|
110
|
+
* `RouterErrors.user_defined_ratelimit_error` (litellm/types/router.py),
|
|
111
|
+
* the body prefix of every deployment tpm/rpm cap 429 raised by
|
|
112
|
+
* litellm/router_utils/pre_call_checks/model_rate_limit_check.py:
|
|
113
|
+
* "Deployment over user-defined ratelimit. tpm limit={n}. current
|
|
114
|
+
* usage={n}. id={id}, model_group={g}".
|
|
115
|
+
* - "model rate limit exceeded. tpm/rpm limit" — the `message` field of the
|
|
116
|
+
* same enforcement RateLimitError: "Model rate limit exceeded. TPM
|
|
117
|
+
* limit={n}, current usage={n}" (model_rate_limit_check.py).
|
|
118
|
+
* - "deployment over defined rpm limit" — the usage-based-routing-v2
|
|
119
|
+
* strategy wording "Deployment over defined rpm limit={n}. current
|
|
120
|
+
* usage={n}" (litellm/router_strategy/lowest_tpm_rpm_v2.py). There is NO
|
|
121
|
+
* tpm sibling in any known release — v2 TPM exhaustion surfaces as the
|
|
122
|
+
* `no_deployments_available` wording below (the deployment is filtered
|
|
123
|
+
* from the candidate set rather than erroring in the increment path).
|
|
124
|
+
* - "no deployments available for selected model" —
|
|
125
|
+
* `RouterErrors.no_deployments_available`, the RouterRateLimitError
|
|
126
|
+
* message when every deployment for the model group is cooling down:
|
|
127
|
+
* "No deployments available for selected model, Try again in {n}
|
|
128
|
+
* seconds…" (litellm/types/router.py).
|
|
129
|
+
* - "litellm rate limit handler" — the v1 parallel-request limiter's
|
|
130
|
+
* ProxyRateLimitError detail prefix: "LiteLLM Rate Limit Handler for
|
|
131
|
+
* rate limit type = {t}. Crossed TPM / RPM / Max Parallel Request
|
|
132
|
+
* Limit. current rpm: {n}, rpm limit: {n}…"
|
|
133
|
+
* (litellm/proxy/hooks/parallel_request_limiter.py).
|
|
134
|
+
* - "crossed tpm / rpm" — `CommonProxyErrors
|
|
135
|
+
* .max_parallel_request_limit_reached.value` = "Crossed TPM / RPM / Max
|
|
136
|
+
* Parallel Request Limit" (litellm/proxy/_types.py), interpolated into
|
|
137
|
+
* BOTH v1 detail shapes, so it also covers the zero-limit branch's
|
|
138
|
+
* "Max parallel request limit reached {additional_details}" message.
|
|
139
|
+
* - "max parallel request limit reached" — the standalone prefix
|
|
140
|
+
* `raise_rate_limit_error` builds when a key's limit is set to 0
|
|
141
|
+
* (parallel_request_limiter.py `error_message`).
|
|
142
|
+
*
|
|
143
|
+
* The proxy-side per-key/team/user limiter (parallel_request_limiter_v3.py
|
|
144
|
+
* `_handle_rate_limit_error`) is matched by CO-OCCURRENCE instead of a
|
|
145
|
+
* per-descriptor entry — see `isLitellmProxyLocal429`. Its detail shape:
|
|
146
|
+
* "Rate limit exceeded for {descriptor}: {value}. Limit type: {t}. Current
|
|
147
|
+
* limit: {n}, Remaining: {n}. Limit resets at: {ts}". Descriptor keys in
|
|
148
|
+
* source are numerous and growing (api_key / user / team / team_member /
|
|
149
|
+
* organization / end_user / agent / agent_session / model_per_key /
|
|
150
|
+
* model_per_team / model_per_organization / model_per_project / tag_per_key
|
|
151
|
+
* / mcp_per_key / mcp_per_team as of 2026-07), so enumerating them is a
|
|
152
|
+
* treadmill: the pair "rate limit exceeded for " + "limit type:" appears in
|
|
153
|
+
* every v3 body and in no Anthropic error. This is the shape a `tpm_limit`
|
|
154
|
+
* on the per-agent virtual keys (src/litellm/provision.ts) trips.
|
|
155
|
+
*
|
|
156
|
+
* Deliberately EXCLUDED: the bare exception-mapping prefix
|
|
157
|
+
* "litellm.RateLimitError:". LiteLLM wraps FORWARDED upstream 429s with the
|
|
158
|
+
* same prefix on the pass-through, so its presence is NOT evidence the limit
|
|
159
|
+
* was proxy-local — a genuine Anthropic account throttle traversing the
|
|
160
|
+
* proxy must keep its account-scoped classification (see
|
|
161
|
+
* `classify429Detail` in throttle-tier.ts for the tie-break).
|
|
162
|
+
*/
|
|
163
|
+
export const litellmProxyLocal429Signals = [
|
|
164
|
+
'deployment over user-defined ratelimit',
|
|
165
|
+
'model rate limit exceeded. tpm limit',
|
|
166
|
+
'model rate limit exceeded. rpm limit',
|
|
167
|
+
'deployment over defined rpm limit',
|
|
168
|
+
'no deployments available for selected model',
|
|
169
|
+
'litellm rate limit handler',
|
|
170
|
+
'crossed tpm / rpm',
|
|
171
|
+
'max parallel request limit reached',
|
|
172
|
+
]
|
|
173
|
+
|
|
174
|
+
/**
|
|
175
|
+
* The v3 proxy-limiter co-occurrence pair (see the provenance comment on
|
|
176
|
+
* `litellmProxyLocal429Signals`): both substrings appear in every
|
|
177
|
+
* parallel_request_limiter_v3 429 body regardless of descriptor key, and
|
|
178
|
+
* never in an Anthropic error. Exported so tests can pin the rule.
|
|
179
|
+
*/
|
|
180
|
+
export const litellmV3LimiterSignalPair = ['rate limit exceeded for ', 'limit type:'] as const
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* True when `text` carries an EXPLICIT LiteLLM-proxy-local rate-limit marker:
|
|
184
|
+
* one of `litellmProxyLocal429Signals`, OR the v3 limiter co-occurrence pair
|
|
185
|
+
* (`litellmV3LimiterSignalPair` — descriptor-agnostic, so new v3 descriptor
|
|
186
|
+
* keys are covered without a list update). Never throws on weird input. Pure
|
|
187
|
+
* wording detection only — precedence against account-scoped wording is
|
|
188
|
+
* owned by `classify429Detail` (throttle-tier.ts).
|
|
189
|
+
*/
|
|
190
|
+
export function isLitellmProxyLocal429(text: string): boolean {
|
|
191
|
+
if (typeof text !== 'string' || text.length === 0) return false
|
|
192
|
+
const sample = text.length > 16_384 ? text.slice(0, 16_384) : text
|
|
193
|
+
const lower = sample.toLowerCase()
|
|
194
|
+
if (litellmProxyLocal429Signals.some(s => lower.includes(s))) return true
|
|
195
|
+
return litellmV3LimiterSignalPair.every(s => lower.includes(s))
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
/**
|
|
199
|
+
* Best-effort extraction of the limit detail LiteLLM embeds in its
|
|
200
|
+
* proxy-local 429 bodies, for instrumentation (the `rate_limit_429_classified`
|
|
201
|
+
* runtime metric). All fields null when nothing parseable is found — never
|
|
202
|
+
* throws. Shapes covered (same provenance as `litellmProxyLocal429Signals`):
|
|
203
|
+
*
|
|
204
|
+
* - "tpm limit=8000. current usage=8241" (model_rate_limit_check body)
|
|
205
|
+
* - "TPM limit=8000, current usage=8241" (model_rate_limit_check message)
|
|
206
|
+
* - "Deployment over defined rpm limit=60. current usage=61" (v2 strategy)
|
|
207
|
+
* - "Limit type: tokens. Current limit: 8000 … Limit resets at:
|
|
208
|
+
* 2026-07-12 08:05:00 UTC" (parallel_request_limiter_v3)
|
|
209
|
+
* - "Try again in 27.5 seconds" (RouterRateLimitError cooldown)
|
|
210
|
+
*
|
|
211
|
+
* CAVEAT — the v3 "Limit resets at" timestamp is NOT reliably UTC despite
|
|
212
|
+
* the literal: litellm formats it with a naive `datetime.fromtimestamp(...)
|
|
213
|
+
* .strftime("%Y-%m-%d %H:%M:%S UTC")` (parallel_request_limiter_v3.py
|
|
214
|
+
* `_handle_rate_limit_error`), i.e. proxy-LOCAL wall-clock time with a
|
|
215
|
+
* hard-coded "UTC" suffix. We parse it as UTC (nothing better is possible
|
|
216
|
+
* from the body alone), so `resetAtMs` — and the metric fields derived from
|
|
217
|
+
* it — can be skewed by the proxy host's UTC offset when the proxy doesn't
|
|
218
|
+
* run on UTC. Treat v3-derived resets as approximate; don't chase phantom
|
|
219
|
+
* clock drift from these metrics. The rate-limit windows are ≤1 minute
|
|
220
|
+
* anyway, so the absolute timestamp is informational.
|
|
221
|
+
*/
|
|
222
|
+
export function parseLitellmLimitDetail(
|
|
223
|
+
text: string,
|
|
224
|
+
parseTimeNow: Date = new Date(),
|
|
225
|
+
): {
|
|
226
|
+
limitType: string | null
|
|
227
|
+
limit: number | null
|
|
228
|
+
currentUsage: number | null
|
|
229
|
+
resetAtMs: number | null
|
|
230
|
+
} {
|
|
231
|
+
const empty = { limitType: null, limit: null, currentUsage: null, resetAtMs: null }
|
|
232
|
+
if (typeof text !== 'string' || text.length === 0) return empty
|
|
233
|
+
const sample = text.length > 16_384 ? text.slice(0, 16_384) : text
|
|
234
|
+
const lower = sample.toLowerCase()
|
|
235
|
+
|
|
236
|
+
// "tpm limit=8000" / "rpm limit=60" (body), "TPM limit=8000" (message),
|
|
237
|
+
// and the v1 limiter's colon form "rpm limit: 60".
|
|
238
|
+
let limitType: string | null = null
|
|
239
|
+
let limit: number | null = null
|
|
240
|
+
const eqLimit = lower.match(/\b([tr]pm)[ _]limit[=:]\s*(\d+)/)
|
|
241
|
+
if (eqLimit) {
|
|
242
|
+
limitType = eqLimit[1]
|
|
243
|
+
limit = Number(eqLimit[2])
|
|
244
|
+
}
|
|
245
|
+
// v3 limiter: "Limit type: tokens. Current limit: 8000"
|
|
246
|
+
if (limitType == null) {
|
|
247
|
+
const v3Type = lower.match(/limit type:\s*(tokens|requests|max_parallel_requests)/)
|
|
248
|
+
if (v3Type) limitType = v3Type[1]
|
|
249
|
+
}
|
|
250
|
+
if (limit == null) {
|
|
251
|
+
const v3Limit = lower.match(/current limit:\s*(\d+)/)
|
|
252
|
+
if (v3Limit) limit = Number(v3Limit[1])
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// "current usage=8241" (body/message, both `=` and `= ` never occur with
|
|
256
|
+
// separators other than the literal shown in source).
|
|
257
|
+
let currentUsage: number | null = null
|
|
258
|
+
const usage = lower.match(/current usage=(\d+)/)
|
|
259
|
+
if (usage) currentUsage = Number(usage[1])
|
|
260
|
+
|
|
261
|
+
// Reset hints LiteLLM emits that the Anthropic-shaped parseResetTime does
|
|
262
|
+
// not cover: "Limit resets at: 2026-07-12 08:05:00 UTC" (v3 limiter — note
|
|
263
|
+
// the space separator, not ISO 'T') and "Try again in 27.5 seconds"
|
|
264
|
+
// (RouterRateLimitError).
|
|
265
|
+
let resetAtMs: number | null = null
|
|
266
|
+
const resetsAt = sample.match(
|
|
267
|
+
/limit resets at:\s*(\d{4}-\d{2}-\d{2}) (\d{2}:\d{2}:\d{2}) UTC/i,
|
|
268
|
+
)
|
|
269
|
+
if (resetsAt) {
|
|
270
|
+
const d = new Date(`${resetsAt[1]}T${resetsAt[2]}Z`)
|
|
271
|
+
if (!Number.isNaN(d.getTime())) resetAtMs = d.getTime()
|
|
272
|
+
}
|
|
273
|
+
if (resetAtMs == null) {
|
|
274
|
+
const tryAgain = lower.match(/try again in\s+(\d+(?:\.\d+)?)\s*seconds/)
|
|
275
|
+
if (tryAgain) {
|
|
276
|
+
const secs = Number(tryAgain[1])
|
|
277
|
+
if (Number.isFinite(secs) && secs > 0 && secs < 7 * 24 * 3600) {
|
|
278
|
+
resetAtMs = parseTimeNow.getTime() + Math.round(secs * 1000)
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
return {
|
|
284
|
+
limitType,
|
|
285
|
+
limit: limit != null && Number.isFinite(limit) ? limit : null,
|
|
286
|
+
currentUsage:
|
|
287
|
+
currentUsage != null && Number.isFinite(currentUsage) ? currentUsage : null,
|
|
288
|
+
resetAtMs,
|
|
289
|
+
}
|
|
290
|
+
}
|
|
291
|
+
|
|
95
292
|
// ─── Detection ───────────────────────────────────────────────────────────────
|
|
96
293
|
|
|
97
294
|
/**
|
|
@@ -138,6 +335,23 @@ export function detectModelUnavailable(
|
|
|
138
335
|
: { kind: 'overload', raw: stderr }
|
|
139
336
|
}
|
|
140
337
|
|
|
338
|
+
// ── 0.5. LiteLLM-proxy-LOCAL 429 — never a quota signal ─────────────────
|
|
339
|
+
// A 429 raised by the LiteLLM proxy's own limiters (`tpm_limit`/`rpm_limit`
|
|
340
|
+
// caps, router cooldown) never reached Anthropic — nothing about the
|
|
341
|
+
// account is exhausted, so it must classify to the calm retryable kind
|
|
342
|
+
// BEFORE the quota substrings run (some LiteLLM bodies contain the word
|
|
343
|
+
// "limit", which a negation-blind quota match could seize on). Runs after
|
|
344
|
+
// step 0 deliberately: account-affirming wording, when present, can only
|
|
345
|
+
// have originated upstream (LiteLLM never emits it) and wins — see the
|
|
346
|
+
// tie-break note on `classify429Detail` (throttle-tier.ts). Uses the full
|
|
347
|
+
// matcher (list + v3 co-occurrence pair) so every descriptor is covered.
|
|
348
|
+
if (isLitellmProxyLocal429(sample)) {
|
|
349
|
+
const resetAt = parseResetTime(sample)
|
|
350
|
+
return resetAt !== undefined
|
|
351
|
+
? { kind: 'overload', resetAt, raw: stderr }
|
|
352
|
+
: { kind: 'overload', raw: stderr }
|
|
353
|
+
}
|
|
354
|
+
|
|
141
355
|
// ── 1. Quota / billing exhaustion ──────────────────────────────────────
|
|
142
356
|
const quotaSignals = [
|
|
143
357
|
'out of extra usage',
|
|
@@ -192,10 +192,14 @@ export interface FleetRollInfo {
|
|
|
192
192
|
/**
|
|
193
193
|
* Trigger attribution (#3031 PR 2): "soft-avoid" = proactive
|
|
194
194
|
* serving-preference roll off an account APPROACHING its limits;
|
|
195
|
-
* "hard-exhaustion" = the probe saw a genuine quota wall
|
|
196
|
-
*
|
|
195
|
+
* "hard-exhaustion" = the probe saw a genuine quota wall;
|
|
196
|
+
* "model-tier-wall" = the flagship-tier (7d_oi) canary saw the account
|
|
197
|
+
* walled on the premium tier (#3176). Absent on pre-PR-2 brokers —
|
|
198
|
+
* rendered as hard exhaustion.
|
|
197
199
|
*/
|
|
198
|
-
reason?: "soft-avoid" | "hard-exhaustion";
|
|
200
|
+
reason?: "soft-avoid" | "hard-exhaustion" | "model-tier-wall";
|
|
201
|
+
/** #3176 — the binding tier bucket, set only when reason is model-tier-wall. */
|
|
202
|
+
bucket?: string;
|
|
199
203
|
}
|
|
200
204
|
|
|
201
205
|
export type FleetRollAnnounceDecision =
|
|
@@ -241,9 +245,17 @@ export function buildFleetRollMessage(roll: FleetRollInfo, now: number): string
|
|
|
241
245
|
? ` (resets ${formatRelative(new Date(roll.exhausted_until), new Date(now))})`
|
|
242
246
|
: "";
|
|
243
247
|
const softAvoid = roll.reason === "soft-avoid";
|
|
248
|
+
// #3176 — a model-tier wall binds on the flagship (premium) 7d_oi bucket, not
|
|
249
|
+
// the 5h/7d windows (which read healthy — the whole point of the bug), so it
|
|
250
|
+
// has no window/pct to cite. Name the flagship tier explicitly and reassure
|
|
251
|
+
// that opus/haiku on the walled account are unaffected (the mark is
|
|
252
|
+
// tier-scoped), rather than rendering a generic, misleading "quota window".
|
|
253
|
+
const modelTierWall = roll.reason === "model-tier-wall";
|
|
244
254
|
const causeLine = softAvoid
|
|
245
255
|
? `Proactive switch — \`${codeSpanSafe(roll.from)}\` is approaching its limits (${winLabel}${pctPart})${resetPart}, so the fleet moved early instead of hitting the wall.`
|
|
246
|
-
:
|
|
256
|
+
: modelTierWall
|
|
257
|
+
? `Flagship (premium) tier weekly limit reached on \`${codeSpanSafe(roll.from)}\`${resetPart} — opus/haiku on that account are unaffected.`
|
|
258
|
+
: `${winLabel}${pctPart} on \`${codeSpanSafe(roll.from)}\`${resetPart}.`;
|
|
247
259
|
return [
|
|
248
260
|
`🔁 **Switched fleet to \`${codeSpanSafe(roll.to)}\`**`,
|
|
249
261
|
``,
|
|
@@ -164,6 +164,53 @@ export type RuntimeMetricEvent =
|
|
|
164
164
|
agent: string
|
|
165
165
|
prompt_key: string
|
|
166
166
|
}
|
|
167
|
+
/**
|
|
168
|
+
* Every terminal rate-limit-family operator event (429 burst / 529
|
|
169
|
+
* overload), classified by origin BEFORE the gateway acts on it — fires
|
|
170
|
+
* even when the user-facing card is cooldown-suppressed, so the count is
|
|
171
|
+
* honest. `classification` says where the limit lives (see
|
|
172
|
+
* `RateLimit429Classification` in throttle-tier.ts): `account-scoped` =
|
|
173
|
+
* Anthropic throttled the account (throttle tier ran), `litellm-local` =
|
|
174
|
+
* the LiteLLM proxy's own tpm/rpm/router limiter tripped (calm path, no
|
|
175
|
+
* account attribution), `generic-transient` = other server-side 429/529
|
|
176
|
+
* wording. `action` is what the gateway DECIDED (throttle / failover /
|
|
177
|
+
* calm), emitted PRE-execution: the actual failover fire is dedup-gated
|
|
178
|
+
* downstream (fleetFallbackGate), so N long-reset 429s inside one dedup
|
|
179
|
+
* window emit N `action: 'failover'` metrics for ONE real roll — count
|
|
180
|
+
* decisions here, count rolls via the broker/fallback announcements.
|
|
181
|
+
* Correlating `account-scoped` fires against fleet token throughput is
|
|
182
|
+
* the operator's evidence base for setting LiteLLM `tpm_limit` caps; the
|
|
183
|
+
* `litellm-local` count then shows those caps actually absorbing load.
|
|
184
|
+
* Limit/reset fields are best-effort parses of the error body (null when
|
|
185
|
+
* absent).
|
|
186
|
+
*/
|
|
187
|
+
| {
|
|
188
|
+
kind: 'rate_limit_429_classified'
|
|
189
|
+
agent: string
|
|
190
|
+
classification: 'account-scoped' | 'litellm-local' | 'generic-transient'
|
|
191
|
+
action: 'throttle' | 'failover' | 'calm'
|
|
192
|
+
reset_at_ms: number | null
|
|
193
|
+
reset_in_ms: number | null
|
|
194
|
+
limit_type: string | null
|
|
195
|
+
limit: number | null
|
|
196
|
+
current_usage: number | null
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* A litellm-local throttle NOTICE actually posted (litellm-local-notice.ts
|
|
200
|
+
* — the debounced "fleet token limiter engaged" message). Distinct from
|
|
201
|
+
* `rate_limit_429_classified`, which fires on EVERY classified 429: this
|
|
202
|
+
* fires once per sent notice, and `suppressed_count` is how many
|
|
203
|
+
* litellm-local 429s the cooldown window silently absorbed since the
|
|
204
|
+
* previous notice (0 on the first) — the debounce's effectiveness in one
|
|
205
|
+
* number. `window_ms` is the resolved cooldown window (default 15 min;
|
|
206
|
+
* channels.telegram.litellm_notice.window_ms).
|
|
207
|
+
*/
|
|
208
|
+
| {
|
|
209
|
+
kind: 'litellm_local_429_notice'
|
|
210
|
+
agent: string
|
|
211
|
+
suppressed_count: number
|
|
212
|
+
window_ms: number
|
|
213
|
+
}
|
|
167
214
|
|
|
168
215
|
/**
|
|
169
216
|
* The JSONL sink lives under the runtime state dir so it's per-agent
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { describe, expect, it } from 'vitest'
|
|
2
|
-
import { createSendGate, type Clock } from './send-gate.js'
|
|
2
|
+
import { createSendGate, SEND_GATE_SHED, type Clock } from './send-gate.js'
|
|
3
3
|
import { isFloodWaitActiveError } from './retry-api-call.js'
|
|
4
4
|
|
|
5
5
|
/**
|
|
@@ -73,8 +73,10 @@ describe('send-gate PR2: cosmetic shedding', () => {
|
|
|
73
73
|
|
|
74
74
|
const res = await gate.gate(fn('cosmetic'), { priorityClass: 'cosmetic' })
|
|
75
75
|
|
|
76
|
-
// Shed resolves
|
|
77
|
-
|
|
76
|
+
// Shed resolves the distinguishable sentinel (#3110 F1 — never a bare
|
|
77
|
+
// undefined, which is reserved for benign no-op drops); fn never ran;
|
|
78
|
+
// the shed counter moved.
|
|
79
|
+
expect(res).toBe(SEND_GATE_SHED)
|
|
78
80
|
expect(calls).toHaveLength(0)
|
|
79
81
|
expect(gate.stats().global.shed).toBe(1)
|
|
80
82
|
expect(gate.stats().global.sent).toBe(0)
|
|
@@ -95,7 +97,7 @@ describe('send-gate PR2: cosmetic shedding', () => {
|
|
|
95
97
|
const shed = await gate.gate(fn('b'), { chat_id: '5', priorityClass: 'cosmetic' })
|
|
96
98
|
|
|
97
99
|
expect(calls.map((c) => c.label)).toEqual(['a'])
|
|
98
|
-
expect(shed).
|
|
100
|
+
expect(shed).toBe(SEND_GATE_SHED)
|
|
99
101
|
expect(gate.stats().global.shed).toBe(1)
|
|
100
102
|
})
|
|
101
103
|
|
|
@@ -128,7 +130,7 @@ describe('send-gate PR2: cosmetic shedding', () => {
|
|
|
128
130
|
priorityClass: 'cosmetic',
|
|
129
131
|
})
|
|
130
132
|
|
|
131
|
-
expect(res).
|
|
133
|
+
expect(res).toBe(SEND_GATE_SHED)
|
|
132
134
|
expect(calls).toHaveLength(0)
|
|
133
135
|
expect(gate.stats().global.shed).toBe(1)
|
|
134
136
|
})
|
|
@@ -366,7 +368,7 @@ describe('send-gate PR2: H1 cross-chat message_id isolation', () => {
|
|
|
366
368
|
priorityClass: 'cosmetic',
|
|
367
369
|
})
|
|
368
370
|
|
|
369
|
-
expect(rA).
|
|
371
|
+
expect(rA).toBe(SEND_GATE_SHED) // A shed
|
|
370
372
|
expect(rB).toBe('B-edit') // B sent
|
|
371
373
|
expect(calls.map((c) => c.label)).toEqual(['B-edit'])
|
|
372
374
|
expect(gate.stats().global.shed).toBe(1)
|
|
@@ -568,7 +570,7 @@ describe('send-gate PR2: opening a window from a 429, and persistence hook', ()
|
|
|
568
570
|
expect(scopes).toEqual(['chat:7', 'global', 'group:7', 'msg-edit:7:3'])
|
|
569
571
|
// After the window opens, a later cosmetic on the same chat sheds.
|
|
570
572
|
const shed = await gate.gate(async () => 'x', { chat_id: '7', priorityClass: 'cosmetic' })
|
|
571
|
-
expect(shed).
|
|
573
|
+
expect(shed).toBe(SEND_GATE_SHED)
|
|
572
574
|
expect(gate.stats().global.shed).toBe(1)
|
|
573
575
|
})
|
|
574
576
|
})
|
|
@@ -118,6 +118,32 @@ export type PriorityClass = 'critical' | 'useful' | 'cosmetic'
|
|
|
118
118
|
*/
|
|
119
119
|
export const UNTAGGED_SEND_CLASS: PriorityClass = 'critical'
|
|
120
120
|
|
|
121
|
+
/**
|
|
122
|
+
* Distinguishable resolution value for a SHED call (#3110 review F1).
|
|
123
|
+
*
|
|
124
|
+
* `undefined` was overloaded three ways at the robustApiCall seam: a gate
|
|
125
|
+
* SHED (cosmetic call dropped under pressure — did NOT land), a gate no-op
|
|
126
|
+
* drop (identical payload already on screen — benign), and the retry
|
|
127
|
+
* policy's swallowed benign 400s ("message is not modified" — also benign).
|
|
128
|
+
* A caller that needs to know "did my edit land?" (the draft stream's
|
|
129
|
+
* shed-honesty handling in stream-controller.ts) could not tell these
|
|
130
|
+
* apart, and treating every `undefined` as a shed froze multi-piece
|
|
131
|
+
* streams on perfectly healthy chats.
|
|
132
|
+
*
|
|
133
|
+
* A shed now resolves THIS sentinel instead. `Symbol.for` keys it in the
|
|
134
|
+
* global symbol registry so duplicated module instances (server + gateway
|
|
135
|
+
* builds) agree on identity. The no-op drop and coalesced-revert drop keep
|
|
136
|
+
* resolving `undefined` — for those the payload IS on screen. Callers that
|
|
137
|
+
* ignore the result (reactions, typing, fire-and-forget card edits) are
|
|
138
|
+
* unaffected either way.
|
|
139
|
+
*/
|
|
140
|
+
export const SEND_GATE_SHED: unique symbol = Symbol.for('switchroom.send-gate.shed')
|
|
141
|
+
|
|
142
|
+
/** True when a gate result is the {@link SEND_GATE_SHED} sentinel. */
|
|
143
|
+
export function isSendGateShed(value: unknown): value is typeof SEND_GATE_SHED {
|
|
144
|
+
return value === SEND_GATE_SHED
|
|
145
|
+
}
|
|
146
|
+
|
|
121
147
|
/**
|
|
122
148
|
* Extra metadata a call site can attach so the gate can key the right buckets.
|
|
123
149
|
* All fields optional — a call with none still passes the global bucket. These
|
|
@@ -159,8 +185,9 @@ export interface BucketCounters {
|
|
|
159
185
|
dropped: number
|
|
160
186
|
/**
|
|
161
187
|
* Cosmetic calls shed under pressure — no token free OR a flood window open
|
|
162
|
-
* (part3-design §2). A shed resolves
|
|
163
|
-
*
|
|
188
|
+
* (part3-design §2). A shed resolves the SEND_GATE_SHED sentinel (F1 —
|
|
189
|
+
* distinguishable from a benign no-op drop's `undefined`); the next send
|
|
190
|
+
* carries full state.
|
|
164
191
|
*/
|
|
165
192
|
shed: number
|
|
166
193
|
/** Useful calls dropped because they exceeded their queue TTL (part3-design §2). */
|
|
@@ -913,7 +940,9 @@ export function createSendGate(config: SendGateConfig): SendGate {
|
|
|
913
940
|
const msgWait = state.suppressedUntilMs > now ? state.suppressedUntilMs - now : 0
|
|
914
941
|
if (wait > 0 || msgWait > 0) {
|
|
915
942
|
counters.shed++
|
|
916
|
-
|
|
943
|
+
// Distinguishable from the no-op drop below (undefined): a shed did
|
|
944
|
+
// NOT land, and edit-driving callers must be able to tell (F1).
|
|
945
|
+
return Promise.resolve(SEND_GATE_SHED as unknown as T)
|
|
917
946
|
}
|
|
918
947
|
}
|
|
919
948
|
|
|
@@ -991,7 +1020,8 @@ export function createSendGate(config: SendGateConfig): SendGate {
|
|
|
991
1020
|
const outcome = await admitPriority(bucketsFor(opts), priority)
|
|
992
1021
|
if (outcome.result === 'shed') {
|
|
993
1022
|
counters.shed++
|
|
994
|
-
|
|
1023
|
+
// See SEND_GATE_SHED: distinguishable "did not land" resolution (F1).
|
|
1024
|
+
return SEND_GATE_SHED as unknown as T
|
|
995
1025
|
}
|
|
996
1026
|
if (outcome.result === 'expired') {
|
|
997
1027
|
counters.expired++
|
|
@@ -41,7 +41,7 @@ function isMultiAgentEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
|
|
|
41
41
|
return env.PROGRESS_CARD_MULTI_AGENT !== '0'
|
|
42
42
|
}
|
|
43
43
|
import { classifyClaudeError, type OperatorEventKind } from './operator-events.js'
|
|
44
|
-
import { isTransientUpstreamSignal } from './model-unavailable.js'
|
|
44
|
+
import { isLitellmProxyLocal429, isTransientUpstreamSignal } from './model-unavailable.js'
|
|
45
45
|
import { createToolLabelSidecar, type ToolLabelSidecar, type SidecarOptions } from './tool-label-sidecar.js'
|
|
46
46
|
import { isModelSentinel } from './model-label.js'
|
|
47
47
|
|
|
@@ -684,9 +684,21 @@ export function detectErrorInTranscriptLine(
|
|
|
684
684
|
// model-unavailable.ts) — an ambiguous 429 that merely says "limit" stays
|
|
685
685
|
// quota-exhausted, biasing toward surfacing a real wall. Other statuses fall
|
|
686
686
|
// through to the shared classifier.
|
|
687
|
+
//
|
|
688
|
+
// A 429 carrying LiteLLM-proxy-LOCAL limiter wording ("Deployment over
|
|
689
|
+
// user-defined ratelimit", "Rate limit exceeded for api_key: …" — the
|
|
690
|
+
// canonical `litellmProxyLocal429Signals` list) is the proxy's own
|
|
691
|
+
// `tpm_limit`/`rpm_limit` cap tripping BEFORE the request reached
|
|
692
|
+
// Anthropic. Nothing about the account is exhausted — blanket-labeling it
|
|
693
|
+
// quota-exhausted would fire the model-unavailable card + mark-exhausted +
|
|
694
|
+
// fleet failover for a purely proxy-local condition. It is rate-limited
|
|
695
|
+
// (calm path). Precedence when BOTH wordings appear is owned by
|
|
696
|
+
// `classify429Detail` gateway-side; here both branches yield the same
|
|
697
|
+
// 'rate-limited' kind, so order is immaterial.
|
|
687
698
|
const kind: OperatorEventKind =
|
|
688
699
|
status === 429
|
|
689
|
-
? isTransientUpstreamSignal(`${text}\n${errStr}`)
|
|
700
|
+
? isTransientUpstreamSignal(`${text}\n${errStr}`) ||
|
|
701
|
+
isLitellmProxyLocal429(`${text}\n${errStr}`)
|
|
690
702
|
? 'rate-limited'
|
|
691
703
|
: 'quota-exhausted'
|
|
692
704
|
: classifyClaudeError({ type: errStr, status, message: text })
|