switchroom 0.18.15 → 0.18.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +16 -0
- package/dist/auth-broker/index.js +445 -10
- package/dist/cli/notion-write-pretool.mjs +16 -0
- package/dist/cli/switchroom.js +654 -479
- package/dist/host-control/main.js +20 -1
- package/dist/vault/approvals/kernel-server.js +16 -0
- package/dist/vault/broker/server.js +16 -0
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +81 -139
- package/telegram-plugin/bridge/bridge.ts +7 -1
- package/telegram-plugin/dist/bridge/bridge.js +26 -1
- package/telegram-plugin/dist/gateway/gateway.js +1758 -661
- package/telegram-plugin/dist/server.js +26 -1
- package/telegram-plugin/draft-stream.ts +78 -3
- package/telegram-plugin/fleet-fallback-resume.ts +26 -3
- package/telegram-plugin/gateway/approval-hold.ts +49 -0
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +64 -22
- package/telegram-plugin/gateway/effort-command.ts +9 -7
- package/telegram-plugin/gateway/gateway.ts +627 -291
- package/telegram-plugin/gateway/linear-activity.ts +20 -4
- package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
- package/telegram-plugin/gateway/model-command.ts +96 -18
- package/telegram-plugin/gateway/pending-session-command.ts +10 -8
- package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
- package/telegram-plugin/gateway/session-model-file.ts +141 -172
- package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
- package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
- package/telegram-plugin/litellm-local-notice.ts +189 -0
- package/telegram-plugin/llm-error-present.ts +436 -0
- package/telegram-plugin/operator-events.ts +7 -1
- package/telegram-plugin/permission-title.ts +172 -10
- package/telegram-plugin/premium-recovery.ts +101 -0
- package/telegram-plugin/quota-watch.ts +16 -4
- package/telegram-plugin/raw-error-scrub.ts +73 -0
- package/telegram-plugin/retry-api-call.ts +8 -2
- package/telegram-plugin/runtime-metrics.ts +16 -0
- package/telegram-plugin/send-gate-degraded.test.ts +161 -8
- package/telegram-plugin/send-gate-observability.test.ts +140 -0
- package/telegram-plugin/send-gate-observability.ts +65 -20
- package/telegram-plugin/send-gate.test.ts +143 -1
- package/telegram-plugin/send-gate.ts +246 -23
- package/telegram-plugin/session-tail.ts +16 -0
- package/telegram-plugin/shared/local-time.ts +69 -0
- package/telegram-plugin/stream-controller.ts +143 -20
- package/telegram-plugin/stream-reply-handler.ts +12 -2
- package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
- package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
- package/telegram-plugin/tests/bot-api.harness.ts +7 -2
- package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
- package/telegram-plugin/tests/draft-stream.test.ts +110 -1
- package/telegram-plugin/tests/effort-command.test.ts +4 -4
- package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +5 -4
- package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
- package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
- package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
- package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
- package/telegram-plugin/tests/model-command.test.ts +84 -1
- package/telegram-plugin/tests/permission-title.test.ts +167 -4
- package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
- package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
- package/telegram-plugin/tests/quota-watch.test.ts +21 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +8 -3
- package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
- package/telegram-plugin/tests/session-model-file.test.ts +7 -155
- package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
- package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
- package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
- package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
- package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
- package/telegram-plugin/tests/worker-activity-feed.test.ts +212 -2
- package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
- package/telegram-plugin/tier-downgrade.ts +198 -0
- package/telegram-plugin/tool-activity-summary.ts +99 -0
- package/telegram-plugin/worker-activity-feed.ts +543 -368
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* litellm-local-notice.ts — debounced operator notice for LiteLLM-proxy-local
|
|
3
|
+
* 429s (pure module: state machine + text + config parsing, no IPC, no bot,
|
|
4
|
+
* no clock except the injected `now`).
|
|
5
|
+
*
|
|
6
|
+
* When an agent routes through the LiteLLM gateway and trips the proxy's OWN
|
|
7
|
+
* `tpm_limit`/`rpm_limit` limiter, `classify429Detail` (throttle-tier.ts)
|
|
8
|
+
* classifies the terminal 429 `litellm-local` and the gateway takes the calm
|
|
9
|
+
* path — no broker mark, no failover, no throttle tier (the request never
|
|
10
|
+
* reached Anthropic, so the condition says nothing about the account). But
|
|
11
|
+
* pre-notice the user-facing surface was the GENERIC "🚦 Rate limited" card,
|
|
12
|
+
* which reads like an Anthropic problem. This module owns the honest
|
|
13
|
+
* replacement:
|
|
14
|
+
*
|
|
15
|
+
* - ONE calm notice per agent per cooldown window (default 15 min,
|
|
16
|
+
* operator-tunable via channels.telegram.litellm_notice.window_ms in
|
|
17
|
+
* switchroom.yaml, projected into access.json by scaffold).
|
|
18
|
+
* - Further litellm-local 429s inside the window are counted SILENTLY;
|
|
19
|
+
* the first notice after the window expires carries "throttled N more
|
|
20
|
+
* times since the last notice".
|
|
21
|
+
* - A notice only ever fires on an actual throttle event — a quiet agent
|
|
22
|
+
* posts nothing (the state machine is evaluate-on-event, no timers).
|
|
23
|
+
*
|
|
24
|
+
* The copy must make clear this is the fleet token limiter (LiteLLM
|
|
25
|
+
* `tpm_limit`/`rpm_limit`), NOT an Anthropic account limit, and that the
|
|
26
|
+
* turn retries — no action needed.
|
|
27
|
+
*
|
|
28
|
+
* Side-effect sequencing lives in gateway/litellm-local-notice-wiring.ts.
|
|
29
|
+
* This module MUST NOT change classification, failover, or quota-ledger
|
|
30
|
+
* behavior — it only renders and debounces the notice.
|
|
31
|
+
*/
|
|
32
|
+
|
|
33
|
+
import { escapeMarkdown } from './card-format.js'
|
|
34
|
+
import type { RateLimit429Classification } from './throttle-tier.js'
|
|
35
|
+
import type { RuntimeMetricEvent } from './runtime-metrics.js'
|
|
36
|
+
|
|
37
|
+
// ─── Cooldown window config ──────────────────────────────────────────────────
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* Default per-agent notice cooldown. 15 minutes: long enough that a sustained
|
|
41
|
+
* burst produces a handful of notices per hour at most, short enough that the
|
|
42
|
+
* operator still sees the limiter working while it's working.
|
|
43
|
+
*/
|
|
44
|
+
export const LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT = 15 * 60_000
|
|
45
|
+
|
|
46
|
+
/**
|
|
47
|
+
* Resolve the cooldown window from the raw access-file value
|
|
48
|
+
* (`access.litellmNoticeWindowMs`, projected by scaffold from
|
|
49
|
+
* channels.telegram.litellm_notice.window_ms). Any non-finite, non-positive,
|
|
50
|
+
* or non-number value falls back to the default — an operator typo must
|
|
51
|
+
* never produce a zero-width window (notice storm) or a NaN comparison
|
|
52
|
+
* (notices never fire again).
|
|
53
|
+
*/
|
|
54
|
+
export function parseLitellmNoticeWindowMs(raw: unknown): number {
|
|
55
|
+
if (typeof raw !== 'number') return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
|
|
56
|
+
if (!Number.isFinite(raw) || raw <= 0) return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
|
|
57
|
+
return raw
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
// ─── Per-agent cooldown state machine ────────────────────────────────────────
|
|
61
|
+
|
|
62
|
+
export interface LitellmLocalNoticeState {
|
|
63
|
+
/** agent → unix ms of the last notice sent for it. */
|
|
64
|
+
lastSentAtMsByAgent: Record<string, number>
|
|
65
|
+
/** agent → litellm-local 429s counted silently since the last notice. */
|
|
66
|
+
suppressedCountByAgent: Record<string, number>
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export function initialLitellmLocalNoticeState(): LitellmLocalNoticeState {
|
|
70
|
+
return { lastSentAtMsByAgent: {}, suppressedCountByAgent: {} }
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface LitellmLocalNoticeVerdict {
|
|
74
|
+
send: boolean
|
|
75
|
+
/** Silent throttle events since the previous notice (0 on the first). Only
|
|
76
|
+
* meaningful when `send` is true — the renderer adds the "N more times"
|
|
77
|
+
* line when > 0. */
|
|
78
|
+
suppressedSinceLastNotice: number
|
|
79
|
+
next: LitellmLocalNoticeState
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Evaluate one litellm-local 429 for `agent` at `now`.
|
|
84
|
+
*
|
|
85
|
+
* - First event ever (or ≥ windowMs since the last notice) → send; the
|
|
86
|
+
* verdict carries the silently-counted events since the previous notice
|
|
87
|
+
* and the counter resets.
|
|
88
|
+
* - Inside the window → suppress; the counter increments.
|
|
89
|
+
*
|
|
90
|
+
* Boundary is INCLUSIVE on expiry (elapsed === windowMs sends), matching
|
|
91
|
+
* `evaluateThrottleNotice` in throttle-tier.ts. Pure: returns the next
|
|
92
|
+
* state, never mutates `prev`.
|
|
93
|
+
*/
|
|
94
|
+
export function evaluateLitellmLocalNotice(
|
|
95
|
+
prev: LitellmLocalNoticeState,
|
|
96
|
+
agent: string,
|
|
97
|
+
now: number,
|
|
98
|
+
windowMs: number = LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT,
|
|
99
|
+
): LitellmLocalNoticeVerdict {
|
|
100
|
+
const last = prev.lastSentAtMsByAgent[agent]
|
|
101
|
+
if (last == null || now - last >= windowMs) {
|
|
102
|
+
return {
|
|
103
|
+
send: true,
|
|
104
|
+
suppressedSinceLastNotice: prev.suppressedCountByAgent[agent] ?? 0,
|
|
105
|
+
next: {
|
|
106
|
+
lastSentAtMsByAgent: { ...prev.lastSentAtMsByAgent, [agent]: now },
|
|
107
|
+
suppressedCountByAgent: { ...prev.suppressedCountByAgent, [agent]: 0 },
|
|
108
|
+
},
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
return {
|
|
112
|
+
send: false,
|
|
113
|
+
suppressedSinceLastNotice: 0,
|
|
114
|
+
next: {
|
|
115
|
+
lastSentAtMsByAgent: prev.lastSentAtMsByAgent,
|
|
116
|
+
suppressedCountByAgent: {
|
|
117
|
+
...prev.suppressedCountByAgent,
|
|
118
|
+
[agent]: (prev.suppressedCountByAgent[agent] ?? 0) + 1,
|
|
119
|
+
},
|
|
120
|
+
},
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
// ─── Notice rendering ────────────────────────────────────────────────────────
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* The ONE calm notice for a litellm-local 429. Markdown (not HTML) per the
|
|
128
|
+
* card-format.ts convention. Copy contract (operator spec):
|
|
129
|
+
* - names the fleet token limiter (LiteLLM `tpm_limit`/`rpm_limit`)
|
|
130
|
+
* - explicitly NOT an Anthropic account limit — nothing exhausted, no
|
|
131
|
+
* account touched
|
|
132
|
+
* - the turn retries automatically; no action needed
|
|
133
|
+
* - after a window with silent suppressions, says "throttled N more
|
|
134
|
+
* time(s) since the last notice"
|
|
135
|
+
*/
|
|
136
|
+
export function renderLitellmLocalNotice(opts: {
|
|
137
|
+
agent: string
|
|
138
|
+
/** Silent litellm-local 429s since the previous notice (0 on the first). */
|
|
139
|
+
suppressedSinceLastNotice: number
|
|
140
|
+
}): string {
|
|
141
|
+
const agent = escapeMarkdown(opts.agent)
|
|
142
|
+
const n = opts.suppressedSinceLastNotice
|
|
143
|
+
const lines = [
|
|
144
|
+
`🚦 **Fleet token limiter engaged** — **${agent}** hit the local LiteLLM proxy cap (\`tpm_limit\`/\`rpm_limit\`).`,
|
|
145
|
+
`This is switchroom's own fleet limiter smoothing a burst, not an Anthropic account limit — nothing is exhausted and no account was touched.`,
|
|
146
|
+
]
|
|
147
|
+
if (n > 0) {
|
|
148
|
+
lines.push(`Throttled ${n} more ${n === 1 ? 'time' : 'times'} since the last notice.`)
|
|
149
|
+
}
|
|
150
|
+
lines.push(`_The turn retries automatically — no action needed._`)
|
|
151
|
+
return lines.join('\n')
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// ─── Metric builder ──────────────────────────────────────────────────────────
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Build the `litellm_local_429_notice` runtime metric for one SENT notice.
|
|
158
|
+
* Distinct from `rate_limit_429_classified` (which fires on EVERY classified
|
|
159
|
+
* event): this one fires only when a notice actually posts, and carries how
|
|
160
|
+
* many events the window silently absorbed — the debounce's effectiveness in
|
|
161
|
+
* one number. Pure builder so the payload shape is unit-testable; the wiring
|
|
162
|
+
* emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
|
|
163
|
+
*/
|
|
164
|
+
export function buildLitellmLocalNoticeMetric(opts: {
|
|
165
|
+
agent: string
|
|
166
|
+
suppressedCount: number
|
|
167
|
+
windowMs: number
|
|
168
|
+
}): Extract<RuntimeMetricEvent, { kind: 'litellm_local_429_notice' }> {
|
|
169
|
+
return {
|
|
170
|
+
kind: 'litellm_local_429_notice',
|
|
171
|
+
agent: opts.agent,
|
|
172
|
+
suppressed_count: opts.suppressedCount,
|
|
173
|
+
window_ms: opts.windowMs,
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// ─── Classification guard ────────────────────────────────────────────────────
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* True only for the `litellm-local` classification. The wiring runs this
|
|
181
|
+
* guard so "a notice never fires for non-litellm-local classifications" is a
|
|
182
|
+
* deterministic mechanism, not caller discipline — account-scoped 429s keep
|
|
183
|
+
* the throttle tier, generic transients keep the calm rate-limited card.
|
|
184
|
+
*/
|
|
185
|
+
export function isLitellmLocalNoticeEligible(
|
|
186
|
+
classification: RateLimit429Classification | null | undefined,
|
|
187
|
+
): classification is 'litellm-local' {
|
|
188
|
+
return classification === 'litellm-local'
|
|
189
|
+
}
|
|
@@ -0,0 +1,436 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* llm-error-present.ts — humanized, cross-surface-deduped presentation of an
|
|
3
|
+
* LLM/API error, so a raw `b'{"type":"error",…}'` line never reaches a user.
|
|
4
|
+
*
|
|
5
|
+
* THE PROBLEM this closes
|
|
6
|
+
* -----------------------
|
|
7
|
+
* One Anthropic error JSONL line fans out to THREE independent render surfaces,
|
|
8
|
+
* none consulting each other, each historically printing the raw `detail`
|
|
9
|
+
* string with its `b'{…}'` byte-blob attached:
|
|
10
|
+
* 1. the agent turn-end "done" card (tool-activity-summary result block),
|
|
11
|
+
* 2. the operator-event "🚦 Rate limited" card (raw `detail`), and
|
|
12
|
+
* 3. the reply/answer passthrough (the synthetic-assistant error text relayed
|
|
13
|
+
* as the turn's answer).
|
|
14
|
+
*
|
|
15
|
+
* This module owns the ONE clean rendering + the ONE dedup authority the three
|
|
16
|
+
* surfaces consult, so a fan-out produces EXACTLY ONE user-facing message with
|
|
17
|
+
* NO raw JSON. It reuses the existing classifiers rather than re-deriving them:
|
|
18
|
+
* `classify429Detail` (throttle-tier.ts), `detectModelUnavailable` /
|
|
19
|
+
* `parseResetTime` (model-unavailable.ts), `isLitellmProxyLocal429` /
|
|
20
|
+
* `isTransientUpstreamSignal` (model-unavailable.ts), `classifyClaudeError`
|
|
21
|
+
* (operator-events.ts). `coreText` is ALWAYS built from a per-kind template —
|
|
22
|
+
* never from the raw `detail` — and `stripRawErrorBytes` is the belt-and-braces
|
|
23
|
+
* scrub so even the enrichment `reason` can carry no JSON.
|
|
24
|
+
*
|
|
25
|
+
* Pure module: no IPC, no bot, no FS. The only mutable state is the
|
|
26
|
+
* `ErrorPresenceGate` singleton (an in-memory dedup ledger), driven by an
|
|
27
|
+
* injectable `now`.
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import {
|
|
31
|
+
detectModelUnavailable,
|
|
32
|
+
isLitellmProxyLocal429,
|
|
33
|
+
parseResetTime,
|
|
34
|
+
} from './model-unavailable.js'
|
|
35
|
+
import { classify429Detail } from './throttle-tier.js'
|
|
36
|
+
import { classifyClaudeError } from './operator-events.js'
|
|
37
|
+
import type { InlineKeyboardMarkup } from './operator-events.js'
|
|
38
|
+
import { stripRawErrorBytes, extractRequestId } from './raw-error-scrub.js'
|
|
39
|
+
import { fmtLocalClock, tzAbbrev } from './shared/local-time.js'
|
|
40
|
+
|
|
41
|
+
export { stripRawErrorBytes, extractRequestId } from './raw-error-scrub.js'
|
|
42
|
+
|
|
43
|
+
// ─── Types ──────────────────────────────────────────────────────────────────
|
|
44
|
+
|
|
45
|
+
export type LlmErrorKind =
|
|
46
|
+
| 'rate_limit'
|
|
47
|
+
| 'overload_529'
|
|
48
|
+
| 'quota_wall'
|
|
49
|
+
| 'auth'
|
|
50
|
+
| 'transient'
|
|
51
|
+
| 'unknown'
|
|
52
|
+
|
|
53
|
+
export type LlmErrorSource = 'anthropic' | 'litellm-local' | 'network'
|
|
54
|
+
|
|
55
|
+
export interface ParsedLlmError {
|
|
56
|
+
kind: LlmErrorKind
|
|
57
|
+
/** Human, JSON-STRIPPED one-liner built from a per-kind template. Never raw. */
|
|
58
|
+
coreText: string
|
|
59
|
+
/** Parsed reset instant, when the source carried one. */
|
|
60
|
+
resetAt?: Date
|
|
61
|
+
/** Parsed retry-after window in ms, when the source carried one. */
|
|
62
|
+
retryAfterMs?: number
|
|
63
|
+
/** Resolved model id, when the source named one. */
|
|
64
|
+
model?: string
|
|
65
|
+
/** Anthropic `request_id`, when present — the strongest dedup key. */
|
|
66
|
+
requestId?: string
|
|
67
|
+
source: LlmErrorSource
|
|
68
|
+
/** True when the harness is still retrying this error internally (mid-retry). */
|
|
69
|
+
autoRetrying: boolean
|
|
70
|
+
/** True when the failure is final (NOT an in-flight retry). */
|
|
71
|
+
terminal: boolean
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Optional retry-state annotations Claude Code stamps on a retried error line. */
|
|
75
|
+
export interface LlmErrorRetryState {
|
|
76
|
+
retryAttempt: number | null
|
|
77
|
+
maxRetries: number | null
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// ─── model extraction ────────────────────────────────────────────────────────
|
|
81
|
+
|
|
82
|
+
/** Pull a resolved model id (`claude-…`, `sr-…`) out of a raw string, or undefined. */
|
|
83
|
+
function extractModel(raw: string): string | undefined {
|
|
84
|
+
const m = raw.match(/["']?model["']?\s*[=:]\s*["']?((?:claude|sr)[A-Za-z0-9._-]+)/i)
|
|
85
|
+
return m ? m[1] : undefined
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// ─── Parser ──────────────────────────────────────────────────────────────────
|
|
89
|
+
|
|
90
|
+
const TRANSIENT_KINDS: ReadonlySet<LlmErrorKind> = new Set<LlmErrorKind>([
|
|
91
|
+
'rate_limit',
|
|
92
|
+
'overload_529',
|
|
93
|
+
'transient',
|
|
94
|
+
])
|
|
95
|
+
|
|
96
|
+
/** The always-actionable kinds — NEVER silenced, even inside a collapse window. */
|
|
97
|
+
export function isActionableKind(kind: LlmErrorKind): boolean {
|
|
98
|
+
return kind === 'auth' || kind === 'quota_wall'
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Parse a raw error string (a session-tail `detail`, an isApiErrorMessage
|
|
103
|
+
* text, or a raw stderr line) into a structured, JSON-free ParsedLlmError.
|
|
104
|
+
* Reuses the existing wording classifiers; maps their kinds into this union.
|
|
105
|
+
*/
|
|
106
|
+
export function parseLlmError(
|
|
107
|
+
raw: string,
|
|
108
|
+
retryState?: LlmErrorRetryState,
|
|
109
|
+
): ParsedLlmError {
|
|
110
|
+
const text = typeof raw === 'string' ? raw : ''
|
|
111
|
+
const requestId = extractRequestId(text)
|
|
112
|
+
const model = extractModel(text)
|
|
113
|
+
|
|
114
|
+
// Reset / retry-after best-effort (Anthropic + relative wordings).
|
|
115
|
+
const resetAt = parseResetTime(text)
|
|
116
|
+
const retryAfterMs =
|
|
117
|
+
resetAt != null ? Math.max(0, resetAt.getTime() - Date.now()) : undefined
|
|
118
|
+
|
|
119
|
+
const { kind, source } = classifyKindAndSource(text)
|
|
120
|
+
|
|
121
|
+
// Retry / terminal semantics. Only the transient family can be mid-retry;
|
|
122
|
+
// auth / quota_wall / model_unavailable are terminal by construction.
|
|
123
|
+
let autoRetrying = false
|
|
124
|
+
let terminal = true
|
|
125
|
+
if (TRANSIENT_KINDS.has(kind)) {
|
|
126
|
+
const { retryAttempt, maxRetries } = retryState ?? { retryAttempt: null, maxRetries: null }
|
|
127
|
+
if (retryAttempt != null && maxRetries != null) {
|
|
128
|
+
autoRetrying = retryAttempt < maxRetries
|
|
129
|
+
terminal = retryAttempt >= maxRetries
|
|
130
|
+
} else {
|
|
131
|
+
// No retry annotation: treat as a surfaced (terminal) failure — Claude
|
|
132
|
+
// writes the user-facing error shape only after its own retries are done.
|
|
133
|
+
autoRetrying = false
|
|
134
|
+
terminal = true
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
kind,
|
|
140
|
+
coreText: buildCoreText(kind, source),
|
|
141
|
+
...(resetAt != null ? { resetAt } : {}),
|
|
142
|
+
...(retryAfterMs != null ? { retryAfterMs } : {}),
|
|
143
|
+
...(model != null ? { model } : {}),
|
|
144
|
+
...(requestId != null ? { requestId } : {}),
|
|
145
|
+
source,
|
|
146
|
+
autoRetrying,
|
|
147
|
+
terminal,
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
function classifyKindAndSource(text: string): { kind: LlmErrorKind; source: LlmErrorSource } {
|
|
152
|
+
const lower = text.toLowerCase()
|
|
153
|
+
|
|
154
|
+
// 1. Auth — always terminal, always actionable.
|
|
155
|
+
const claudeKind = classifyClaudeError({ message: text, type: text })
|
|
156
|
+
if (claudeKind === 'credentials-expired' || claudeKind === 'credentials-invalid') {
|
|
157
|
+
return { kind: 'auth', source: 'anthropic' }
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
// 2. LiteLLM-proxy-LOCAL 429 — the proxy's own limiter, never Anthropic.
|
|
161
|
+
// Checked before the generic quota/overload matchers because some LiteLLM
|
|
162
|
+
// bodies contain the word "limit" that a quota matcher could seize on.
|
|
163
|
+
if (isLitellmProxyLocal429(text)) {
|
|
164
|
+
return { kind: 'rate_limit', source: 'litellm-local' }
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
// 3. Model-unavailable detector owns quota-wall / overload / network wording.
|
|
168
|
+
const mu = detectModelUnavailable(text)
|
|
169
|
+
if (mu != null) {
|
|
170
|
+
if (mu.kind === 'quota_exhausted') return { kind: 'quota_wall', source: 'anthropic' }
|
|
171
|
+
if (mu.kind === 'network') return { kind: 'transient', source: 'network' }
|
|
172
|
+
// mu.kind === 'overload' | 'rate_limited' — split 529 overload from a 429.
|
|
173
|
+
if (
|
|
174
|
+
lower.includes('529') ||
|
|
175
|
+
lower.includes('overloaded')
|
|
176
|
+
) {
|
|
177
|
+
return { kind: 'overload_529', source: 'anthropic' }
|
|
178
|
+
}
|
|
179
|
+
// A 429-family transient. Three-way classify picks the source nuance; all
|
|
180
|
+
// land on the calm rate_limit kind here.
|
|
181
|
+
const c = classify429Detail(text)
|
|
182
|
+
return {
|
|
183
|
+
kind: 'rate_limit',
|
|
184
|
+
source: c === 'litellm-local' ? 'litellm-local' : 'anthropic',
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// 4. Credit balance is a quota-family wall (actionable).
|
|
189
|
+
if (claudeKind === 'credit-exhausted') {
|
|
190
|
+
return { kind: 'quota_wall', source: 'anthropic' }
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// 5. Bare rate-limit classification from the operator taxonomy.
|
|
194
|
+
if (claudeKind === 'rate-limited') {
|
|
195
|
+
return { kind: 'rate_limit', source: 'anthropic' }
|
|
196
|
+
}
|
|
197
|
+
if (claudeKind === 'unknown-5xx') {
|
|
198
|
+
return { kind: 'overload_529', source: 'anthropic' }
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
return { kind: 'unknown', source: 'anthropic' }
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
function buildCoreText(kind: LlmErrorKind, source: LlmErrorSource): string {
|
|
205
|
+
switch (kind) {
|
|
206
|
+
case 'rate_limit':
|
|
207
|
+
return source === 'litellm-local'
|
|
208
|
+
? 'Hit the local proxy rate limit — retrying automatically.'
|
|
209
|
+
: 'Rate limited by Anthropic — retrying automatically.'
|
|
210
|
+
case 'overload_529':
|
|
211
|
+
return 'Anthropic is overloaded (529) — retrying automatically.'
|
|
212
|
+
case 'quota_wall':
|
|
213
|
+
return 'Usage limit reached on this Claude subscription.'
|
|
214
|
+
case 'auth':
|
|
215
|
+
return 'Claude login needs re-authentication.'
|
|
216
|
+
case 'transient':
|
|
217
|
+
return source === 'network'
|
|
218
|
+
? "Couldn't reach Anthropic (network) — retrying automatically."
|
|
219
|
+
: 'A temporary upstream hiccup — retrying automatically.'
|
|
220
|
+
case 'unknown':
|
|
221
|
+
return 'The model returned an error.'
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
// ─── Rendering ───────────────────────────────────────────────────────────────
|
|
226
|
+
|
|
227
|
+
export interface RenderedLlmError {
|
|
228
|
+
text: string
|
|
229
|
+
keyboard?: InlineKeyboardMarkup
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Format a reset instant in the operator's local tz plus a relative tail, e.g.
|
|
234
|
+
* `clears ~4:52pm AEST (~in 38m)`. Returns '' when there is no reset to show.
|
|
235
|
+
*/
|
|
236
|
+
export function formatResetLocal(resetAt: Date | undefined, tz: string, now: Date = new Date()): string {
|
|
237
|
+
if (resetAt == null) return ''
|
|
238
|
+
const ms = resetAt.getTime()
|
|
239
|
+
if (!Number.isFinite(ms)) return ''
|
|
240
|
+
const clock = fmtLocalClock(ms, tz)
|
|
241
|
+
const abbrev = tzAbbrev(ms, tz)
|
|
242
|
+
const rel = formatRelativeTail(ms - now.getTime())
|
|
243
|
+
return rel ? `clears ~${clock} ${abbrev} (${rel})` : `clears ~${clock} ${abbrev}`
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
function formatRelativeTail(deltaMs: number): string {
|
|
247
|
+
if (deltaMs <= 0) return '~now'
|
|
248
|
+
const totalMin = Math.round(deltaMs / 60_000)
|
|
249
|
+
if (totalMin < 1) return '~in <1m'
|
|
250
|
+
if (totalMin < 60) return `~in ${totalMin}m`
|
|
251
|
+
const hours = Math.floor(totalMin / 60)
|
|
252
|
+
const mins = totalMin % 60
|
|
253
|
+
if (hours < 24) return mins > 0 ? `~in ${hours}h ${mins}m` : `~in ${hours}h`
|
|
254
|
+
const days = Math.floor(hours / 24)
|
|
255
|
+
const remH = hours % 24
|
|
256
|
+
return remH > 0 ? `~in ${days}d ${remH}h` : `~in ${days}d`
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
/**
|
|
260
|
+
* Render ONE clean card for a parsed LLM error. Action buttons for the
|
|
261
|
+
* actionable kinds (auth → Reauth, quota_wall → Wait/switch). `agent` is used
|
|
262
|
+
* in the callback_data (URL-encoded) and the headline; `tz` localizes the reset.
|
|
263
|
+
*/
|
|
264
|
+
export function renderLlmError(
|
|
265
|
+
parsed: ParsedLlmError,
|
|
266
|
+
agent: string,
|
|
267
|
+
tz: string,
|
|
268
|
+
now: Date = new Date(),
|
|
269
|
+
): RenderedLlmError {
|
|
270
|
+
const safeAgent = escapeAgent(agent)
|
|
271
|
+
const emoji = kindEmoji(parsed.kind)
|
|
272
|
+
const lines: string[] = [`${emoji} ${parsed.coreText} (**${safeAgent}**)`]
|
|
273
|
+
|
|
274
|
+
const resetLine = formatResetLocal(parsed.resetAt, tz, now)
|
|
275
|
+
if (resetLine) lines.push(`_${resetLine}_`)
|
|
276
|
+
|
|
277
|
+
if (parsed.model) lines.push(`_model: ${escapeAgent(parsed.model)}_`)
|
|
278
|
+
|
|
279
|
+
const text = lines.join('\n')
|
|
280
|
+
|
|
281
|
+
switch (parsed.kind) {
|
|
282
|
+
case 'auth':
|
|
283
|
+
return {
|
|
284
|
+
text,
|
|
285
|
+
keyboard: {
|
|
286
|
+
inline_keyboard: [
|
|
287
|
+
[
|
|
288
|
+
{ text: '🔐 Reauth now', callback_data: `op:reauth:${encodeURIComponent(agent)}` },
|
|
289
|
+
{ text: '❌ Dismiss', callback_data: `op:dismiss:${encodeURIComponent(agent)}` },
|
|
290
|
+
],
|
|
291
|
+
],
|
|
292
|
+
},
|
|
293
|
+
}
|
|
294
|
+
case 'quota_wall':
|
|
295
|
+
return {
|
|
296
|
+
text,
|
|
297
|
+
keyboard: {
|
|
298
|
+
inline_keyboard: [
|
|
299
|
+
[{ text: '⏳ Wait', callback_data: `op:dismiss:${encodeURIComponent(agent)}` }],
|
|
300
|
+
],
|
|
301
|
+
},
|
|
302
|
+
}
|
|
303
|
+
default:
|
|
304
|
+
return { text }
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
function kindEmoji(kind: LlmErrorKind): string {
|
|
309
|
+
switch (kind) {
|
|
310
|
+
case 'rate_limit':
|
|
311
|
+
return '🚦'
|
|
312
|
+
case 'overload_529':
|
|
313
|
+
return '🔥'
|
|
314
|
+
case 'quota_wall':
|
|
315
|
+
return '⚠️'
|
|
316
|
+
case 'auth':
|
|
317
|
+
return '🔑'
|
|
318
|
+
case 'transient':
|
|
319
|
+
return '🌐'
|
|
320
|
+
case 'unknown':
|
|
321
|
+
return '⚠️'
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/** Minimal markdown-safe agent rendering (mirrors operator-events escapeMarkdown intent). */
|
|
326
|
+
function escapeAgent(s: string): string {
|
|
327
|
+
return s.replace(/([_*`\[\]])/g, '\\$1')
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
// ─── Cross-surface dedup: ErrorPresenceGate ──────────────────────────────────
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* How long ONE terminal error owns the "already surfaced" claim across all
|
|
334
|
+
* three surfaces. Distinct from — and additional to — the 5-minute per-kind
|
|
335
|
+
* `shouldEmitOperatorEvent` cooldown (operator-events.ts): that debounces an
|
|
336
|
+
* error STORM on one surface; this collapses ONE error's FAN-OUT across
|
|
337
|
+
* surfaces within a single turn.
|
|
338
|
+
*/
|
|
339
|
+
export const ERROR_COLLAPSE_WINDOW_MS = 60_000
|
|
340
|
+
|
|
341
|
+
/**
|
|
342
|
+
* A tiny in-memory dedup ledger. The FIRST surface to `claim(key)` within the
|
|
343
|
+
* collapse window wins (returns true) and renders; every later surface loses
|
|
344
|
+
* (returns false) and suppresses. Key preference: the Anthropic `request_id`
|
|
345
|
+
* when present (exact), else a `${kind}:${agent}:${windowBucket}` coarse key so
|
|
346
|
+
* two distinct-but-unlabelled errors of the same kind within 60s still collapse.
|
|
347
|
+
*/
|
|
348
|
+
export class ErrorPresenceGate {
|
|
349
|
+
private readonly claims = new Map<string, number>()
|
|
350
|
+
|
|
351
|
+
/** Build the dedup key for a parsed error + agent. */
|
|
352
|
+
keyFor(parsed: Pick<ParsedLlmError, 'kind' | 'requestId'>, agent: string, now: number): string {
|
|
353
|
+
if (parsed.requestId) return `rid:${parsed.requestId}`
|
|
354
|
+
const bucket = Math.floor(now / ERROR_COLLAPSE_WINDOW_MS)
|
|
355
|
+
return `${parsed.kind}:${agent}:${bucket}`
|
|
356
|
+
}
|
|
357
|
+
|
|
358
|
+
/**
|
|
359
|
+
* Attempt to claim ownership of `key`. Returns true for the first caller
|
|
360
|
+
* within the window, false for every subsequent caller. Expired claims are
|
|
361
|
+
* pruned lazily on each call.
|
|
362
|
+
*/
|
|
363
|
+
claim(key: string, now: number = Date.now()): boolean {
|
|
364
|
+
this.prune(now)
|
|
365
|
+
const existing = this.claims.get(key)
|
|
366
|
+
if (existing != null && now - existing < ERROR_COLLAPSE_WINDOW_MS) {
|
|
367
|
+
return false
|
|
368
|
+
}
|
|
369
|
+
this.claims.set(key, now)
|
|
370
|
+
return true
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
/** True when `key` is currently claimed (does NOT claim). */
|
|
374
|
+
isClaimed(key: string, now: number = Date.now()): boolean {
|
|
375
|
+
const existing = this.claims.get(key)
|
|
376
|
+
return existing != null && now - existing < ERROR_COLLAPSE_WINDOW_MS
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
private prune(now: number): void {
|
|
380
|
+
for (const [k, at] of this.claims) {
|
|
381
|
+
if (now - at >= ERROR_COLLAPSE_WINDOW_MS) this.claims.delete(k)
|
|
382
|
+
}
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
/** Test-only: forget every claim. */
|
|
386
|
+
reset(): void {
|
|
387
|
+
this.claims.clear()
|
|
388
|
+
}
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/** The process-wide gate the three surfaces consult. */
|
|
392
|
+
export const errorPresenceGate = new ErrorPresenceGate()
|
|
393
|
+
|
|
394
|
+
/**
|
|
395
|
+
* The single decision a surface makes before rendering a parsed LLM error.
|
|
396
|
+
* Returns:
|
|
397
|
+
* - 'render' — this surface owns the humanized card, render it.
|
|
398
|
+
* - 'suppress' — another surface already owns it (dedup) OR it is a
|
|
399
|
+
* transient error the harness is still auto-retrying — stay silent.
|
|
400
|
+
*
|
|
401
|
+
* ACTIONABLE kinds (auth / quota_wall) ALWAYS return 'render' — they are never
|
|
402
|
+
* deduped away and never silenced mid-retry, because the operator must always
|
|
403
|
+
* see (and be able to act on) a login/quota wall.
|
|
404
|
+
*/
|
|
405
|
+
export function decideErrorSurface(
|
|
406
|
+
parsed: ParsedLlmError,
|
|
407
|
+
agent: string,
|
|
408
|
+
opts: { claim: boolean; now?: number; gate?: ErrorPresenceGate } = { claim: true },
|
|
409
|
+
): 'render' | 'suppress' {
|
|
410
|
+
const now = opts.now ?? Date.now()
|
|
411
|
+
const gate = opts.gate ?? errorPresenceGate
|
|
412
|
+
|
|
413
|
+
if (isActionableKind(parsed.kind)) {
|
|
414
|
+
// Always render — but still record the claim so a redundant transient
|
|
415
|
+
// surface for the same key stays collapsed.
|
|
416
|
+
if (opts.claim) gate.claim(gate.keyFor(parsed, agent, now), now)
|
|
417
|
+
return 'render'
|
|
418
|
+
}
|
|
419
|
+
|
|
420
|
+
// Auto-retry silence: a transient error still being retried internally is
|
|
421
|
+
// NOT surfaced on ANY surface until it goes terminal. NOTE: in the current
|
|
422
|
+
// production wiring the operator surface reaches here only AFTER session-tail
|
|
423
|
+
// (readNew → `errEvent.terminal || !errEvent.transient`) has already dropped
|
|
424
|
+
// in-flight transients, and the gateway calls `parseLlmError(detail)` with no
|
|
425
|
+
// retryState (the raw retry annotations don't survive the IPC hop) — so a
|
|
426
|
+
// production `parsed.autoRetrying` is always false here. This branch is the
|
|
427
|
+
// module's own guarantee for ANY caller that DOES pass retryState (unit-tested
|
|
428
|
+
// as such); prod silence is owned upstream at session-tail, not re-derived here.
|
|
429
|
+
if (parsed.autoRetrying && !parsed.terminal) return 'suppress'
|
|
430
|
+
|
|
431
|
+
const key = gate.keyFor(parsed, agent, now)
|
|
432
|
+
if (opts.claim) {
|
|
433
|
+
return gate.claim(key, now) ? 'render' : 'suppress'
|
|
434
|
+
}
|
|
435
|
+
return gate.isClaimed(key, now) ? 'suppress' : 'render'
|
|
436
|
+
}
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
*/
|
|
14
14
|
|
|
15
15
|
import { escapeMarkdown } from './format.js'
|
|
16
|
+
import { stripRawErrorBytes } from './raw-error-scrub.js'
|
|
16
17
|
|
|
17
18
|
// ─── Taxonomy ────────────────────────────────────────────────────────────────
|
|
18
19
|
|
|
@@ -216,7 +217,12 @@ export interface RenderResult {
|
|
|
216
217
|
*/
|
|
217
218
|
export function renderOperatorEvent(ev: OperatorEvent): RenderResult {
|
|
218
219
|
const agent = escapeMarkdown(ev.agent)
|
|
219
|
-
|
|
220
|
+
// #llm-error-surfacing: NEVER let a raw API-error byte-blob (`· b'{…}'`,
|
|
221
|
+
// trailing `{"type":"error"…}` JSON, `API Error:` prefix) reach a user. A
|
|
222
|
+
// clean human detail passes through unchanged; only smuggled raw bytes are
|
|
223
|
+
// stripped. This is the belt-and-braces scrub for every card kind — the
|
|
224
|
+
// rate-limited card in particular used to relay the raw synthetic-error text.
|
|
225
|
+
const detail = escapeMarkdown(stripRawErrorBytes(ev.detail))
|
|
220
226
|
|
|
221
227
|
switch (ev.kind) {
|
|
222
228
|
case 'credentials-expired':
|