switchroom 0.18.15 → 0.18.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/dist/agent-scheduler/index.js +16 -0
  2. package/dist/auth-broker/index.js +445 -10
  3. package/dist/cli/notion-write-pretool.mjs +16 -0
  4. package/dist/cli/switchroom.js +654 -479
  5. package/dist/host-control/main.js +20 -1
  6. package/dist/vault/approvals/kernel-server.js +16 -0
  7. package/dist/vault/broker/server.js +16 -0
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/bridge/bridge.ts +7 -1
  11. package/telegram-plugin/dist/bridge/bridge.js +26 -1
  12. package/telegram-plugin/dist/gateway/gateway.js +1758 -661
  13. package/telegram-plugin/dist/server.js +26 -1
  14. package/telegram-plugin/draft-stream.ts +78 -3
  15. package/telegram-plugin/fleet-fallback-resume.ts +26 -3
  16. package/telegram-plugin/gateway/approval-hold.ts +49 -0
  17. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +64 -22
  18. package/telegram-plugin/gateway/effort-command.ts +9 -7
  19. package/telegram-plugin/gateway/gateway.ts +627 -291
  20. package/telegram-plugin/gateway/linear-activity.ts +20 -4
  21. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  22. package/telegram-plugin/gateway/model-command.ts +96 -18
  23. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  24. package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
  25. package/telegram-plugin/gateway/session-model-file.ts +141 -172
  26. package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
  27. package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
  28. package/telegram-plugin/litellm-local-notice.ts +189 -0
  29. package/telegram-plugin/llm-error-present.ts +436 -0
  30. package/telegram-plugin/operator-events.ts +7 -1
  31. package/telegram-plugin/permission-title.ts +172 -10
  32. package/telegram-plugin/premium-recovery.ts +101 -0
  33. package/telegram-plugin/quota-watch.ts +16 -4
  34. package/telegram-plugin/raw-error-scrub.ts +73 -0
  35. package/telegram-plugin/retry-api-call.ts +8 -2
  36. package/telegram-plugin/runtime-metrics.ts +16 -0
  37. package/telegram-plugin/send-gate-degraded.test.ts +161 -8
  38. package/telegram-plugin/send-gate-observability.test.ts +140 -0
  39. package/telegram-plugin/send-gate-observability.ts +65 -20
  40. package/telegram-plugin/send-gate.test.ts +143 -1
  41. package/telegram-plugin/send-gate.ts +246 -23
  42. package/telegram-plugin/session-tail.ts +16 -0
  43. package/telegram-plugin/shared/local-time.ts +69 -0
  44. package/telegram-plugin/stream-controller.ts +143 -20
  45. package/telegram-plugin/stream-reply-handler.ts +12 -2
  46. package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
  47. package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
  48. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  49. package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
  50. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  51. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  52. package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
  53. package/telegram-plugin/tests/flood-windows-persistence.test.ts +5 -4
  54. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  55. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  56. package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
  57. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  58. package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
  59. package/telegram-plugin/tests/model-command.test.ts +84 -1
  60. package/telegram-plugin/tests/permission-title.test.ts +167 -4
  61. package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
  62. package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
  63. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  64. package/telegram-plugin/tests/reaction-gate-routing.test.ts +8 -3
  65. package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
  66. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  67. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  68. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  69. package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
  70. package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
  71. package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
  72. package/telegram-plugin/tests/worker-activity-feed.test.ts +212 -2
  73. package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
  74. package/telegram-plugin/tier-downgrade.ts +198 -0
  75. package/telegram-plugin/tool-activity-summary.ts +99 -0
  76. package/telegram-plugin/worker-activity-feed.ts +543 -368
@@ -0,0 +1,189 @@
1
+ /**
2
+ * litellm-local-notice.ts — debounced operator notice for LiteLLM-proxy-local
3
+ * 429s (pure module: state machine + text + config parsing, no IPC, no bot,
4
+ * no clock except the injected `now`).
5
+ *
6
+ * When an agent routes through the LiteLLM gateway and trips the proxy's OWN
7
+ * `tpm_limit`/`rpm_limit` limiter, `classify429Detail` (throttle-tier.ts)
8
+ * classifies the terminal 429 `litellm-local` and the gateway takes the calm
9
+ * path — no broker mark, no failover, no throttle tier (the request never
10
+ * reached Anthropic, so the condition says nothing about the account). But
11
+ * pre-notice the user-facing surface was the GENERIC "🚦 Rate limited" card,
12
+ * which reads like an Anthropic problem. This module owns the honest
13
+ * replacement:
14
+ *
15
+ * - ONE calm notice per agent per cooldown window (default 15 min,
16
+ * operator-tunable via channels.telegram.litellm_notice.window_ms in
17
+ * switchroom.yaml, projected into access.json by scaffold).
18
+ * - Further litellm-local 429s inside the window are counted SILENTLY;
19
+ * the first notice after the window expires carries "throttled N more
20
+ * times since the last notice".
21
+ * - A notice only ever fires on an actual throttle event — a quiet agent
22
+ * posts nothing (the state machine is evaluate-on-event, no timers).
23
+ *
24
+ * The copy must make clear this is the fleet token limiter (LiteLLM
25
+ * `tpm_limit`/`rpm_limit`), NOT an Anthropic account limit, and that the
26
+ * turn retries — no action needed.
27
+ *
28
+ * Side-effect sequencing lives in gateway/litellm-local-notice-wiring.ts.
29
+ * This module MUST NOT change classification, failover, or quota-ledger
30
+ * behavior — it only renders and debounces the notice.
31
+ */
32
+
33
+ import { escapeMarkdown } from './card-format.js'
34
+ import type { RateLimit429Classification } from './throttle-tier.js'
35
+ import type { RuntimeMetricEvent } from './runtime-metrics.js'
36
+
37
+ // ─── Cooldown window config ──────────────────────────────────────────────────
38
+
39
+ /**
40
+ * Default per-agent notice cooldown. 15 minutes: long enough that a sustained
41
+ * burst produces a handful of notices per hour at most, short enough that the
42
+ * operator still sees the limiter working while it's working.
43
+ */
44
+ export const LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT = 15 * 60_000
45
+
46
+ /**
47
+ * Resolve the cooldown window from the raw access-file value
48
+ * (`access.litellmNoticeWindowMs`, projected by scaffold from
49
+ * channels.telegram.litellm_notice.window_ms). Any non-finite, non-positive,
50
+ * or non-number value falls back to the default — an operator typo must
51
+ * never produce a zero-width window (notice storm) or a NaN comparison
52
+ * (notices never fire again).
53
+ */
54
+ export function parseLitellmNoticeWindowMs(raw: unknown): number {
55
+ if (typeof raw !== 'number') return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
56
+ if (!Number.isFinite(raw) || raw <= 0) return LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT
57
+ return raw
58
+ }
59
+
60
+ // ─── Per-agent cooldown state machine ────────────────────────────────────────
61
+
62
+ export interface LitellmLocalNoticeState {
63
+ /** agent → unix ms of the last notice sent for it. */
64
+ lastSentAtMsByAgent: Record<string, number>
65
+ /** agent → litellm-local 429s counted silently since the last notice. */
66
+ suppressedCountByAgent: Record<string, number>
67
+ }
68
+
69
+ export function initialLitellmLocalNoticeState(): LitellmLocalNoticeState {
70
+ return { lastSentAtMsByAgent: {}, suppressedCountByAgent: {} }
71
+ }
72
+
73
+ export interface LitellmLocalNoticeVerdict {
74
+ send: boolean
75
+ /** Silent throttle events since the previous notice (0 on the first). Only
76
+ * meaningful when `send` is true — the renderer adds the "N more times"
77
+ * line when > 0. */
78
+ suppressedSinceLastNotice: number
79
+ next: LitellmLocalNoticeState
80
+ }
81
+
82
+ /**
83
+ * Evaluate one litellm-local 429 for `agent` at `now`.
84
+ *
85
+ * - First event ever (or ≥ windowMs since the last notice) → send; the
86
+ * verdict carries the silently-counted events since the previous notice
87
+ * and the counter resets.
88
+ * - Inside the window → suppress; the counter increments.
89
+ *
90
+ * Boundary is INCLUSIVE on expiry (elapsed === windowMs sends), matching
91
+ * `evaluateThrottleNotice` in throttle-tier.ts. Pure: returns the next
92
+ * state, never mutates `prev`.
93
+ */
94
+ export function evaluateLitellmLocalNotice(
95
+ prev: LitellmLocalNoticeState,
96
+ agent: string,
97
+ now: number,
98
+ windowMs: number = LITELLM_LOCAL_NOTICE_WINDOW_MS_DEFAULT,
99
+ ): LitellmLocalNoticeVerdict {
100
+ const last = prev.lastSentAtMsByAgent[agent]
101
+ if (last == null || now - last >= windowMs) {
102
+ return {
103
+ send: true,
104
+ suppressedSinceLastNotice: prev.suppressedCountByAgent[agent] ?? 0,
105
+ next: {
106
+ lastSentAtMsByAgent: { ...prev.lastSentAtMsByAgent, [agent]: now },
107
+ suppressedCountByAgent: { ...prev.suppressedCountByAgent, [agent]: 0 },
108
+ },
109
+ }
110
+ }
111
+ return {
112
+ send: false,
113
+ suppressedSinceLastNotice: 0,
114
+ next: {
115
+ lastSentAtMsByAgent: prev.lastSentAtMsByAgent,
116
+ suppressedCountByAgent: {
117
+ ...prev.suppressedCountByAgent,
118
+ [agent]: (prev.suppressedCountByAgent[agent] ?? 0) + 1,
119
+ },
120
+ },
121
+ }
122
+ }
123
+
124
+ // ─── Notice rendering ────────────────────────────────────────────────────────
125
+
126
+ /**
127
+ * The ONE calm notice for a litellm-local 429. Markdown (not HTML) per the
128
+ * card-format.ts convention. Copy contract (operator spec):
129
+ * - names the fleet token limiter (LiteLLM `tpm_limit`/`rpm_limit`)
130
+ * - explicitly NOT an Anthropic account limit — nothing exhausted, no
131
+ * account touched
132
+ * - the turn retries automatically; no action needed
133
+ * - after a window with silent suppressions, says "throttled N more
134
+ * time(s) since the last notice"
135
+ */
136
+ export function renderLitellmLocalNotice(opts: {
137
+ agent: string
138
+ /** Silent litellm-local 429s since the previous notice (0 on the first). */
139
+ suppressedSinceLastNotice: number
140
+ }): string {
141
+ const agent = escapeMarkdown(opts.agent)
142
+ const n = opts.suppressedSinceLastNotice
143
+ const lines = [
144
+ `🚦 **Fleet token limiter engaged** — **${agent}** hit the local LiteLLM proxy cap (\`tpm_limit\`/\`rpm_limit\`).`,
145
+ `This is switchroom's own fleet limiter smoothing a burst, not an Anthropic account limit — nothing is exhausted and no account was touched.`,
146
+ ]
147
+ if (n > 0) {
148
+ lines.push(`Throttled ${n} more ${n === 1 ? 'time' : 'times'} since the last notice.`)
149
+ }
150
+ lines.push(`_The turn retries automatically — no action needed._`)
151
+ return lines.join('\n')
152
+ }
153
+
154
+ // ─── Metric builder ──────────────────────────────────────────────────────────
155
+
156
+ /**
157
+ * Build the `litellm_local_429_notice` runtime metric for one SENT notice.
158
+ * Distinct from `rate_limit_429_classified` (which fires on EVERY classified
159
+ * event): this one fires only when a notice actually posts, and carries how
160
+ * many events the window silently absorbed — the debounce's effectiveness in
161
+ * one number. Pure builder so the payload shape is unit-testable; the wiring
162
+ * emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
163
+ */
164
+ export function buildLitellmLocalNoticeMetric(opts: {
165
+ agent: string
166
+ suppressedCount: number
167
+ windowMs: number
168
+ }): Extract<RuntimeMetricEvent, { kind: 'litellm_local_429_notice' }> {
169
+ return {
170
+ kind: 'litellm_local_429_notice',
171
+ agent: opts.agent,
172
+ suppressed_count: opts.suppressedCount,
173
+ window_ms: opts.windowMs,
174
+ }
175
+ }
176
+
177
+ // ─── Classification guard ────────────────────────────────────────────────────
178
+
179
+ /**
180
+ * True only for the `litellm-local` classification. The wiring runs this
181
+ * guard so "a notice never fires for non-litellm-local classifications" is a
182
+ * deterministic mechanism, not caller discipline — account-scoped 429s keep
183
+ * the throttle tier, generic transients keep the calm rate-limited card.
184
+ */
185
+ export function isLitellmLocalNoticeEligible(
186
+ classification: RateLimit429Classification | null | undefined,
187
+ ): classification is 'litellm-local' {
188
+ return classification === 'litellm-local'
189
+ }
@@ -0,0 +1,436 @@
1
+ /**
2
+ * llm-error-present.ts — humanized, cross-surface-deduped presentation of an
3
+ * LLM/API error, so a raw `b'{"type":"error",…}'` line never reaches a user.
4
+ *
5
+ * THE PROBLEM this closes
6
+ * -----------------------
7
+ * One Anthropic error JSONL line fans out to THREE independent render surfaces,
8
+ * none consulting each other, each historically printing the raw `detail`
9
+ * string with its `b'{…}'` byte-blob attached:
10
+ * 1. the agent turn-end "done" card (tool-activity-summary result block),
11
+ * 2. the operator-event "🚦 Rate limited" card (raw `detail`), and
12
+ * 3. the reply/answer passthrough (the synthetic-assistant error text relayed
13
+ * as the turn's answer).
14
+ *
15
+ * This module owns the ONE clean rendering + the ONE dedup authority the three
16
+ * surfaces consult, so a fan-out produces EXACTLY ONE user-facing message with
17
+ * NO raw JSON. It reuses the existing classifiers rather than re-deriving them:
18
+ * `classify429Detail` (throttle-tier.ts), `detectModelUnavailable` /
19
+ * `parseResetTime` (model-unavailable.ts), `isLitellmProxyLocal429` /
20
+ * `isTransientUpstreamSignal` (model-unavailable.ts), `classifyClaudeError`
21
+ * (operator-events.ts). `coreText` is ALWAYS built from a per-kind template —
22
+ * never from the raw `detail` — and `stripRawErrorBytes` is the belt-and-braces
23
+ * scrub so even the enrichment `reason` can carry no JSON.
24
+ *
25
+ * Pure module: no IPC, no bot, no FS. The only mutable state is the
26
+ * `ErrorPresenceGate` singleton (an in-memory dedup ledger), driven by an
27
+ * injectable `now`.
28
+ */
29
+
30
+ import {
31
+ detectModelUnavailable,
32
+ isLitellmProxyLocal429,
33
+ parseResetTime,
34
+ } from './model-unavailable.js'
35
+ import { classify429Detail } from './throttle-tier.js'
36
+ import { classifyClaudeError } from './operator-events.js'
37
+ import type { InlineKeyboardMarkup } from './operator-events.js'
38
+ import { stripRawErrorBytes, extractRequestId } from './raw-error-scrub.js'
39
+ import { fmtLocalClock, tzAbbrev } from './shared/local-time.js'
40
+
41
+ export { stripRawErrorBytes, extractRequestId } from './raw-error-scrub.js'
42
+
43
+ // ─── Types ──────────────────────────────────────────────────────────────────
44
+
45
+ export type LlmErrorKind =
46
+ | 'rate_limit'
47
+ | 'overload_529'
48
+ | 'quota_wall'
49
+ | 'auth'
50
+ | 'transient'
51
+ | 'unknown'
52
+
53
+ export type LlmErrorSource = 'anthropic' | 'litellm-local' | 'network'
54
+
55
+ export interface ParsedLlmError {
56
+ kind: LlmErrorKind
57
+ /** Human, JSON-STRIPPED one-liner built from a per-kind template. Never raw. */
58
+ coreText: string
59
+ /** Parsed reset instant, when the source carried one. */
60
+ resetAt?: Date
61
+ /** Parsed retry-after window in ms, when the source carried one. */
62
+ retryAfterMs?: number
63
+ /** Resolved model id, when the source named one. */
64
+ model?: string
65
+ /** Anthropic `request_id`, when present — the strongest dedup key. */
66
+ requestId?: string
67
+ source: LlmErrorSource
68
+ /** True when the harness is still retrying this error internally (mid-retry). */
69
+ autoRetrying: boolean
70
+ /** True when the failure is final (NOT an in-flight retry). */
71
+ terminal: boolean
72
+ }
73
+
74
+ /** Optional retry-state annotations Claude Code stamps on a retried error line. */
75
+ export interface LlmErrorRetryState {
76
+ retryAttempt: number | null
77
+ maxRetries: number | null
78
+ }
79
+
80
+ // ─── model extraction ────────────────────────────────────────────────────────
81
+
82
+ /** Pull a resolved model id (`claude-…`, `sr-…`) out of a raw string, or undefined. */
83
+ function extractModel(raw: string): string | undefined {
84
+ const m = raw.match(/["']?model["']?\s*[=:]\s*["']?((?:claude|sr)[A-Za-z0-9._-]+)/i)
85
+ return m ? m[1] : undefined
86
+ }
87
+
88
+ // ─── Parser ──────────────────────────────────────────────────────────────────
89
+
90
+ const TRANSIENT_KINDS: ReadonlySet<LlmErrorKind> = new Set<LlmErrorKind>([
91
+ 'rate_limit',
92
+ 'overload_529',
93
+ 'transient',
94
+ ])
95
+
96
+ /** The always-actionable kinds — NEVER silenced, even inside a collapse window. */
97
+ export function isActionableKind(kind: LlmErrorKind): boolean {
98
+ return kind === 'auth' || kind === 'quota_wall'
99
+ }
100
+
101
+ /**
102
+ * Parse a raw error string (a session-tail `detail`, an isApiErrorMessage
103
+ * text, or a raw stderr line) into a structured, JSON-free ParsedLlmError.
104
+ * Reuses the existing wording classifiers; maps their kinds into this union.
105
+ */
106
+ export function parseLlmError(
107
+ raw: string,
108
+ retryState?: LlmErrorRetryState,
109
+ ): ParsedLlmError {
110
+ const text = typeof raw === 'string' ? raw : ''
111
+ const requestId = extractRequestId(text)
112
+ const model = extractModel(text)
113
+
114
+ // Reset / retry-after best-effort (Anthropic + relative wordings).
115
+ const resetAt = parseResetTime(text)
116
+ const retryAfterMs =
117
+ resetAt != null ? Math.max(0, resetAt.getTime() - Date.now()) : undefined
118
+
119
+ const { kind, source } = classifyKindAndSource(text)
120
+
121
+ // Retry / terminal semantics. Only the transient family can be mid-retry;
122
+ // auth / quota_wall / model_unavailable are terminal by construction.
123
+ let autoRetrying = false
124
+ let terminal = true
125
+ if (TRANSIENT_KINDS.has(kind)) {
126
+ const { retryAttempt, maxRetries } = retryState ?? { retryAttempt: null, maxRetries: null }
127
+ if (retryAttempt != null && maxRetries != null) {
128
+ autoRetrying = retryAttempt < maxRetries
129
+ terminal = retryAttempt >= maxRetries
130
+ } else {
131
+ // No retry annotation: treat as a surfaced (terminal) failure — Claude
132
+ // writes the user-facing error shape only after its own retries are done.
133
+ autoRetrying = false
134
+ terminal = true
135
+ }
136
+ }
137
+
138
+ return {
139
+ kind,
140
+ coreText: buildCoreText(kind, source),
141
+ ...(resetAt != null ? { resetAt } : {}),
142
+ ...(retryAfterMs != null ? { retryAfterMs } : {}),
143
+ ...(model != null ? { model } : {}),
144
+ ...(requestId != null ? { requestId } : {}),
145
+ source,
146
+ autoRetrying,
147
+ terminal,
148
+ }
149
+ }
150
+
151
+ function classifyKindAndSource(text: string): { kind: LlmErrorKind; source: LlmErrorSource } {
152
+ const lower = text.toLowerCase()
153
+
154
+ // 1. Auth — always terminal, always actionable.
155
+ const claudeKind = classifyClaudeError({ message: text, type: text })
156
+ if (claudeKind === 'credentials-expired' || claudeKind === 'credentials-invalid') {
157
+ return { kind: 'auth', source: 'anthropic' }
158
+ }
159
+
160
+ // 2. LiteLLM-proxy-LOCAL 429 — the proxy's own limiter, never Anthropic.
161
+ // Checked before the generic quota/overload matchers because some LiteLLM
162
+ // bodies contain the word "limit" that a quota matcher could seize on.
163
+ if (isLitellmProxyLocal429(text)) {
164
+ return { kind: 'rate_limit', source: 'litellm-local' }
165
+ }
166
+
167
+ // 3. Model-unavailable detector owns quota-wall / overload / network wording.
168
+ const mu = detectModelUnavailable(text)
169
+ if (mu != null) {
170
+ if (mu.kind === 'quota_exhausted') return { kind: 'quota_wall', source: 'anthropic' }
171
+ if (mu.kind === 'network') return { kind: 'transient', source: 'network' }
172
+ // mu.kind === 'overload' | 'rate_limited' — split 529 overload from a 429.
173
+ if (
174
+ lower.includes('529') ||
175
+ lower.includes('overloaded')
176
+ ) {
177
+ return { kind: 'overload_529', source: 'anthropic' }
178
+ }
179
+ // A 429-family transient. Three-way classify picks the source nuance; all
180
+ // land on the calm rate_limit kind here.
181
+ const c = classify429Detail(text)
182
+ return {
183
+ kind: 'rate_limit',
184
+ source: c === 'litellm-local' ? 'litellm-local' : 'anthropic',
185
+ }
186
+ }
187
+
188
+ // 4. Credit balance is a quota-family wall (actionable).
189
+ if (claudeKind === 'credit-exhausted') {
190
+ return { kind: 'quota_wall', source: 'anthropic' }
191
+ }
192
+
193
+ // 5. Bare rate-limit classification from the operator taxonomy.
194
+ if (claudeKind === 'rate-limited') {
195
+ return { kind: 'rate_limit', source: 'anthropic' }
196
+ }
197
+ if (claudeKind === 'unknown-5xx') {
198
+ return { kind: 'overload_529', source: 'anthropic' }
199
+ }
200
+
201
+ return { kind: 'unknown', source: 'anthropic' }
202
+ }
203
+
204
+ function buildCoreText(kind: LlmErrorKind, source: LlmErrorSource): string {
205
+ switch (kind) {
206
+ case 'rate_limit':
207
+ return source === 'litellm-local'
208
+ ? 'Hit the local proxy rate limit — retrying automatically.'
209
+ : 'Rate limited by Anthropic — retrying automatically.'
210
+ case 'overload_529':
211
+ return 'Anthropic is overloaded (529) — retrying automatically.'
212
+ case 'quota_wall':
213
+ return 'Usage limit reached on this Claude subscription.'
214
+ case 'auth':
215
+ return 'Claude login needs re-authentication.'
216
+ case 'transient':
217
+ return source === 'network'
218
+ ? "Couldn't reach Anthropic (network) — retrying automatically."
219
+ : 'A temporary upstream hiccup — retrying automatically.'
220
+ case 'unknown':
221
+ return 'The model returned an error.'
222
+ }
223
+ }
224
+
225
+ // ─── Rendering ───────────────────────────────────────────────────────────────
226
+
227
+ export interface RenderedLlmError {
228
+ text: string
229
+ keyboard?: InlineKeyboardMarkup
230
+ }
231
+
232
+ /**
233
+ * Format a reset instant in the operator's local tz plus a relative tail, e.g.
234
+ * `clears ~4:52pm AEST (~in 38m)`. Returns '' when there is no reset to show.
235
+ */
236
+ export function formatResetLocal(resetAt: Date | undefined, tz: string, now: Date = new Date()): string {
237
+ if (resetAt == null) return ''
238
+ const ms = resetAt.getTime()
239
+ if (!Number.isFinite(ms)) return ''
240
+ const clock = fmtLocalClock(ms, tz)
241
+ const abbrev = tzAbbrev(ms, tz)
242
+ const rel = formatRelativeTail(ms - now.getTime())
243
+ return rel ? `clears ~${clock} ${abbrev} (${rel})` : `clears ~${clock} ${abbrev}`
244
+ }
245
+
246
+ function formatRelativeTail(deltaMs: number): string {
247
+ if (deltaMs <= 0) return '~now'
248
+ const totalMin = Math.round(deltaMs / 60_000)
249
+ if (totalMin < 1) return '~in <1m'
250
+ if (totalMin < 60) return `~in ${totalMin}m`
251
+ const hours = Math.floor(totalMin / 60)
252
+ const mins = totalMin % 60
253
+ if (hours < 24) return mins > 0 ? `~in ${hours}h ${mins}m` : `~in ${hours}h`
254
+ const days = Math.floor(hours / 24)
255
+ const remH = hours % 24
256
+ return remH > 0 ? `~in ${days}d ${remH}h` : `~in ${days}d`
257
+ }
258
+
259
+ /**
260
+ * Render ONE clean card for a parsed LLM error. Action buttons for the
261
+ * actionable kinds (auth → Reauth, quota_wall → Wait/switch). `agent` is used
262
+ * in the callback_data (URL-encoded) and the headline; `tz` localizes the reset.
263
+ */
264
+ export function renderLlmError(
265
+ parsed: ParsedLlmError,
266
+ agent: string,
267
+ tz: string,
268
+ now: Date = new Date(),
269
+ ): RenderedLlmError {
270
+ const safeAgent = escapeAgent(agent)
271
+ const emoji = kindEmoji(parsed.kind)
272
+ const lines: string[] = [`${emoji} ${parsed.coreText} (**${safeAgent}**)`]
273
+
274
+ const resetLine = formatResetLocal(parsed.resetAt, tz, now)
275
+ if (resetLine) lines.push(`_${resetLine}_`)
276
+
277
+ if (parsed.model) lines.push(`_model: ${escapeAgent(parsed.model)}_`)
278
+
279
+ const text = lines.join('\n')
280
+
281
+ switch (parsed.kind) {
282
+ case 'auth':
283
+ return {
284
+ text,
285
+ keyboard: {
286
+ inline_keyboard: [
287
+ [
288
+ { text: '🔐 Reauth now', callback_data: `op:reauth:${encodeURIComponent(agent)}` },
289
+ { text: '❌ Dismiss', callback_data: `op:dismiss:${encodeURIComponent(agent)}` },
290
+ ],
291
+ ],
292
+ },
293
+ }
294
+ case 'quota_wall':
295
+ return {
296
+ text,
297
+ keyboard: {
298
+ inline_keyboard: [
299
+ [{ text: '⏳ Wait', callback_data: `op:dismiss:${encodeURIComponent(agent)}` }],
300
+ ],
301
+ },
302
+ }
303
+ default:
304
+ return { text }
305
+ }
306
+ }
307
+
308
+ function kindEmoji(kind: LlmErrorKind): string {
309
+ switch (kind) {
310
+ case 'rate_limit':
311
+ return '🚦'
312
+ case 'overload_529':
313
+ return '🔥'
314
+ case 'quota_wall':
315
+ return '⚠️'
316
+ case 'auth':
317
+ return '🔑'
318
+ case 'transient':
319
+ return '🌐'
320
+ case 'unknown':
321
+ return '⚠️'
322
+ }
323
+ }
324
+
325
+ /** Minimal markdown-safe agent rendering (mirrors operator-events escapeMarkdown intent). */
326
+ function escapeAgent(s: string): string {
327
+ return s.replace(/([_*`\[\]])/g, '\\$1')
328
+ }
329
+
330
+ // ─── Cross-surface dedup: ErrorPresenceGate ──────────────────────────────────
331
+
332
+ /**
333
+ * How long ONE terminal error owns the "already surfaced" claim across all
334
+ * three surfaces. Distinct from — and additional to — the 5-minute per-kind
335
+ * `shouldEmitOperatorEvent` cooldown (operator-events.ts): that debounces an
336
+ * error STORM on one surface; this collapses ONE error's FAN-OUT across
337
+ * surfaces within a single turn.
338
+ */
339
+ export const ERROR_COLLAPSE_WINDOW_MS = 60_000
340
+
341
+ /**
342
+ * A tiny in-memory dedup ledger. The FIRST surface to `claim(key)` within the
343
+ * collapse window wins (returns true) and renders; every later surface loses
344
+ * (returns false) and suppresses. Key preference: the Anthropic `request_id`
345
+ * when present (exact), else a `${kind}:${agent}:${windowBucket}` coarse key so
346
+ * two distinct-but-unlabelled errors of the same kind within 60s still collapse.
347
+ */
348
+ export class ErrorPresenceGate {
349
+ private readonly claims = new Map<string, number>()
350
+
351
+ /** Build the dedup key for a parsed error + agent. */
352
+ keyFor(parsed: Pick<ParsedLlmError, 'kind' | 'requestId'>, agent: string, now: number): string {
353
+ if (parsed.requestId) return `rid:${parsed.requestId}`
354
+ const bucket = Math.floor(now / ERROR_COLLAPSE_WINDOW_MS)
355
+ return `${parsed.kind}:${agent}:${bucket}`
356
+ }
357
+
358
+ /**
359
+ * Attempt to claim ownership of `key`. Returns true for the first caller
360
+ * within the window, false for every subsequent caller. Expired claims are
361
+ * pruned lazily on each call.
362
+ */
363
+ claim(key: string, now: number = Date.now()): boolean {
364
+ this.prune(now)
365
+ const existing = this.claims.get(key)
366
+ if (existing != null && now - existing < ERROR_COLLAPSE_WINDOW_MS) {
367
+ return false
368
+ }
369
+ this.claims.set(key, now)
370
+ return true
371
+ }
372
+
373
+ /** True when `key` is currently claimed (does NOT claim). */
374
+ isClaimed(key: string, now: number = Date.now()): boolean {
375
+ const existing = this.claims.get(key)
376
+ return existing != null && now - existing < ERROR_COLLAPSE_WINDOW_MS
377
+ }
378
+
379
+ private prune(now: number): void {
380
+ for (const [k, at] of this.claims) {
381
+ if (now - at >= ERROR_COLLAPSE_WINDOW_MS) this.claims.delete(k)
382
+ }
383
+ }
384
+
385
+ /** Test-only: forget every claim. */
386
+ reset(): void {
387
+ this.claims.clear()
388
+ }
389
+ }
390
+
391
+ /** The process-wide gate the three surfaces consult. */
392
+ export const errorPresenceGate = new ErrorPresenceGate()
393
+
394
+ /**
395
+ * The single decision a surface makes before rendering a parsed LLM error.
396
+ * Returns:
397
+ * - 'render' — this surface owns the humanized card, render it.
398
+ * - 'suppress' — another surface already owns it (dedup) OR it is a
399
+ * transient error the harness is still auto-retrying — stay silent.
400
+ *
401
+ * ACTIONABLE kinds (auth / quota_wall) ALWAYS return 'render' — they are never
402
+ * deduped away and never silenced mid-retry, because the operator must always
403
+ * see (and be able to act on) a login/quota wall.
404
+ */
405
+ export function decideErrorSurface(
406
+ parsed: ParsedLlmError,
407
+ agent: string,
408
+ opts: { claim: boolean; now?: number; gate?: ErrorPresenceGate } = { claim: true },
409
+ ): 'render' | 'suppress' {
410
+ const now = opts.now ?? Date.now()
411
+ const gate = opts.gate ?? errorPresenceGate
412
+
413
+ if (isActionableKind(parsed.kind)) {
414
+ // Always render — but still record the claim so a redundant transient
415
+ // surface for the same key stays collapsed.
416
+ if (opts.claim) gate.claim(gate.keyFor(parsed, agent, now), now)
417
+ return 'render'
418
+ }
419
+
420
+ // Auto-retry silence: a transient error still being retried internally is
421
+ // NOT surfaced on ANY surface until it goes terminal. NOTE: in the current
422
+ // production wiring the operator surface reaches here only AFTER session-tail
423
+ // (readNew → `errEvent.terminal || !errEvent.transient`) has already dropped
424
+ // in-flight transients, and the gateway calls `parseLlmError(detail)` with no
425
+ // retryState (the raw retry annotations don't survive the IPC hop) — so a
426
+ // production `parsed.autoRetrying` is always false here. This branch is the
427
+ // module's own guarantee for ANY caller that DOES pass retryState (unit-tested
428
+ // as such); prod silence is owned upstream at session-tail, not re-derived here.
429
+ if (parsed.autoRetrying && !parsed.terminal) return 'suppress'
430
+
431
+ const key = gate.keyFor(parsed, agent, now)
432
+ if (opts.claim) {
433
+ return gate.claim(key, now) ? 'render' : 'suppress'
434
+ }
435
+ return gate.isClaimed(key, now) ? 'suppress' : 'render'
436
+ }
@@ -13,6 +13,7 @@
13
13
  */
14
14
 
15
15
  import { escapeMarkdown } from './format.js'
16
+ import { stripRawErrorBytes } from './raw-error-scrub.js'
16
17
 
17
18
  // ─── Taxonomy ────────────────────────────────────────────────────────────────
18
19
 
@@ -216,7 +217,12 @@ export interface RenderResult {
216
217
  */
217
218
  export function renderOperatorEvent(ev: OperatorEvent): RenderResult {
218
219
  const agent = escapeMarkdown(ev.agent)
219
- const detail = escapeMarkdown(ev.detail)
220
+ // #llm-error-surfacing: NEVER let a raw API-error byte-blob (`· b'{…}'`,
221
+ // trailing `{"type":"error"…}` JSON, `API Error:` prefix) reach a user. A
222
+ // clean human detail passes through unchanged; only smuggled raw bytes are
223
+ // stripped. This is the belt-and-braces scrub for every card kind — the
224
+ // rate-limited card in particular used to relay the raw synthetic-error text.
225
+ const detail = escapeMarkdown(stripRawErrorBytes(ev.detail))
220
226
 
221
227
  switch (ev.kind) {
222
228
  case 'credentials-expired':