switchroom 0.18.17 → 0.18.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +13 -0
- package/dist/auth-broker/index.js +13 -0
- package/dist/cli/notion-write-pretool.mjs +13 -0
- package/dist/cli/switchroom.js +605 -479
- package/dist/host-control/main.js +17 -1
- package/dist/vault/approvals/kernel-server.js +13 -0
- package/dist/vault/broker/server.js +13 -0
- package/package.json +1 -1
- package/telegram-plugin/bridge/bridge.ts +7 -1
- package/telegram-plugin/dist/bridge/bridge.js +26 -1
- package/telegram-plugin/dist/gateway/gateway.js +1401 -431
- package/telegram-plugin/dist/server.js +26 -1
- package/telegram-plugin/fleet-fallback-resume.ts +26 -3
- package/telegram-plugin/gateway/approval-hold.ts +49 -0
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +61 -18
- package/telegram-plugin/gateway/gateway.ts +362 -71
- package/telegram-plugin/gateway/linear-activity.ts +20 -4
- package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
- package/telegram-plugin/gateway/session-model-file.ts +103 -0
- package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
- package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
- package/telegram-plugin/llm-error-present.ts +436 -0
- package/telegram-plugin/operator-events.ts +7 -1
- package/telegram-plugin/permission-title.ts +172 -10
- package/telegram-plugin/premium-recovery.ts +101 -0
- package/telegram-plugin/raw-error-scrub.ts +73 -0
- package/telegram-plugin/retry-api-call.ts +8 -2
- package/telegram-plugin/send-gate-degraded.test.ts +152 -1
- package/telegram-plugin/send-gate-observability.test.ts +140 -0
- package/telegram-plugin/send-gate-observability.ts +65 -20
- package/telegram-plugin/send-gate.test.ts +143 -1
- package/telegram-plugin/send-gate.ts +212 -19
- package/telegram-plugin/session-tail.ts +16 -0
- package/telegram-plugin/shared/local-time.ts +69 -0
- package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
- package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
- package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
- package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +3 -2
- package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
- package/telegram-plugin/tests/llm-error-present.test.ts +380 -0
- package/telegram-plugin/tests/permission-title.test.ts +167 -4
- package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
- package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +6 -1
- package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
- package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
- package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
- package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
- package/telegram-plugin/tests/worker-activity-feed.test.ts +5 -2
- package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
- package/telegram-plugin/tier-downgrade.ts +198 -0
- package/telegram-plugin/tool-activity-summary.ts +99 -0
- package/telegram-plugin/worker-activity-feed.ts +509 -409
|
@@ -122,6 +122,45 @@ describe('createFleetFallbackResumeGate — staleness guard', () => {
|
|
|
122
122
|
})
|
|
123
123
|
})
|
|
124
124
|
|
|
125
|
+
describe('createFleetFallbackResumeGate — peek / arm seams (tier-downgrade LOW-3 + LOW-9a)', () => {
|
|
126
|
+
it('peek() evaluates the verdict WITHOUT arming the single-flight latch', () => {
|
|
127
|
+
const clk = fakeClock()
|
|
128
|
+
const gate = createFleetFallbackResumeGate({ nowFn: clk.now })
|
|
129
|
+
// Peeking 'resume' must not record an arm time — a following peek/decide is
|
|
130
|
+
// still 'resume'. This is what lets the gateway do a fallible carrier write
|
|
131
|
+
// between the peek and the arm without leaving a phantom armed latch (LOW-3).
|
|
132
|
+
expect(gate.peek(clk.now())).toBe('resume')
|
|
133
|
+
expect(gate.inspect().lastResumedAtMs).toBe(Number.NEGATIVE_INFINITY)
|
|
134
|
+
expect(gate.peek(clk.now())).toBe('resume')
|
|
135
|
+
// The real restart never fires (write threw) → nothing armed → a genuine
|
|
136
|
+
// later swap still resumes.
|
|
137
|
+
expect(gate.decide(clk.now())).toBe('resume')
|
|
138
|
+
})
|
|
139
|
+
|
|
140
|
+
it('arm() commits the single-flight window so a later peek reports skip-inflight (LOW-9a)', () => {
|
|
141
|
+
const clk = fakeClock()
|
|
142
|
+
const gate = createFleetFallbackResumeGate({ nowFn: clk.now, singleFlightMs: 60_000 })
|
|
143
|
+
// Turn 1: peek 'resume' → do the write → arm.
|
|
144
|
+
expect(gate.peek(clk.now())).toBe('resume')
|
|
145
|
+
gate.arm()
|
|
146
|
+
// Turn 2 (concurrent, same process, 1s later): a resume restart is already
|
|
147
|
+
// armed, so peek reports skip-inflight → the gateway suppresses the give-up.
|
|
148
|
+
clk.advance(1_000)
|
|
149
|
+
expect(gate.peek(clk.now())).toBe('skip-inflight')
|
|
150
|
+
})
|
|
151
|
+
|
|
152
|
+
it('peek(skip-inflight/skip-stale) never arms; only arm() records the window', () => {
|
|
153
|
+
const clk = fakeClock()
|
|
154
|
+
const gate = createFleetFallbackResumeGate({ nowFn: clk.now })
|
|
155
|
+
// A stale peek must not arm (mirrors decide()'s stale behaviour).
|
|
156
|
+
expect(gate.peek(clk.now() - (DEFAULT_RESUME_MAX_AGE_MS + 1))).toBe('skip-stale')
|
|
157
|
+
expect(gate.inspect().lastResumedAtMs).toBe(Number.NEGATIVE_INFINITY)
|
|
158
|
+
// decide() remains equivalent to peek+arm-on-resume.
|
|
159
|
+
expect(gate.decide(clk.now())).toBe('resume')
|
|
160
|
+
expect(gate.peek(clk.now())).toBe('skip-inflight')
|
|
161
|
+
})
|
|
162
|
+
})
|
|
163
|
+
|
|
125
164
|
describe('createFleetFallbackResumeGate — reset / inspect seams', () => {
|
|
126
165
|
it('reset() clears the single-flight arm', () => {
|
|
127
166
|
const clk = fakeClock()
|
|
@@ -103,9 +103,10 @@ describe('#3084 scoped flood-window persistence', () => {
|
|
|
103
103
|
),
|
|
104
104
|
).rejects.toBe(floodErr)
|
|
105
105
|
|
|
106
|
-
// The window is on disk.
|
|
106
|
+
// The window is on disk. #3111: a chat-bound 429 persists ONLY the chat
|
|
107
|
+
// scope — no coincident `global` window that would suppress unrelated chats.
|
|
107
108
|
const persisted = readFloodWindows(winPath, 0)
|
|
108
|
-
expect(persisted.map((w) => w.scopeKey).sort()).toEqual(['chat:7'
|
|
109
|
+
expect(persisted.map((w) => w.scopeKey).sort()).toEqual(['chat:7'])
|
|
109
110
|
|
|
110
111
|
// Gate 2 (simulated restart): built BEFORE any outbound call from the
|
|
111
112
|
// persisted windows. A cosmetic send on chat:7 must still shed.
|
|
@@ -155,9 +155,14 @@ describe('createLinearIssue — behaviour (#2312)', () => {
|
|
|
155
155
|
expect(r.content[0].text).toMatch(/default_team_id/)
|
|
156
156
|
})
|
|
157
157
|
|
|
158
|
-
it('short-circuits to "Already filed" on
|
|
158
|
+
it('short-circuits to "Already filed" only on an EXACT marker match, skipping unrelated top hits', async () => {
|
|
159
159
|
const { fetchImpl, calls } = routingFetch({
|
|
160
|
-
searchIssues: () => ({ searchIssues: { nodes: [
|
|
160
|
+
searchIssues: () => ({ searchIssues: { nodes: [
|
|
161
|
+
// relevance-ranked top hit that merely shares tokens — must be ignored.
|
|
162
|
+
{ id: 'noise', url: 'https://linear.app/acme/issue/ENG-99', title: 'unrelated', description: 'no capture marker here' },
|
|
163
|
+
// the real prior capture, carrying the exact marker for this key.
|
|
164
|
+
{ id: 'old', url: 'https://linear.app/acme/issue/ENG-7', title: 'prior', description: `Y${captureDedupMarker('chat:99')}` },
|
|
165
|
+
] } }),
|
|
161
166
|
teams: oneTeam,
|
|
162
167
|
issueCreate: issueOk,
|
|
163
168
|
})
|
|
@@ -165,12 +170,35 @@ describe('createLinearIssue — behaviour (#2312)', () => {
|
|
|
165
170
|
{ title: 'X', body: 'Y', dedup_key: 'chat:99' },
|
|
166
171
|
{ resolveToken: okToken('t'), fetchImpl, log: () => {} },
|
|
167
172
|
)
|
|
173
|
+
// matched the marker-bearing node, not the higher-ranked noise node.
|
|
168
174
|
expect(r.content[0].text).toBe('Already filed: https://linear.app/acme/issue/ENG-7')
|
|
169
175
|
// only the search ran — no team resolve, no create.
|
|
170
176
|
expect(calls).toHaveLength(1)
|
|
171
177
|
expect(calls[0].query).toMatch(/searchIssues/)
|
|
172
178
|
})
|
|
173
179
|
|
|
180
|
+
it('does NOT dedup when the search returns only unrelated hits (fuzzy false-positive regression)', async () => {
|
|
181
|
+
// Regression for the searchIssues fuzzy-match bug: a novel dedup_key
|
|
182
|
+
// returned an unrelated relevance-ranked hit and wrongly suppressed
|
|
183
|
+
// creation. No node carries this key's marker, so the tool must create.
|
|
184
|
+
const { fetchImpl, calls } = routingFetch({
|
|
185
|
+
searchIssues: () => ({ searchIssues: { nodes: [
|
|
186
|
+
{ id: 'x', url: 'https://linear.app/acme/issue/ENG-56', title: 'ProductOS Job Spec', description: 'shares tokens, but no capture marker' },
|
|
187
|
+
] } }),
|
|
188
|
+
teams: oneTeam,
|
|
189
|
+
issueCreate: issueOk,
|
|
190
|
+
})
|
|
191
|
+
const r = await createLinearIssue(
|
|
192
|
+
{ title: 'Series QA tracker', body: 'Y', dedup_key: 'productos-qa-2026-07-13' },
|
|
193
|
+
{ resolveToken: okToken('t'), fetchImpl, log: () => {} },
|
|
194
|
+
)
|
|
195
|
+
expect(r.content[0].text).toMatch(/Filed:/)
|
|
196
|
+
// fell through past the search to team resolve + create, embedding the marker.
|
|
197
|
+
const create = calls.find((c) => c.query.includes('issueCreate'))!
|
|
198
|
+
expect(create).toBeTruthy()
|
|
199
|
+
expect((create.variables.input as Record<string, unknown>).description).toContain(captureDedupMarker('productos-qa-2026-07-13'))
|
|
200
|
+
})
|
|
201
|
+
|
|
174
202
|
it('falls through to create when the dedup search misses, and embeds the marker', async () => {
|
|
175
203
|
const { fetchImpl, calls } = routingFetch({
|
|
176
204
|
searchIssues: () => ({ searchIssues: { nodes: [] } }),
|
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* llm-error-present.test.ts — outcome-asserting tests for the humanized,
|
|
3
|
+
* cross-surface-deduped LLM-error surfacing (no raw JSON passthrough).
|
|
4
|
+
*
|
|
5
|
+
* Each block is written to FAIL against pre-fix behaviour:
|
|
6
|
+
* - golden fan-out: pre-fix, the raw synthetic-error text fanned out to the
|
|
7
|
+
* reply + done-card surfaces (≥2 messages, raw bytes) — this asserts EXACTLY
|
|
8
|
+
* ONE JSON-free message across all three.
|
|
9
|
+
* - operator-card sanitize: pre-fix, renderOperatorEvent embedded the raw
|
|
10
|
+
* detail verbatim — this asserts the bytes are gone.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect, beforeEach } from 'vitest'
|
|
14
|
+
import {
|
|
15
|
+
parseLlmError,
|
|
16
|
+
renderLlmError,
|
|
17
|
+
stripRawErrorBytes,
|
|
18
|
+
extractRequestId,
|
|
19
|
+
formatResetLocal,
|
|
20
|
+
decideErrorSurface,
|
|
21
|
+
isActionableKind,
|
|
22
|
+
ErrorPresenceGate,
|
|
23
|
+
errorPresenceGate,
|
|
24
|
+
ERROR_COLLAPSE_WINDOW_MS,
|
|
25
|
+
type LlmErrorRetryState,
|
|
26
|
+
} from '../llm-error-present.js'
|
|
27
|
+
import { truncateDetailPreservingRequestId } from '../raw-error-scrub.js'
|
|
28
|
+
import { projectTranscriptLine, detectErrorInTranscriptLine } from '../session-tail.js'
|
|
29
|
+
import { renderOperatorEvent, type OperatorEvent } from '../operator-events.js'
|
|
30
|
+
|
|
31
|
+
// A raw byte-blob every surface must scrub.
|
|
32
|
+
const RAW_BYTES = `b'{"type":"error","error":{"type":"rate_limit_error","message":"rate limit"},"request_id":"req_abc123"}'`
|
|
33
|
+
|
|
34
|
+
function assertNoRawBytes(text: string): void {
|
|
35
|
+
expect(text).not.toContain("b'{")
|
|
36
|
+
expect(text).not.toContain('{"type":"error"')
|
|
37
|
+
expect(text.toLowerCase()).not.toContain('api error:')
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// ─── stripRawErrorBytes ──────────────────────────────────────────────────────
|
|
41
|
+
|
|
42
|
+
describe('stripRawErrorBytes', () => {
|
|
43
|
+
it('removes the Python byte-blob, JSON error object, and API Error prefix', () => {
|
|
44
|
+
const raw = `API Error: 429 Server is temporarily limiting requests · ${RAW_BYTES}`
|
|
45
|
+
const out = stripRawErrorBytes(raw)
|
|
46
|
+
assertNoRawBytes(out)
|
|
47
|
+
expect(out).toContain('Server is temporarily limiting requests')
|
|
48
|
+
})
|
|
49
|
+
|
|
50
|
+
it('passes a clean human string through unchanged (modulo whitespace)', () => {
|
|
51
|
+
expect(stripRawErrorBytes('Invalid API key')).toBe('Invalid API key')
|
|
52
|
+
expect(stripRawErrorBytes('Rate limit exceeded')).toBe('Rate limit exceeded')
|
|
53
|
+
expect(stripRawErrorBytes('')).toBe('')
|
|
54
|
+
})
|
|
55
|
+
|
|
56
|
+
it('strips a bare JSON blob down to empty', () => {
|
|
57
|
+
expect(stripRawErrorBytes('{"type":"error","error":{"type":"overloaded_error"}}')).toBe('')
|
|
58
|
+
})
|
|
59
|
+
})
|
|
60
|
+
|
|
61
|
+
// ─── extractRequestId ────────────────────────────────────────────────────────
|
|
62
|
+
|
|
63
|
+
describe('extractRequestId', () => {
|
|
64
|
+
it('pulls an Anthropic request_id from a raw JSON string', () => {
|
|
65
|
+
expect(extractRequestId(RAW_BYTES)).toBe('req_abc123')
|
|
66
|
+
expect(extractRequestId('no id here')).toBeUndefined()
|
|
67
|
+
})
|
|
68
|
+
})
|
|
69
|
+
|
|
70
|
+
// ─── parser table ────────────────────────────────────────────────────────────
|
|
71
|
+
|
|
72
|
+
describe('parseLlmError — classification table', () => {
|
|
73
|
+
it('rate_limit (Anthropic account throttle)', () => {
|
|
74
|
+
const p = parseLlmError(
|
|
75
|
+
"This request would exceed your account's rate limit. Please try again later.",
|
|
76
|
+
)
|
|
77
|
+
expect(p.kind).toBe('rate_limit')
|
|
78
|
+
expect(p.source).toBe('anthropic')
|
|
79
|
+
assertNoRawBytes(p.coreText)
|
|
80
|
+
})
|
|
81
|
+
|
|
82
|
+
it('overload_529', () => {
|
|
83
|
+
const p = parseLlmError('overloaded_error: Anthropic 529 overloaded')
|
|
84
|
+
expect(p.kind).toBe('overload_529')
|
|
85
|
+
expect(p.source).toBe('anthropic')
|
|
86
|
+
assertNoRawBytes(p.coreText)
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
it('quota_wall with a parsed reset', () => {
|
|
90
|
+
const p = parseLlmError("You've hit your limit · resets 8:50am (Australia/Melbourne)")
|
|
91
|
+
expect(p.kind).toBe('quota_wall')
|
|
92
|
+
expect(p.source).toBe('anthropic')
|
|
93
|
+
expect(p.resetAt).toBeInstanceOf(Date)
|
|
94
|
+
expect(p.terminal).toBe(true)
|
|
95
|
+
assertNoRawBytes(p.coreText)
|
|
96
|
+
})
|
|
97
|
+
|
|
98
|
+
it('litellm-local proxy 429', () => {
|
|
99
|
+
const p = parseLlmError(
|
|
100
|
+
'litellm.RateLimitError: Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241',
|
|
101
|
+
)
|
|
102
|
+
expect(p.kind).toBe('rate_limit')
|
|
103
|
+
expect(p.source).toBe('litellm-local')
|
|
104
|
+
assertNoRawBytes(p.coreText)
|
|
105
|
+
})
|
|
106
|
+
|
|
107
|
+
it('auth (credentials expired)', () => {
|
|
108
|
+
const p = parseLlmError('authentication_error: OAuth token expired, please refresh')
|
|
109
|
+
expect(p.kind).toBe('auth')
|
|
110
|
+
expect(p.source).toBe('anthropic')
|
|
111
|
+
expect(p.terminal).toBe(true)
|
|
112
|
+
expect(p.autoRetrying).toBe(false)
|
|
113
|
+
})
|
|
114
|
+
|
|
115
|
+
it('network', () => {
|
|
116
|
+
const p = parseLlmError('fetch failed: ECONNREFUSED getaddrinfo ENOTFOUND api.anthropic.com')
|
|
117
|
+
expect(p.kind).toBe('transient')
|
|
118
|
+
expect(p.source).toBe('network')
|
|
119
|
+
assertNoRawBytes(p.coreText)
|
|
120
|
+
})
|
|
121
|
+
|
|
122
|
+
it('unknown falls through cleanly', () => {
|
|
123
|
+
const p = parseLlmError('something entirely unrecognizable happened')
|
|
124
|
+
expect(p.kind).toBe('unknown')
|
|
125
|
+
assertNoRawBytes(p.coreText)
|
|
126
|
+
})
|
|
127
|
+
|
|
128
|
+
it('extracts model + requestId, and coreText is JSON-free even from a raw line', () => {
|
|
129
|
+
const raw = `Server is temporarily limiting requests · model=claude-opus-4-8 · ${RAW_BYTES}`
|
|
130
|
+
const p = parseLlmError(raw)
|
|
131
|
+
expect(p.kind).toBe('rate_limit')
|
|
132
|
+
expect(p.model).toBe('claude-opus-4-8')
|
|
133
|
+
expect(p.requestId).toBe('req_abc123')
|
|
134
|
+
assertNoRawBytes(p.coreText)
|
|
135
|
+
})
|
|
136
|
+
})
|
|
137
|
+
|
|
138
|
+
// ─── local-time render ───────────────────────────────────────────────────────
|
|
139
|
+
|
|
140
|
+
describe('renderLlmError / formatResetLocal — local-time rendering', () => {
|
|
141
|
+
const tz = 'Australia/Melbourne'
|
|
142
|
+
// Pin now to a fixed instant. 2026-07-13T06:14:00Z = 4:14pm AEST (winter, +10).
|
|
143
|
+
const now = new Date('2026-07-13T06:14:00Z')
|
|
144
|
+
|
|
145
|
+
it('renders the reset in local wall-clock + relative tail, DST-correct', () => {
|
|
146
|
+
// resetAt = 2026-07-13T06:52:00Z = 4:52pm AEST, ~38m out.
|
|
147
|
+
const reset = new Date('2026-07-13T06:52:00Z')
|
|
148
|
+
const line = formatResetLocal(reset, tz, now)
|
|
149
|
+
expect(line).toContain('4:52pm')
|
|
150
|
+
expect(line).toContain('AEST')
|
|
151
|
+
expect(line).toContain('~in 38m')
|
|
152
|
+
})
|
|
153
|
+
|
|
154
|
+
it('renderLlmError produces a JSON-free card with the local reset', () => {
|
|
155
|
+
const parsed = parseLlmError(
|
|
156
|
+
`Server is temporarily limiting requests · resets 4:52pm (Australia/Melbourne) · ${RAW_BYTES}`,
|
|
157
|
+
)
|
|
158
|
+
const { text } = renderLlmError(parsed, 'gymbro', tz, now)
|
|
159
|
+
assertNoRawBytes(text)
|
|
160
|
+
expect(text).toContain('gymbro')
|
|
161
|
+
expect(text).toContain('AEST')
|
|
162
|
+
})
|
|
163
|
+
|
|
164
|
+
it('auth + quota_wall cards carry action buttons', () => {
|
|
165
|
+
const auth = renderLlmError(parseLlmError('authentication_error: token expired'), 'a', tz, now)
|
|
166
|
+
expect(auth.keyboard?.inline_keyboard.flat().some((b) => b.callback_data?.includes('reauth'))).toBe(true)
|
|
167
|
+
const quota = renderLlmError(
|
|
168
|
+
parseLlmError("You've hit your limit · resets 5pm"),
|
|
169
|
+
'a',
|
|
170
|
+
tz,
|
|
171
|
+
now,
|
|
172
|
+
)
|
|
173
|
+
expect(quota.keyboard?.inline_keyboard.flat().length).toBeGreaterThan(0)
|
|
174
|
+
})
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
// ─── dedup key ───────────────────────────────────────────────────────────────
|
|
178
|
+
|
|
179
|
+
describe('ErrorPresenceGate — dedup key semantics', () => {
|
|
180
|
+
let gate: ErrorPresenceGate
|
|
181
|
+
beforeEach(() => {
|
|
182
|
+
gate = new ErrorPresenceGate()
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
it('same requestId → collapses to 1', () => {
|
|
186
|
+
const p = { kind: 'rate_limit' as const, requestId: 'req_x' }
|
|
187
|
+
const now = 1_000_000
|
|
188
|
+
const k = gate.keyFor(p, 'agent', now)
|
|
189
|
+
expect(gate.claim(k, now)).toBe(true)
|
|
190
|
+
expect(gate.claim(gate.keyFor(p, 'agent', now + 5_000), now + 5_000)).toBe(false)
|
|
191
|
+
})
|
|
192
|
+
|
|
193
|
+
it('different requestId, same kind → 2 distinct claims', () => {
|
|
194
|
+
const now = 1_000_000
|
|
195
|
+
expect(gate.claim(gate.keyFor({ kind: 'rate_limit', requestId: 'a' }, 'g', now), now)).toBe(true)
|
|
196
|
+
expect(gate.claim(gate.keyFor({ kind: 'rate_limit', requestId: 'b' }, 'g', now), now)).toBe(true)
|
|
197
|
+
})
|
|
198
|
+
|
|
199
|
+
it('no requestId, same kind: <window → 1, >window → 2', () => {
|
|
200
|
+
const now = 5 * ERROR_COLLAPSE_WINDOW_MS
|
|
201
|
+
const p = { kind: 'rate_limit' as const }
|
|
202
|
+
expect(gate.claim(gate.keyFor(p, 'g', now), now)).toBe(true)
|
|
203
|
+
// Within window (same bucket) → suppressed.
|
|
204
|
+
const within = now + ERROR_COLLAPSE_WINDOW_MS - 1
|
|
205
|
+
expect(gate.claim(gate.keyFor(p, 'g', within), within)).toBe(false)
|
|
206
|
+
// Past window (next bucket) → new claim.
|
|
207
|
+
const past = now + ERROR_COLLAPSE_WINDOW_MS + 1
|
|
208
|
+
expect(gate.claim(gate.keyFor(p, 'g', past), past)).toBe(true)
|
|
209
|
+
})
|
|
210
|
+
})
|
|
211
|
+
|
|
212
|
+
// ─── auto-retry silence ──────────────────────────────────────────────────────
|
|
213
|
+
//
|
|
214
|
+
// UNIT test of the module's own auto-retry-silence CONTRACT: given a retryState
|
|
215
|
+
// with attempt<max, a transient error is suppressed. This is NOT exercised by
|
|
216
|
+
// the production gateway wiring (it calls `parseLlmError(detail)` without a
|
|
217
|
+
// retryState, and session-tail already drops in-flight transients upstream — see
|
|
218
|
+
// the note in decideErrorSurface). These cases pin the module guarantee for any
|
|
219
|
+
// caller that DOES thread retryState; they do not imply the gateway does.
|
|
220
|
+
|
|
221
|
+
describe('decideErrorSurface — auto-retry silence (module contract, retryState-driven)', () => {
|
|
222
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
223
|
+
|
|
224
|
+
it('a transient 529 mid-retry (attempt<max) is silenced on all surfaces', () => {
|
|
225
|
+
const retry: LlmErrorRetryState = { retryAttempt: 1, maxRetries: 5 }
|
|
226
|
+
const p = parseLlmError('overloaded_error 529 overloaded', retry)
|
|
227
|
+
expect(p.autoRetrying).toBe(true)
|
|
228
|
+
expect(p.terminal).toBe(false)
|
|
229
|
+
expect(decideErrorSurface(p, 'g', { claim: true })).toBe('suppress')
|
|
230
|
+
})
|
|
231
|
+
|
|
232
|
+
it('the same 529 at attempt>=max renders exactly one card', () => {
|
|
233
|
+
const retry: LlmErrorRetryState = { retryAttempt: 5, maxRetries: 5 }
|
|
234
|
+
const p = parseLlmError('overloaded_error 529 overloaded', retry)
|
|
235
|
+
expect(p.autoRetrying).toBe(false)
|
|
236
|
+
expect(p.terminal).toBe(true)
|
|
237
|
+
expect(decideErrorSurface(p, 'g', { claim: true })).toBe('render')
|
|
238
|
+
// A second surface for the same terminal error collapses.
|
|
239
|
+
expect(decideErrorSurface(p, 'g', { claim: true })).toBe('suppress')
|
|
240
|
+
})
|
|
241
|
+
})
|
|
242
|
+
|
|
243
|
+
// ─── actionable never suppressed ─────────────────────────────────────────────
|
|
244
|
+
|
|
245
|
+
describe('decideErrorSurface — actionable kinds never suppressed', () => {
|
|
246
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
247
|
+
|
|
248
|
+
it('auth + quota_wall always render even inside an active collapse window', () => {
|
|
249
|
+
const auth = parseLlmError('authentication_error: credentials expired')
|
|
250
|
+
const quota = parseLlmError("You've hit your limit · resets 5pm")
|
|
251
|
+
expect(isActionableKind(auth.kind)).toBe(true)
|
|
252
|
+
expect(isActionableKind(quota.kind)).toBe(true)
|
|
253
|
+
// Pre-claim the same-kind window key to simulate a busy collapse window.
|
|
254
|
+
const now = 2_000_000
|
|
255
|
+
errorPresenceGate.claim(errorPresenceGate.keyFor(auth, 'g', now), now)
|
|
256
|
+
errorPresenceGate.claim(errorPresenceGate.keyFor(quota, 'g', now), now)
|
|
257
|
+
expect(decideErrorSurface(auth, 'g', { claim: true, now })).toBe('render')
|
|
258
|
+
expect(decideErrorSurface(quota, 'g', { claim: true, now })).toBe('render')
|
|
259
|
+
})
|
|
260
|
+
})
|
|
261
|
+
|
|
262
|
+
// ─── golden fan-out ──────────────────────────────────────────────────────────
|
|
263
|
+
|
|
264
|
+
describe('golden fan-out — one 429 line → exactly one JSON-free user message', () => {
|
|
265
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
266
|
+
|
|
267
|
+
// The v2.1.x synthetic-assistant usage/rate-limit shape: the raw error text
|
|
268
|
+
// lives inside message.content[].text, with the byte-blob attached.
|
|
269
|
+
const line = JSON.stringify({
|
|
270
|
+
type: 'assistant',
|
|
271
|
+
message: {
|
|
272
|
+
role: 'assistant',
|
|
273
|
+
content: [
|
|
274
|
+
{
|
|
275
|
+
type: 'text',
|
|
276
|
+
text: `Server is temporarily limiting requests (not your usage limit) · resets 4:52pm (Australia/Melbourne) · ${RAW_BYTES}`,
|
|
277
|
+
},
|
|
278
|
+
],
|
|
279
|
+
},
|
|
280
|
+
error: 'rate_limit',
|
|
281
|
+
isApiErrorMessage: true,
|
|
282
|
+
apiErrorStatus: 429,
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
it('starves the reply + done-card surfaces and renders one operator card', () => {
|
|
286
|
+
const userMessages: string[] = []
|
|
287
|
+
|
|
288
|
+
// Surfaces #1 (done-card recap) and #3 (reply passthrough) both derive from
|
|
289
|
+
// the projected transcript `text` events. Post-fix: NO text/thinking event
|
|
290
|
+
// is projected for an isApiErrorMessage line, so neither surface can relay
|
|
291
|
+
// the raw bytes as the turn's answer.
|
|
292
|
+
const projected = projectTranscriptLine(line)
|
|
293
|
+
for (const ev of projected) {
|
|
294
|
+
if (ev.kind === 'text') userMessages.push(`reply:${ev.text}`)
|
|
295
|
+
}
|
|
296
|
+
expect(projected.some((e) => e.kind === 'text')).toBe(false)
|
|
297
|
+
|
|
298
|
+
// Surface #2 (operator card): the SAME line is independently classified and
|
|
299
|
+
// rendered as the ONE humanized card. This mirrors the production gateway
|
|
300
|
+
// call exactly — `parseLlmError(detail)` with NO retryState (session-tail
|
|
301
|
+
// has already dropped any in-flight transient before forwarding, so the
|
|
302
|
+
// event that reaches the gateway is terminal by construction).
|
|
303
|
+
const detected = detectErrorInTranscriptLine(line)
|
|
304
|
+
expect(detected).not.toBeNull()
|
|
305
|
+
expect(detected!.terminal).toBe(true)
|
|
306
|
+
const parsed = parseLlmError(detected!.detail)
|
|
307
|
+
if (decideErrorSurface(parsed, 'gymbro', { claim: true }) === 'render') {
|
|
308
|
+
userMessages.push(renderLlmError(parsed, 'gymbro', 'Australia/Melbourne').text)
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
// EXACTLY ONE user-facing message, and it carries no raw bytes.
|
|
312
|
+
expect(userMessages).toHaveLength(1)
|
|
313
|
+
assertNoRawBytes(userMessages[0])
|
|
314
|
+
})
|
|
315
|
+
})
|
|
316
|
+
|
|
317
|
+
// ─── operator-card sanitize (surface #2 regression) ──────────────────────────
|
|
318
|
+
|
|
319
|
+
describe('renderOperatorEvent — never relays raw error bytes', () => {
|
|
320
|
+
function ev(kind: OperatorEvent['kind'], detail: string): OperatorEvent {
|
|
321
|
+
return { kind, agent: 'g', detail, suggestedActions: [], firstSeenAt: new Date() }
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
it('strips raw bytes from a rate-limited detail', () => {
|
|
325
|
+
const { text } = renderOperatorEvent(ev('rate-limited', `Rate limited · ${RAW_BYTES}`))
|
|
326
|
+
assertNoRawBytes(text)
|
|
327
|
+
expect(text).toContain('Rate limited')
|
|
328
|
+
})
|
|
329
|
+
|
|
330
|
+
it('strips raw bytes from an unknown-4xx detail', () => {
|
|
331
|
+
const { text } = renderOperatorEvent(ev('unknown-4xx', `API Error: 400 bad · ${RAW_BYTES}`))
|
|
332
|
+
assertNoRawBytes(text)
|
|
333
|
+
})
|
|
334
|
+
|
|
335
|
+
it('leaves a clean detail intact', () => {
|
|
336
|
+
const { text } = renderOperatorEvent(ev('credentials-expired', 'token expired 2026-04-01'))
|
|
337
|
+
expect(text).toContain('token expired 2026-04-01')
|
|
338
|
+
})
|
|
339
|
+
})
|
|
340
|
+
|
|
341
|
+
// ─── request_id survives the 1000-char bridge truncation (MEDIUM) ─────────────
|
|
342
|
+
|
|
343
|
+
describe('truncateDetailPreservingRequestId — exact dedup key survives truncation', () => {
|
|
344
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
345
|
+
|
|
346
|
+
// A realistic long Anthropic rate-limit body whose request_id sits PAST char
|
|
347
|
+
// 1000 (the bridge's OPERATOR_EVENT_DETAIL_MAX). `filler` pads the human part.
|
|
348
|
+
function longRateLimitLine(requestId: string): string {
|
|
349
|
+
const filler = 'Server is temporarily limiting requests (not your usage limit). '.repeat(30)
|
|
350
|
+
return `${filler} b'{"type":"error","error":{"type":"rate_limit_error","message":"rate limited"},"request_id":"${requestId}"}'`
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
it('preserves the request_id that a naive slice(0,1000) would drop', () => {
|
|
354
|
+
const line = longRateLimitLine('req_deadbeef01')
|
|
355
|
+
expect(line.length).toBeGreaterThan(1000)
|
|
356
|
+
// A naive slice loses the trailing id...
|
|
357
|
+
expect(extractRequestId(line.slice(0, 1000))).toBeUndefined()
|
|
358
|
+
// ...but the preserving truncation keeps it, within budget.
|
|
359
|
+
const truncated = truncateDetailPreservingRequestId(line, 1000)
|
|
360
|
+
expect(truncated.length).toBeLessThanOrEqual(1000)
|
|
361
|
+
expect(extractRequestId(truncated)).toBe('req_deadbeef01')
|
|
362
|
+
})
|
|
363
|
+
|
|
364
|
+
it('two DISTINCT request_ids past char 1000 render TWO messages (exact key still fires)', () => {
|
|
365
|
+
const now = 9_000_000
|
|
366
|
+
const messages: string[] = []
|
|
367
|
+
for (const rid of ['req_aaaaaaaa', 'req_bbbbbbbb']) {
|
|
368
|
+
// Simulate the bridge hop: truncate-preserving, THEN parse gateway-side.
|
|
369
|
+
const detail = truncateDetailPreservingRequestId(longRateLimitLine(rid), 1000)
|
|
370
|
+
const parsed = parseLlmError(detail)
|
|
371
|
+
expect(parsed.requestId).toBe(rid) // exact key preserved, not coarse fallback
|
|
372
|
+
if (decideErrorSurface(parsed, 'gymbro', { claim: true, now }) === 'render') {
|
|
373
|
+
messages.push(renderLlmError(parsed, 'gymbro', 'UTC', new Date(now)).text)
|
|
374
|
+
}
|
|
375
|
+
}
|
|
376
|
+
// Pre-fix (id truncated away) both fall to the SAME `${kind}:${agent}:${bucket}`
|
|
377
|
+
// key and collapse to ONE. With the id preserved, the exact keys differ → TWO.
|
|
378
|
+
expect(messages).toHaveLength(2)
|
|
379
|
+
})
|
|
380
|
+
})
|