switchroom 0.18.17 → 0.18.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +13 -0
- package/dist/auth-broker/index.js +13 -0
- package/dist/cli/notion-write-pretool.mjs +13 -0
- package/dist/cli/switchroom.js +605 -479
- package/dist/host-control/main.js +17 -1
- package/dist/vault/approvals/kernel-server.js +13 -0
- package/dist/vault/broker/server.js +13 -0
- package/package.json +1 -1
- package/telegram-plugin/bridge/bridge.ts +7 -1
- package/telegram-plugin/dist/bridge/bridge.js +26 -1
- package/telegram-plugin/dist/gateway/gateway.js +1544 -619
- package/telegram-plugin/dist/server.js +32 -1
- package/telegram-plugin/fleet-fallback-resume.ts +26 -3
- package/telegram-plugin/format.ts +137 -213
- package/telegram-plugin/gateway/approval-hold.ts +49 -0
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +61 -18
- package/telegram-plugin/gateway/gateway.ts +399 -85
- package/telegram-plugin/gateway/linear-activity.ts +20 -4
- package/telegram-plugin/gateway/outbound-send-path.ts +9 -7
- package/telegram-plugin/gateway/premium-recovery-wiring.ts +122 -0
- package/telegram-plugin/gateway/session-model-file.ts +103 -0
- package/telegram-plugin/gateway/tier-downgrade-wiring.ts +121 -0
- package/telegram-plugin/gateway/unhandled-rejection-policy.ts +14 -1
- package/telegram-plugin/llm-error-present.ts +474 -0
- package/telegram-plugin/operator-events.ts +7 -1
- package/telegram-plugin/permission-title.ts +172 -10
- package/telegram-plugin/premium-recovery.ts +101 -0
- package/telegram-plugin/raw-error-scrub.ts +73 -0
- package/telegram-plugin/retry-api-call.ts +8 -2
- package/telegram-plugin/send-gate-degraded.test.ts +152 -1
- package/telegram-plugin/send-gate-observability.test.ts +140 -0
- package/telegram-plugin/send-gate-observability.ts +65 -20
- package/telegram-plugin/send-gate.test.ts +143 -1
- package/telegram-plugin/send-gate.ts +212 -19
- package/telegram-plugin/session-tail.ts +16 -0
- package/telegram-plugin/shared/local-time.ts +69 -0
- package/telegram-plugin/stream-reply-handler.ts +5 -14
- package/telegram-plugin/tests/approval-hold-harness.ts +6 -6
- package/telegram-plugin/tests/approval-hold-outcome.test.ts +10 -2
- package/telegram-plugin/tests/bridge-dead-watchdog.test.ts +61 -0
- package/telegram-plugin/tests/fleet-fallback-resume.test.ts +39 -0
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +3 -2
- package/telegram-plugin/tests/format-consistency.test.ts +68 -53
- package/telegram-plugin/tests/formatting-parse-regression.test.ts +5 -6
- package/telegram-plugin/tests/formatting-torture-set.ts +1 -1
- package/telegram-plugin/tests/gateway-outbound-redact.test.ts +26 -0
- package/telegram-plugin/tests/linear-create-issue.test.ts +30 -2
- package/telegram-plugin/tests/llm-error-present.test.ts +481 -0
- package/telegram-plugin/tests/outbound-send-path.test.ts +4 -3
- package/telegram-plugin/tests/paragraph-normalizer.test.ts +42 -100
- package/telegram-plugin/tests/permission-title.test.ts +167 -4
- package/telegram-plugin/tests/premium-recovery-wiring.test.ts +150 -0
- package/telegram-plugin/tests/premium-recovery.test.ts +165 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +6 -1
- package/telegram-plugin/tests/retry-api-call.test.ts +21 -0
- package/telegram-plugin/tests/stream-reply-handler.test.ts +9 -12
- package/telegram-plugin/tests/telegram-format.test.ts +86 -31
- package/telegram-plugin/tests/tier-downgrade-wiring.test.ts +165 -0
- package/telegram-plugin/tests/tier-downgrade.test.ts +141 -0
- package/telegram-plugin/tests/turn-flush-safety.test.ts +17 -21
- package/telegram-plugin/tests/unhandled-rejection-policy.test.ts +27 -1
- package/telegram-plugin/tests/worker-activity-feed.test.ts +5 -2
- package/telegram-plugin/tests/worker-feed-coalesce.test.ts +492 -0
- package/telegram-plugin/tier-downgrade.ts +198 -0
- package/telegram-plugin/tool-activity-summary.ts +99 -0
- package/telegram-plugin/turn-flush-safety.ts +4 -3
- package/telegram-plugin/worker-activity-feed.ts +509 -409
|
@@ -0,0 +1,481 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* llm-error-present.test.ts — outcome-asserting tests for the humanized,
|
|
3
|
+
* cross-surface-deduped LLM-error surfacing (no raw JSON passthrough).
|
|
4
|
+
*
|
|
5
|
+
* Each block is written to FAIL against pre-fix behaviour:
|
|
6
|
+
* - golden fan-out: pre-fix, the raw synthetic-error text fanned out to the
|
|
7
|
+
* reply + done-card surfaces (≥2 messages, raw bytes) — this asserts EXACTLY
|
|
8
|
+
* ONE JSON-free message across all three.
|
|
9
|
+
* - operator-card sanitize: pre-fix, renderOperatorEvent embedded the raw
|
|
10
|
+
* detail verbatim — this asserts the bytes are gone.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect, beforeEach } from 'vitest'
|
|
14
|
+
import {
|
|
15
|
+
parseLlmError,
|
|
16
|
+
renderLlmError,
|
|
17
|
+
renderLlmErrorSafe,
|
|
18
|
+
stripRawErrorBytes,
|
|
19
|
+
extractRequestId,
|
|
20
|
+
formatResetLocal,
|
|
21
|
+
decideErrorSurface,
|
|
22
|
+
isActionableKind,
|
|
23
|
+
ErrorPresenceGate,
|
|
24
|
+
errorPresenceGate,
|
|
25
|
+
ERROR_COLLAPSE_WINDOW_MS,
|
|
26
|
+
type LlmErrorRetryState,
|
|
27
|
+
} from '../llm-error-present.js'
|
|
28
|
+
import { truncateDetailPreservingRequestId } from '../raw-error-scrub.js'
|
|
29
|
+
import { projectTranscriptLine, detectErrorInTranscriptLine } from '../session-tail.js'
|
|
30
|
+
import { renderOperatorEvent, type OperatorEvent } from '../operator-events.js'
|
|
31
|
+
import { redact } from '../secret-detect/redact.js'
|
|
32
|
+
|
|
33
|
+
// A raw byte-blob every surface must scrub.
|
|
34
|
+
const RAW_BYTES = `b'{"type":"error","error":{"type":"rate_limit_error","message":"rate limit"},"request_id":"req_abc123"}'`
|
|
35
|
+
|
|
36
|
+
function assertNoRawBytes(text: string): void {
|
|
37
|
+
expect(text).not.toContain("b'{")
|
|
38
|
+
expect(text).not.toContain('{"type":"error"')
|
|
39
|
+
expect(text.toLowerCase()).not.toContain('api error:')
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
// ─── stripRawErrorBytes ──────────────────────────────────────────────────────
|
|
43
|
+
|
|
44
|
+
describe('stripRawErrorBytes', () => {
|
|
45
|
+
it('removes the Python byte-blob, JSON error object, and API Error prefix', () => {
|
|
46
|
+
const raw = `API Error: 429 Server is temporarily limiting requests · ${RAW_BYTES}`
|
|
47
|
+
const out = stripRawErrorBytes(raw)
|
|
48
|
+
assertNoRawBytes(out)
|
|
49
|
+
expect(out).toContain('Server is temporarily limiting requests')
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
it('passes a clean human string through unchanged (modulo whitespace)', () => {
|
|
53
|
+
expect(stripRawErrorBytes('Invalid API key')).toBe('Invalid API key')
|
|
54
|
+
expect(stripRawErrorBytes('Rate limit exceeded')).toBe('Rate limit exceeded')
|
|
55
|
+
expect(stripRawErrorBytes('')).toBe('')
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
it('strips a bare JSON blob down to empty', () => {
|
|
59
|
+
expect(stripRawErrorBytes('{"type":"error","error":{"type":"overloaded_error"}}')).toBe('')
|
|
60
|
+
})
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
// ─── extractRequestId ────────────────────────────────────────────────────────
|
|
64
|
+
|
|
65
|
+
describe('extractRequestId', () => {
|
|
66
|
+
it('pulls an Anthropic request_id from a raw JSON string', () => {
|
|
67
|
+
expect(extractRequestId(RAW_BYTES)).toBe('req_abc123')
|
|
68
|
+
expect(extractRequestId('no id here')).toBeUndefined()
|
|
69
|
+
})
|
|
70
|
+
})
|
|
71
|
+
|
|
72
|
+
// ─── parser table ────────────────────────────────────────────────────────────
|
|
73
|
+
|
|
74
|
+
describe('parseLlmError — classification table', () => {
|
|
75
|
+
it('rate_limit (Anthropic account throttle)', () => {
|
|
76
|
+
const p = parseLlmError(
|
|
77
|
+
"This request would exceed your account's rate limit. Please try again later.",
|
|
78
|
+
)
|
|
79
|
+
expect(p.kind).toBe('rate_limit')
|
|
80
|
+
expect(p.source).toBe('anthropic')
|
|
81
|
+
assertNoRawBytes(p.coreText)
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
it('overload_529', () => {
|
|
85
|
+
const p = parseLlmError('overloaded_error: Anthropic 529 overloaded')
|
|
86
|
+
expect(p.kind).toBe('overload_529')
|
|
87
|
+
expect(p.source).toBe('anthropic')
|
|
88
|
+
assertNoRawBytes(p.coreText)
|
|
89
|
+
})
|
|
90
|
+
|
|
91
|
+
it('quota_wall with a parsed reset', () => {
|
|
92
|
+
const p = parseLlmError("You've hit your limit · resets 8:50am (Australia/Melbourne)")
|
|
93
|
+
expect(p.kind).toBe('quota_wall')
|
|
94
|
+
expect(p.source).toBe('anthropic')
|
|
95
|
+
expect(p.resetAt).toBeInstanceOf(Date)
|
|
96
|
+
expect(p.terminal).toBe(true)
|
|
97
|
+
assertNoRawBytes(p.coreText)
|
|
98
|
+
})
|
|
99
|
+
|
|
100
|
+
it('litellm-local proxy 429', () => {
|
|
101
|
+
const p = parseLlmError(
|
|
102
|
+
'litellm.RateLimitError: Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241',
|
|
103
|
+
)
|
|
104
|
+
expect(p.kind).toBe('rate_limit')
|
|
105
|
+
expect(p.source).toBe('litellm-local')
|
|
106
|
+
assertNoRawBytes(p.coreText)
|
|
107
|
+
})
|
|
108
|
+
|
|
109
|
+
it('auth (credentials expired)', () => {
|
|
110
|
+
const p = parseLlmError('authentication_error: OAuth token expired, please refresh')
|
|
111
|
+
expect(p.kind).toBe('auth')
|
|
112
|
+
expect(p.source).toBe('anthropic')
|
|
113
|
+
expect(p.terminal).toBe(true)
|
|
114
|
+
expect(p.autoRetrying).toBe(false)
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
it('network', () => {
|
|
118
|
+
const p = parseLlmError('fetch failed: ECONNREFUSED getaddrinfo ENOTFOUND api.anthropic.com')
|
|
119
|
+
expect(p.kind).toBe('transient')
|
|
120
|
+
expect(p.source).toBe('network')
|
|
121
|
+
assertNoRawBytes(p.coreText)
|
|
122
|
+
})
|
|
123
|
+
|
|
124
|
+
it('unknown falls through cleanly', () => {
|
|
125
|
+
const p = parseLlmError('something entirely unrecognizable happened')
|
|
126
|
+
expect(p.kind).toBe('unknown')
|
|
127
|
+
assertNoRawBytes(p.coreText)
|
|
128
|
+
})
|
|
129
|
+
|
|
130
|
+
it('extracts model + requestId, and coreText is JSON-free even from a raw line', () => {
|
|
131
|
+
const raw = `Server is temporarily limiting requests · model=claude-opus-4-8 · ${RAW_BYTES}`
|
|
132
|
+
const p = parseLlmError(raw)
|
|
133
|
+
expect(p.kind).toBe('rate_limit')
|
|
134
|
+
expect(p.model).toBe('claude-opus-4-8')
|
|
135
|
+
expect(p.requestId).toBe('req_abc123')
|
|
136
|
+
assertNoRawBytes(p.coreText)
|
|
137
|
+
})
|
|
138
|
+
})
|
|
139
|
+
|
|
140
|
+
// ─── local-time render ───────────────────────────────────────────────────────
|
|
141
|
+
|
|
142
|
+
describe('renderLlmError / formatResetLocal — local-time rendering', () => {
|
|
143
|
+
const tz = 'Australia/Melbourne'
|
|
144
|
+
// Pin now to a fixed instant. 2026-07-13T06:14:00Z = 4:14pm AEST (winter, +10).
|
|
145
|
+
const now = new Date('2026-07-13T06:14:00Z')
|
|
146
|
+
|
|
147
|
+
it('renders the reset in local wall-clock + relative tail, DST-correct', () => {
|
|
148
|
+
// resetAt = 2026-07-13T06:52:00Z = 4:52pm AEST, ~38m out.
|
|
149
|
+
const reset = new Date('2026-07-13T06:52:00Z')
|
|
150
|
+
const line = formatResetLocal(reset, tz, now)
|
|
151
|
+
expect(line).toContain('4:52pm')
|
|
152
|
+
expect(line).toContain('AEST')
|
|
153
|
+
expect(line).toContain('~in 38m')
|
|
154
|
+
})
|
|
155
|
+
|
|
156
|
+
it('renderLlmError produces a JSON-free card with the local reset', () => {
|
|
157
|
+
const parsed = parseLlmError(
|
|
158
|
+
`Server is temporarily limiting requests · resets 4:52pm (Australia/Melbourne) · ${RAW_BYTES}`,
|
|
159
|
+
)
|
|
160
|
+
const { text } = renderLlmError(parsed, 'gymbro', tz, now)
|
|
161
|
+
assertNoRawBytes(text)
|
|
162
|
+
expect(text).toContain('gymbro')
|
|
163
|
+
expect(text).toContain('AEST')
|
|
164
|
+
})
|
|
165
|
+
|
|
166
|
+
// FIX 1 (Ken, CPO, 2026-07): the dead auth/quota action buttons are GONE —
|
|
167
|
+
// renderLlmError never returns an inline_keyboard; the actionable classes
|
|
168
|
+
// carry a plain-text recommendation line instead.
|
|
169
|
+
it('auth card recommends re-authentication in text, with NO action buttons', () => {
|
|
170
|
+
const auth = renderLlmError(parseLlmError('authentication_error: token expired'), 'a', tz, now)
|
|
171
|
+
expect((auth as { keyboard?: unknown }).keyboard).toBeUndefined()
|
|
172
|
+
expect(auth.text.toLowerCase()).toContain('re-authenticate')
|
|
173
|
+
})
|
|
174
|
+
|
|
175
|
+
it('quota_wall card recommends switch/wait in text (naming the reset), NO buttons', () => {
|
|
176
|
+
const quota = renderLlmError(parseLlmError("You've hit your limit · resets 5pm"), 'a', tz, now)
|
|
177
|
+
expect((quota as { keyboard?: unknown }).keyboard).toBeUndefined()
|
|
178
|
+
const lower = quota.text.toLowerCase()
|
|
179
|
+
expect(lower).toContain('switch to another account')
|
|
180
|
+
expect(lower).toContain('wait for the quota to reset')
|
|
181
|
+
// The reset instant is named in the recommendation line.
|
|
182
|
+
expect(quota.text).toContain('AEST')
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
it('transient (rate_limit) card has neither buttons nor an action recommendation', () => {
|
|
186
|
+
const rl = renderLlmError(parseLlmError('rate_limit_error: slow down'), 'a', tz, now)
|
|
187
|
+
expect((rl as { keyboard?: unknown }).keyboard).toBeUndefined()
|
|
188
|
+
expect(rl.text.toLowerCase()).not.toContain('re-authenticate')
|
|
189
|
+
expect(rl.text).not.toContain('→')
|
|
190
|
+
})
|
|
191
|
+
|
|
192
|
+
// FIX 3 (crash guard): an invalid IANA timezone throws a RangeError out of the
|
|
193
|
+
// raw renderer (local-time.ts's "never throws" claim is false for tz
|
|
194
|
+
// construction). renderLlmErrorSafe MUST swallow it and degrade — asserting the
|
|
195
|
+
// OUTCOME (no throw + a usable message), not just that the branch ran.
|
|
196
|
+
it('renderLlmError DOES throw on an invalid tz with a reset present (documents the hazard)', () => {
|
|
197
|
+
const parsed = parseLlmError("You've hit your limit · resets 5pm")
|
|
198
|
+
expect(parsed.resetAt).toBeDefined()
|
|
199
|
+
expect(() => renderLlmError(parsed, 'gymbro', 'Not/AZone', now)).toThrow()
|
|
200
|
+
})
|
|
201
|
+
|
|
202
|
+
it('renderLlmErrorSafe does NOT throw on an invalid tz and returns a usable line', () => {
|
|
203
|
+
const parsed = parseLlmError("You've hit your limit · resets 5pm")
|
|
204
|
+
let out: { text: string } | undefined
|
|
205
|
+
expect(() => {
|
|
206
|
+
out = renderLlmErrorSafe(parsed, 'gymbro', 'Not/AZone', now)
|
|
207
|
+
}).not.toThrow()
|
|
208
|
+
expect(out?.text).toContain('gymbro')
|
|
209
|
+
assertNoRawBytes(out!.text)
|
|
210
|
+
})
|
|
211
|
+
})
|
|
212
|
+
|
|
213
|
+
// ─── FIX 2: operator-card text is scrubbed by the REAL redactor ───────────────
|
|
214
|
+
//
|
|
215
|
+
// The gateway sends operator-event cards via a raw bot.api call that bypasses
|
|
216
|
+
// the normal outbound redact chokepoint; it now routes the rendered text through
|
|
217
|
+
// the same redact() the reply path uses. These tests assert the OUTCOME: a
|
|
218
|
+
// synthetic bearer token / sk- key / url-embedded credential planted in an error
|
|
219
|
+
// detail does NOT survive into the redacted card text. stripRawErrorBytes alone
|
|
220
|
+
// (a JSON-shape scrub) does NOT catch these — redact() is required.
|
|
221
|
+
describe('operator-card secret redaction (FIX 2)', () => {
|
|
222
|
+
const now = new Date('2026-07-13T06:14:00Z')
|
|
223
|
+
// Runtime-assembled so no contiguous Anthropic-token literal lands in source
|
|
224
|
+
// (check-no-pii-secrets discipline). Resolves to a real sk-ant-shaped key that
|
|
225
|
+
// redact()'s anthropic_api_key pattern masks.
|
|
226
|
+
const BEARER = ['sk', 'ant', 'api03-ABCDEF1234567890abcdefGHIJKLMN'].join('-')
|
|
227
|
+
const URL_SECRET = 'https://user:hunter2pass@api.anthropic.com/v1/x?api_key=abc123secretval456'
|
|
228
|
+
|
|
229
|
+
const mkEvent = (kind: OperatorEvent['kind'], detail: string): OperatorEvent => ({
|
|
230
|
+
agent: 'gymbro',
|
|
231
|
+
kind,
|
|
232
|
+
detail,
|
|
233
|
+
suggestedActions: [],
|
|
234
|
+
firstSeenAt: now,
|
|
235
|
+
})
|
|
236
|
+
|
|
237
|
+
// Mirrors emitGatewayOperatorEvent's transform: redact the DETAIL first (via
|
|
238
|
+
// the real redact()), THEN render — so the scrub happens BEFORE the renderer's
|
|
239
|
+
// escapeMarkdown, exactly as production now does. Redacting the already-escaped
|
|
240
|
+
// final text would let url-query-param secrets (`api_key=…`) slip past.
|
|
241
|
+
const renderCardAsSent = (ev: OperatorEvent): string =>
|
|
242
|
+
renderOperatorEvent({ ...ev, detail: redact(ev.detail) }).text
|
|
243
|
+
|
|
244
|
+
it('stripRawErrorBytes alone LEAKS a bearer/api-key (proves redact is needed)', () => {
|
|
245
|
+
// Guard test: the shape-scrub inside the renderer is JSON-shape-only. If this
|
|
246
|
+
// ever stops leaking, the shape-scrub grew secret awareness.
|
|
247
|
+
expect(stripRawErrorBytes(`auth failed: Bearer ${BEARER}`)).toContain(BEARER)
|
|
248
|
+
})
|
|
249
|
+
|
|
250
|
+
it('pre-fix render (no redact) LEAKS the bearer; the production transform scrubs it', () => {
|
|
251
|
+
const ev = mkEvent('credentials-expired', `token rejected: Bearer ${BEARER}`)
|
|
252
|
+
// Renderer alone (pre-fix path): secret survives.
|
|
253
|
+
expect(renderOperatorEvent(ev).text).toContain(BEARER)
|
|
254
|
+
// Production transform (redact detail → render): secret is gone.
|
|
255
|
+
expect(renderCardAsSent(ev)).not.toContain(BEARER)
|
|
256
|
+
})
|
|
257
|
+
|
|
258
|
+
it('masks a url-embedded credential in a credit-exhausted card', () => {
|
|
259
|
+
const ev = mkEvent('credit-exhausted', `billing check: ${URL_SECRET}`)
|
|
260
|
+
const sent = renderCardAsSent(ev)
|
|
261
|
+
expect(sent).not.toContain('hunter2pass')
|
|
262
|
+
expect(sent).not.toContain('abc123secretval456')
|
|
263
|
+
})
|
|
264
|
+
|
|
265
|
+
it('masks a bearer key in an unknown-4xx card', () => {
|
|
266
|
+
const ev = mkEvent('unknown-4xx', `API Error: 400 · x-api-key ${BEARER}`)
|
|
267
|
+
expect(renderCardAsSent(ev)).not.toContain(BEARER)
|
|
268
|
+
})
|
|
269
|
+
|
|
270
|
+
it('the humanized (renderLlmError) card never relays raw detail — no secret to leak', () => {
|
|
271
|
+
// The humanized card's coreText is a per-kind TEMPLATE, never the raw detail,
|
|
272
|
+
// so a secret in the detail cannot reach it even before redaction.
|
|
273
|
+
const parsed = parseLlmError(`rate_limit_error · Bearer ${BEARER}`)
|
|
274
|
+
expect(renderLlmError(parsed, 'gymbro', 'Australia/Melbourne', now).text).not.toContain(BEARER)
|
|
275
|
+
})
|
|
276
|
+
})
|
|
277
|
+
|
|
278
|
+
// ─── dedup key ───────────────────────────────────────────────────────────────
|
|
279
|
+
|
|
280
|
+
describe('ErrorPresenceGate — dedup key semantics', () => {
|
|
281
|
+
let gate: ErrorPresenceGate
|
|
282
|
+
beforeEach(() => {
|
|
283
|
+
gate = new ErrorPresenceGate()
|
|
284
|
+
})
|
|
285
|
+
|
|
286
|
+
it('same requestId → collapses to 1', () => {
|
|
287
|
+
const p = { kind: 'rate_limit' as const, requestId: 'req_x' }
|
|
288
|
+
const now = 1_000_000
|
|
289
|
+
const k = gate.keyFor(p, 'agent', now)
|
|
290
|
+
expect(gate.claim(k, now)).toBe(true)
|
|
291
|
+
expect(gate.claim(gate.keyFor(p, 'agent', now + 5_000), now + 5_000)).toBe(false)
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
it('different requestId, same kind → 2 distinct claims', () => {
|
|
295
|
+
const now = 1_000_000
|
|
296
|
+
expect(gate.claim(gate.keyFor({ kind: 'rate_limit', requestId: 'a' }, 'g', now), now)).toBe(true)
|
|
297
|
+
expect(gate.claim(gate.keyFor({ kind: 'rate_limit', requestId: 'b' }, 'g', now), now)).toBe(true)
|
|
298
|
+
})
|
|
299
|
+
|
|
300
|
+
it('no requestId, same kind: <window → 1, >window → 2', () => {
|
|
301
|
+
const now = 5 * ERROR_COLLAPSE_WINDOW_MS
|
|
302
|
+
const p = { kind: 'rate_limit' as const }
|
|
303
|
+
expect(gate.claim(gate.keyFor(p, 'g', now), now)).toBe(true)
|
|
304
|
+
// Within window (same bucket) → suppressed.
|
|
305
|
+
const within = now + ERROR_COLLAPSE_WINDOW_MS - 1
|
|
306
|
+
expect(gate.claim(gate.keyFor(p, 'g', within), within)).toBe(false)
|
|
307
|
+
// Past window (next bucket) → new claim.
|
|
308
|
+
const past = now + ERROR_COLLAPSE_WINDOW_MS + 1
|
|
309
|
+
expect(gate.claim(gate.keyFor(p, 'g', past), past)).toBe(true)
|
|
310
|
+
})
|
|
311
|
+
})
|
|
312
|
+
|
|
313
|
+
// ─── auto-retry silence ──────────────────────────────────────────────────────
|
|
314
|
+
//
|
|
315
|
+
// UNIT test of the module's own auto-retry-silence CONTRACT: given a retryState
|
|
316
|
+
// with attempt<max, a transient error is suppressed. This is NOT exercised by
|
|
317
|
+
// the production gateway wiring (it calls `parseLlmError(detail)` without a
|
|
318
|
+
// retryState, and session-tail already drops in-flight transients upstream — see
|
|
319
|
+
// the note in decideErrorSurface). These cases pin the module guarantee for any
|
|
320
|
+
// caller that DOES thread retryState; they do not imply the gateway does.
|
|
321
|
+
|
|
322
|
+
describe('decideErrorSurface — auto-retry silence (module contract, retryState-driven)', () => {
|
|
323
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
324
|
+
|
|
325
|
+
it('a transient 529 mid-retry (attempt<max) is silenced on all surfaces', () => {
|
|
326
|
+
const retry: LlmErrorRetryState = { retryAttempt: 1, maxRetries: 5 }
|
|
327
|
+
const p = parseLlmError('overloaded_error 529 overloaded', retry)
|
|
328
|
+
expect(p.autoRetrying).toBe(true)
|
|
329
|
+
expect(p.terminal).toBe(false)
|
|
330
|
+
expect(decideErrorSurface(p, 'g', { claim: true })).toBe('suppress')
|
|
331
|
+
})
|
|
332
|
+
|
|
333
|
+
it('the same 529 at attempt>=max renders exactly one card', () => {
|
|
334
|
+
const retry: LlmErrorRetryState = { retryAttempt: 5, maxRetries: 5 }
|
|
335
|
+
const p = parseLlmError('overloaded_error 529 overloaded', retry)
|
|
336
|
+
expect(p.autoRetrying).toBe(false)
|
|
337
|
+
expect(p.terminal).toBe(true)
|
|
338
|
+
expect(decideErrorSurface(p, 'g', { claim: true })).toBe('render')
|
|
339
|
+
// A second surface for the same terminal error collapses.
|
|
340
|
+
expect(decideErrorSurface(p, 'g', { claim: true })).toBe('suppress')
|
|
341
|
+
})
|
|
342
|
+
})
|
|
343
|
+
|
|
344
|
+
// ─── actionable never suppressed ─────────────────────────────────────────────
|
|
345
|
+
|
|
346
|
+
describe('decideErrorSurface — actionable kinds never suppressed', () => {
|
|
347
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
348
|
+
|
|
349
|
+
it('auth + quota_wall always render even inside an active collapse window', () => {
|
|
350
|
+
const auth = parseLlmError('authentication_error: credentials expired')
|
|
351
|
+
const quota = parseLlmError("You've hit your limit · resets 5pm")
|
|
352
|
+
expect(isActionableKind(auth.kind)).toBe(true)
|
|
353
|
+
expect(isActionableKind(quota.kind)).toBe(true)
|
|
354
|
+
// Pre-claim the same-kind window key to simulate a busy collapse window.
|
|
355
|
+
const now = 2_000_000
|
|
356
|
+
errorPresenceGate.claim(errorPresenceGate.keyFor(auth, 'g', now), now)
|
|
357
|
+
errorPresenceGate.claim(errorPresenceGate.keyFor(quota, 'g', now), now)
|
|
358
|
+
expect(decideErrorSurface(auth, 'g', { claim: true, now })).toBe('render')
|
|
359
|
+
expect(decideErrorSurface(quota, 'g', { claim: true, now })).toBe('render')
|
|
360
|
+
})
|
|
361
|
+
})
|
|
362
|
+
|
|
363
|
+
// ─── golden fan-out ──────────────────────────────────────────────────────────
|
|
364
|
+
|
|
365
|
+
describe('golden fan-out — one 429 line → exactly one JSON-free user message', () => {
|
|
366
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
367
|
+
|
|
368
|
+
// The v2.1.x synthetic-assistant usage/rate-limit shape: the raw error text
|
|
369
|
+
// lives inside message.content[].text, with the byte-blob attached.
|
|
370
|
+
const line = JSON.stringify({
|
|
371
|
+
type: 'assistant',
|
|
372
|
+
message: {
|
|
373
|
+
role: 'assistant',
|
|
374
|
+
content: [
|
|
375
|
+
{
|
|
376
|
+
type: 'text',
|
|
377
|
+
text: `Server is temporarily limiting requests (not your usage limit) · resets 4:52pm (Australia/Melbourne) · ${RAW_BYTES}`,
|
|
378
|
+
},
|
|
379
|
+
],
|
|
380
|
+
},
|
|
381
|
+
error: 'rate_limit',
|
|
382
|
+
isApiErrorMessage: true,
|
|
383
|
+
apiErrorStatus: 429,
|
|
384
|
+
})
|
|
385
|
+
|
|
386
|
+
it('starves the reply + done-card surfaces and renders one operator card', () => {
|
|
387
|
+
const userMessages: string[] = []
|
|
388
|
+
|
|
389
|
+
// Surfaces #1 (done-card recap) and #3 (reply passthrough) both derive from
|
|
390
|
+
// the projected transcript `text` events. Post-fix: NO text/thinking event
|
|
391
|
+
// is projected for an isApiErrorMessage line, so neither surface can relay
|
|
392
|
+
// the raw bytes as the turn's answer.
|
|
393
|
+
const projected = projectTranscriptLine(line)
|
|
394
|
+
for (const ev of projected) {
|
|
395
|
+
if (ev.kind === 'text') userMessages.push(`reply:${ev.text}`)
|
|
396
|
+
}
|
|
397
|
+
expect(projected.some((e) => e.kind === 'text')).toBe(false)
|
|
398
|
+
|
|
399
|
+
// Surface #2 (operator card): the SAME line is independently classified and
|
|
400
|
+
// rendered as the ONE humanized card. This mirrors the production gateway
|
|
401
|
+
// call exactly — `parseLlmError(detail)` with NO retryState (session-tail
|
|
402
|
+
// has already dropped any in-flight transient before forwarding, so the
|
|
403
|
+
// event that reaches the gateway is terminal by construction).
|
|
404
|
+
const detected = detectErrorInTranscriptLine(line)
|
|
405
|
+
expect(detected).not.toBeNull()
|
|
406
|
+
expect(detected!.terminal).toBe(true)
|
|
407
|
+
const parsed = parseLlmError(detected!.detail)
|
|
408
|
+
if (decideErrorSurface(parsed, 'gymbro', { claim: true }) === 'render') {
|
|
409
|
+
userMessages.push(renderLlmError(parsed, 'gymbro', 'Australia/Melbourne').text)
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
// EXACTLY ONE user-facing message, and it carries no raw bytes.
|
|
413
|
+
expect(userMessages).toHaveLength(1)
|
|
414
|
+
assertNoRawBytes(userMessages[0])
|
|
415
|
+
})
|
|
416
|
+
})
|
|
417
|
+
|
|
418
|
+
// ─── operator-card sanitize (surface #2 regression) ──────────────────────────
|
|
419
|
+
|
|
420
|
+
describe('renderOperatorEvent — never relays raw error bytes', () => {
|
|
421
|
+
function ev(kind: OperatorEvent['kind'], detail: string): OperatorEvent {
|
|
422
|
+
return { kind, agent: 'g', detail, suggestedActions: [], firstSeenAt: new Date() }
|
|
423
|
+
}
|
|
424
|
+
|
|
425
|
+
it('strips raw bytes from a rate-limited detail', () => {
|
|
426
|
+
const { text } = renderOperatorEvent(ev('rate-limited', `Rate limited · ${RAW_BYTES}`))
|
|
427
|
+
assertNoRawBytes(text)
|
|
428
|
+
expect(text).toContain('Rate limited')
|
|
429
|
+
})
|
|
430
|
+
|
|
431
|
+
it('strips raw bytes from an unknown-4xx detail', () => {
|
|
432
|
+
const { text } = renderOperatorEvent(ev('unknown-4xx', `API Error: 400 bad · ${RAW_BYTES}`))
|
|
433
|
+
assertNoRawBytes(text)
|
|
434
|
+
})
|
|
435
|
+
|
|
436
|
+
it('leaves a clean detail intact', () => {
|
|
437
|
+
const { text } = renderOperatorEvent(ev('credentials-expired', 'token expired 2026-04-01'))
|
|
438
|
+
expect(text).toContain('token expired 2026-04-01')
|
|
439
|
+
})
|
|
440
|
+
})
|
|
441
|
+
|
|
442
|
+
// ─── request_id survives the 1000-char bridge truncation (MEDIUM) ─────────────
|
|
443
|
+
|
|
444
|
+
describe('truncateDetailPreservingRequestId — exact dedup key survives truncation', () => {
|
|
445
|
+
beforeEach(() => errorPresenceGate.reset())
|
|
446
|
+
|
|
447
|
+
// A realistic long Anthropic rate-limit body whose request_id sits PAST char
|
|
448
|
+
// 1000 (the bridge's OPERATOR_EVENT_DETAIL_MAX). `filler` pads the human part.
|
|
449
|
+
function longRateLimitLine(requestId: string): string {
|
|
450
|
+
const filler = 'Server is temporarily limiting requests (not your usage limit). '.repeat(30)
|
|
451
|
+
return `${filler} b'{"type":"error","error":{"type":"rate_limit_error","message":"rate limited"},"request_id":"${requestId}"}'`
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
it('preserves the request_id that a naive slice(0,1000) would drop', () => {
|
|
455
|
+
const line = longRateLimitLine('req_deadbeef01')
|
|
456
|
+
expect(line.length).toBeGreaterThan(1000)
|
|
457
|
+
// A naive slice loses the trailing id...
|
|
458
|
+
expect(extractRequestId(line.slice(0, 1000))).toBeUndefined()
|
|
459
|
+
// ...but the preserving truncation keeps it, within budget.
|
|
460
|
+
const truncated = truncateDetailPreservingRequestId(line, 1000)
|
|
461
|
+
expect(truncated.length).toBeLessThanOrEqual(1000)
|
|
462
|
+
expect(extractRequestId(truncated)).toBe('req_deadbeef01')
|
|
463
|
+
})
|
|
464
|
+
|
|
465
|
+
it('two DISTINCT request_ids past char 1000 render TWO messages (exact key still fires)', () => {
|
|
466
|
+
const now = 9_000_000
|
|
467
|
+
const messages: string[] = []
|
|
468
|
+
for (const rid of ['req_aaaaaaaa', 'req_bbbbbbbb']) {
|
|
469
|
+
// Simulate the bridge hop: truncate-preserving, THEN parse gateway-side.
|
|
470
|
+
const detail = truncateDetailPreservingRequestId(longRateLimitLine(rid), 1000)
|
|
471
|
+
const parsed = parseLlmError(detail)
|
|
472
|
+
expect(parsed.requestId).toBe(rid) // exact key preserved, not coarse fallback
|
|
473
|
+
if (decideErrorSurface(parsed, 'gymbro', { claim: true, now }) === 'render') {
|
|
474
|
+
messages.push(renderLlmError(parsed, 'gymbro', 'UTC', new Date(now)).text)
|
|
475
|
+
}
|
|
476
|
+
}
|
|
477
|
+
// Pre-fix (id truncated away) both fall to the SAME `${kind}:${agent}:${bucket}`
|
|
478
|
+
// key and collapse to ONE. With the id preserved, the exact keys differ → TWO.
|
|
479
|
+
expect(messages).toHaveLength(2)
|
|
480
|
+
})
|
|
481
|
+
})
|
|
@@ -4,7 +4,6 @@ import {
|
|
|
4
4
|
normalizeParagraphBreaks,
|
|
5
5
|
normalizePunctuation,
|
|
6
6
|
stripExcessBold,
|
|
7
|
-
addParagraphSpacers,
|
|
8
7
|
splitMarkdownChunks,
|
|
9
8
|
hardSliceToCap,
|
|
10
9
|
RICH_MESSAGE_MAX_CHARS,
|
|
@@ -55,8 +54,10 @@ function referenceNormalize(rawText: string): { text: string; voiceReplaced: num
|
|
|
55
54
|
return { text, voiceReplaced }
|
|
56
55
|
}
|
|
57
56
|
|
|
58
|
-
function referenceEffectiveText(text: string,
|
|
59
|
-
|
|
57
|
+
function referenceEffectiveText(text: string, _literalText: boolean): string {
|
|
58
|
+
// The NBSP paragraph-spacer pass was removed in the #2669 follow-up; both
|
|
59
|
+
// paths now pass the normalized text through unchanged.
|
|
60
|
+
return text
|
|
60
61
|
}
|
|
61
62
|
|
|
62
63
|
function referenceChunks(
|