switchroom 0.18.14 → 0.18.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent-scheduler/index.js +3 -0
  2. package/dist/auth-broker/index.js +473 -49
  3. package/dist/cli/notion-write-pretool.mjs +3 -0
  4. package/dist/cli/switchroom.js +1200 -1067
  5. package/dist/host-control/main.js +56 -51
  6. package/dist/vault/approvals/kernel-server.js +19 -12
  7. package/dist/vault/broker/server.js +675 -668
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/dist/bridge/bridge.js +21 -0
  11. package/telegram-plugin/dist/gateway/gateway.js +531 -259
  12. package/telegram-plugin/dist/server.js +22 -1
  13. package/telegram-plugin/draft-stream.ts +78 -3
  14. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
  15. package/telegram-plugin/gateway/effort-command.ts +9 -7
  16. package/telegram-plugin/gateway/gateway.ts +310 -219
  17. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  18. package/telegram-plugin/gateway/model-command.ts +96 -18
  19. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  20. package/telegram-plugin/gateway/session-model-file.ts +38 -172
  21. package/telegram-plugin/litellm-local-notice.ts +189 -0
  22. package/telegram-plugin/model-unavailable.ts +214 -0
  23. package/telegram-plugin/quota-watch.ts +16 -4
  24. package/telegram-plugin/runtime-metrics.ts +47 -0
  25. package/telegram-plugin/send-gate-degraded.test.ts +9 -7
  26. package/telegram-plugin/send-gate.ts +34 -4
  27. package/telegram-plugin/session-tail.ts +14 -2
  28. package/telegram-plugin/stream-controller.ts +143 -20
  29. package/telegram-plugin/stream-reply-handler.ts +12 -2
  30. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  31. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  32. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  33. package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
  34. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  35. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  36. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  37. package/telegram-plugin/tests/model-command.test.ts +84 -1
  38. package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
  39. package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
  40. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  41. package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
  42. package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
  43. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  44. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  45. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  46. package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
  47. package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
  48. package/telegram-plugin/throttle-tier.ts +98 -1
  49. package/telegram-plugin/worker-activity-feed.ts +83 -8
@@ -12,6 +12,8 @@
12
12
 
13
13
  import { describe, it, expect } from 'vitest'
14
14
  import {
15
+ build429ClassifiedMetric,
16
+ classify429Detail,
15
17
  decideThrottleTier,
16
18
  evaluateThrottleNotice,
17
19
  isAccountScopedThrottle,
@@ -136,6 +138,180 @@ describe('decideThrottleTier — decision matrix', () => {
136
138
  })
137
139
  })
138
140
 
141
+ // Verbatim LiteLLM proxy-local 429 shapes (provenance on
142
+ // `litellmProxyLocal429Signals`, model-unavailable.ts).
143
+ const LITELLM_TPM_CAP =
144
+ 'Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241. ' +
145
+ 'id=abc123def, model_group=claude-fable-5'
146
+ const LITELLM_KEY_LIMIT =
147
+ 'Rate limit exceeded for api_key: hashed-key-1a2b3c. Limit type: tokens. ' +
148
+ 'Current limit: 8000, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC'
149
+ const LITELLM_ROUTER_COOLDOWN =
150
+ 'No deployments available for selected model, Try again in 27.5 seconds. ' +
151
+ "Passed model=claude-fable-5. pre-call-checks=False, cooldown_list=['abc123def']"
152
+ // v3 limiter, descriptor NOT in the enumerated signal list — matched by the
153
+ // litellmV3LimiterSignalPair co-occurrence rule (model-unavailable.ts).
154
+ const LITELLM_TEAM_MODEL_LIMIT =
155
+ 'Rate limit exceeded for model_per_team: team-1a2b3c:claude-fable-5. ' +
156
+ 'Limit type: tokens. Current limit: 50000, Remaining: 0. ' +
157
+ 'Limit resets at: 2026-07-12 08:05:00 UTC'
158
+
159
+ describe('decideThrottleTier — LiteLLM-proxy-local 429s never enter the tier', () => {
160
+ // OUTCOME pin: `none` is what keeps a proxy-local cap trip off the
161
+ // account-scoped machinery — no broker mark-throttled, no throttle-tier
162
+ // runner fire, no failover escalation. The gateway additionally gates on
163
+ // classify429Detail, but the decision module must agree.
164
+ it.each([
165
+ ['deployment tpm cap', LITELLM_TPM_CAP],
166
+ ['virtual-key tpm_limit', LITELLM_KEY_LIMIT],
167
+ ['router cooldown', LITELLM_ROUTER_COOLDOWN],
168
+ ['team model cap (non-enumerated v3 descriptor)', LITELLM_TEAM_MODEL_LIMIT],
169
+ ])('%s → none (calm path owns it)', (_name, detail) => {
170
+ expect(decideThrottleTier({ detail, now: NOW, thresholdMs: THRESHOLD })).toEqual({
171
+ action: 'none',
172
+ })
173
+ })
174
+ })
175
+
176
+ describe('classify429Detail — three-way origin classification', () => {
177
+ it('classifies LiteLLM limiter wordings as litellm-local', () => {
178
+ expect(classify429Detail(LITELLM_TPM_CAP)).toBe('litellm-local')
179
+ expect(classify429Detail(LITELLM_KEY_LIMIT)).toBe('litellm-local')
180
+ expect(classify429Detail(LITELLM_ROUTER_COOLDOWN)).toBe('litellm-local')
181
+ // Non-enumerated v3 descriptor — the co-occurrence pair, not a list
182
+ // entry, carries this one.
183
+ expect(classify429Detail(LITELLM_TEAM_MODEL_LIMIT)).toBe('litellm-local')
184
+ })
185
+
186
+ it('classifies Anthropic account-affirming wording as account-scoped', () => {
187
+ expect(classify429Detail(TRANSIENT('resets in 3m'))).toBe('account-scoped')
188
+ expect(
189
+ classify429Detail('would exceed your account’s rate limit'),
190
+ ).toBe('account-scoped')
191
+ })
192
+
193
+ it('classifies server-side / bare rate-limit wording as generic-transient', () => {
194
+ expect(
195
+ classify429Detail('Server is temporarily limiting requests (not your usage limit)'),
196
+ ).toBe('generic-transient')
197
+ expect(classify429Detail('overloaded_error 529')).toBe('generic-transient')
198
+ expect(classify429Detail('rate_limit_error: try again later')).toBe('generic-transient')
199
+ })
200
+
201
+ it('TIE-BREAK: account-affirming wording wins over LiteLLM wording when both appear', () => {
202
+ // The pass-through shape: LiteLLM wraps a FORWARDED upstream Anthropic
203
+ // account 429 (exception mapping / limiter prose in the same body).
204
+ // LiteLLM never emits the account wording itself, so its presence means
205
+ // Anthropic really throttled the account — the broker throttle mark and
206
+ // the throttle tier must still run.
207
+ const mixed =
208
+ `litellm.RateLimitError: ${LITELLM_TPM_CAP} — upstream said: ` +
209
+ "This request would exceed your account's rate limit. Please try again later."
210
+ expect(classify429Detail(mixed)).toBe('account-scoped')
211
+ // And the tier decision still engages (not 'none').
212
+ expect(
213
+ decideThrottleTier({ detail: mixed, now: NOW, thresholdMs: THRESHOLD }).action,
214
+ ).not.toBe('none')
215
+ })
216
+
217
+ it('a bare litellm.RateLimitError wrapper WITHOUT limiter wording stays generic-transient', () => {
218
+ // The exception-mapping prefix alone is not proxy-local evidence.
219
+ expect(
220
+ classify429Detail('litellm.RateLimitError: RateLimitError: 429 try again later'),
221
+ ).toBe('generic-transient')
222
+ })
223
+
224
+ it('never throws on weird input', () => {
225
+ expect(classify429Detail('')).toBe('generic-transient')
226
+ expect(classify429Detail(undefined as unknown as string)).toBe('generic-transient')
227
+ expect(classify429Detail(9000 as unknown as string)).toBe('generic-transient')
228
+ expect(classify429Detail('A'.repeat(200_000))).toBe('generic-transient')
229
+ })
230
+ })
231
+
232
+ describe('build429ClassifiedMetric — instrumentation payload', () => {
233
+ it('litellm-local deployment cap → limit fields + calm action', () => {
234
+ const m = build429ClassifiedMetric({
235
+ agent: 'carrie',
236
+ detail: LITELLM_TPM_CAP,
237
+ classification: 'litellm-local',
238
+ action: 'calm',
239
+ now: NOW,
240
+ })
241
+ expect(m).toEqual({
242
+ kind: 'rate_limit_429_classified',
243
+ agent: 'carrie',
244
+ classification: 'litellm-local',
245
+ action: 'calm',
246
+ reset_at_ms: null,
247
+ reset_in_ms: null,
248
+ limit_type: 'tpm',
249
+ limit: 8000,
250
+ current_usage: 8241,
251
+ })
252
+ })
253
+
254
+ it('litellm-local NON-enumerated v3 descriptor (model_per_team) → full metric payload', () => {
255
+ // Finding-pin: a team-level model cap must produce the metric (it used
256
+ // to miss every signal → quota-exhausted → no metric at all).
257
+ const m = build429ClassifiedMetric({
258
+ agent: 'carrie',
259
+ detail: LITELLM_TEAM_MODEL_LIMIT,
260
+ classification: 'litellm-local',
261
+ action: 'calm',
262
+ now: NOW,
263
+ })
264
+ expect(m.kind).toBe('rate_limit_429_classified')
265
+ expect(m.classification).toBe('litellm-local')
266
+ expect(m.limit_type).toBe('tokens')
267
+ expect(m.limit).toBe(50_000)
268
+ expect(m.reset_at_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
269
+ })
270
+
271
+ it('litellm-local v3 key limit → tokens limit type + parsed "Limit resets at" UTC reset', () => {
272
+ const m = build429ClassifiedMetric({
273
+ agent: 'carrie',
274
+ detail: LITELLM_KEY_LIMIT,
275
+ classification: 'litellm-local',
276
+ action: 'calm',
277
+ now: NOW,
278
+ })
279
+ expect(m.limit_type).toBe('tokens')
280
+ expect(m.limit).toBe(8000)
281
+ expect(m.reset_at_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
282
+ expect(m.reset_in_ms).toBe(Date.UTC(2026, 6, 12, 8, 5, 0) - NOW)
283
+ })
284
+
285
+ it('account-scoped 429 → Anthropic-parsed reset, no limit fields', () => {
286
+ const m = build429ClassifiedMetric({
287
+ agent: 'carrie',
288
+ detail: TRANSIENT('resets in 3m'),
289
+ classification: 'account-scoped',
290
+ action: 'throttle',
291
+ now: NOW,
292
+ })
293
+ expect(m.classification).toBe('account-scoped')
294
+ expect(m.action).toBe('throttle')
295
+ expect(m.reset_at_ms).toBe(NOW + 3 * 60_000)
296
+ expect(m.reset_in_ms).toBe(3 * 60_000)
297
+ expect(m.limit_type).toBeNull()
298
+ expect(m.limit).toBeNull()
299
+ expect(m.current_usage).toBeNull()
300
+ })
301
+
302
+ it('never throws on weird detail; nulls out unparseable resets', () => {
303
+ const m = build429ClassifiedMetric({
304
+ agent: 'carrie',
305
+ detail: undefined as unknown as string,
306
+ classification: 'generic-transient',
307
+ action: 'calm',
308
+ now: NOW,
309
+ })
310
+ expect(m.reset_at_ms).toBeNull()
311
+ expect(m.reset_in_ms).toBeNull()
312
+ })
313
+ })
314
+
139
315
  describe('isAccountScopedThrottle — the tier gate', () => {
140
316
  it("matches account-affirming wording (both apostrophe variants + 'not your account')", () => {
141
317
  expect(isAccountScopedThrottle("would exceed your account's rate limit")).toBe(true)
@@ -7,6 +7,7 @@ import {
7
7
  type BotApiForWorkerFeed,
8
8
  } from '../worker-activity-feed.js'
9
9
  import { STATUS_ROLLING_LINES, STATUS_LINE_MAX } from '../status-no-truncate.js'
10
+ import { SEND_GATE_SHED } from '../send-gate.js'
10
11
 
11
12
  describe('isWorkerActivityFeedEnabled (default ON)', () => {
12
13
  it('defaults to true when the env var is unset', () => {
@@ -1336,3 +1337,209 @@ describe('narrative dedup — non-adjacent repeats collapse (A,B,A)', () => {
1336
1337
  expect(last.text).toContain('step-repeat')
1337
1338
  })
1338
1339
  })
1340
+
1341
+ // ─── send-gate shed contract (#3174) ────────────────────────────────────────
1342
+ //
1343
+ // The feed's send/edit adapters transit the deterministic send gate
1344
+ // (telegram-plugin/send-gate.ts, #3084), wired at the robustApiCall layer. The
1345
+ // gate SHEDS a call — resolves `undefined` WITHOUT hitting the API — when a
1346
+ // flood window is open (or a `useful` send exceeds its queue TTL, or a cosmetic
1347
+ // edit finds no free token). Before this fix the first-paint path dereferenced
1348
+ // `sent.message_id` on that `undefined`, throwing every heartbeat tick (~6s) for
1349
+ // the whole ban — 578 `undefined is not an object (evaluating 'sent.message_id')`
1350
+ // crashes in one 6h flood ban on 2026-07-12. These tests pin the shed contract:
1351
+ // (a) an open flood window parks the handle and makes ZERO api calls; (b) an
1352
+ // undefined send is treated as NOT-delivered (no crash, no phantom message id);
1353
+ // (c)/(d) an undefined edit is not recorded as on-screen (shed honesty, mirroring
1354
+ // PR #3173) so the update re-sends once the gate clears.
1355
+
1356
+ interface GateBot extends BotApiForWorkerFeed {
1357
+ /** Times sendMessage was INVOKED (attempts), delivered or shed. */
1358
+ sendCalls: number
1359
+ /** Times editMessageText was INVOKED (attempts), delivered or shed. */
1360
+ editCalls: number
1361
+ /** DELIVERED sends (gate admitted). */
1362
+ sent: Array<{ text: string }>
1363
+ /** DELIVERED edits (gate admitted). */
1364
+ edits: Array<{ messageId: number; text: string }>
1365
+ /**
1366
+ * One-shot: shed the next send even with the window closed. Faithful to the
1367
+ * gate: a worker-feed SEND is `useful` (never `cosmetic`), so a gate that
1368
+ * can't admit it resolves `undefined` (queue-TTL `expired`) — NEVER the
1369
+ * cosmetic-only SEND_GATE_SHED sentinel. The feed's send-shed detection is
1370
+ * structural (`typeof sent.message_id !== 'number'`), catching either.
1371
+ */
1372
+ shedNextSend: boolean
1373
+ /**
1374
+ * One-shot: shed the next EDIT even with the window closed. Faithful to the
1375
+ * gate: worker-feed edits are `cosmetic`, so a gate shed resolves the
1376
+ * distinguishable SEND_GATE_SHED sentinel (#3110 F1) — NOT `undefined` (which
1377
+ * the gate reserves for a benign no-op drop whose payload IS on screen).
1378
+ */
1379
+ shedNextEdit: boolean
1380
+ }
1381
+
1382
+ /**
1383
+ * A bot adapter that models the send gate: it SHEDS (without recording a
1384
+ * delivery) whenever the injected flood probe reports an open window, or a
1385
+ * one-shot `shedNext*` flag is set (the race where the window opens between the
1386
+ * feed's probe read and the send). Faithful to the post-#3110 contract: a shed
1387
+ * EDIT resolves the SEND_GATE_SHED sentinel (cosmetic shed), a shed SEND
1388
+ * resolves `undefined` (`useful` queue-TTL expiry). A DELIVERED edit resolves a
1389
+ * non-shed value (`{}` — grammy returns `true`/`Message`).
1390
+ */
1391
+ function makeGateBot(floodRemaining: () => number): GateBot {
1392
+ let nextId = 1000
1393
+ const gb: GateBot = {
1394
+ sendCalls: 0,
1395
+ editCalls: 0,
1396
+ sent: [],
1397
+ edits: [],
1398
+ shedNextSend: false,
1399
+ shedNextEdit: false,
1400
+ sendMessage: async (_chatId, text) => {
1401
+ gb.sendCalls++
1402
+ if (floodRemaining() > 0 || gb.shedNextSend) {
1403
+ gb.shedNextSend = false
1404
+ return undefined as unknown as { message_id: number }
1405
+ }
1406
+ gb.sent.push({ text })
1407
+ return { message_id: nextId++ }
1408
+ },
1409
+ editMessageText: async (_chatId, messageId, text) => {
1410
+ gb.editCalls++
1411
+ if (floodRemaining() > 0 || gb.shedNextEdit) {
1412
+ gb.shedNextEdit = false
1413
+ return SEND_GATE_SHED as unknown as undefined
1414
+ }
1415
+ gb.edits.push({ messageId, text })
1416
+ return {}
1417
+ },
1418
+ }
1419
+ return gb
1420
+ }
1421
+
1422
+ describe('worker-feed send-gate shed contract', () => {
1423
+ it('makes ZERO api calls while a flood window is open, then paints full state after it closes', async () => {
1424
+ let clock = 0
1425
+ const untilTs = 6 * 60 * 60 * 1000 // a 6h ban, as observed 2026-07-12
1426
+ const probe = () => Math.max(0, untilTs - clock)
1427
+ const bot = makeGateBot(probe)
1428
+ const feed = createWorkerActivityFeed({
1429
+ bot,
1430
+ now: () => clock,
1431
+ firstPaintMinMs: 0,
1432
+ floodWaitRemainingMs: probe,
1433
+ })
1434
+
1435
+ // Drive the heartbeat cadence hammering the feed for the whole ban.
1436
+ for (let i = 0; i < 30; i++) {
1437
+ clock += 6000
1438
+ await feed.update('w1', 'chat', view({ elapsedMs: clock }))
1439
+ }
1440
+ // Parked on the window — never even called the API (the whole point).
1441
+ expect(bot.sendCalls).toBe(0)
1442
+ expect(bot.sent).toHaveLength(0)
1443
+ expect(feed.has('w1')).toBe(false)
1444
+
1445
+ // Window closes: the next tick paints full state exactly once.
1446
+ clock = untilTs + 1000
1447
+ await feed.update('w1', 'chat', view({ elapsedMs: clock }))
1448
+ expect(bot.sendCalls).toBe(1)
1449
+ expect(bot.sent).toHaveLength(1)
1450
+ expect(bot.sent[0].text).toContain('🛠 **Worker**')
1451
+ expect(feed.has('w1')).toBe(true)
1452
+ })
1453
+
1454
+ it('treats an undefined send result as not-delivered — no crash, no phantom message id', async () => {
1455
+ let clock = 10_000
1456
+ const logs: string[] = []
1457
+ const bot = makeGateBot(() => 0) // probe reads clear …
1458
+ bot.shedNextSend = true // … but the gate sheds this send (race)
1459
+ const feed = createWorkerActivityFeed({
1460
+ bot,
1461
+ now: () => clock,
1462
+ firstPaintMinMs: 0,
1463
+ floodWaitRemainingMs: () => 0,
1464
+ log: (m) => logs.push(m),
1465
+ })
1466
+
1467
+ await feed.update('w1', 'chat', view())
1468
+ expect(bot.sendCalls).toBe(1) // it attempted the send
1469
+ expect(bot.sent).toHaveLength(0) // but nothing was delivered
1470
+ expect(feed.messageIdOf('w1')).toBeNull() // and no phantom message id recorded
1471
+ expect(feed.has('w1')).toBe(false)
1472
+ // Regression pin: the old code dereferenced `undefined.message_id` and logged
1473
+ // it as a hard "send failed"; the shed is now a recognized, clean skip.
1474
+ expect(logs.some((l) => l.startsWith('worker-feed: send failed'))).toBe(false)
1475
+ expect(logs.some((l) => l.includes('shed by send gate'))).toBe(true)
1476
+
1477
+ // The next paint (gate no longer shedding) lands cleanly.
1478
+ clock = 20_000
1479
+ await feed.update('w1', 'chat', view())
1480
+ expect(bot.sent).toHaveLength(1)
1481
+ expect(feed.has('w1')).toBe(true)
1482
+ })
1483
+
1484
+ it('does not record a shed cosmetic edit as on-screen — the update re-sends after the gate clears', async () => {
1485
+ let clock = 10_000
1486
+ const bot = makeGateBot(() => 0)
1487
+ const feed = createWorkerActivityFeed({
1488
+ bot,
1489
+ now: () => clock,
1490
+ firstPaintMinMs: 0,
1491
+ minEditIntervalMs: 0,
1492
+ floodWaitRemainingMs: () => 0,
1493
+ })
1494
+
1495
+ await feed.update('w1', 'chat', view({ toolCount: 1 }))
1496
+ expect(bot.sent).toHaveLength(1)
1497
+
1498
+ // Gate sheds the edit (undefined) — it is NOT on screen.
1499
+ clock = 20_000
1500
+ bot.shedNextEdit = true
1501
+ await feed.update('w1', 'chat', view({ toolCount: 2 }))
1502
+ expect(bot.editCalls).toBe(1)
1503
+ expect(bot.edits).toHaveLength(0)
1504
+
1505
+ // Shed honesty: the SAME update re-sends once the gate clears — the shed
1506
+ // payload was never recorded as `lastBody`, so it is not dropped as a dup.
1507
+ clock = 30_000
1508
+ await feed.update('w1', 'chat', view({ toolCount: 2 }))
1509
+ expect(bot.edits).toHaveLength(1)
1510
+ expect(bot.edits[0].text).toContain('2 tools')
1511
+ })
1512
+
1513
+ it('does not falsely finalize on a shed terminal edit — re-drives the finalize once the gate clears', async () => {
1514
+ let clock = 10_000
1515
+ const bot = makeGateBot(() => 0)
1516
+ const feed = createWorkerActivityFeed({
1517
+ bot,
1518
+ now: () => clock,
1519
+ firstPaintMinMs: 0,
1520
+ minEditIntervalMs: 0,
1521
+ floodWaitRemainingMs: () => 0,
1522
+ })
1523
+
1524
+ await feed.update('w1', 'chat', view({ toolCount: 1 }))
1525
+ expect(bot.sent).toHaveLength(1)
1526
+
1527
+ // Gate sheds the terminal edit (undefined).
1528
+ clock = 20_000
1529
+ bot.shedNextEdit = true
1530
+ await feed.finish('w1', view({ state: 'done', toolCount: 5 }))
1531
+ expect(bot.editCalls).toBe(1)
1532
+ expect(bot.edits).toHaveLength(0)
1533
+ // NOT finalized: the handle survives (pendingFinish staged) rather than
1534
+ // being torn down with the card frozen on its last running render.
1535
+ expect(feed.has('w1')).toBe(true)
1536
+
1537
+ // Re-drive with the gate clear: the terminal recap lands and the handle
1538
+ // finalizes.
1539
+ clock = 30_000
1540
+ await feed.finish('w1', view({ state: 'done', toolCount: 5 }))
1541
+ expect(bot.edits).toHaveLength(1)
1542
+ expect(bot.edits[0].text).toContain('_done · 5 tools')
1543
+ expect(feed.has('w1')).toBe(false)
1544
+ })
1545
+ })
@@ -33,7 +33,12 @@
33
33
 
34
34
  import { escapeMarkdown } from './card-format.js'
35
35
  import { formatResetRelative } from './quota-check.js'
36
- import { parseResetTime } from './model-unavailable.js'
36
+ import {
37
+ isLitellmProxyLocal429,
38
+ parseLitellmLimitDetail,
39
+ parseResetTime,
40
+ } from './model-unavailable.js'
41
+ import type { RuntimeMetricEvent } from './runtime-metrics.js'
37
42
 
38
43
  // ─── Account-scoped throttle wording ─────────────────────────────────────────
39
44
 
@@ -65,6 +70,98 @@ export function isAccountScopedThrottle(text: string): boolean {
65
70
  return accountScopedThrottleSignals.some((s) => lower.includes(s))
66
71
  }
67
72
 
73
+ // ─── Three-way 429 classification ────────────────────────────────────────────
74
+
75
+ /**
76
+ * Where a terminal 429-family failure originated:
77
+ * - `account-scoped` — Anthropic throttled THIS account ("would exceed
78
+ * your account's rate limit"). Eligible for the throttle tier below
79
+ * (broker mark-throttled / failover).
80
+ * - `litellm-local` — the LiteLLM proxy's OWN limiter tripped
81
+ * (`tpm_limit`/`rpm_limit` cap, router cooldown — see
82
+ * `litellmProxyLocal429Signals` in model-unavailable.ts). The request
83
+ * never reached Anthropic; account state must not be touched.
84
+ * - `generic-transient` — everything else in the rate-limit family
85
+ * (server-side 429/529 wording, bare `rate_limit_error`). Calm path.
86
+ */
87
+ export type RateLimit429Classification =
88
+ | 'account-scoped'
89
+ | 'litellm-local'
90
+ | 'generic-transient'
91
+
92
+ /**
93
+ * Classify a terminal `rate-limited` operator event's detail text.
94
+ *
95
+ * TIE-BREAK (both wordings present): account-scoped wins, and only on its
96
+ * EXPLICIT wording. Rationale: LiteLLM never emits the account-affirming
97
+ * strings itself, so when they co-occur with LiteLLM wording the detail is a
98
+ * genuine upstream Anthropic account 429 that traversed (and was wrapped by)
99
+ * the proxy — e.g. the pass-through's "litellm.RateLimitError: …would exceed
100
+ * your account's rate limit…" exception mapping. Classifying that as
101
+ * proxy-local would drop the broker throttle mark and walk every retry
102
+ * straight back into the same account throttle. The reverse risk is nil: a
103
+ * purely proxy-local 429 (limiter fired BEFORE any upstream call) cannot
104
+ * contain Anthropic's account wording.
105
+ *
106
+ * BOUND: operator events forwarded over IPC carry `detail` truncated to
107
+ * 1000 chars (bridge.ts `sendOperatorEvent`'s `.slice(0, 1000)`;
108
+ * `OPERATOR_EVENT_DETAIL_MAX` in gateway/ipc-server.ts), so on that path the
109
+ * tie-break sees only the first 1000 chars of the body. In practice both
110
+ * wordings sit well inside that window (real Anthropic and LiteLLM bodies
111
+ * are <400 chars), and a truncated-away account marker would merely
112
+ * downgrade account-scoped → litellm-local/generic-transient — the calm
113
+ * path, never a wrong account mark.
114
+ *
115
+ * Never throws on weird input (both matchers are total).
116
+ */
117
+ export function classify429Detail(text: string): RateLimit429Classification {
118
+ if (isAccountScopedThrottle(text)) return 'account-scoped'
119
+ if (isLitellmProxyLocal429(text)) return 'litellm-local'
120
+ return 'generic-transient'
121
+ }
122
+
123
+ /**
124
+ * Build the `rate_limit_429_classified` runtime metric for one terminal
125
+ * rate-limited operator event — the instrumentation that lets an operator
126
+ * correlate Anthropic ACCOUNT 429s with fleet TPM (the prerequisite for
127
+ * enabling LiteLLM `tpm_limit` caps; see docs/auth.md § LiteLLM-proxy-local
128
+ * 429s). Pure builder so the payload shape is unit-testable; the gateway
129
+ * emits the result via emitRuntimeMetric (PostHog + JSONL dual sink).
130
+ *
131
+ * `action` is what the gateway decided for this event:
132
+ * - `throttle` / `failover` — the account-scoped throttle tier's decision
133
+ * - `calm` — the existing calm rate-limited path (litellm-local and
134
+ * generic-transient always land here; no broker mark, no failover)
135
+ *
136
+ * Reset detail is best-effort: the Anthropic-shaped `parseResetTime` first,
137
+ * then LiteLLM's own shapes ("Limit resets at: … UTC" / "Try again in Ns")
138
+ * via `parseLitellmLimitDetail`. Limit fields parse only from LiteLLM
139
+ * wording (Anthropic bodies carry no numeric limit).
140
+ */
141
+ export function build429ClassifiedMetric(opts: {
142
+ agent: string
143
+ detail: string
144
+ classification: RateLimit429Classification
145
+ action: 'throttle' | 'failover' | 'calm'
146
+ now: number
147
+ }): Extract<RuntimeMetricEvent, { kind: 'rate_limit_429_classified' }> {
148
+ const detail = typeof opts.detail === 'string' ? opts.detail : ''
149
+ const litellm = parseLitellmLimitDetail(detail, new Date(opts.now))
150
+ const anthropicResetMs = parseResetTime(detail, new Date(opts.now))?.getTime() ?? null
151
+ const resetAtMs = anthropicResetMs ?? litellm.resetAtMs
152
+ return {
153
+ kind: 'rate_limit_429_classified',
154
+ agent: opts.agent,
155
+ classification: opts.classification,
156
+ action: opts.action,
157
+ reset_at_ms: resetAtMs,
158
+ reset_in_ms: resetAtMs != null ? Math.max(0, resetAtMs - opts.now) : null,
159
+ limit_type: litellm.limitType,
160
+ limit: litellm.limit,
161
+ current_usage: litellm.currentUsage,
162
+ }
163
+ }
164
+
68
165
  /**
69
166
  * Default retry-in-place ceiling: a transient 429 whose reset is within this
70
167
  * window waits on the account instead of failing the fleet over. Override
@@ -39,6 +39,7 @@ import {
39
39
  } from './card-format.js'
40
40
  import { STATUS_ROLLING_LINES } from './status-no-truncate.js'
41
41
  import { renderStatusCard, formatStepSuffix } from './tool-activity-summary.js'
42
+ import { isSendGateShed } from './send-gate.js'
42
43
 
43
44
  /** Worker-activity feed is ON by default; an operator opts out with
44
45
  * SWITCHROOM_WORKER_ACTIVITY_FEED=0. */
@@ -195,6 +196,24 @@ export interface WorkerActivityFeedOpts {
195
196
  firstPaintMinMs?: number
196
197
  /** stderr-style log sink. Defaults to noop. */
197
198
  log?: (msg: string) => void
199
+ /**
200
+ * Remaining ms of the currently-open per-bot flood window, read from the
201
+ * SAME persisted marker the gateway's `robustApiCall` and held-card sweep
202
+ * consult (`makeFloodWaitProbe(FLOOD_STATE_PATH)`). Two independent reads of
203
+ * one source of truth — not a second notion of "is the channel open".
204
+ *
205
+ * Why the feed needs it (#3084 follow-up): the feed's send/edit adapters run
206
+ * through the send gate, which SHEDS (resolves `undefined`) any call made
207
+ * while a flood window is open. Without this probe the heartbeat re-fires a
208
+ * shed send every `heartbeatTickMs` (~6s) for the WHOLE ban — thousands of
209
+ * gate admissions, and (pre-fix) a `sent.message_id` crash on the `undefined`
210
+ * every tick. With it, a running/first-paint tick that sees an open window
211
+ * parks the handle in cooldown for the window's remaining and makes ZERO api
212
+ * calls until it closes, mirroring the held-card sweep's pre-send probe.
213
+ * Defaults to `() => 0` (no window) so tests and non-gateway callers are
214
+ * unchanged.
215
+ */
216
+ floodWaitRemainingMs?: () => number
198
217
  /**
199
218
  * Heartbeat timer factory. Injectable for tests. Defaults to the real
200
219
  * `setInterval`, `.unref()`'d so it never keeps the process alive.
@@ -363,6 +382,7 @@ export interface WorkerActivityFeed {
363
382
  export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerActivityFeed {
364
383
  const log = opts.log ?? (() => {})
365
384
  const nowFn = opts.now ?? Date.now
385
+ const floodWaitRemainingMs = opts.floodWaitRemainingMs ?? (() => 0)
366
386
  const minEditInterval = opts.minEditIntervalMs ?? 2500
367
387
  const firstPaintMin = opts.firstPaintMinMs ?? 8000
368
388
  const heartbeatTickMs = opts.heartbeatTickMs ?? 6000
@@ -419,6 +439,22 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
419
439
  log(`worker-feed: ${label} 429 — backing off ${retryAfter}s`)
420
440
  }
421
441
 
442
+ /**
443
+ * Park a handle in cooldown for the remaining of an open flood window (if
444
+ * any), so the heartbeat's `nowFn() < h.cooldownUntil` guard suppresses every
445
+ * further send/edit until the ban closes. Returns true when a window was open
446
+ * (caller should abandon the current attempt). Two reads of the SAME on-disk
447
+ * marker `robustApiCall` gates on — never a second notion of "channel open".
448
+ */
449
+ function parkIfFloodWindowOpen(h: WorkerHandle): boolean {
450
+ const remaining = floodWaitRemainingMs()
451
+ if (remaining <= 0) return false
452
+ // Only extend the cooldown — never shorten one a 429 already set longer.
453
+ const until = nowFn() + remaining + COOLDOWN_JITTER_MS
454
+ if (until > h.cooldownUntil) h.cooldownUntil = until
455
+ return true
456
+ }
457
+
422
458
  function accumulateNarrative(h: WorkerHandle, view: WorkerActivityView): void {
423
459
  const line = view.latestSummary.trim()
424
460
  if (line.length === 0) return
@@ -453,6 +489,12 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
453
489
  h.lastView = merged
454
490
  if (h.dispatchAtMs == null) h.dispatchAtMs = nowFn() - view.elapsedMs
455
491
  if (nowFn() < h.cooldownUntil) return
492
+ // A flood window is open: the send gate would SHED every call made now
493
+ // (resolving `undefined`). Park in cooldown for the window's remaining and
494
+ // make ZERO api calls until it closes — the heartbeat re-drives the paint
495
+ // with full state on the first tick past the window. Same source of truth
496
+ // as robustApiCall's own pre-call probe.
497
+ if (parkIfFloodWindowOpen(h)) return
456
498
  const body = renderWorkerActivity(merged, liveSuffix)
457
499
 
458
500
  // First paint: hold off until the worker has run long enough to be
@@ -461,6 +503,17 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
461
503
  if (view.elapsedMs < firstPaintMin) return
462
504
  try {
463
505
  const sent = await opts.bot.sendMessage(h.chatId, body, sendOptsFor(h))
506
+ // Shed contract (#3084): the gate resolves `undefined` for a send it
507
+ // shed (open flood window) or dropped as stale (`useful` TTL). That is
508
+ // NOT a delivered message — dereferencing `sent.message_id` here was
509
+ // the `undefined is not an object` crash that fired every ~6s for a
510
+ // whole 6h ban. Treat it as not-delivered: record no message_id, park
511
+ // on any open window, and let the heartbeat re-drive the paint.
512
+ if (sent == null || typeof sent.message_id !== 'number') {
513
+ parkIfFloodWindowOpen(h)
514
+ log(`worker-feed: first paint shed by send gate agent=${h.agentId} — not delivered`)
515
+ return
516
+ }
464
517
  h.messageId = sent.message_id
465
518
  h.lastBody = body
466
519
  h.lastEditAt = nowFn()
@@ -480,7 +533,19 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
480
533
  if (nowFn() - h.lastEditAt < minEditInterval) return
481
534
 
482
535
  try {
483
- await opts.bot.editMessageText(h.chatId, h.messageId, body, sendOptsFor(h))
536
+ const res = await opts.bot.editMessageText(h.chatId, h.messageId, body, sendOptsFor(h))
537
+ // Shed honesty (#3084, mirrors #3173): a cosmetic edit the gate shed
538
+ // resolves the distinguishable SEND_GATE_SHED sentinel (#3110 F1 — NOT a
539
+ // bare `undefined`, which the gate reserves for a benign no-op drop whose
540
+ // payload IS already on screen). The shed payload is NOT on screen, so do
541
+ // not record it as `lastBody`, or the next tick's dedup would skip
542
+ // re-sending the very update the gate dropped. Park on any open window and
543
+ // let the heartbeat re-drive once it closes. A benign `undefined`/`true`
544
+ // falls through below and is correctly recorded as delivered.
545
+ if (isSendGateShed(res)) {
546
+ parkIfFloodWindowOpen(h)
547
+ return
548
+ }
484
549
  h.lastBody = body
485
550
  h.lastEditAt = nowFn()
486
551
  log(
@@ -534,12 +599,11 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
534
599
  h.pendingFinish = null
535
600
  return
536
601
  }
537
- if (nowFn() < h.cooldownUntil) {
538
- // Honour the flood-wait; a terminal edit isn't worth a ban. But
539
- // unlike the prior "stale but harmless" surrender, STAGE the terminal
540
- // view so the heartbeat re-drives the finalize edit the instant the
541
- // cooldown expires — a transport hiccup can no longer leave a
542
- // finished worker's card stuck on its last running render.
602
+ // A flood window is open (or a prior 429 set a cooldown): honour it — a
603
+ // terminal edit isn't worth extending a ban. STAGE the terminal view so the
604
+ // heartbeat re-drives the finalize the instant the window/cooldown expires,
605
+ // so a finished worker's card can't get stuck on its last running render.
606
+ if (parkIfFloodWindowOpen(h) || nowFn() < h.cooldownUntil) {
543
607
  h.pendingFinish = view
544
608
  return
545
609
  }
@@ -549,7 +613,18 @@ export function createWorkerActivityFeed(opts: WorkerActivityFeedOpts): WorkerAc
549
613
  return
550
614
  }
551
615
  try {
552
- await opts.bot.editMessageText(h.chatId, h.messageId, body, sendOptsFor(h))
616
+ const res = await opts.bot.editMessageText(h.chatId, h.messageId, body, sendOptsFor(h))
617
+ // Shed honesty (#3084): a shed terminal edit resolves the distinguishable
618
+ // SEND_GATE_SHED sentinel (#3110 F1 — NOT a bare `undefined`, which is the
619
+ // gate's benign no-op drop) — it did NOT land. Keep `pendingFinish` staged
620
+ // so the heartbeat re-drives it once the window closes; do NOT clear it or
621
+ // record `lastBody` (that would be a false finalization — card frozen on
622
+ // its running render while we believe it's done). Park on any open window.
623
+ if (isSendGateShed(res)) {
624
+ parkIfFloodWindowOpen(h)
625
+ h.pendingFinish = view
626
+ return
627
+ }
553
628
  h.lastBody = body
554
629
  h.lastEditAt = nowFn()
555
630
  h.pendingFinish = null