switchroom 0.18.14 → 0.18.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/agent-scheduler/index.js +3 -0
  2. package/dist/auth-broker/index.js +473 -49
  3. package/dist/cli/notion-write-pretool.mjs +3 -0
  4. package/dist/cli/switchroom.js +1200 -1067
  5. package/dist/host-control/main.js +56 -51
  6. package/dist/vault/approvals/kernel-server.js +19 -12
  7. package/dist/vault/broker/server.js +675 -668
  8. package/package.json +1 -1
  9. package/profiles/_base/start.sh.hbs +81 -139
  10. package/telegram-plugin/dist/bridge/bridge.js +21 -0
  11. package/telegram-plugin/dist/gateway/gateway.js +531 -259
  12. package/telegram-plugin/dist/server.js +22 -1
  13. package/telegram-plugin/draft-stream.ts +78 -3
  14. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
  15. package/telegram-plugin/gateway/effort-command.ts +9 -7
  16. package/telegram-plugin/gateway/gateway.ts +310 -219
  17. package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
  18. package/telegram-plugin/gateway/model-command.ts +96 -18
  19. package/telegram-plugin/gateway/pending-session-command.ts +10 -8
  20. package/telegram-plugin/gateway/session-model-file.ts +38 -172
  21. package/telegram-plugin/litellm-local-notice.ts +189 -0
  22. package/telegram-plugin/model-unavailable.ts +214 -0
  23. package/telegram-plugin/quota-watch.ts +16 -4
  24. package/telegram-plugin/runtime-metrics.ts +47 -0
  25. package/telegram-plugin/send-gate-degraded.test.ts +9 -7
  26. package/telegram-plugin/send-gate.ts +34 -4
  27. package/telegram-plugin/session-tail.ts +14 -2
  28. package/telegram-plugin/stream-controller.ts +143 -20
  29. package/telegram-plugin/stream-reply-handler.ts +12 -2
  30. package/telegram-plugin/tests/bot-api.harness.ts +7 -2
  31. package/telegram-plugin/tests/draft-stream.test.ts +110 -1
  32. package/telegram-plugin/tests/effort-command.test.ts +4 -4
  33. package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
  34. package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
  35. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
  36. package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
  37. package/telegram-plugin/tests/model-command.test.ts +84 -1
  38. package/telegram-plugin/tests/model-unavailable.test.ts +187 -0
  39. package/telegram-plugin/tests/operator-events-session-tail.test.ts +55 -0
  40. package/telegram-plugin/tests/quota-watch.test.ts +21 -0
  41. package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
  42. package/telegram-plugin/tests/runtime-metrics.test.ts +24 -0
  43. package/telegram-plugin/tests/session-model-file.test.ts +7 -155
  44. package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
  45. package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
  46. package/telegram-plugin/tests/throttle-tier.test.ts +176 -0
  47. package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
  48. package/telegram-plugin/throttle-tier.ts +98 -1
  49. package/telegram-plugin/worker-activity-feed.ts +83 -8
@@ -14,10 +14,52 @@ import { describe, it, expect } from 'vitest'
14
14
  import {
15
15
  detectModelUnavailable,
16
16
  formatModelUnavailableCard,
17
+ isLitellmProxyLocal429,
18
+ parseLitellmLimitDetail,
17
19
  resolveModelUnavailableFromOperatorEvent,
18
20
  type ModelUnavailableDetection,
19
21
  } from '../model-unavailable.js'
20
22
 
23
+ // Real LiteLLM proxy-local 429 bodies, verbatim shapes from BerriAI/litellm
24
+ // source (see the provenance comment on `litellmProxyLocal429Signals`).
25
+ const LITELLM_DEPLOYMENT_CAP_BODY =
26
+ 'Deployment over user-defined ratelimit. tpm limit=8000. current usage=8241. ' +
27
+ 'id=abc123def, model_group=claude-fable-5'
28
+ const LITELLM_DEPLOYMENT_CAP_MESSAGE =
29
+ 'litellm.RateLimitError: Model rate limit exceeded. TPM limit=8000, current usage=8241'
30
+ const LITELLM_V2_STRATEGY_BODY =
31
+ "Deployment over defined rpm limit=60. current usage=61. id=abc123def, " +
32
+ "model_group=claude-fable-5. Get the model info by calling 'router.get_model_info(id)"
33
+ const LITELLM_ROUTER_COOLDOWN_BODY =
34
+ 'No deployments available for selected model, Try again in 27.5 seconds. ' +
35
+ "Passed model=claude-fable-5. pre-call-checks=False, cooldown_list=['abc123def']"
36
+ const LITELLM_V3_KEY_LIMIT_BODY =
37
+ 'Rate limit exceeded for api_key: hashed-key-1a2b3c. Limit type: tokens. ' +
38
+ 'Current limit: 8000, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC'
39
+ // v1 shape A — the ProxyRateLimitError detail (parallel_request_limiter.py
40
+ // ~line 124; the interpolated CommonProxyErrors.max_parallel_request_limit_
41
+ // reached.value is "Crossed TPM / RPM / Max Parallel Request Limit").
42
+ const LITELLM_V1_PARALLEL_BODY =
43
+ 'LiteLLM Rate Limit Handler for rate limit type = requests. ' +
44
+ 'Crossed TPM / RPM / Max Parallel Request Limit. ' +
45
+ 'current rpm: 61, rpm limit: 60, current tpm: 100, tpm limit: 8000, ' +
46
+ 'current max_parallel_requests: 1, max_parallel_requests: 10'
47
+ // v1 shape B — raise_rate_limit_error's zero-limit branch: the standalone
48
+ // "Max parallel request limit reached" prefix + additional_details
49
+ // (parallel_request_limiter.py ~lines 88 + 183).
50
+ const LITELLM_V1_ZERO_LIMIT_BODY =
51
+ 'Max parallel request limit reached Crossed TPM / RPM / Max Parallel ' +
52
+ 'Request Limit. Hit limit for tokens. Current limits: ' +
53
+ 'max_parallel_requests: 10, tpm_limit: 0, rpm_limit: 60'
54
+ // v3 limiter with a descriptor OUTSIDE any enumerated list — covered by the
55
+ // litellmV3LimiterSignalPair co-occurrence rule (v3 descriptor keys also
56
+ // include organization / team_member / model_per_team / agent / tag_per_key
57
+ // / mcp_per_key and keep growing, so enumeration is a treadmill).
58
+ const LITELLM_V3_TEAM_MODEL_BODY =
59
+ 'Rate limit exceeded for model_per_team: team-1a2b3c:claude-fable-5. ' +
60
+ 'Limit type: tokens. Current limit: 50000, Remaining: 0. ' +
61
+ 'Limit resets at: 2026-07-12 08:05:00 UTC'
62
+
21
63
  // ─── detectModelUnavailable ──────────────────────────────────────────────────
22
64
 
23
65
  describe('detectModelUnavailable — quota / billing strings', () => {
@@ -117,6 +159,151 @@ describe('detectModelUnavailable — transient upstream 429 vs account quota (#2
117
159
  })
118
160
  })
119
161
 
162
+ describe('detectModelUnavailable — LiteLLM-proxy-LOCAL 429s (never quota)', () => {
163
+ // A 429 raised by LiteLLM's OWN limiter never reached Anthropic — it must
164
+ // classify to the calm retryable kind, never quota_exhausted (which drives
165
+ // the scary card + mark-exhausted + fleet failover for an account that was
166
+ // never touched). Bodies are verbatim LiteLLM source shapes.
167
+ it.each([
168
+ ['deployment tpm cap (model_rate_limit_check body)', LITELLM_DEPLOYMENT_CAP_BODY],
169
+ ['deployment tpm cap (RateLimitError message)', LITELLM_DEPLOYMENT_CAP_MESSAGE],
170
+ ['usage-based-routing v2 rpm wording', LITELLM_V2_STRATEGY_BODY],
171
+ ['router cooldown (RouterRateLimitError)', LITELLM_ROUTER_COOLDOWN_BODY],
172
+ ['virtual-key tpm_limit (parallel_request_limiter_v3)', LITELLM_V3_KEY_LIMIT_BODY],
173
+ ['team model cap — non-enumerated v3 descriptor (co-occurrence rule)', LITELLM_V3_TEAM_MODEL_BODY],
174
+ ['v1 parallel-request limiter (Rate Limit Handler detail)', LITELLM_V1_PARALLEL_BODY],
175
+ ['v1 zero-limit branch (Max parallel request limit reached prefix)', LITELLM_V1_ZERO_LIMIT_BODY],
176
+ ])('classifies %s as overload, NOT quota_exhausted', (_name, body) => {
177
+ const d = detectModelUnavailable(body)
178
+ expect(d?.kind).toBe('overload')
179
+ })
180
+
181
+ it('a rate-limited operator event with a LiteLLM-local body resolves to NO model-unavailable card', () => {
182
+ // End-to-end seam the gateway uses: resolveModelUnavailableFromOperatorEvent
183
+ // returning null is what keeps the calm 🚦 card AND blocks the
184
+ // auto-fallback branch (fallback only fires when a detection resolves).
185
+ for (const body of [
186
+ LITELLM_DEPLOYMENT_CAP_BODY,
187
+ LITELLM_V3_KEY_LIMIT_BODY,
188
+ LITELLM_ROUTER_COOLDOWN_BODY,
189
+ ]) {
190
+ expect(
191
+ resolveModelUnavailableFromOperatorEvent({ kind: 'rate-limited', detail: body }),
192
+ ).toBeNull()
193
+ }
194
+ })
195
+ })
196
+
197
+ describe('isLitellmProxyLocal429 — signal matching', () => {
198
+ it('matches every canonical LiteLLM limiter wording', () => {
199
+ for (const body of [
200
+ LITELLM_DEPLOYMENT_CAP_BODY,
201
+ LITELLM_DEPLOYMENT_CAP_MESSAGE,
202
+ LITELLM_V2_STRATEGY_BODY,
203
+ LITELLM_ROUTER_COOLDOWN_BODY,
204
+ LITELLM_V3_KEY_LIMIT_BODY,
205
+ LITELLM_V1_PARALLEL_BODY,
206
+ LITELLM_V1_ZERO_LIMIT_BODY,
207
+ ]) {
208
+ expect(isLitellmProxyLocal429(body)).toBe(true)
209
+ }
210
+ })
211
+
212
+ it('matches ANY v3 descriptor via the co-occurrence pair, not an enumerated list', () => {
213
+ // model_per_team is not in litellmProxyLocal429Signals — only the
214
+ // "rate limit exceeded for " + "limit type:" pair catches it. Same for
215
+ // any descriptor litellm adds later.
216
+ expect(isLitellmProxyLocal429(LITELLM_V3_TEAM_MODEL_BODY)).toBe(true)
217
+ expect(
218
+ isLitellmProxyLocal429(
219
+ 'Rate limit exceeded for organization: org-1a2b3c. Limit type: requests. ' +
220
+ 'Current limit: 100, Remaining: 0. Limit resets at: 2026-07-12 08:05:00 UTC',
221
+ ),
222
+ ).toBe(true)
223
+ // HALF the pair is not enough — "rate limit exceeded for" prose without
224
+ // the v3 "Limit type:" field must not classify proxy-local.
225
+ expect(
226
+ isLitellmProxyLocal429('Rate limit exceeded for this account, try later'),
227
+ ).toBe(false)
228
+ expect(isLitellmProxyLocal429('Limit type: tokens')).toBe(false)
229
+ })
230
+
231
+ it('does NOT match Anthropic account/wall/server wordings', () => {
232
+ expect(
233
+ isLitellmProxyLocal429(
234
+ "This request would exceed your account's rate limit. Please try again later.",
235
+ ),
236
+ ).toBe(false)
237
+ expect(
238
+ isLitellmProxyLocal429('Server is temporarily limiting requests (not your usage limit)'),
239
+ ).toBe(false)
240
+ expect(isLitellmProxyLocal429("You've hit your limit · resets 8:50am")).toBe(false)
241
+ })
242
+
243
+ it('does NOT match the bare litellm.RateLimitError exception-mapping prefix', () => {
244
+ // The pass-through wraps FORWARDED upstream 429s with the same prefix —
245
+ // it is not evidence the limit was proxy-local.
246
+ expect(
247
+ isLitellmProxyLocal429(
248
+ "litellm.RateLimitError: RateLimitError: This request would exceed your account's rate limit.",
249
+ ),
250
+ ).toBe(false)
251
+ })
252
+
253
+ it('never throws on weird input', () => {
254
+ expect(isLitellmProxyLocal429('')).toBe(false)
255
+ expect(isLitellmProxyLocal429(undefined as unknown as string)).toBe(false)
256
+ expect(isLitellmProxyLocal429(42 as unknown as string)).toBe(false)
257
+ // Slice guard: signal past the 16KB sample is not scanned.
258
+ expect(isLitellmProxyLocal429('A'.repeat(100_000) + LITELLM_DEPLOYMENT_CAP_BODY)).toBe(false)
259
+ })
260
+ })
261
+
262
+ describe('parseLitellmLimitDetail — instrumentation extraction', () => {
263
+ const NOW = new Date(Date.UTC(2026, 6, 12, 8, 0, 0))
264
+
265
+ it('extracts tpm limit + current usage from the deployment-cap body', () => {
266
+ const d = parseLitellmLimitDetail(LITELLM_DEPLOYMENT_CAP_BODY, NOW)
267
+ expect(d.limitType).toBe('tpm')
268
+ expect(d.limit).toBe(8000)
269
+ expect(d.currentUsage).toBe(8241)
270
+ expect(d.resetAtMs).toBeNull()
271
+ })
272
+
273
+ it('extracts rpm limit from the v2 strategy body', () => {
274
+ const d = parseLitellmLimitDetail(LITELLM_V2_STRATEGY_BODY, NOW)
275
+ expect(d.limitType).toBe('rpm')
276
+ expect(d.limit).toBe(60)
277
+ expect(d.currentUsage).toBe(61)
278
+ })
279
+
280
+ it('extracts the v3 limiter limit type, limit, and "Limit resets at" UTC timestamp', () => {
281
+ const d = parseLitellmLimitDetail(LITELLM_V3_KEY_LIMIT_BODY, NOW)
282
+ expect(d.limitType).toBe('tokens')
283
+ expect(d.limit).toBe(8000)
284
+ expect(d.resetAtMs).toBe(Date.UTC(2026, 6, 12, 8, 5, 0))
285
+ })
286
+
287
+ it('extracts the router-cooldown "Try again in N seconds" relative reset', () => {
288
+ const d = parseLitellmLimitDetail(LITELLM_ROUTER_COOLDOWN_BODY, NOW)
289
+ expect(d.resetAtMs).toBe(NOW.getTime() + 27_500)
290
+ })
291
+
292
+ it('extracts the v1 limiter colon-form rpm limit', () => {
293
+ const d = parseLitellmLimitDetail(LITELLM_V1_PARALLEL_BODY, NOW)
294
+ expect(d.limitType).toBe('rpm')
295
+ expect(d.limit).toBe(60)
296
+ })
297
+
298
+ it('returns all-null on non-LiteLLM prose and never throws on weird input', () => {
299
+ const empty = { limitType: null, limit: null, currentUsage: null, resetAtMs: null }
300
+ expect(parseLitellmLimitDetail('Please try again later.', NOW)).toEqual(empty)
301
+ expect(parseLitellmLimitDetail('', NOW)).toEqual(empty)
302
+ expect(parseLitellmLimitDetail(undefined as unknown as string, NOW)).toEqual(empty)
303
+ expect(parseLitellmLimitDetail({} as unknown as string, NOW)).toEqual(empty)
304
+ })
305
+ })
306
+
120
307
  describe('detectModelUnavailable — network failures', () => {
121
308
  it('classifies ECONNREFUSED', () => {
122
309
  expect(detectModelUnavailable('connect ECONNREFUSED 1.2.3.4:443')?.kind).toBe('network')
@@ -198,6 +198,61 @@ describe('detectErrorInTranscriptLine — error detection', () => {
198
198
  expect(detection).toBeNull()
199
199
  })
200
200
 
201
+ // A 429 whose body carries LiteLLM-proxy-LOCAL limiter wording is the
202
+ // proxy's own tpm_limit/rpm_limit cap tripping BEFORE the request reached
203
+ // Anthropic. Nothing about the account is exhausted — it must take the
204
+ // calm rate-limited path (no model-unavailable card, no mark-exhausted,
205
+ // no fleet failover). Prerequisite for enabling litellm tpm caps on the
206
+ // fleet: without this, every cap trip would bench a healthy account.
207
+ it('classifies a LiteLLM-proxy-local 429 as rate-limited, NOT quota-exhausted', () => {
208
+ const litellmBodies = [
209
+ // Deployment tpm cap (model_rate_limit_check.py body, verbatim shape).
210
+ 'API Error: 429 Deployment over user-defined ratelimit. tpm limit=8000. ' +
211
+ 'current usage=8241. id=abc123def, model_group=claude-fable-5',
212
+ // Virtual-key tpm_limit (parallel_request_limiter_v3.py).
213
+ 'API Error: 429 Rate limit exceeded for api_key: hashed-key-1a2b3c. ' +
214
+ 'Limit type: tokens. Current limit: 8000, Remaining: 0. ' +
215
+ 'Limit resets at: 2026-07-12 08:05:00 UTC',
216
+ // Router cooldown (RouterRateLimitError).
217
+ 'API Error: 429 No deployments available for selected model, ' +
218
+ 'Try again in 27.5 seconds. Passed model=claude-fable-5.',
219
+ // Team-level model cap — a v3 descriptor OUTSIDE the enumerated signal
220
+ // list, covered only by the co-occurrence pair
221
+ // (litellmV3LimiterSignalPair). Pre-fix this missed every signal →
222
+ // quota-exhausted → scary card + failover.
223
+ 'API Error: 429 Rate limit exceeded for model_per_team: ' +
224
+ 'team-1a2b3c:claude-fable-5. Limit type: tokens. ' +
225
+ 'Current limit: 50000, Remaining: 0. ' +
226
+ 'Limit resets at: 2026-07-12 08:05:00 UTC',
227
+ ]
228
+ for (const text of litellmBodies) {
229
+ const line = JSON.stringify({
230
+ type: 'assistant',
231
+ message: {
232
+ role: 'assistant',
233
+ model: '<synthetic>',
234
+ content: [{ type: 'text', text }],
235
+ },
236
+ error: 'rate_limit',
237
+ isApiErrorMessage: true,
238
+ apiErrorStatus: 429,
239
+ })
240
+ const result = detectErrorInTranscriptLine(line)
241
+ expect(result).not.toBeNull()
242
+ expect(result!.kind).toBe('rate-limited')
243
+ expect(result!.transient).toBe(true)
244
+ // End-to-end: the resolver must NOT produce a model-unavailable card —
245
+ // null is what keeps the calm 🚦 card and blocks the auto-fallback
246
+ // branch in emitGatewayOperatorEvent.
247
+ expect(
248
+ resolveModelUnavailableFromOperatorEvent({
249
+ kind: result!.kind,
250
+ detail: result!.detail,
251
+ }),
252
+ ).toBeNull()
253
+ }
254
+ })
255
+
201
256
  // Guard against over-correcting: a GENUINE quota wall (no transient marker)
202
257
  // must STILL be quota-exhausted AND still resolve to a card.
203
258
  it('a genuine quota-wall 429 still produces the quota-exhausted card', () => {
@@ -1170,4 +1170,25 @@ describe("buildFleetRollMessage — reason attribution (#3031 PR 2 reason field)
1170
1170
  expect(msg).toContain("7-day window at 96% on");
1171
1171
  }
1172
1172
  });
1173
+
1174
+ it("model-tier-wall (#3176) names the flagship tier and reassures opus/haiku are unaffected — not a generic quota window", () => {
1175
+ // A tier wall binds on the 7d_oi bucket; window/pct are absent (5h/7d read
1176
+ // healthy). exhausted_until carries the tier reset.
1177
+ const roll: FleetRollInfo = {
1178
+ from: "alice@example.com",
1179
+ to: "bob@example.com",
1180
+ at: NOW - 60_000,
1181
+ reason: "model-tier-wall",
1182
+ bucket: "seven_day_overage_included",
1183
+ exhausted_until: NOW + 2.8 * 60 * 60 * 1000,
1184
+ };
1185
+ const msg = buildFleetRollMessage(roll, NOW);
1186
+ expect(msg).toContain("Flagship (premium) tier weekly limit reached");
1187
+ expect(msg).toContain("opus/haiku on that account are unaffected");
1188
+ // Must NOT mislead as a generic 5h/7d window roll.
1189
+ expect(msg).not.toContain("quota window");
1190
+ expect(msg).not.toContain("Proactive switch");
1191
+ // The tier reset is surfaced.
1192
+ expect(msg).toContain("resets");
1193
+ });
1173
1194
  });
@@ -29,7 +29,7 @@
29
29
  import { describe, it, expect, vi } from 'vitest'
30
30
  import { readFileSync } from 'node:fs'
31
31
  import { fileURLToPath } from 'node:url'
32
- import { createSendGate, type Clock } from '../send-gate.js'
32
+ import { createSendGate, SEND_GATE_SHED, type Clock } from '../send-gate.js'
33
33
  import { createRetryApiCall } from '../retry-api-call.js'
34
34
  import { errors } from './fake-bot-api.js'
35
35
 
@@ -95,7 +95,7 @@ describe('#3155 reactions route through the send gate (cosmetic)', () => {
95
95
  // resolves undefined (fire-and-forget callers .catch nothing), and the gate
96
96
  // counts the shed.
97
97
  expect(setMessageReaction).not.toHaveBeenCalled()
98
- expect(result).toBeUndefined()
98
+ expect(result).toBe(SEND_GATE_SHED) // shed sentinel (#3110 F1)
99
99
  expect(sendGate.stats().global.shed).toBe(1)
100
100
  })
101
101
 
@@ -91,6 +91,30 @@ describe('runtime-metrics — JSONL sink', () => {
91
91
  expect(typeof parsed.ts).toBe('number')
92
92
  })
93
93
 
94
+ it('rate_limit_429_classified carries classification + action + limit/reset detail', () => {
95
+ emitRuntimeMetric({
96
+ kind: 'rate_limit_429_classified',
97
+ agent: 'carrie',
98
+ classification: 'litellm-local',
99
+ action: 'calm',
100
+ reset_at_ms: 1_783_850_700_000,
101
+ reset_in_ms: 300_000,
102
+ limit_type: 'tpm',
103
+ limit: 8000,
104
+ current_usage: 8241,
105
+ })
106
+ const parsed = JSON.parse(readFileSync(metricsPath, 'utf-8').trim())
107
+ expect(parsed.kind).toBe('rate_limit_429_classified')
108
+ expect(parsed.agent).toBe('carrie')
109
+ expect(parsed.classification).toBe('litellm-local')
110
+ expect(parsed.action).toBe('calm')
111
+ expect(parsed.reset_in_ms).toBe(300_000)
112
+ expect(parsed.limit_type).toBe('tpm')
113
+ expect(parsed.limit).toBe(8000)
114
+ expect(parsed.current_usage).toBe(8241)
115
+ expect(typeof parsed.ts).toBe('number')
116
+ })
117
+
94
118
  it('appends — does not overwrite — across calls', () => {
95
119
  for (let i = 0; i < 5; i++) {
96
120
  emitRuntimeMetric({
@@ -1,10 +1,12 @@
1
1
  /**
2
- * Durable session-model file helpers (session-model-file.ts) — the gateway
3
- * side of the stickiness contract (reference/rfcs/session-model-stickiness.md).
2
+ * Session-model file helpers (session-model-file.ts) — the gateway side of the
3
+ * session-scoped /model contract (reference/rfcs/session-model-stickiness.md
4
+ * §0.1, rev 4 — consume-once). The `.relaunch-model-intent` subsystem and the
5
+ * crashloop counter were retired with rev 4; their tests are gone.
4
6
  */
5
7
 
6
8
  import { describe, it, expect, beforeEach, afterEach } from 'vitest'
7
- import { mkdtempSync, rmSync, readFileSync, writeFileSync, existsSync } from 'node:fs'
9
+ import { mkdtempSync, rmSync, writeFileSync, existsSync } from 'node:fs'
8
10
  import { join } from 'node:path'
9
11
  import { tmpdir } from 'node:os'
10
12
  import {
@@ -15,17 +17,8 @@ import {
15
17
  readSessionModelFileRaw,
16
18
  restoreSessionModelFileRaw,
17
19
  clearSessionModelFile,
18
- clearSessionModelBootAttempts,
19
- SESSION_MODEL_BOOT_ATTEMPTS_FILE,
20
- writeRelaunchModelIntent,
21
- clearRelaunchModelIntent,
22
20
  readConfiguredDefaultModel,
23
- intentForRestartReason,
24
- readRelaunchModelIntent,
25
- clearStaleGatewayShutdownIntent,
26
- GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX,
27
21
  SESSION_MODEL_FILE,
28
- RELAUNCH_MODEL_INTENT_FILE,
29
22
  CONFIGURED_DEFAULT_MODEL_FILE,
30
23
  parseSessionEffort,
31
24
  writeSessionEffortFile,
@@ -87,86 +80,6 @@ describe('rollback snapshot (scheduleModelRelaunch dispatch failure)', () => {
87
80
  })
88
81
  })
89
82
 
90
- describe('relaunch intent', () => {
91
- it('writes one-line JSON with intent, reason, and embedded ts (the freshness clock)', () => {
92
- writeRelaunchModelIntent(dir, 'keep', 'user: /new from chat')
93
- const raw = readFileSync(join(dir, RELAUNCH_MODEL_INTENT_FILE), 'utf8')
94
- const parsed = JSON.parse(raw)
95
- expect(parsed.intent).toBe('keep')
96
- expect(parsed.reason).toBe('user: /new from chat')
97
- expect(Math.abs(Date.now() - parsed.ts)).toBeLessThan(5000)
98
- })
99
-
100
- it('last-writer-wins and clearable', () => {
101
- writeRelaunchModelIntent(dir, 'keep', 'a')
102
- writeRelaunchModelIntent(dir, 'revert', 'b')
103
- expect(JSON.parse(readFileSync(join(dir, RELAUNCH_MODEL_INTENT_FILE), 'utf8')).intent).toBe('revert')
104
- clearRelaunchModelIntent(dir)
105
- expect(existsSync(join(dir, RELAUNCH_MODEL_INTENT_FILE))).toBe(false)
106
- })
107
- })
108
-
109
- describe('intentForRestartReason — the triggerSelfRestart per-reason table (RFC §3)', () => {
110
- it.each([
111
- 'schedule-restart-immediate',
112
- 'restart-drain-cap-forced',
113
- 'turn-complete-pending-restart',
114
- 'fleet-fallback-resume',
115
- 'sr-to-claude-model-switch',
116
- ])('switchroom-managed relaunch %s → keep', (reason) => {
117
- expect(intentForRestartReason(reason)).toBe('keep')
118
- })
119
-
120
- it('inline-button-restart → keep (#3039: a restart is not "clear my model")', () => {
121
- expect(intentForRestartReason('inline-button-restart')).toBe('keep')
122
- })
123
-
124
- it('unknown gateway reasons default to keep (only gateway code calls triggerSelfRestart; crashes never do)', () => {
125
- expect(intentForRestartReason('some-future-recovery-path')).toBe('keep')
126
- })
127
- })
128
-
129
- describe('readRelaunchModelIntent', () => {
130
- it('round-trips a stamped intent; null when absent / corrupt / malformed', () => {
131
- expect(readRelaunchModelIntent(dir)).toBeNull()
132
- writeRelaunchModelIntent(dir, 'keep', 'watchdog recovery')
133
- const rec = readRelaunchModelIntent(dir)!
134
- expect(rec.intent).toBe('keep')
135
- expect(rec.reason).toBe('watchdog recovery')
136
- expect(Math.abs(Date.now() - rec.ts)).toBeLessThan(5000)
137
- writeFileSync(join(dir, RELAUNCH_MODEL_INTENT_FILE), '{broken')
138
- expect(readRelaunchModelIntent(dir)).toBeNull()
139
- writeFileSync(join(dir, RELAUNCH_MODEL_INTENT_FILE), '{"intent":"maybe","reason":"x","ts":1}\n')
140
- expect(readRelaunchModelIntent(dir)).toBeNull()
141
- })
142
- })
143
-
144
- describe('clearStaleGatewayShutdownIntent (#3018 finding 4 — a gateway-only bounce must not leave a keep stamp)', () => {
145
- it('clears ONLY a gateway-shutdown-stamped intent and reports it', () => {
146
- writeRelaunchModelIntent(
147
- dir,
148
- 'keep',
149
- `${GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX} graceful SIGTERM shutdown (deploy/rolling restart) — preserving user-chosen session model`,
150
- )
151
- expect(clearStaleGatewayShutdownIntent(dir)).toBe(true)
152
- expect(existsSync(join(dir, RELAUNCH_MODEL_INTENT_FILE))).toBe(false)
153
- // Idempotent: a second call finds nothing.
154
- expect(clearStaleGatewayShutdownIntent(dir)).toBe(false)
155
- })
156
-
157
- it('never touches a triggerSelfRestart / user-slash stamp (un-prefixed reason)', () => {
158
- writeRelaunchModelIntent(dir, 'keep', 'sr-to-claude-model-switch')
159
- expect(clearStaleGatewayShutdownIntent(dir)).toBe(false)
160
- expect(readRelaunchModelIntent(dir)!.reason).toBe('sr-to-claude-model-switch')
161
- })
162
-
163
- it('is a safe no-op on an absent or corrupt intent file', () => {
164
- expect(clearStaleGatewayShutdownIntent(dir)).toBe(false)
165
- writeFileSync(join(dir, RELAUNCH_MODEL_INTENT_FILE), 'not json')
166
- expect(clearStaleGatewayShutdownIntent(dir)).toBe(false)
167
- })
168
- })
169
-
170
83
  describe('readConfiguredDefaultModel', () => {
171
84
  it('reads the trimmed value; null when absent or empty', () => {
172
85
  expect(readConfiguredDefaultModel(dir)).toBeNull()
@@ -181,9 +94,9 @@ describe('readConfiguredDefaultModel', () => {
181
94
  })
182
95
  })
183
96
 
184
- // ─── #3039: durable session-effort carrier ───────────────────────────────────
97
+ // ─── #3186: consume-once session-effort carrier (queued-command boot apply) ──
185
98
 
186
- describe('session-effort file helpers (#3039)', () => {
99
+ describe('session-effort file helpers (#3186)', () => {
187
100
  let dir: string
188
101
  beforeEach(() => {
189
102
  dir = mkdtempSync(join(tmpdir(), 'sr-session-effort-'))
@@ -218,64 +131,3 @@ describe('session-effort file helpers (#3039)', () => {
218
131
  expect(readSessionEffortFile(dir)).toBeNull()
219
132
  })
220
133
  })
221
-
222
- describe('intentForRestartReason is keep-for-everything (#3039)', () => {
223
- it('every reason keeps — restarts never clear a user model choice', () => {
224
- for (const reason of [
225
- 'inline-button-restart',
226
- 'user: /restart from chat',
227
- 'schedule-restart-immediate',
228
- 'anything-else',
229
- ]) {
230
- expect(intentForRestartReason(reason)).toBe('keep')
231
- }
232
- })
233
- })
234
-
235
- describe('clearSessionModelBootAttempts (#3043 item 2: bridge-register clears the crashloop counter)', () => {
236
- const bootAttemptsPath = () => join(dir, SESSION_MODEL_BOOT_ATTEMPTS_FILE)
237
-
238
- // Faithful model of start.sh.hbs "Override crashloop self-heal": each boot
239
- // within 150s of the previous stamp bumps the count; a stale stamp resets to
240
- // 1. Reproduced here so the accumulate-vs-reset OUTCOME is asserted, not just
241
- // the delete call.
242
- function simulateBootStamp(nowSec: number): number {
243
- let count = 0
244
- let prev = 0
245
- if (existsSync(bootAttemptsPath())) {
246
- const [c, p] = readFileSync(bootAttemptsPath(), 'utf8').trim().split(/\s+/)
247
- count = Number(c) || 0
248
- prev = Number(p) || 0
249
- }
250
- count = nowSec - prev < 150 ? count + 1 : 1
251
- writeFileSync(bootAttemptsPath(), `${count} ${nowSec}\n`)
252
- return count
253
- }
254
-
255
- it('WITHOUT a bridge register, three fast boots accumulate toward the 3-strike clear', () => {
256
- // Three operator hand-bounces, each <150s apart, with no register between.
257
- expect(simulateBootStamp(1000)).toBe(1)
258
- expect(simulateBootStamp(1010)).toBe(2)
259
- expect(simulateBootStamp(1020)).toBe(3) // start.sh would now clear a HEALTHY override — the false positive
260
- })
261
-
262
- it('a bridge register between boots resets the counter, so a healthy agent never reaches 3', () => {
263
- expect(simulateBootStamp(1000)).toBe(1)
264
- // Boot 1 came all the way up and the bridge registered → gateway clears it.
265
- clearSessionModelBootAttempts(dir)
266
- expect(existsSync(bootAttemptsPath())).toBe(false)
267
-
268
- // Next fast boot starts fresh at 1 (no accumulation), and register clears again.
269
- expect(simulateBootStamp(1010)).toBe(1)
270
- clearSessionModelBootAttempts(dir)
271
- expect(simulateBootStamp(1020)).toBe(1)
272
- // The healthy agent's counter never climbs to the 3-strike clear.
273
- const [count] = readFileSync(bootAttemptsPath(), 'utf8').trim().split(/\s+/)
274
- expect(Number(count)).toBeLessThan(3)
275
- })
276
-
277
- it('is best-effort — no throw when the counter file is absent', () => {
278
- expect(existsSync(bootAttemptsPath())).toBe(false)
279
- expect(() => clearSessionModelBootAttempts(dir)).not.toThrow()
280
- })
281
- })