switchroom 0.18.15 → 0.18.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +3 -0
- package/dist/auth-broker/index.js +432 -10
- package/dist/cli/notion-write-pretool.mjs +3 -0
- package/dist/cli/switchroom.js +50 -1
- package/dist/host-control/main.js +4 -1
- package/dist/vault/approvals/kernel-server.js +3 -0
- package/dist/vault/broker/server.js +3 -0
- package/package.json +1 -1
- package/profiles/_base/start.sh.hbs +81 -139
- package/telegram-plugin/dist/gateway/gateway.js +386 -259
- package/telegram-plugin/draft-stream.ts +78 -3
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +3 -4
- package/telegram-plugin/gateway/effort-command.ts +9 -7
- package/telegram-plugin/gateway/gateway.ts +265 -220
- package/telegram-plugin/gateway/litellm-local-notice-wiring.ts +200 -0
- package/telegram-plugin/gateway/model-command.ts +96 -18
- package/telegram-plugin/gateway/pending-session-command.ts +10 -8
- package/telegram-plugin/gateway/session-model-file.ts +38 -172
- package/telegram-plugin/litellm-local-notice.ts +189 -0
- package/telegram-plugin/quota-watch.ts +16 -4
- package/telegram-plugin/runtime-metrics.ts +16 -0
- package/telegram-plugin/send-gate-degraded.test.ts +9 -7
- package/telegram-plugin/send-gate.ts +34 -4
- package/telegram-plugin/stream-controller.ts +143 -20
- package/telegram-plugin/stream-reply-handler.ts +12 -2
- package/telegram-plugin/tests/bot-api.harness.ts +7 -2
- package/telegram-plugin/tests/draft-stream.test.ts +110 -1
- package/telegram-plugin/tests/effort-command.test.ts +4 -4
- package/telegram-plugin/tests/flood-windows-persistence.test.ts +2 -2
- package/telegram-plugin/tests/gateway-pending-command-wiring.test.ts +33 -19
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +47 -127
- package/telegram-plugin/tests/litellm-local-notice.test.ts +417 -0
- package/telegram-plugin/tests/model-command.test.ts +84 -1
- package/telegram-plugin/tests/quota-watch.test.ts +21 -0
- package/telegram-plugin/tests/reaction-gate-routing.test.ts +2 -2
- package/telegram-plugin/tests/session-model-file.test.ts +7 -155
- package/telegram-plugin/tests/stream-controller-send-gate.test.ts +521 -0
- package/telegram-plugin/tests/stream-reply-handler.test.ts +44 -0
- package/telegram-plugin/tests/worker-activity-feed.test.ts +207 -0
- package/telegram-plugin/worker-activity-feed.ts +83 -8
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* litellm-local-notice-wiring.ts — side-effect runner for the debounced
|
|
3
|
+
* litellm-local 429 notice.
|
|
4
|
+
*
|
|
5
|
+
* The state machine, notice text, config parsing, and metric payload live in
|
|
6
|
+
* ../litellm-local-notice.ts (pure). This module owns the sequencing, with
|
|
7
|
+
* every dependency injected so the wiring is unit-testable without importing
|
|
8
|
+
* gateway.ts (same shape as throttle-tier-wiring.ts):
|
|
9
|
+
*
|
|
10
|
+
* 1. Guard — only the `litellm-local` classification is eligible
|
|
11
|
+
* (isLitellmLocalNoticeEligible). account-scoped keeps the throttle
|
|
12
|
+
* tier; generic-transient keeps the calm rate-limited card. The guard
|
|
13
|
+
* lives HERE (not in the caller) so "a notice never fires for
|
|
14
|
+
* non-litellm-local classifications" is mechanism, not discipline.
|
|
15
|
+
* 2. Per-agent cooldown — evaluate against the resolved window
|
|
16
|
+
* (deps.windowMs(), re-read per event so an operator-tuned
|
|
17
|
+
* channels.telegram.litellm_notice.window_ms takes effect after
|
|
18
|
+
* apply+restart without re-creating the runner). Suppressed events are
|
|
19
|
+
* counted silently.
|
|
20
|
+
* 3. ONE calm notice — broadcast to every authorized chat via the injected
|
|
21
|
+
* send (the gateway binds it to swallowingApiCall, the standard
|
|
22
|
+
* retry-wrapped path). The window is armed, the
|
|
23
|
+
* `litellm_local_429_notice` metric emitted, and "posted" logged ONLY
|
|
24
|
+
* when at least one send was actually ISSUED — an empty allowFrom or a
|
|
25
|
+
* send callback that throws for every chat leaves the state untouched,
|
|
26
|
+
* so the next event retries instead of silently claiming delivery.
|
|
27
|
+
*
|
|
28
|
+
* Deliberately NO broker calls, NO quota-ledger writes, NO failover, NO
|
|
29
|
+
* retry nudge: the litellm-local calm path's invariant is that account state
|
|
30
|
+
* is never touched (the request never reached Anthropic). Claude Code's own
|
|
31
|
+
* retry handles the turn.
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
import {
|
|
35
|
+
buildLitellmLocalNoticeMetric,
|
|
36
|
+
evaluateLitellmLocalNotice,
|
|
37
|
+
initialLitellmLocalNoticeState,
|
|
38
|
+
isLitellmLocalNoticeEligible,
|
|
39
|
+
renderLitellmLocalNotice,
|
|
40
|
+
type LitellmLocalNoticeState,
|
|
41
|
+
} from '../litellm-local-notice.js'
|
|
42
|
+
import type { RateLimit429Classification } from '../throttle-tier.js'
|
|
43
|
+
import type { RuntimeMetricEvent } from '../runtime-metrics.js'
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* The user-facing surface decision for one terminal non-account-scoped
|
|
47
|
+
* `rate-limited` operator event — the gateway's ordering contract with the
|
|
48
|
+
* shared per-agent-per-kind card cooldown (shouldEmitOperatorEvent in
|
|
49
|
+
* operator-events.ts), extracted so it is pinnable by tests:
|
|
50
|
+
*
|
|
51
|
+
* - `litellm-local` resolves BEFORE the shared gate is consulted. Two
|
|
52
|
+
* load-bearing consequences: (a) it never ARMS `${agent}:rate-limited`
|
|
53
|
+
* (shouldEmitOperatorEvent's true-return records the timestamp), so a
|
|
54
|
+
* proxy-cap burst can't suppress a later genuine transient's card; and
|
|
55
|
+
* (b) it is never SUPPRESSED by a cooldown that a recent 529 / generic
|
|
56
|
+
* card armed — the notice runner owns its own per-agent debounce
|
|
57
|
+
* instead.
|
|
58
|
+
* - Every other classification consults (and on true, arms) the shared
|
|
59
|
+
* gate exactly once — the caller must NOT re-consult it downstream.
|
|
60
|
+
*/
|
|
61
|
+
export function decideRateLimitedSurface(opts: {
|
|
62
|
+
classification: RateLimit429Classification
|
|
63
|
+
agent: string
|
|
64
|
+
/** The shared cooldown gate — gateway binds
|
|
65
|
+
* (a) => shouldEmitOperatorEvent(a, 'rate-limited'). Consulted (and on
|
|
66
|
+
* true, armed) only for non-litellm-local classifications. */
|
|
67
|
+
shouldEmitCard: (agent: string) => boolean
|
|
68
|
+
}): 'litellm-local-notice' | 'generic-card' | 'cooldown-suppressed' {
|
|
69
|
+
if (isLitellmLocalNoticeEligible(opts.classification)) return 'litellm-local-notice'
|
|
70
|
+
return opts.shouldEmitCard(opts.agent) ? 'generic-card' : 'cooldown-suppressed'
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export interface LitellmLocalNoticeRunnerDeps {
|
|
74
|
+
/** Chats the notice broadcasts to (access.allowFrom, resolved per call). */
|
|
75
|
+
listNoticeChats(): Array<string | number>
|
|
76
|
+
/** Fire-and-forget rich send (gateway wraps swallowingApiCall). */
|
|
77
|
+
sendNotice(chatId: string | number, markdown: string): void
|
|
78
|
+
/** Resolved cooldown window (ms), re-read per event —
|
|
79
|
+
* parseLitellmNoticeWindowMs(loadAccess().litellmNoticeWindowMs). */
|
|
80
|
+
windowMs(): number
|
|
81
|
+
/** Runtime-metric sink (gateway binds emitRuntimeMetric). */
|
|
82
|
+
emitMetric(event: RuntimeMetricEvent): void
|
|
83
|
+
log(msg: string): void
|
|
84
|
+
now?: () => number
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/**
|
|
88
|
+
* What one event produced:
|
|
89
|
+
* - `sent` — a notice was issued to ≥1 chat (window armed, metric
|
|
90
|
+
* emitted). The gateway records the event into the
|
|
91
|
+
* operator-event history ONLY on this outcome, so a
|
|
92
|
+
* suppressed low-stakes proxy throttle never overwrites a
|
|
93
|
+
* more important most-recent event (e.g.
|
|
94
|
+
* credentials-expired) in /status enrichment.
|
|
95
|
+
* - `suppressed` — inside the cooldown window; counted silently.
|
|
96
|
+
* - `skipped` — ineligible classification, no authorized recipients,
|
|
97
|
+
* every send failed synchronously, or an internal error
|
|
98
|
+
* (logged). No debounce state change.
|
|
99
|
+
*/
|
|
100
|
+
export type LitellmLocalNoticeOutcome = 'sent' | 'suppressed' | 'skipped'
|
|
101
|
+
|
|
102
|
+
export interface LitellmLocalNoticeRunner {
|
|
103
|
+
/**
|
|
104
|
+
* Run the notice path for one terminal rate-limited operator event.
|
|
105
|
+
* No-ops (returns 'skipped') unless `classification` is `litellm-local`.
|
|
106
|
+
* Never throws.
|
|
107
|
+
*/
|
|
108
|
+
onRateLimited(
|
|
109
|
+
classification: RateLimit429Classification,
|
|
110
|
+
agent: string,
|
|
111
|
+
): LitellmLocalNoticeOutcome
|
|
112
|
+
/** Test/debug view of internal state. */
|
|
113
|
+
inspect(): { state: LitellmLocalNoticeState }
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export function createLitellmLocalNoticeRunner(
|
|
117
|
+
deps: LitellmLocalNoticeRunnerDeps,
|
|
118
|
+
): LitellmLocalNoticeRunner {
|
|
119
|
+
const now = deps.now ?? (() => Date.now())
|
|
120
|
+
// In-memory only — the debounce resets on gateway restart, so a sustained
|
|
121
|
+
// throttle can produce one extra notice per restart. Accepted trade-off:
|
|
122
|
+
// persisting a courtesy-surface cooldown isn't worth state-file plumbing,
|
|
123
|
+
// and the failure mode is one duplicate calm message, not a storm.
|
|
124
|
+
let state = initialLitellmLocalNoticeState()
|
|
125
|
+
|
|
126
|
+
function onRateLimited(
|
|
127
|
+
classification: RateLimit429Classification,
|
|
128
|
+
agent: string,
|
|
129
|
+
): LitellmLocalNoticeOutcome {
|
|
130
|
+
try {
|
|
131
|
+
if (!isLitellmLocalNoticeEligible(classification)) return 'skipped'
|
|
132
|
+
const windowMs = deps.windowMs()
|
|
133
|
+
const verdict = evaluateLitellmLocalNotice(state, agent, now(), windowMs)
|
|
134
|
+
if (!verdict.send) {
|
|
135
|
+
state = verdict.next
|
|
136
|
+
deps.log(
|
|
137
|
+
`[litellm-local-notice] suppressed (cooldown) agent=${agent} ` +
|
|
138
|
+
`suppressedSoFar=${state.suppressedCountByAgent[agent] ?? 0}`,
|
|
139
|
+
)
|
|
140
|
+
return 'suppressed'
|
|
141
|
+
}
|
|
142
|
+
const chats = deps.listNoticeChats()
|
|
143
|
+
const markdown = renderLitellmLocalNotice({
|
|
144
|
+
agent,
|
|
145
|
+
suppressedSinceLastNotice: verdict.suppressedSinceLastNotice,
|
|
146
|
+
})
|
|
147
|
+
let issued = 0
|
|
148
|
+
for (const chatId of chats) {
|
|
149
|
+
try {
|
|
150
|
+
deps.sendNotice(chatId, markdown)
|
|
151
|
+
issued++
|
|
152
|
+
} catch (err) {
|
|
153
|
+
deps.log(
|
|
154
|
+
`[litellm-local-notice] send failed chat=${chatId} agent=${agent}: ` +
|
|
155
|
+
`${(err as Error)?.message ?? err}`,
|
|
156
|
+
)
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
if (issued === 0) {
|
|
160
|
+
// Nothing actually went out (no authorized chats, or every send
|
|
161
|
+
// threw synchronously). Do NOT arm the window, emit the metric, or
|
|
162
|
+
// claim "posted" — the next event retries instead of the debounce
|
|
163
|
+
// silently absorbing events no one will ever hear about.
|
|
164
|
+
deps.log(
|
|
165
|
+
`[litellm-local-notice] no sends issued (chats=${chats.length}) agent=${agent} — ` +
|
|
166
|
+
`window not armed, will retry on the next event`,
|
|
167
|
+
)
|
|
168
|
+
return 'skipped'
|
|
169
|
+
}
|
|
170
|
+
state = verdict.next
|
|
171
|
+
deps.emitMetric(
|
|
172
|
+
buildLitellmLocalNoticeMetric({
|
|
173
|
+
agent,
|
|
174
|
+
suppressedCount: verdict.suppressedSinceLastNotice,
|
|
175
|
+
windowMs,
|
|
176
|
+
}),
|
|
177
|
+
)
|
|
178
|
+
deps.log(
|
|
179
|
+
`[litellm-local-notice] posted agent=${agent} chats=${issued} ` +
|
|
180
|
+
`suppressedSinceLast=${verdict.suppressedSinceLastNotice} windowMs=${windowMs}`,
|
|
181
|
+
)
|
|
182
|
+
return 'sent'
|
|
183
|
+
} catch (err) {
|
|
184
|
+
// The notice is a courtesy surface — a failure here must never break
|
|
185
|
+
// the operator-event path that invoked it. The log itself is wrapped
|
|
186
|
+
// too: a dead stderr sink must not void the never-throws contract.
|
|
187
|
+
try {
|
|
188
|
+
deps.log(
|
|
189
|
+
`[litellm-local-notice] error agent=${agent}: ${(err as Error)?.message ?? err}`,
|
|
190
|
+
)
|
|
191
|
+
} catch { /* sink dead — nothing left to do */ }
|
|
192
|
+
return 'skipped'
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
return {
|
|
197
|
+
onRateLimited,
|
|
198
|
+
inspect: () => ({ state }),
|
|
199
|
+
}
|
|
200
|
+
}
|
|
@@ -108,6 +108,82 @@ export function parseModelCommand(text: string): ParsedModelCommand | null {
|
|
|
108
108
|
return { kind: 'set', model: arg }
|
|
109
109
|
}
|
|
110
110
|
|
|
111
|
+
/**
|
|
112
|
+
* The busy state a typed `/model` disposition is decided against. Two
|
|
113
|
+
* independent signals are folded here (#3177): `currentTurnActive` is the
|
|
114
|
+
* gateway's per-turn atom (`currentTurn !== null`), which is NULLED at
|
|
115
|
+
* `turn_end`; `turnInFlight` is the authoritative delivery-machine +
|
|
116
|
+
* pending-approval gate (`turnInFlightForGate()`). The atom can read idle
|
|
117
|
+
* while the session is still busy (the "recovered-late" / premature-turn-end
|
|
118
|
+
* window seen in finn's 2026-07-12 log), so a switch decided on the atom ALONE
|
|
119
|
+
* takes the direct-inject path and types `/model <name>` into a still-busy
|
|
120
|
+
* pane, where claude swallows it as literal text — a silent no-op. Folding
|
|
121
|
+
* BOTH signals means a session busy by EITHER measure ack+queues instead.
|
|
122
|
+
*/
|
|
123
|
+
export interface ModelCommandContext {
|
|
124
|
+
/** Gateway per-turn atom is non-null (`currentTurn !== null`). */
|
|
125
|
+
currentTurnActive: boolean
|
|
126
|
+
/** Authoritative busy gate (`turnInFlightForGate()`): machine-in-turn OR a pending approval. */
|
|
127
|
+
turnInFlight: boolean
|
|
128
|
+
/** Interactive picker menu is enabled (`SWITCHROOM_MODEL_MENU !== '0'`). */
|
|
129
|
+
menuEnabled: boolean
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* True when the session is busy by EITHER busy signal. A typed `/model` set
|
|
134
|
+
* MUST NOT take the direct-inject path while this holds — it would type into a
|
|
135
|
+
* busy pane and be swallowed as text. See ModelCommandContext.
|
|
136
|
+
*/
|
|
137
|
+
export function isModelCommandBusy(ctx: Pick<ModelCommandContext, 'currentTurnActive' | 'turnInFlight'>): boolean {
|
|
138
|
+
return ctx.currentTurnActive || ctx.turnInFlight
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* What the gateway should do with a parsed `/model` command. Pure so the
|
|
143
|
+
* routing decision is unit-testable without booting the bot. Every parsed
|
|
144
|
+
* shape maps to exactly one visible action — there is NO branch that silently
|
|
145
|
+
* does nothing (the #3177 swallow):
|
|
146
|
+
*
|
|
147
|
+
* - `menu` — bare `/model` with the picker enabled → render the dashboard.
|
|
148
|
+
* - `queue` — a `set` while busy (by EITHER signal) → ack + enqueue for
|
|
149
|
+
* apply-on-idle. `target` is the alias-expanded token.
|
|
150
|
+
* - `apply` — run the handler now (idle `set`, or `show`/`help` text paths).
|
|
151
|
+
*/
|
|
152
|
+
export type ModelCommandDisposition =
|
|
153
|
+
| { kind: 'menu' }
|
|
154
|
+
| { kind: 'queue'; target: string }
|
|
155
|
+
| { kind: 'apply'; parsed: ParsedModelCommand }
|
|
156
|
+
|
|
157
|
+
export function planModelCommand(
|
|
158
|
+
parsed: ParsedModelCommand,
|
|
159
|
+
ctx: ModelCommandContext,
|
|
160
|
+
): ModelCommandDisposition {
|
|
161
|
+
if (parsed.kind === 'show' && ctx.menuEnabled) return { kind: 'menu' }
|
|
162
|
+
if (parsed.kind === 'set' && isModelCommandBusy(ctx)) {
|
|
163
|
+
return { kind: 'queue', target: expandSrAlias(parsed.model) }
|
|
164
|
+
}
|
|
165
|
+
return { kind: 'apply', parsed }
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* The unconditional durable log line the gateway writes at `/model` entry,
|
|
170
|
+
* BEFORE any branch or reply attempt (#3177 guarantee (b)). Even if every
|
|
171
|
+
* downstream reply is shed/dropped, this line — plus the paired history row —
|
|
172
|
+
* guarantees a typed `/model` is never invisible. Kept pure + shape-stable so
|
|
173
|
+
* the format is testable and greppable (`grep 'gw /model received'`).
|
|
174
|
+
*/
|
|
175
|
+
export function modelCommandReceiptLine(
|
|
176
|
+
agent: string,
|
|
177
|
+
parsed: ParsedModelCommand,
|
|
178
|
+
busy: boolean,
|
|
179
|
+
): string {
|
|
180
|
+
const arg =
|
|
181
|
+
parsed.kind === 'set' ? parsed.model
|
|
182
|
+
: parsed.kind === 'show' ? '(show)'
|
|
183
|
+
: '(help)'
|
|
184
|
+
return `telegram gateway: gw /model received agent=${agent} kind=${parsed.kind} arg=${arg} busy=${busy}`
|
|
185
|
+
}
|
|
186
|
+
|
|
111
187
|
export interface ModelCommandDeps {
|
|
112
188
|
/** Inject primitive — wired to injectSlashCommand in the gateway. */
|
|
113
189
|
inject: (agent: string, command: string) => Promise<InjectResult>
|
|
@@ -175,7 +251,7 @@ export interface ModelCommandReply {
|
|
|
175
251
|
}
|
|
176
252
|
|
|
177
253
|
const PERSIST_NOTE =
|
|
178
|
-
'
|
|
254
|
+
'_Session-only — this override lasts until the agent’s next restart, then reverts to the configured \`model:\`. \`/model default\` clears it now. To change the default permanently, set \`model:\` in switchroom.yaml._'
|
|
179
255
|
|
|
180
256
|
function helpText(deps: ModelCommandDeps, reason?: string): ModelCommandReply {
|
|
181
257
|
const srAliasExamples = Object.keys(SR_MODEL_ALIASES).map(a => `\`${a}\``).join(' · ')
|
|
@@ -507,18 +583,19 @@ export const SR_MODEL_ALIASES: Record<string, string> = {
|
|
|
507
583
|
|
|
508
584
|
/** Expand a short alias (case-insensitive) to its full sr-* id, or return the original. */
|
|
509
585
|
/**
|
|
510
|
-
* #3042 review blocker 2a: can `token` be trusted for a
|
|
511
|
-
* `.session-model` persist WITHOUT a live confirmation from claude?
|
|
586
|
+
* #3042 review blocker 2a: can `token` be trusted for a boot-applied
|
|
587
|
+
* `.session-model` carrier persist WITHOUT a live confirmation from claude?
|
|
512
588
|
*
|
|
513
589
|
* A queued typed `/model <arg>` that is persisted at shutdown was never
|
|
514
|
-
* validated by claude's picker
|
|
515
|
-
*
|
|
516
|
-
*
|
|
517
|
-
* tokens with a switchroom-known
|
|
518
|
-
* Claude aliases (claude resolves
|
|
519
|
-
*
|
|
520
|
-
* arbitrary `sr-*` ids typed by
|
|
521
|
-
* session to verify, so the operator is
|
|
590
|
+
* validated by claude's picker. Under the consume-once carrier (rev 4) a
|
|
591
|
+
* garbage-but-shape-valid token (e.g. `claude-nonexistnet-9`) can crash at
|
|
592
|
+
* most ONE boot before the carrier is gone and the agent reverts — but we
|
|
593
|
+
* still avoid even that crash-boot. Only tokens with a switchroom-known
|
|
594
|
+
* meaning are offline-trustable: the static Claude aliases (claude resolves
|
|
595
|
+
* them itself) and the curated sr-* alias TARGETS (present in the LiteLLM
|
|
596
|
+
* config by construction). Full `claude-*` / arbitrary `sr-*` ids typed by
|
|
597
|
+
* hand are refused — they need the live session to verify, so the operator is
|
|
598
|
+
* asked to re-issue after boot.
|
|
522
599
|
*/
|
|
523
600
|
export function isOfflineTrustedModelToken(token: string): boolean {
|
|
524
601
|
// #3043 item 1: the live accept path normalizes case (isClaudeModel /
|
|
@@ -833,18 +910,19 @@ export interface ModelCallbackOutcome {
|
|
|
833
910
|
/**
|
|
834
911
|
* The canonical `claude --model` token (alias or full `claude-*` id) for a
|
|
835
912
|
* Claude selection, when derivable — distinct from `selectedModel` (a display
|
|
836
|
-
* name for /status).
|
|
837
|
-
*
|
|
838
|
-
*
|
|
913
|
+
* name for /status). Session-scoped (rev 4): a live Claude selection persists
|
|
914
|
+
* NO carrier (the switch applies in-session and reverts on the next boot);
|
|
915
|
+
* the gateway uses this token ONLY on an sr-* → Claude transition, writing it
|
|
916
|
+
* to the consume-once `.session-model` carrier so that transition's own
|
|
917
|
+
* apply-relaunch boots the tapped model (then reverts on the following
|
|
839
918
|
* restart). Absent when the target has no derivable token.
|
|
840
919
|
*/
|
|
841
920
|
selectedModelToken?: string
|
|
842
921
|
/**
|
|
843
922
|
* True when the confirmed selection was the "Default (recommended)" row —
|
|
844
|
-
* i.e. the session is now on the configured default and any
|
|
845
|
-
* `.session-model`
|
|
846
|
-
*
|
|
847
|
-
* model on the next keep-relaunch).
|
|
923
|
+
* i.e. the session is now on the configured default and any leftover
|
|
924
|
+
* `.session-model` carrier must be CLEARED (a stale carrier would be
|
|
925
|
+
* consumed — mis-applied — by the next boot).
|
|
848
926
|
*/
|
|
849
927
|
clearedDefault?: boolean
|
|
850
928
|
/** Short toast for answerCallbackQuery. */
|
|
@@ -244,15 +244,17 @@ export interface PendingCommandCardEdit {
|
|
|
244
244
|
}
|
|
245
245
|
|
|
246
246
|
/**
|
|
247
|
-
* What the gateway should
|
|
248
|
-
*
|
|
249
|
-
*
|
|
250
|
-
*
|
|
251
|
-
* `.session-effort
|
|
252
|
-
*
|
|
247
|
+
* What the gateway should record for a queued command that can't apply live
|
|
248
|
+
* because the session is going away (gateway shutdown or a pending session
|
|
249
|
+
* relaunch). Instead of telling the operator to re-issue, the choice is
|
|
250
|
+
* persisted to the CONSUME-ONCE boot carriers (`.session-model` /
|
|
251
|
+
* `.session-effort`, #3184/#3186): start.sh applies each on the single boot
|
|
252
|
+
* that reads it — the queued command's apply-relaunch — and then deletes it,
|
|
253
|
+
* so the queued command still deterministically applies via the relaunch and
|
|
254
|
+
* any SUBSEQUENT restart reverts to the configured default.
|
|
253
255
|
*
|
|
254
|
-
* - 'model' — persist `arg` as the `.session-model`
|
|
255
|
-
* - 'effort' — persist `arg` as the `.session-effort`
|
|
256
|
+
* - 'model' — persist `arg` as the `.session-model` carrier
|
|
257
|
+
* - 'effort' — persist `arg` as the `.session-effort` carrier
|
|
256
258
|
* - 'clear-model' — the queued command was `/model default`: clear the carrier
|
|
257
259
|
* - 'clear-effort' — the queued command was `/effort default`: clear the carrier
|
|
258
260
|
* - null — not offline-resolvable (a `mdl:s:<tag>` menu selection
|
|
@@ -1,32 +1,31 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
2
|
+
* Session-scoped `/model` carrier — file helpers shared by the gateway.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* Contract: reference/rfcs/session-model-stickiness.md §0.1 (rev 4, operator
|
|
5
|
+
* decision 2026-07-12 — SESSION-SCOPED, superseding the rev-3 keep-by-default).
|
|
6
|
+
* Files in the bind-mounted agent state dir:
|
|
6
7
|
*
|
|
7
|
-
* - `.session-model` —
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
* (
|
|
13
|
-
*
|
|
8
|
+
* - `.session-model` — a CONSUME-ONCE carrier written ONLY immediately
|
|
9
|
+
* before a relaunch that applies the switch (the sr-* / sr→Claude paths
|
|
10
|
+
* and a queued /model persisted at graceful shutdown). One-line JSON
|
|
11
|
+
* `{"model","configuredDefaultAtWrite","ts"}`. start.sh applies it on the
|
|
12
|
+
* single boot that reads it and then deletes it; every SUBSEQUENT restart
|
|
13
|
+
* (deploy, /restart, /new, watchdog recovery, crash, raw docker restart)
|
|
14
|
+
* finds no carrier and boots the configured default. Live Claude switches
|
|
15
|
+
* do NOT write a carrier — they apply in-session and the explicit
|
|
16
|
+
* `claude --model <configured>` flag reverts them on the next boot.
|
|
17
|
+
* Cleared live by `/model default`; invalidation (corruption / the
|
|
18
|
+
* configured yaml default changed at the apply-boot) drops it and notifies
|
|
19
|
+
* the operator chat via `.session-model-alert`.
|
|
14
20
|
*
|
|
15
|
-
* - `.
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
23
|
-
*
|
|
24
|
-
* - `.session-effort` — the DURABLE effort override (#3039), same shape
|
|
25
|
-
* and lifecycle as `.session-model` with `level` in place of `model`:
|
|
26
|
-
* `{"level","configuredDefaultAtWrite","ts"}`. Written on every
|
|
27
|
-
* positively-confirmed `/effort` apply, resolved by start.sh into the
|
|
28
|
-
* relaunch's `--effort`, cleared only by `/effort default` or
|
|
29
|
-
* invalidation (with a boot alert).
|
|
21
|
+
* - `.session-effort` — the CONSUME-ONCE effort carrier (#3186, session-
|
|
22
|
+
* scoped like `.session-model`). Same shape with `level` in place of
|
|
23
|
+
* `model`: `{"level","configuredDefaultAtWrite","ts"}`. Written ONLY by
|
|
24
|
+
* the queued-command shutdown persist (a mid-turn `/effort` carried
|
|
25
|
+
* across the bounce); start.sh resolves it into the apply-boot's
|
|
26
|
+
* `--effort` and deletes it, so any subsequent restart reverts to the
|
|
27
|
+
* configured `thinking_effort`. Live `/effort` applies record in memory
|
|
28
|
+
* only. Cleared live by `/effort default`; invalidation alerts at boot.
|
|
30
29
|
*
|
|
31
30
|
* The `model` token is always a canonical `claude --model` token (alias,
|
|
32
31
|
* `claude-*` id, or `sr-*` id) — NEVER a display label like "Opus 4.8".
|
|
@@ -38,12 +37,7 @@ import { join } from 'node:path'
|
|
|
38
37
|
import { isValidModelArg } from './model-command.js'
|
|
39
38
|
|
|
40
39
|
export const SESSION_MODEL_FILE = '.session-model'
|
|
41
|
-
export const RELAUNCH_MODEL_INTENT_FILE = '.relaunch-model-intent'
|
|
42
40
|
export const CONFIGURED_DEFAULT_MODEL_FILE = '.configured-default-model'
|
|
43
|
-
/** Crashloop self-heal counter (start.sh stamps `<count> <epoch>` per fast boot). */
|
|
44
|
-
export const SESSION_MODEL_BOOT_ATTEMPTS_FILE = '.session-model-boot-attempts'
|
|
45
|
-
|
|
46
|
-
export type RelaunchModelIntent = 'keep' | 'revert'
|
|
47
41
|
|
|
48
42
|
export interface SessionModelRecord {
|
|
49
43
|
model: string
|
|
@@ -51,21 +45,6 @@ export interface SessionModelRecord {
|
|
|
51
45
|
ts: number
|
|
52
46
|
}
|
|
53
47
|
|
|
54
|
-
/**
|
|
55
|
-
* Classify a triggerSelfRestart reason into the intent the boot should honor.
|
|
56
|
-
*
|
|
57
|
-
* Since #3039 (operator contract 2026-07-11) EVERY restart reason keeps the
|
|
58
|
-
* override: a restart — user-tapped, watchdog, deploy, or crash — is not
|
|
59
|
-
* "clear my model". The override is cleared only by explicit user action
|
|
60
|
-
* (`/model default`) or invalidation at boot, both of which run their own
|
|
61
|
-
* paths. The 'revert' intent value remains recognised by start.sh (and this
|
|
62
|
-
* classifier's signature keeps it) so an older gateway's stamp still parses,
|
|
63
|
-
* but current code never emits it.
|
|
64
|
-
*/
|
|
65
|
-
export function intentForRestartReason(_reason: string): RelaunchModelIntent {
|
|
66
|
-
return 'keep'
|
|
67
|
-
}
|
|
68
|
-
|
|
69
48
|
function atomicWrite(path: string, content: string): void {
|
|
70
49
|
const tmp = `${path}.tmp-${process.pid}-${Date.now()}`
|
|
71
50
|
writeFileSync(tmp, content, 'utf8')
|
|
@@ -104,8 +83,8 @@ export function parseSessionModel(text: string): SessionModelRecord | null {
|
|
|
104
83
|
}
|
|
105
84
|
|
|
106
85
|
/**
|
|
107
|
-
* Write the
|
|
108
|
-
* callers must pass a `claude --model` token, never a display label
|
|
86
|
+
* Write the consume-once session-model carrier. Throws on a non-canonical
|
|
87
|
+
* token — callers must pass a `claude --model` token, never a display label
|
|
109
88
|
* (regression guard for the "Opus 4.5 persisted" class).
|
|
110
89
|
*/
|
|
111
90
|
export function writeSessionModelFile(
|
|
@@ -137,37 +116,10 @@ export function readSessionModelFile(agentDir: string): SessionModelRecord | nul
|
|
|
137
116
|
return raw == null ? null : parseSessionModel(raw)
|
|
138
117
|
}
|
|
139
118
|
|
|
140
|
-
/** Delete the
|
|
119
|
+
/** Delete the session-model carrier (`/model default`, rollback). Best-effort. */
|
|
141
120
|
export function clearSessionModelFile(agentDir: string): void {
|
|
142
121
|
try {
|
|
143
122
|
rmSync(join(agentDir, SESSION_MODEL_FILE), { force: true })
|
|
144
|
-
// #3042 item 4: also drop the kept-alert dedup sentinel so a future
|
|
145
|
-
// override of the same name re-alerts on its first kept boot.
|
|
146
|
-
rmSync(join(agentDir, '.session-model-kept-notified'), { force: true })
|
|
147
|
-
rmSync(join(agentDir, SESSION_MODEL_BOOT_ATTEMPTS_FILE), { force: true })
|
|
148
|
-
} catch {
|
|
149
|
-
/* best-effort */
|
|
150
|
-
}
|
|
151
|
-
}
|
|
152
|
-
|
|
153
|
-
/**
|
|
154
|
-
* #3043 item 2: clear ONLY the crashloop boot-attempts counter — a positive
|
|
155
|
-
* health signal from the gateway, not a carrier change.
|
|
156
|
-
*
|
|
157
|
-
* start.sh's self-heal (start.sh.hbs "Override crashloop self-heal") increments
|
|
158
|
-
* the counter on every boot that re-enters within 150s with the override still
|
|
159
|
-
* active, and clears a healthy override after 3 fast boots. That window can't
|
|
160
|
-
* tell a genuine crashloop from three quick OPERATOR hand-bounces of a healthy
|
|
161
|
-
* agent — both look like fast successive boots — so three deliberate restarts
|
|
162
|
-
* would spuriously wipe a working override. A boot that reaches bridge
|
|
163
|
-
* registration is proven healthy (the session came all the way up and the
|
|
164
|
-
* bridge connected), so the gateway deletes the counter there. Only boots that
|
|
165
|
-
* genuinely FAIL before the bridge registers now accumulate toward the 3-strike
|
|
166
|
-
* clear. Best-effort; absent file is fine.
|
|
167
|
-
*/
|
|
168
|
-
export function clearSessionModelBootAttempts(agentDir: string): void {
|
|
169
|
-
try {
|
|
170
|
-
rmSync(join(agentDir, SESSION_MODEL_BOOT_ATTEMPTS_FILE), { force: true })
|
|
171
123
|
} catch {
|
|
172
124
|
/* best-effort */
|
|
173
125
|
}
|
|
@@ -186,94 +138,6 @@ export function restoreSessionModelFileRaw(agentDir: string, raw: string | null)
|
|
|
186
138
|
}
|
|
187
139
|
}
|
|
188
140
|
|
|
189
|
-
/**
|
|
190
|
-
* Stamp the one-shot relaunch intent. MUST be called synchronously BEFORE
|
|
191
|
-
* the restart signal/dispatch it describes (write-before-kill invariant —
|
|
192
|
-
* the next start.sh boot reads this to decide keep vs revert). Best-effort:
|
|
193
|
-
* a failed write means the boot falls back to the default (keep, #3039) —
|
|
194
|
-
* the user's choice is preserved either way.
|
|
195
|
-
*/
|
|
196
|
-
export function writeRelaunchModelIntent(
|
|
197
|
-
agentDir: string,
|
|
198
|
-
intent: RelaunchModelIntent,
|
|
199
|
-
reason: string,
|
|
200
|
-
): void {
|
|
201
|
-
try {
|
|
202
|
-
atomicWrite(
|
|
203
|
-
join(agentDir, RELAUNCH_MODEL_INTENT_FILE),
|
|
204
|
-
`${JSON.stringify({ intent, reason, ts: Date.now() })}\n`,
|
|
205
|
-
)
|
|
206
|
-
} catch (err) {
|
|
207
|
-
process.stderr.write(
|
|
208
|
-
`telegram gateway: relaunch-model-intent write failed (boot will revert): ${(err as Error)?.message ?? String(err)}\n`,
|
|
209
|
-
)
|
|
210
|
-
}
|
|
211
|
-
}
|
|
212
|
-
|
|
213
|
-
/**
|
|
214
|
-
* Reason prefix the gateway's SIGTERM/SIGINT shutdown handler stamps on its
|
|
215
|
-
* deploy-survival keep-intent (#3017/#3018). Distinguishable on purpose:
|
|
216
|
-
* a gateway-only bounce (supervisor relaunch, bare gateway-unit restart)
|
|
217
|
-
* leaves that stamp UNCONSUMED on disk — start.sh only runs on a container
|
|
218
|
-
* boot — and the next gateway boot uses this prefix to recognise and clear
|
|
219
|
-
* the stale stamp (see clearStaleGatewayShutdownIntent).
|
|
220
|
-
*/
|
|
221
|
-
export const GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX = 'gateway-shutdown:'
|
|
222
|
-
|
|
223
|
-
export interface RelaunchModelIntentRecord {
|
|
224
|
-
intent: RelaunchModelIntent
|
|
225
|
-
reason: string
|
|
226
|
-
ts: number
|
|
227
|
-
}
|
|
228
|
-
|
|
229
|
-
/** Parsed `.relaunch-model-intent`, or null when absent / corrupt / malformed. */
|
|
230
|
-
export function readRelaunchModelIntent(agentDir: string): RelaunchModelIntentRecord | null {
|
|
231
|
-
try {
|
|
232
|
-
const raw = readFileSync(join(agentDir, RELAUNCH_MODEL_INTENT_FILE), 'utf8')
|
|
233
|
-
const parsed = JSON.parse(raw) as Partial<RelaunchModelIntentRecord>
|
|
234
|
-
if (
|
|
235
|
-
(parsed.intent !== 'keep' && parsed.intent !== 'revert') ||
|
|
236
|
-
typeof parsed.reason !== 'string' ||
|
|
237
|
-
typeof parsed.ts !== 'number'
|
|
238
|
-
) {
|
|
239
|
-
return null
|
|
240
|
-
}
|
|
241
|
-
return { intent: parsed.intent, reason: parsed.reason, ts: parsed.ts }
|
|
242
|
-
} catch {
|
|
243
|
-
return null
|
|
244
|
-
}
|
|
245
|
-
}
|
|
246
|
-
|
|
247
|
-
/**
|
|
248
|
-
* Boot-time cleanup for the gateway-only-bounce hole (#3018 finding 4).
|
|
249
|
-
*
|
|
250
|
-
* A container-level stop/deploy consumes `.relaunch-model-intent` in start.sh
|
|
251
|
-
* BEFORE any gateway boots. So if a freshly-booting GATEWAY still sees an
|
|
252
|
-
* intent that a gateway shutdown handler stamped (reason carries
|
|
253
|
-
* GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX), the preceding bounce was
|
|
254
|
-
* gateway-only — the container never restarted and the stamp is stale.
|
|
255
|
-
* Left in place, it could convert a genuine crash within the 10-min
|
|
256
|
-
* freshness window into a "keep", breaking the crash-reverts policy.
|
|
257
|
-
* Clear it. Never touches a triggerSelfRestart / user-slash stamp (those
|
|
258
|
-
* use their own un-prefixed reasons and precede a container bounce).
|
|
259
|
-
* Returns true when a stale stamp was cleared.
|
|
260
|
-
*/
|
|
261
|
-
export function clearStaleGatewayShutdownIntent(agentDir: string): boolean {
|
|
262
|
-
const rec = readRelaunchModelIntent(agentDir)
|
|
263
|
-
if (rec == null || !rec.reason.startsWith(GATEWAY_SHUTDOWN_INTENT_REASON_PREFIX)) return false
|
|
264
|
-
clearRelaunchModelIntent(agentDir)
|
|
265
|
-
return true
|
|
266
|
-
}
|
|
267
|
-
|
|
268
|
-
/** Remove a stamped intent (rollback of a failed dispatch). Best-effort. */
|
|
269
|
-
export function clearRelaunchModelIntent(agentDir: string): void {
|
|
270
|
-
try {
|
|
271
|
-
rmSync(join(agentDir, RELAUNCH_MODEL_INTENT_FILE), { force: true })
|
|
272
|
-
} catch {
|
|
273
|
-
/* best-effort */
|
|
274
|
-
}
|
|
275
|
-
}
|
|
276
|
-
|
|
277
141
|
/**
|
|
278
142
|
* The resolved configured default start.sh recorded this boot
|
|
279
143
|
* (`.configured-default-model`, written before override resolution — the
|
|
@@ -288,12 +152,13 @@ export function readConfiguredDefaultModel(agentDir: string): string | null {
|
|
|
288
152
|
}
|
|
289
153
|
}
|
|
290
154
|
|
|
291
|
-
// ───
|
|
155
|
+
// ─── Consume-once session-effort carrier (#3186) ────────────────────────────
|
|
292
156
|
//
|
|
293
|
-
// The `/effort` sibling of `.session-model
|
|
294
|
-
//
|
|
295
|
-
//
|
|
296
|
-
//
|
|
157
|
+
// The `/effort` sibling of `.session-model`, session-scoped. Written ONLY by
|
|
158
|
+
// the queued-command shutdown persist (a mid-turn /effort carried across the
|
|
159
|
+
// bounce); start.sh applies it on the single boot that reads it (`--effort
|
|
160
|
+
// <level>`) and deletes it. Cleared live by `/effort default`; invalidation
|
|
161
|
+
// (configured `thinking_effort:` changed / corrupt file) alerts once at boot.
|
|
297
162
|
|
|
298
163
|
export const SESSION_EFFORT_FILE = '.session-effort'
|
|
299
164
|
|
|
@@ -330,8 +195,9 @@ export function parseSessionEffort(text: string): SessionEffortRecord | null {
|
|
|
330
195
|
}
|
|
331
196
|
|
|
332
197
|
/**
|
|
333
|
-
* Write the
|
|
334
|
-
* the value is passed verbatim to `claude --effort` at the next boot
|
|
198
|
+
* Write the consume-once effort carrier. Throws on a non-allowlisted level —
|
|
199
|
+
* the value is passed verbatim to `claude --effort` at the next boot (the
|
|
200
|
+
* apply-boot, which consumes the file).
|
|
335
201
|
*/
|
|
336
202
|
export function writeSessionEffortFile(
|
|
337
203
|
agentDir: string,
|
|
@@ -347,7 +213,7 @@ export function writeSessionEffortFile(
|
|
|
347
213
|
)
|
|
348
214
|
}
|
|
349
215
|
|
|
350
|
-
/** Parsed
|
|
216
|
+
/** Parsed effort carrier, or null when absent/corrupt. */
|
|
351
217
|
export function readSessionEffortFile(agentDir: string): SessionEffortRecord | null {
|
|
352
218
|
try {
|
|
353
219
|
return parseSessionEffort(readFileSync(join(agentDir, SESSION_EFFORT_FILE), 'utf8'))
|
|
@@ -356,7 +222,7 @@ export function readSessionEffortFile(agentDir: string): SessionEffortRecord | n
|
|
|
356
222
|
}
|
|
357
223
|
}
|
|
358
224
|
|
|
359
|
-
/** Delete the
|
|
225
|
+
/** Delete the effort carrier (`/effort default`, leftover hygiene). Best-effort. */
|
|
360
226
|
export function clearSessionEffortFile(agentDir: string): void {
|
|
361
227
|
try {
|
|
362
228
|
rmSync(join(agentDir, SESSION_EFFORT_FILE), { force: true })
|