talon-agent 5.4.0 → 5.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +2 -0
  2. package/package.json +1 -1
  3. package/prompts/identity.md +2 -0
  4. package/prompts/telegram.md +2 -1
  5. package/src/app.ts +12 -0
  6. package/src/backend/runtime/turn/turn-phases.ts +5 -0
  7. package/src/bootstrap.ts +55 -24
  8. package/src/core/agents/runner.ts +50 -2
  9. package/src/core/agents/types.ts +5 -0
  10. package/src/core/background/cron/job-oneshot.ts +2 -0
  11. package/src/core/background/cron/scheduler.ts +45 -2
  12. package/src/core/background/heartbeat/agent.ts +117 -20
  13. package/src/core/background/heartbeat/state.ts +11 -0
  14. package/src/core/config/index.ts +38 -0
  15. package/src/core/daemon/respawn.ts +34 -10
  16. package/src/core/daemon/signals.ts +111 -0
  17. package/src/core/engine/backend-controller/index.ts +1 -0
  18. package/src/core/engine/backend-controller/pool.ts +11 -0
  19. package/src/core/engine/backend-router/headroom.ts +267 -0
  20. package/src/core/engine/backend-router/index.ts +52 -0
  21. package/src/core/engine/backend-router/ledger.ts +248 -0
  22. package/src/core/engine/backend-router/router.ts +322 -0
  23. package/src/core/engine/backend-router/usage.ts +80 -0
  24. package/src/core/engine/gateway-actions/agents/control.ts +2 -1
  25. package/src/core/engine/gateway-actions/models.ts +74 -37
  26. package/src/core/tools/ops/models.ts +2 -2
  27. package/src/frontend/presentation/plan-usage-report.ts +29 -38
  28. package/src/frontend/presentation/reports.ts +5 -1
  29. package/src/frontend/telegram/formatting.ts +39 -0
  30. package/src/frontend/terminal/builtins/status.ts +29 -1
  31. package/src/index.ts +7 -0
  32. package/src/util/log.ts +1 -0
@@ -0,0 +1,248 @@
1
+ /**
2
+ * Local rolling token ledger — the headroom signal for backends with no
3
+ * account usage API.
4
+ *
5
+ * Claude and Codex report subscription windows; `agy` has no account
6
+ * endpoint and `openai-agents` has no plan at all. Without a second signal
7
+ * the router would treat those as infinitely fresh and pile every background
8
+ * run onto them. So Talon counts what it spends itself: every chat turn,
9
+ * one-shot and sub-agent folds its token total into a per-backend ledger,
10
+ * and `headroom.ts` reads that against the operator's soft budget
11
+ * (`config.backendBudgets`).
12
+ *
13
+ * The ledger is deliberately a *local estimate*, not accounting: it only
14
+ * sees what this daemon ran, and a plan API always takes precedence when
15
+ * one exists. Entries older than the widest window (24h) are pruned on
16
+ * every read and write, which also bounds the file.
17
+ *
18
+ * Persistence is `~/.talon/data/backend-ledger.json`, written atomically so
19
+ * a restart mid-write can't leave a truncated file — a zeroed ledger would
20
+ * silently hand a spent backend a full headroom score.
21
+ */
22
+
23
+ import { readFile } from "node:fs/promises";
24
+ import { resolve } from "node:path";
25
+ import { dirs } from "../../../util/paths.js";
26
+ import { logWarn } from "../../../util/log.js";
27
+ import { writePrivateJson } from "../../mesh/persist.js";
28
+
29
+ /** Widest window the ledger answers for; everything older is dropped. */
30
+ export const LEDGER_RETENTION_MS = 24 * 60 * 60_000;
31
+ /** The short window, as `backendBudgets.tokensPer5h` measures it. */
32
+ export const LEDGER_SHORT_WINDOW_MS = 5 * 60 * 60_000;
33
+
34
+ /** How long a write is coalesced for, so a burst of turns is one fsync. */
35
+ const FLUSH_DEBOUNCE_MS = 2_000;
36
+
37
+ /** One recorded spend: when it happened and how many tokens it cost. */
38
+ interface LedgerEntry {
39
+ /** Epoch ms. */
40
+ readonly t: number;
41
+ /** Total tokens (input + output + cache + thinking, as the backend counts). */
42
+ readonly n: number;
43
+ }
44
+
45
+ interface LedgerFile {
46
+ readonly version: 1;
47
+ readonly backends: Record<string, LedgerEntry[]>;
48
+ }
49
+
50
+ interface LedgerState {
51
+ entries: Map<string, LedgerEntry[]>;
52
+ loaded: boolean;
53
+ loading: Promise<void> | null;
54
+ flushTimer: ReturnType<typeof setTimeout> | null;
55
+ dirty: boolean;
56
+ /** Overridable for tests; resolved lazily so TALON_HOME changes are seen. */
57
+ path: string | null;
58
+ }
59
+
60
+ const state: LedgerState = {
61
+ entries: new Map(),
62
+ loaded: false,
63
+ loading: null,
64
+ flushTimer: null,
65
+ dirty: false,
66
+ path: null,
67
+ };
68
+
69
+ function ledgerPath(): string {
70
+ return state.path ?? resolve(dirs.data, "backend-ledger.json");
71
+ }
72
+
73
+ /** Drop entries that have aged out of the widest window. */
74
+ function prune(list: LedgerEntry[], now: number): LedgerEntry[] {
75
+ const floor = now - LEDGER_RETENTION_MS;
76
+ // Entries are appended in time order, so the survivors are a suffix —
77
+ // but a clock step backwards can break that, hence a filter not a slice.
78
+ return list.filter((e) => e.t > floor);
79
+ }
80
+
81
+ function parseLedger(raw: unknown): Map<string, LedgerEntry[]> {
82
+ const out = new Map<string, LedgerEntry[]>();
83
+ if (!raw || typeof raw !== "object") return out;
84
+ const file = raw as Partial<LedgerFile>;
85
+ if (file.version !== 1 || !file.backends) return out;
86
+ const now = Date.now();
87
+ for (const [id, list] of Object.entries(file.backends)) {
88
+ if (!Array.isArray(list)) continue;
89
+ const clean = list.filter(
90
+ (e): e is LedgerEntry =>
91
+ Boolean(e) &&
92
+ typeof (e as LedgerEntry).t === "number" &&
93
+ typeof (e as LedgerEntry).n === "number" &&
94
+ Number.isFinite((e as LedgerEntry).t) &&
95
+ Number.isFinite((e as LedgerEntry).n),
96
+ );
97
+ const pruned = prune(clean, now);
98
+ if (pruned.length > 0) out.set(id, pruned);
99
+ }
100
+ return out;
101
+ }
102
+
103
+ /**
104
+ * Load the persisted ledger once per process. Idempotent and safe to call
105
+ * concurrently — every caller awaits the same read. A missing or corrupt
106
+ * file starts an empty ledger rather than failing a routing decision.
107
+ */
108
+ export async function loadBackendLedger(): Promise<void> {
109
+ if (state.loaded) return;
110
+ if (state.loading) return state.loading;
111
+ state.loading = (async () => {
112
+ try {
113
+ const raw = await readFile(ledgerPath(), "utf8");
114
+ const parsed = parseLedger(JSON.parse(raw));
115
+ // In-process records taken while the read was in flight win: merge
116
+ // rather than replace, so a turn that landed during boot isn't lost.
117
+ for (const [id, list] of parsed) {
118
+ state.entries.set(id, [...list, ...(state.entries.get(id) ?? [])]);
119
+ }
120
+ } catch {
121
+ /* no ledger yet, or unreadable — start empty */
122
+ }
123
+ state.loaded = true;
124
+ state.loading = null;
125
+ })();
126
+ return state.loading;
127
+ }
128
+
129
+ function scheduleFlush(): void {
130
+ state.dirty = true;
131
+ if (state.flushTimer) return;
132
+ const timer = setTimeout(() => {
133
+ state.flushTimer = null;
134
+ void flushBackendLedger();
135
+ }, FLUSH_DEBOUNCE_MS);
136
+ timer.unref?.();
137
+ state.flushTimer = timer;
138
+ }
139
+
140
+ /** Write the ledger out now. Exported so shutdown and tests can force it. */
141
+ export async function flushBackendLedger(): Promise<void> {
142
+ if (!state.dirty) return;
143
+ state.dirty = false;
144
+ const now = Date.now();
145
+ const backends: Record<string, LedgerEntry[]> = {};
146
+ for (const [id, list] of state.entries) {
147
+ const pruned = prune(list, now);
148
+ state.entries.set(id, pruned);
149
+ if (pruned.length > 0) backends[id] = pruned;
150
+ }
151
+ try {
152
+ await writePrivateJson(ledgerPath(), {
153
+ version: 1,
154
+ backends,
155
+ } satisfies LedgerFile);
156
+ } catch (err) {
157
+ logWarn(
158
+ "router",
159
+ `Could not persist the backend ledger: ${err instanceof Error ? err.message : String(err)}`,
160
+ );
161
+ }
162
+ }
163
+
164
+ /**
165
+ * Fold one run's token spend into a backend's ledger.
166
+ *
167
+ * Called from the shared turn accounting and from every one-shot completion
168
+ * path, so *all* backends accumulate a ledger — the plan API simply wins over
169
+ * it where one exists. Cheap and synchronous: the disk write is debounced.
170
+ */
171
+ export function recordBackendUsage(
172
+ backendId: string,
173
+ tokens: number,
174
+ at = Date.now(),
175
+ ): void {
176
+ if (!backendId) return;
177
+ if (!Number.isFinite(tokens) || tokens <= 0) return;
178
+ // A first record before the file has been read would be overwritten by the
179
+ // load; kick the read off here so the merge in loadBackendLedger keeps it.
180
+ if (!state.loaded && !state.loading) void loadBackendLedger();
181
+ const list = state.entries.get(backendId) ?? [];
182
+ list.push({ t: at, n: Math.round(tokens) });
183
+ state.entries.set(backendId, prune(list, at));
184
+ scheduleFlush();
185
+ }
186
+
187
+ /**
188
+ * Fold a completed run's usage into the ledger. The same four fields every
189
+ * background path reports (`OneShotUsage`, `TaskUsage`), summed — cache
190
+ * reads included, because they still count against a subscription window.
191
+ */
192
+ export function recordBackendRunUsage(
193
+ backendId: string,
194
+ usage:
195
+ | {
196
+ inputTokens?: number;
197
+ outputTokens?: number;
198
+ cacheRead?: number;
199
+ cacheWrite?: number;
200
+ }
201
+ | undefined
202
+ | null,
203
+ at = Date.now(),
204
+ ): void {
205
+ if (!usage) return;
206
+ const total =
207
+ (usage.inputTokens ?? 0) +
208
+ (usage.outputTokens ?? 0) +
209
+ (usage.cacheRead ?? 0) +
210
+ (usage.cacheWrite ?? 0);
211
+ recordBackendUsage(backendId, total, at);
212
+ }
213
+
214
+ /** Tokens a backend spent inside a window ending now. */
215
+ export function tokensInWindow(
216
+ backendId: string,
217
+ windowMs: number,
218
+ now = Date.now(),
219
+ ): number {
220
+ const list = state.entries.get(backendId);
221
+ if (!list || list.length === 0) return 0;
222
+ const floor = now - windowMs;
223
+ let total = 0;
224
+ for (const entry of list) if (entry.t > floor) total += entry.n;
225
+ return total;
226
+ }
227
+
228
+ /** Both windows the budget schema knows about, for one backend. */
229
+ export function ledgerUsage(
230
+ backendId: string,
231
+ now = Date.now(),
232
+ ): { tokens5h: number; tokensDay: number } {
233
+ return {
234
+ tokens5h: tokensInWindow(backendId, LEDGER_SHORT_WINDOW_MS, now),
235
+ tokensDay: tokensInWindow(backendId, LEDGER_RETENTION_MS, now),
236
+ };
237
+ }
238
+
239
+ /** Test seam — point the ledger at a temp file and start from empty. */
240
+ export function resetBackendLedgerForTest(path?: string): void {
241
+ if (state.flushTimer) clearTimeout(state.flushTimer);
242
+ state.entries = new Map();
243
+ state.loaded = false;
244
+ state.loading = null;
245
+ state.flushTimer = null;
246
+ state.dirty = false;
247
+ state.path = path ?? null;
248
+ }
@@ -0,0 +1,322 @@
1
+ /**
2
+ * Plan-aware backend routing for background work.
3
+ *
4
+ * Background work — `spawn_agent` sub-agents, cron `query` jobs, the
5
+ * heartbeat — used to inherit whichever backend the chat happened to be on.
6
+ * On a multi-subscription install that is a good way to burn one plan to its
7
+ * ceiling while another sits idle. When nothing is pinned, this picks the
8
+ * backend with the most headroom instead.
9
+ *
10
+ * The rules, in order:
11
+ *
12
+ * 1. An explicit backend (or a model, which pins its backend) always wins.
13
+ * Routing is what happens in the *absence* of a choice, never over one.
14
+ * 2. `config.router.enabled: false` returns the caller's own backend —
15
+ * byte-identical to pre-router Talon.
16
+ * 3. Otherwise: rank the candidates by headroom, skip anyone at or above
17
+ * `ceilingPercent`, and take the top one.
18
+ *
19
+ * Only backends that can actually host an isolated run are candidates, and
20
+ * routing never boots a cold provider to find out: a backend qualifies when
21
+ * it is pooled with a `background` slot, or when the operator gave it a local
22
+ * budget (`backendBudgets`), which is both an opt-in and the only way it
23
+ * would have a headroom signal. The caller's own backend is always in the
24
+ * running, so there is always an answer.
25
+ */
26
+
27
+ import type { TalonConfig } from "../../config/index.js";
28
+ import type { ReasoningEffortLevel } from "../../types.js";
29
+ import { log } from "../../../util/log.js";
30
+ import {
31
+ acquireBackendInstance,
32
+ getPoolConfig,
33
+ getPooledBackend,
34
+ listAvailableBackends,
35
+ } from "../backend-controller/index.js";
36
+ import {
37
+ formatHeadroom,
38
+ getBackendHeadroom,
39
+ hasBudget,
40
+ type BackendHeadroom,
41
+ } from "./headroom.js";
42
+
43
+ /** Default for `config.router.ceilingPercent`. */
44
+ export const DEFAULT_CEILING_PERCENT = 85;
45
+
46
+ /** Which background subsystem is asking. Logged, and nothing else — yet. */
47
+ export type RoutePurpose = "subagent" | "cron" | "heartbeat";
48
+
49
+ /**
50
+ * A coarse shape-of-work hint. It never outranks headroom: a class either
51
+ * *vetoes* backends that cannot do the job at all, or breaks a tie.
52
+ */
53
+ export type TaskClass = "coding" | "mechanical" | "reasoning";
54
+
55
+ /**
56
+ * The one place the task-class opinions live. `require` is a veto (applied
57
+ * only when at least one required backend is a candidate, so a deployment
58
+ * without it still gets an answer); `prefer` is tie-break order.
59
+ *
60
+ * - coding — agentic edit/test loops; Codex and Claude are the two
61
+ * with real harnesses behind them.
62
+ * - mechanical — sweeps and reformatting; cheapest first (agy's flash
63
+ * tier, then Claude's haiku tier).
64
+ * - reasoning — deep thinking (also where `xhigh` effort lands); Claude
65
+ * is the only backend Talon drives an opus-class model on.
66
+ */
67
+ const TASK_CLASS_RULES: Record<
68
+ TaskClass,
69
+ { readonly prefer: readonly string[]; readonly require?: readonly string[] }
70
+ > = {
71
+ coding: { prefer: ["codex", "claude"] },
72
+ mechanical: { prefer: ["agy", "claude"] },
73
+ reasoning: { prefer: ["claude"], require: ["claude"] },
74
+ };
75
+
76
+ /** Ranking priority of a headroom source — measured beats unmeasured. */
77
+ const SOURCE_RANK: Record<BackendHeadroom["source"], number> = {
78
+ plan: 0,
79
+ ledger: 1,
80
+ none: 2,
81
+ };
82
+
83
+ export interface RouteHints {
84
+ readonly taskClass?: TaskClass;
85
+ readonly effort?: ReasoningEffortLevel;
86
+ }
87
+
88
+ export interface RouteRequest {
89
+ readonly purpose: RoutePurpose;
90
+ /** An explicit backend from the caller. Wins outright. */
91
+ readonly requestedBackendId?: string;
92
+ /** An explicit model. Pins whatever backend is serving it. */
93
+ readonly requestedModel?: string;
94
+ /** The backend the caller would have used before routing existed. */
95
+ readonly chatBackendId: string;
96
+ /** Defaults to the config the backend pool was initialised with. */
97
+ readonly config?: TalonConfig;
98
+ readonly hints?: RouteHints;
99
+ }
100
+
101
+ export interface RouteDecision {
102
+ readonly backendId: string;
103
+ /** Only set when the caller pinned one — model choice stays downstream. */
104
+ readonly model?: string;
105
+ /** Human-readable: `pinned`, `disabled`, or why this backend won. */
106
+ readonly reason: string;
107
+ /** True when headroom actually chose this, rather than a pin or a default. */
108
+ readonly routed: boolean;
109
+ /** What the decision saw, for the spawn reply and the log line. */
110
+ readonly headroom?: BackendHeadroom;
111
+ }
112
+
113
+ /** The task class a reasoning-effort level implies, if any. */
114
+ export function taskClassForEffort(
115
+ effort: ReasoningEffortLevel | undefined,
116
+ ): TaskClass | undefined {
117
+ return effort === "xhigh" || effort === "high" ? "reasoning" : undefined;
118
+ }
119
+
120
+ /** `config.router`, with the documented defaults filled in. */
121
+ function routerSettings(config: TalonConfig | undefined): {
122
+ enabled: boolean;
123
+ ceilingPercent: number;
124
+ } {
125
+ return {
126
+ enabled: config?.router?.enabled ?? true,
127
+ ceilingPercent: config?.router?.ceilingPercent ?? DEFAULT_CEILING_PERCENT,
128
+ };
129
+ }
130
+
131
+ /**
132
+ * Can this backend host an isolated run, without booting it to find out?
133
+ * A pooled instance answers from its capability slots; a cold one qualifies
134
+ * only on an explicit local budget (see the module comment).
135
+ */
136
+ function isCandidate(
137
+ id: string,
138
+ config: TalonConfig | undefined,
139
+ chatBackendId: string,
140
+ ): boolean {
141
+ if (id === chatBackendId) return true;
142
+ const pooled = getPooledBackend(id);
143
+ if (pooled) return Boolean(pooled.background);
144
+ return hasBudget(config, id);
145
+ }
146
+
147
+ /** Percent of the tightest window — what the ceiling is applied to. */
148
+ function limitingPercent(entry: BackendHeadroom): number {
149
+ return entry.limiting?.percent ?? 0;
150
+ }
151
+
152
+ /** Comparator: headroom first, then the documented tie-breaks. */
153
+ function rank(
154
+ a: BackendHeadroom,
155
+ b: BackendHeadroom,
156
+ chatBackendId: string,
157
+ prefer: readonly string[],
158
+ ): number {
159
+ // Headroom to 3dp: two backends a thousandth apart are a tie, not a winner.
160
+ const byHeadroom =
161
+ Math.round(b.headroom * 1000) - Math.round(a.headroom * 1000);
162
+ if (byHeadroom !== 0) return byHeadroom;
163
+ const bySource = SOURCE_RANK[a.source] - SOURCE_RANK[b.source];
164
+ if (bySource !== 0) return bySource;
165
+ const preferIndex = (id: string): number => {
166
+ const i = prefer.indexOf(id);
167
+ return i === -1 ? prefer.length : i;
168
+ };
169
+ const byPrefer = preferIndex(a.id) - preferIndex(b.id);
170
+ if (byPrefer !== 0) return byPrefer;
171
+ // Cache warmth: a sub-agent is isolated, but the caller's backend is
172
+ // already up and its MCP servers are already spawned.
173
+ if (a.id === chatBackendId) return -1;
174
+ if (b.id === chatBackendId) return 1;
175
+ return a.id.localeCompare(b.id);
176
+ }
177
+
178
+ /**
179
+ * Apply a task class's hard requirement, when a candidate still satisfies it.
180
+ * Runs on the post-ceiling set, so a spent required backend is already gone.
181
+ */
182
+ function applyVeto(
183
+ entries: BackendHeadroom[],
184
+ required: readonly string[] | undefined,
185
+ ): BackendHeadroom[] {
186
+ if (!required || required.length === 0) return entries;
187
+ const kept = entries.filter((e) => required.includes(e.id));
188
+ return kept.length > 0 ? kept : entries;
189
+ }
190
+
191
+ /** Drop candidates at or above the ceiling — unless that drops all of them. */
192
+ function applyCeiling(
193
+ entries: BackendHeadroom[],
194
+ ceilingPercent: number,
195
+ ): { kept: BackendHeadroom[]; allOverCeiling: boolean } {
196
+ const kept = entries.filter((e) => limitingPercent(e) < ceilingPercent);
197
+ if (kept.length > 0) return { kept, allOverCeiling: false };
198
+ // Everything is spent. Running the least-bad one beats running nothing:
199
+ // the background subsystems have no queue to defer into.
200
+ return { kept: entries, allOverCeiling: true };
201
+ }
202
+
203
+ function decisionFor(
204
+ winner: BackendHeadroom,
205
+ allOverCeiling: boolean,
206
+ ): RouteDecision {
207
+ const reason = allOverCeiling
208
+ ? `every backend is over the ceiling — least spent: ${formatHeadroom(winner)}`
209
+ : `most headroom ${Math.round(winner.headroom * 100)}%`;
210
+ return {
211
+ backendId: winner.id,
212
+ reason,
213
+ routed: true,
214
+ headroom: winner,
215
+ };
216
+ }
217
+
218
+ function logDecision(
219
+ request: RouteRequest,
220
+ decision: RouteDecision,
221
+ candidates: BackendHeadroom[],
222
+ ): void {
223
+ const seen = candidates.map((c) => `${c.id}=${formatHeadroom(c)}`).join(", ");
224
+ log(
225
+ "router",
226
+ `${request.purpose}: → ${decision.backendId} (${decision.reason})` +
227
+ (seen ? ` | candidates: ${seen}` : ""),
228
+ );
229
+ }
230
+
231
+ /**
232
+ * Choose the backend a piece of background work should run on.
233
+ *
234
+ * Never throws and never returns nothing: every path ends at a real backend
235
+ * id, falling back to the caller's own.
236
+ */
237
+ export async function chooseBackend(
238
+ request: RouteRequest,
239
+ ): Promise<RouteDecision> {
240
+ const { chatBackendId, purpose } = request;
241
+
242
+ if (request.requestedBackendId || request.requestedModel) {
243
+ const decision: RouteDecision = {
244
+ backendId: request.requestedBackendId ?? chatBackendId,
245
+ reason: "pinned",
246
+ routed: false,
247
+ ...(request.requestedModel ? { model: request.requestedModel } : {}),
248
+ };
249
+ log("router", `${purpose}: → ${decision.backendId} (pinned)`);
250
+ return decision;
251
+ }
252
+
253
+ const config = request.config ?? getPoolConfig() ?? undefined;
254
+ const settings = routerSettings(config);
255
+ if (!settings.enabled) {
256
+ return { backendId: chatBackendId, reason: "disabled", routed: false };
257
+ }
258
+
259
+ const ids = listAvailableBackends(config)
260
+ .filter(({ id }) => isCandidate(id, config, chatBackendId))
261
+ .map(({ id, label }) => ({ id, label }));
262
+ if (ids.length === 0) {
263
+ return { backendId: chatBackendId, reason: "no candidates", routed: false };
264
+ }
265
+
266
+ const measured = await Promise.all(
267
+ ids.map(({ id, label }) => getBackendHeadroom(id, label, config)),
268
+ );
269
+
270
+ const rules = request.hints?.taskClass
271
+ ? TASK_CLASS_RULES[request.hints.taskClass]
272
+ : undefined;
273
+ // Ceiling first, THEN the veto: a hard task-class requirement must not be
274
+ // able to send work to a backend that is out of plan. A lesser model that
275
+ // runs beats the right one that rate-limits.
276
+ const { kept, allOverCeiling } = applyCeiling(
277
+ measured,
278
+ settings.ceilingPercent,
279
+ );
280
+ const eligible = applyVeto(kept, rules?.require);
281
+ const ordered = [...eligible].sort((a, b) =>
282
+ rank(a, b, chatBackendId, rules?.prefer ?? []),
283
+ );
284
+ const winner = ordered[0];
285
+ if (!winner) {
286
+ return { backendId: chatBackendId, reason: "no candidates", routed: false };
287
+ }
288
+
289
+ const decision = decisionFor(winner, allOverCeiling);
290
+ logDecision(request, decision, measured);
291
+ return decision;
292
+ }
293
+
294
+ /**
295
+ * The default model for a backend the router just picked.
296
+ *
297
+ * A routed run cannot carry the caller's model across — a model id is
298
+ * backend-specific. Config's `backendDefaults` wins (it is the operator's
299
+ * answer to "what should this provider run"), then the backend's own
300
+ * canonical default. Resolves `null` when neither exists, which the call
301
+ * sites read as "stay where you were".
302
+ */
303
+ export async function resolveRoutedModel(
304
+ backendId: string,
305
+ config?: TalonConfig,
306
+ ): Promise<string | null> {
307
+ const settings = config ?? getPoolConfig() ?? undefined;
308
+ const configured = settings?.backendDefaults?.[backendId];
309
+ if (configured) return configured;
310
+ try {
311
+ const pooled = getPooledBackend(backendId);
312
+ if (pooled) return (await pooled.models?.getDefaultModelId()) ?? null;
313
+ const acquired = await acquireBackendInstance(backendId);
314
+ try {
315
+ return (await acquired.backend.models?.getDefaultModelId()) ?? null;
316
+ } finally {
317
+ await acquired.release();
318
+ }
319
+ } catch {
320
+ return null;
321
+ }
322
+ }
@@ -0,0 +1,80 @@
1
+ /**
2
+ * One usage snapshot per backend — what `/usage`, the `plan_usage` tool and
3
+ * `list_backends` all read.
4
+ *
5
+ * It lives in core rather than in the frontend presentation layer because
6
+ * the gateway tools need it too, and core cannot import a frontend. The
7
+ * split is: this module *gathers* (plan windows, headroom, and the reason
8
+ * there is nothing to show), renderers *format*.
9
+ *
10
+ * Nothing here boots a backend. Reading a number should never cost a
11
+ * subprocess per idle provider, so a backend that isn't running is listed
12
+ * with a reason instead of being woken or omitted.
13
+ */
14
+
15
+ import type { PlanUsage } from "../../agent-runtime/capabilities.js";
16
+ import type { TalonConfig } from "../../config/index.js";
17
+ import {
18
+ getPooledBackend,
19
+ listAvailableBackends,
20
+ } from "../backend-controller/index.js";
21
+ import { getBackendHeadroom, type BackendHeadroom } from "./headroom.js";
22
+
23
+ export interface BackendUsageSnapshot {
24
+ readonly id: string;
25
+ readonly label: string;
26
+ /** Raw plan windows, when this backend reported any. */
27
+ readonly plan?: PlanUsage;
28
+ /** Always present — every backend gets a comparable headroom figure. */
29
+ readonly headroom: BackendHeadroom;
30
+ /** Why there is no `plan`. Absent when there is one. */
31
+ readonly note?: string;
32
+ }
33
+
34
+ /** The reason a backend has no plan windows to show. */
35
+ function noteFor(id: string, headroom: BackendHeadroom): string {
36
+ const backend = getPooledBackend(id);
37
+ if (!backend) return "not running";
38
+ if (!backend.usage?.getPlanUsage) return "no plan limits on this backend";
39
+ if (headroom.source === "ledger") return "tracked against a local budget";
40
+ return "no usage information available";
41
+ }
42
+
43
+ /**
44
+ * Every exposed backend, in config order, with its plan (where it has one)
45
+ * and its headroom (always).
46
+ *
47
+ * `force` skips the 60s headroom cache — a person looking at `/usage` wants
48
+ * the number now, where a routing decision a second after another one does
49
+ * not.
50
+ */
51
+ export async function collectBackendUsage(
52
+ config: TalonConfig | undefined,
53
+ options?: { force?: boolean },
54
+ ): Promise<BackendUsageSnapshot[]> {
55
+ const backends = listAvailableBackends(config);
56
+ return Promise.all(
57
+ backends.map(async ({ id, label }) => {
58
+ const headroom = await getBackendHeadroom(id, label, config, options);
59
+ if (headroom.plan && headroom.plan.windows.length > 0) {
60
+ return { id, label, headroom, plan: headroom.plan };
61
+ }
62
+ return { id, label, headroom, note: noteFor(id, headroom) };
63
+ }),
64
+ );
65
+ }
66
+
67
+ /**
68
+ * Reorder a snapshot list so one backend comes first, everything else
69
+ * keeping config order. The `plan_usage` tool leads with the chat's own
70
+ * backend so callers that only read the first entry see what they used to.
71
+ */
72
+ export function leadWith(
73
+ entries: BackendUsageSnapshot[],
74
+ id: string,
75
+ ): BackendUsageSnapshot[] {
76
+ const index = entries.findIndex((e) => e.id === id);
77
+ if (index <= 0) return entries;
78
+ const lead = entries[index] as BackendUsageSnapshot;
79
+ return [lead, ...entries.filter((_, i) => i !== index)];
80
+ }
@@ -208,7 +208,8 @@ export const agentControlHandlers: SharedActionHandlers = {
208
208
  ok: true,
209
209
  text:
210
210
  `Spawned agent "${parsed.label}" (id: ${outcome.agentId})\n` +
211
- `Backend: ${outcome.backendId}/${outcome.model}\n` +
211
+ `Backend: ${outcome.backendId}/${outcome.model}` +
212
+ `${outcome.routing ? ` (routed: ${outcome.routing})` : ""}\n` +
212
213
  `Timeout: ${timeoutS}s\n` +
213
214
  `It runs in the background. You will be woken with its report — ` +
214
215
  `carry on with what you were doing.`,