talon-agent 5.4.1 → 5.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/package.json +2 -2
- package/prompts/identity.md +2 -0
- package/prompts/telegram.md +2 -1
- package/src/app.ts +8 -0
- package/src/backend/runtime/turn/turn-phases.ts +5 -0
- package/src/bootstrap.ts +55 -24
- package/src/core/agents/runner.ts +50 -2
- package/src/core/agents/types.ts +5 -0
- package/src/core/background/cron/job-oneshot.ts +2 -0
- package/src/core/background/cron/scheduler.ts +45 -2
- package/src/core/background/heartbeat/agent.ts +117 -20
- package/src/core/background/heartbeat/state.ts +11 -0
- package/src/core/config/index.ts +38 -0
- package/src/core/engine/backend-controller/index.ts +1 -0
- package/src/core/engine/backend-controller/pool.ts +11 -0
- package/src/core/engine/backend-router/headroom.ts +267 -0
- package/src/core/engine/backend-router/index.ts +52 -0
- package/src/core/engine/backend-router/ledger.ts +248 -0
- package/src/core/engine/backend-router/router.ts +322 -0
- package/src/core/engine/backend-router/usage.ts +80 -0
- package/src/core/engine/gateway-actions/agents/control.ts +2 -1
- package/src/core/engine/gateway-actions/models.ts +74 -37
- package/src/core/tools/ops/models.ts +2 -2
- package/src/frontend/presentation/plan-usage-report.ts +29 -38
- package/src/frontend/presentation/reports.ts +5 -1
- package/src/frontend/telegram/formatting.ts +39 -0
- package/src/frontend/telegram/handlers/access.ts +41 -0
- package/src/frontend/telegram/handlers/index.ts +1 -0
- package/src/frontend/telegram/index.ts +7 -1
- package/src/frontend/terminal/builtins/status.ts +29 -1
- package/src/util/log.ts +1 -0
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Plan-aware backend routing for background work.
|
|
3
|
+
*
|
|
4
|
+
* Background work — `spawn_agent` sub-agents, cron `query` jobs, the
|
|
5
|
+
* heartbeat — used to inherit whichever backend the chat happened to be on.
|
|
6
|
+
* On a multi-subscription install that is a good way to burn one plan to its
|
|
7
|
+
* ceiling while another sits idle. When nothing is pinned, this picks the
|
|
8
|
+
* backend with the most headroom instead.
|
|
9
|
+
*
|
|
10
|
+
* The rules, in order:
|
|
11
|
+
*
|
|
12
|
+
* 1. An explicit backend (or a model, which pins its backend) always wins.
|
|
13
|
+
* Routing is what happens in the *absence* of a choice, never over one.
|
|
14
|
+
* 2. `config.router.enabled: false` returns the caller's own backend —
|
|
15
|
+
* byte-identical to pre-router Talon.
|
|
16
|
+
* 3. Otherwise: rank the candidates by headroom, skip anyone at or above
|
|
17
|
+
* `ceilingPercent`, and take the top one.
|
|
18
|
+
*
|
|
19
|
+
* Only backends that can actually host an isolated run are candidates, and
|
|
20
|
+
* routing never boots a cold provider to find out: a backend qualifies when
|
|
21
|
+
* it is pooled with a `background` slot, or when the operator gave it a local
|
|
22
|
+
* budget (`backendBudgets`), which is both an opt-in and the only way it
|
|
23
|
+
* would have a headroom signal. The caller's own backend is always in the
|
|
24
|
+
* running, so there is always an answer.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import type { TalonConfig } from "../../config/index.js";
|
|
28
|
+
import type { ReasoningEffortLevel } from "../../types.js";
|
|
29
|
+
import { log } from "../../../util/log.js";
|
|
30
|
+
import {
|
|
31
|
+
acquireBackendInstance,
|
|
32
|
+
getPoolConfig,
|
|
33
|
+
getPooledBackend,
|
|
34
|
+
listAvailableBackends,
|
|
35
|
+
} from "../backend-controller/index.js";
|
|
36
|
+
import {
|
|
37
|
+
formatHeadroom,
|
|
38
|
+
getBackendHeadroom,
|
|
39
|
+
hasBudget,
|
|
40
|
+
type BackendHeadroom,
|
|
41
|
+
} from "./headroom.js";
|
|
42
|
+
|
|
43
|
+
/** Default for `config.router.ceilingPercent`. */
|
|
44
|
+
export const DEFAULT_CEILING_PERCENT = 85;
|
|
45
|
+
|
|
46
|
+
/** Which background subsystem is asking. Logged, and nothing else — yet. */
|
|
47
|
+
export type RoutePurpose = "subagent" | "cron" | "heartbeat";
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* A coarse shape-of-work hint. It never outranks headroom: a class either
|
|
51
|
+
* *vetoes* backends that cannot do the job at all, or breaks a tie.
|
|
52
|
+
*/
|
|
53
|
+
export type TaskClass = "coding" | "mechanical" | "reasoning";
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* The one place the task-class opinions live. `require` is a veto (applied
|
|
57
|
+
* only when at least one required backend is a candidate, so a deployment
|
|
58
|
+
* without it still gets an answer); `prefer` is tie-break order.
|
|
59
|
+
*
|
|
60
|
+
* - coding — agentic edit/test loops; Codex and Claude are the two
|
|
61
|
+
* with real harnesses behind them.
|
|
62
|
+
* - mechanical — sweeps and reformatting; cheapest first (agy's flash
|
|
63
|
+
* tier, then Claude's haiku tier).
|
|
64
|
+
* - reasoning — deep thinking (also where `xhigh` effort lands); Claude
|
|
65
|
+
* is the only backend Talon drives an opus-class model on.
|
|
66
|
+
*/
|
|
67
|
+
const TASK_CLASS_RULES: Record<
|
|
68
|
+
TaskClass,
|
|
69
|
+
{ readonly prefer: readonly string[]; readonly require?: readonly string[] }
|
|
70
|
+
> = {
|
|
71
|
+
coding: { prefer: ["codex", "claude"] },
|
|
72
|
+
mechanical: { prefer: ["agy", "claude"] },
|
|
73
|
+
reasoning: { prefer: ["claude"], require: ["claude"] },
|
|
74
|
+
};
|
|
75
|
+
|
|
76
|
+
/** Ranking priority of a headroom source — measured beats unmeasured. */
|
|
77
|
+
const SOURCE_RANK: Record<BackendHeadroom["source"], number> = {
|
|
78
|
+
plan: 0,
|
|
79
|
+
ledger: 1,
|
|
80
|
+
none: 2,
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
export interface RouteHints {
|
|
84
|
+
readonly taskClass?: TaskClass;
|
|
85
|
+
readonly effort?: ReasoningEffortLevel;
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export interface RouteRequest {
|
|
89
|
+
readonly purpose: RoutePurpose;
|
|
90
|
+
/** An explicit backend from the caller. Wins outright. */
|
|
91
|
+
readonly requestedBackendId?: string;
|
|
92
|
+
/** An explicit model. Pins whatever backend is serving it. */
|
|
93
|
+
readonly requestedModel?: string;
|
|
94
|
+
/** The backend the caller would have used before routing existed. */
|
|
95
|
+
readonly chatBackendId: string;
|
|
96
|
+
/** Defaults to the config the backend pool was initialised with. */
|
|
97
|
+
readonly config?: TalonConfig;
|
|
98
|
+
readonly hints?: RouteHints;
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
export interface RouteDecision {
|
|
102
|
+
readonly backendId: string;
|
|
103
|
+
/** Only set when the caller pinned one — model choice stays downstream. */
|
|
104
|
+
readonly model?: string;
|
|
105
|
+
/** Human-readable: `pinned`, `disabled`, or why this backend won. */
|
|
106
|
+
readonly reason: string;
|
|
107
|
+
/** True when headroom actually chose this, rather than a pin or a default. */
|
|
108
|
+
readonly routed: boolean;
|
|
109
|
+
/** What the decision saw, for the spawn reply and the log line. */
|
|
110
|
+
readonly headroom?: BackendHeadroom;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** The task class a reasoning-effort level implies, if any. */
|
|
114
|
+
export function taskClassForEffort(
|
|
115
|
+
effort: ReasoningEffortLevel | undefined,
|
|
116
|
+
): TaskClass | undefined {
|
|
117
|
+
return effort === "xhigh" || effort === "high" ? "reasoning" : undefined;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** `config.router`, with the documented defaults filled in. */
|
|
121
|
+
function routerSettings(config: TalonConfig | undefined): {
|
|
122
|
+
enabled: boolean;
|
|
123
|
+
ceilingPercent: number;
|
|
124
|
+
} {
|
|
125
|
+
return {
|
|
126
|
+
enabled: config?.router?.enabled ?? true,
|
|
127
|
+
ceilingPercent: config?.router?.ceilingPercent ?? DEFAULT_CEILING_PERCENT,
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Can this backend host an isolated run, without booting it to find out?
|
|
133
|
+
* A pooled instance answers from its capability slots; a cold one qualifies
|
|
134
|
+
* only on an explicit local budget (see the module comment).
|
|
135
|
+
*/
|
|
136
|
+
function isCandidate(
|
|
137
|
+
id: string,
|
|
138
|
+
config: TalonConfig | undefined,
|
|
139
|
+
chatBackendId: string,
|
|
140
|
+
): boolean {
|
|
141
|
+
if (id === chatBackendId) return true;
|
|
142
|
+
const pooled = getPooledBackend(id);
|
|
143
|
+
if (pooled) return Boolean(pooled.background);
|
|
144
|
+
return hasBudget(config, id);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
/** Percent of the tightest window — what the ceiling is applied to. */
|
|
148
|
+
function limitingPercent(entry: BackendHeadroom): number {
|
|
149
|
+
return entry.limiting?.percent ?? 0;
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
/** Comparator: headroom first, then the documented tie-breaks. */
|
|
153
|
+
function rank(
|
|
154
|
+
a: BackendHeadroom,
|
|
155
|
+
b: BackendHeadroom,
|
|
156
|
+
chatBackendId: string,
|
|
157
|
+
prefer: readonly string[],
|
|
158
|
+
): number {
|
|
159
|
+
// Headroom to 3dp: two backends a thousandth apart are a tie, not a winner.
|
|
160
|
+
const byHeadroom =
|
|
161
|
+
Math.round(b.headroom * 1000) - Math.round(a.headroom * 1000);
|
|
162
|
+
if (byHeadroom !== 0) return byHeadroom;
|
|
163
|
+
const bySource = SOURCE_RANK[a.source] - SOURCE_RANK[b.source];
|
|
164
|
+
if (bySource !== 0) return bySource;
|
|
165
|
+
const preferIndex = (id: string): number => {
|
|
166
|
+
const i = prefer.indexOf(id);
|
|
167
|
+
return i === -1 ? prefer.length : i;
|
|
168
|
+
};
|
|
169
|
+
const byPrefer = preferIndex(a.id) - preferIndex(b.id);
|
|
170
|
+
if (byPrefer !== 0) return byPrefer;
|
|
171
|
+
// Cache warmth: a sub-agent is isolated, but the caller's backend is
|
|
172
|
+
// already up and its MCP servers are already spawned.
|
|
173
|
+
if (a.id === chatBackendId) return -1;
|
|
174
|
+
if (b.id === chatBackendId) return 1;
|
|
175
|
+
return a.id.localeCompare(b.id);
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* Apply a task class's hard requirement, when a candidate still satisfies it.
|
|
180
|
+
* Runs on the post-ceiling set, so a spent required backend is already gone.
|
|
181
|
+
*/
|
|
182
|
+
function applyVeto(
|
|
183
|
+
entries: BackendHeadroom[],
|
|
184
|
+
required: readonly string[] | undefined,
|
|
185
|
+
): BackendHeadroom[] {
|
|
186
|
+
if (!required || required.length === 0) return entries;
|
|
187
|
+
const kept = entries.filter((e) => required.includes(e.id));
|
|
188
|
+
return kept.length > 0 ? kept : entries;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** Drop candidates at or above the ceiling — unless that drops all of them. */
|
|
192
|
+
function applyCeiling(
|
|
193
|
+
entries: BackendHeadroom[],
|
|
194
|
+
ceilingPercent: number,
|
|
195
|
+
): { kept: BackendHeadroom[]; allOverCeiling: boolean } {
|
|
196
|
+
const kept = entries.filter((e) => limitingPercent(e) < ceilingPercent);
|
|
197
|
+
if (kept.length > 0) return { kept, allOverCeiling: false };
|
|
198
|
+
// Everything is spent. Running the least-bad one beats running nothing:
|
|
199
|
+
// the background subsystems have no queue to defer into.
|
|
200
|
+
return { kept: entries, allOverCeiling: true };
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
function decisionFor(
|
|
204
|
+
winner: BackendHeadroom,
|
|
205
|
+
allOverCeiling: boolean,
|
|
206
|
+
): RouteDecision {
|
|
207
|
+
const reason = allOverCeiling
|
|
208
|
+
? `every backend is over the ceiling — least spent: ${formatHeadroom(winner)}`
|
|
209
|
+
: `most headroom ${Math.round(winner.headroom * 100)}%`;
|
|
210
|
+
return {
|
|
211
|
+
backendId: winner.id,
|
|
212
|
+
reason,
|
|
213
|
+
routed: true,
|
|
214
|
+
headroom: winner,
|
|
215
|
+
};
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
function logDecision(
|
|
219
|
+
request: RouteRequest,
|
|
220
|
+
decision: RouteDecision,
|
|
221
|
+
candidates: BackendHeadroom[],
|
|
222
|
+
): void {
|
|
223
|
+
const seen = candidates.map((c) => `${c.id}=${formatHeadroom(c)}`).join(", ");
|
|
224
|
+
log(
|
|
225
|
+
"router",
|
|
226
|
+
`${request.purpose}: → ${decision.backendId} (${decision.reason})` +
|
|
227
|
+
(seen ? ` | candidates: ${seen}` : ""),
|
|
228
|
+
);
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
/**
|
|
232
|
+
* Choose the backend a piece of background work should run on.
|
|
233
|
+
*
|
|
234
|
+
* Never throws and never returns nothing: every path ends at a real backend
|
|
235
|
+
* id, falling back to the caller's own.
|
|
236
|
+
*/
|
|
237
|
+
export async function chooseBackend(
|
|
238
|
+
request: RouteRequest,
|
|
239
|
+
): Promise<RouteDecision> {
|
|
240
|
+
const { chatBackendId, purpose } = request;
|
|
241
|
+
|
|
242
|
+
if (request.requestedBackendId || request.requestedModel) {
|
|
243
|
+
const decision: RouteDecision = {
|
|
244
|
+
backendId: request.requestedBackendId ?? chatBackendId,
|
|
245
|
+
reason: "pinned",
|
|
246
|
+
routed: false,
|
|
247
|
+
...(request.requestedModel ? { model: request.requestedModel } : {}),
|
|
248
|
+
};
|
|
249
|
+
log("router", `${purpose}: → ${decision.backendId} (pinned)`);
|
|
250
|
+
return decision;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
const config = request.config ?? getPoolConfig() ?? undefined;
|
|
254
|
+
const settings = routerSettings(config);
|
|
255
|
+
if (!settings.enabled) {
|
|
256
|
+
return { backendId: chatBackendId, reason: "disabled", routed: false };
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
const ids = listAvailableBackends(config)
|
|
260
|
+
.filter(({ id }) => isCandidate(id, config, chatBackendId))
|
|
261
|
+
.map(({ id, label }) => ({ id, label }));
|
|
262
|
+
if (ids.length === 0) {
|
|
263
|
+
return { backendId: chatBackendId, reason: "no candidates", routed: false };
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const measured = await Promise.all(
|
|
267
|
+
ids.map(({ id, label }) => getBackendHeadroom(id, label, config)),
|
|
268
|
+
);
|
|
269
|
+
|
|
270
|
+
const rules = request.hints?.taskClass
|
|
271
|
+
? TASK_CLASS_RULES[request.hints.taskClass]
|
|
272
|
+
: undefined;
|
|
273
|
+
// Ceiling first, THEN the veto: a hard task-class requirement must not be
|
|
274
|
+
// able to send work to a backend that is out of plan. A lesser model that
|
|
275
|
+
// runs beats the right one that rate-limits.
|
|
276
|
+
const { kept, allOverCeiling } = applyCeiling(
|
|
277
|
+
measured,
|
|
278
|
+
settings.ceilingPercent,
|
|
279
|
+
);
|
|
280
|
+
const eligible = applyVeto(kept, rules?.require);
|
|
281
|
+
const ordered = [...eligible].sort((a, b) =>
|
|
282
|
+
rank(a, b, chatBackendId, rules?.prefer ?? []),
|
|
283
|
+
);
|
|
284
|
+
const winner = ordered[0];
|
|
285
|
+
if (!winner) {
|
|
286
|
+
return { backendId: chatBackendId, reason: "no candidates", routed: false };
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
const decision = decisionFor(winner, allOverCeiling);
|
|
290
|
+
logDecision(request, decision, measured);
|
|
291
|
+
return decision;
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/**
|
|
295
|
+
* The default model for a backend the router just picked.
|
|
296
|
+
*
|
|
297
|
+
* A routed run cannot carry the caller's model across — a model id is
|
|
298
|
+
* backend-specific. Config's `backendDefaults` wins (it is the operator's
|
|
299
|
+
* answer to "what should this provider run"), then the backend's own
|
|
300
|
+
* canonical default. Resolves `null` when neither exists, which the call
|
|
301
|
+
* sites read as "stay where you were".
|
|
302
|
+
*/
|
|
303
|
+
export async function resolveRoutedModel(
|
|
304
|
+
backendId: string,
|
|
305
|
+
config?: TalonConfig,
|
|
306
|
+
): Promise<string | null> {
|
|
307
|
+
const settings = config ?? getPoolConfig() ?? undefined;
|
|
308
|
+
const configured = settings?.backendDefaults?.[backendId];
|
|
309
|
+
if (configured) return configured;
|
|
310
|
+
try {
|
|
311
|
+
const pooled = getPooledBackend(backendId);
|
|
312
|
+
if (pooled) return (await pooled.models?.getDefaultModelId()) ?? null;
|
|
313
|
+
const acquired = await acquireBackendInstance(backendId);
|
|
314
|
+
try {
|
|
315
|
+
return (await acquired.backend.models?.getDefaultModelId()) ?? null;
|
|
316
|
+
} finally {
|
|
317
|
+
await acquired.release();
|
|
318
|
+
}
|
|
319
|
+
} catch {
|
|
320
|
+
return null;
|
|
321
|
+
}
|
|
322
|
+
}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* One usage snapshot per backend — what `/usage`, the `plan_usage` tool and
|
|
3
|
+
* `list_backends` all read.
|
|
4
|
+
*
|
|
5
|
+
* It lives in core rather than in the frontend presentation layer because
|
|
6
|
+
* the gateway tools need it too, and core cannot import a frontend. The
|
|
7
|
+
* split is: this module *gathers* (plan windows, headroom, and the reason
|
|
8
|
+
* there is nothing to show), renderers *format*.
|
|
9
|
+
*
|
|
10
|
+
* Nothing here boots a backend. Reading a number should never cost a
|
|
11
|
+
* subprocess per idle provider, so a backend that isn't running is listed
|
|
12
|
+
* with a reason instead of being woken or omitted.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import type { PlanUsage } from "../../agent-runtime/capabilities.js";
|
|
16
|
+
import type { TalonConfig } from "../../config/index.js";
|
|
17
|
+
import {
|
|
18
|
+
getPooledBackend,
|
|
19
|
+
listAvailableBackends,
|
|
20
|
+
} from "../backend-controller/index.js";
|
|
21
|
+
import { getBackendHeadroom, type BackendHeadroom } from "./headroom.js";
|
|
22
|
+
|
|
23
|
+
export interface BackendUsageSnapshot {
|
|
24
|
+
readonly id: string;
|
|
25
|
+
readonly label: string;
|
|
26
|
+
/** Raw plan windows, when this backend reported any. */
|
|
27
|
+
readonly plan?: PlanUsage;
|
|
28
|
+
/** Always present — every backend gets a comparable headroom figure. */
|
|
29
|
+
readonly headroom: BackendHeadroom;
|
|
30
|
+
/** Why there is no `plan`. Absent when there is one. */
|
|
31
|
+
readonly note?: string;
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** The reason a backend has no plan windows to show. */
|
|
35
|
+
function noteFor(id: string, headroom: BackendHeadroom): string {
|
|
36
|
+
const backend = getPooledBackend(id);
|
|
37
|
+
if (!backend) return "not running";
|
|
38
|
+
if (!backend.usage?.getPlanUsage) return "no plan limits on this backend";
|
|
39
|
+
if (headroom.source === "ledger") return "tracked against a local budget";
|
|
40
|
+
return "no usage information available";
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* Every exposed backend, in config order, with its plan (where it has one)
|
|
45
|
+
* and its headroom (always).
|
|
46
|
+
*
|
|
47
|
+
* `force` skips the 60s headroom cache — a person looking at `/usage` wants
|
|
48
|
+
* the number now, where a routing decision a second after another one does
|
|
49
|
+
* not.
|
|
50
|
+
*/
|
|
51
|
+
export async function collectBackendUsage(
|
|
52
|
+
config: TalonConfig | undefined,
|
|
53
|
+
options?: { force?: boolean },
|
|
54
|
+
): Promise<BackendUsageSnapshot[]> {
|
|
55
|
+
const backends = listAvailableBackends(config);
|
|
56
|
+
return Promise.all(
|
|
57
|
+
backends.map(async ({ id, label }) => {
|
|
58
|
+
const headroom = await getBackendHeadroom(id, label, config, options);
|
|
59
|
+
if (headroom.plan && headroom.plan.windows.length > 0) {
|
|
60
|
+
return { id, label, headroom, plan: headroom.plan };
|
|
61
|
+
}
|
|
62
|
+
return { id, label, headroom, note: noteFor(id, headroom) };
|
|
63
|
+
}),
|
|
64
|
+
);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Reorder a snapshot list so one backend comes first, everything else
|
|
69
|
+
* keeping config order. The `plan_usage` tool leads with the chat's own
|
|
70
|
+
* backend so callers that only read the first entry see what they used to.
|
|
71
|
+
*/
|
|
72
|
+
export function leadWith(
|
|
73
|
+
entries: BackendUsageSnapshot[],
|
|
74
|
+
id: string,
|
|
75
|
+
): BackendUsageSnapshot[] {
|
|
76
|
+
const index = entries.findIndex((e) => e.id === id);
|
|
77
|
+
if (index <= 0) return entries;
|
|
78
|
+
const lead = entries[index] as BackendUsageSnapshot;
|
|
79
|
+
return [lead, ...entries.filter((_, i) => i !== index)];
|
|
80
|
+
}
|
|
@@ -208,7 +208,8 @@ export const agentControlHandlers: SharedActionHandlers = {
|
|
|
208
208
|
ok: true,
|
|
209
209
|
text:
|
|
210
210
|
`Spawned agent "${parsed.label}" (id: ${outcome.agentId})\n` +
|
|
211
|
-
`Backend: ${outcome.backendId}/${outcome.model}
|
|
211
|
+
`Backend: ${outcome.backendId}/${outcome.model}` +
|
|
212
|
+
`${outcome.routing ? ` (routed: ${outcome.routing})` : ""}\n` +
|
|
212
213
|
`Timeout: ${timeoutS}s\n` +
|
|
213
214
|
`It runs in the background. You will be woken with its report — ` +
|
|
214
215
|
`carry on with what you were doing.`,
|
|
@@ -10,11 +10,33 @@ import {
|
|
|
10
10
|
getBackendForChat,
|
|
11
11
|
getBackendIdForChat,
|
|
12
12
|
getAvailableBackends,
|
|
13
|
+
getPoolConfig,
|
|
13
14
|
getPooledBackend,
|
|
14
15
|
acquireBackendInstance,
|
|
15
16
|
} from "../backend-controller/index.js";
|
|
17
|
+
import {
|
|
18
|
+
collectBackendUsage,
|
|
19
|
+
formatHeadroom,
|
|
20
|
+
leadWith,
|
|
21
|
+
type BackendUsageSnapshot,
|
|
22
|
+
} from "../backend-router/index.js";
|
|
16
23
|
import type { SharedActionHandlers } from "./types.js";
|
|
17
24
|
|
|
25
|
+
/** One `plan_usage` block: the headroom line, then any plan windows. */
|
|
26
|
+
function usageLines(entry: BackendUsageSnapshot): string[] {
|
|
27
|
+
const head = `- ${entry.label || entry.id}: ${formatHeadroom(entry.headroom)}`;
|
|
28
|
+
if (!entry.plan) {
|
|
29
|
+
return [`${head}${entry.note ? ` (${entry.note})` : ""}`];
|
|
30
|
+
}
|
|
31
|
+
return [
|
|
32
|
+
`${head}${entry.plan.plan ? ` · ${entry.plan.plan}` : ""}`,
|
|
33
|
+
...entry.plan.windows.map(
|
|
34
|
+
(w) =>
|
|
35
|
+
` ${w.label}: ${w.percent}% used${w.resetsAt ? `, resets ${w.resetsAt}` : ""}`,
|
|
36
|
+
),
|
|
37
|
+
];
|
|
38
|
+
}
|
|
39
|
+
|
|
18
40
|
export const modelHandlers: SharedActionHandlers = {
|
|
19
41
|
list_models: async (body, chatId, _backend, chatKey) => {
|
|
20
42
|
const chatIdStr = chatKey;
|
|
@@ -98,52 +120,67 @@ export const modelHandlers: SharedActionHandlers = {
|
|
|
98
120
|
}
|
|
99
121
|
},
|
|
100
122
|
|
|
101
|
-
// Account-level
|
|
102
|
-
//
|
|
123
|
+
// Account-level and fleet-wide: every exposed backend, with the headroom
|
|
124
|
+
// figure the plan-aware router ranks on — a backend with no usage API
|
|
125
|
+
// still answers, from its local budget ledger. The chat's own backend
|
|
126
|
+
// leads so a caller reading only the first entry sees what it used to.
|
|
103
127
|
plan_usage: async (_body, chatId, _backend, chatKey) => {
|
|
104
|
-
const
|
|
105
|
-
const
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
if (!source?.usage?.getPlanUsage)
|
|
109
|
-
return {
|
|
110
|
-
ok: false,
|
|
111
|
-
error: "No configured backend reports subscription rate limits.",
|
|
112
|
-
};
|
|
113
|
-
|
|
114
|
-
const usage = await source.usage.getPlanUsage();
|
|
115
|
-
if (!usage)
|
|
116
|
-
return {
|
|
117
|
-
ok: false,
|
|
118
|
-
error:
|
|
119
|
-
"Plan usage is unavailable — no subscription credentials, or this session authenticates with an API key.",
|
|
120
|
-
};
|
|
121
|
-
|
|
122
|
-
const lines = usage.windows.map(
|
|
123
|
-
(w) =>
|
|
124
|
-
`- ${w.label}: ${w.percent}% used${w.resetsAt ? `, resets ${w.resetsAt}` : ""}`,
|
|
128
|
+
const currentId = getBackendIdForChat(chatKey);
|
|
129
|
+
const entries = leadWith(
|
|
130
|
+
await collectBackendUsage(getPoolConfig() ?? undefined, { force: true }),
|
|
131
|
+
currentId,
|
|
125
132
|
);
|
|
133
|
+
if (entries.length === 0)
|
|
134
|
+
return { ok: false, error: "No backends are available." };
|
|
135
|
+
|
|
136
|
+
const lead = entries[0] as BackendUsageSnapshot;
|
|
126
137
|
return {
|
|
127
138
|
ok: true,
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
139
|
+
// Kept for callers written against the single-backend shape.
|
|
140
|
+
plan: lead.plan?.plan ?? null,
|
|
141
|
+
windows: lead.plan?.windows ?? [],
|
|
142
|
+
backends: entries.map((entry) => ({
|
|
143
|
+
id: entry.id,
|
|
144
|
+
label: entry.label,
|
|
145
|
+
current: entry.id === currentId,
|
|
146
|
+
headroom: Math.round(entry.headroom.headroom * 100) / 100,
|
|
147
|
+
source: entry.headroom.source,
|
|
148
|
+
limiting: entry.headroom.limiting ?? null,
|
|
149
|
+
stale: entry.headroom.stale ?? false,
|
|
150
|
+
plan: entry.plan?.plan ?? null,
|
|
151
|
+
windows: entry.plan?.windows ?? [],
|
|
152
|
+
...(entry.note ? { note: entry.note } : {}),
|
|
153
|
+
})),
|
|
154
|
+
text: `Plan usage and headroom by backend:\n${entries.flatMap(usageLines).join("\n")}`,
|
|
131
155
|
};
|
|
132
156
|
},
|
|
133
157
|
|
|
134
|
-
list_backends: (body, chatId, _backend, chatKey) => {
|
|
158
|
+
list_backends: async (body, chatId, _backend, chatKey) => {
|
|
135
159
|
const currentId = getBackendIdForChat(chatKey);
|
|
136
|
-
const
|
|
137
|
-
|
|
138
|
-
label: b.label,
|
|
139
|
-
current: b.id === currentId,
|
|
140
|
-
}));
|
|
141
|
-
if (backends.length === 0)
|
|
160
|
+
const available = getAvailableBackends();
|
|
161
|
+
if (available.length === 0)
|
|
142
162
|
return { ok: true, backends: [], text: "No backends are available." };
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
);
|
|
163
|
+
// Headroom comes off the router's 60s cache, so listing backends is
|
|
164
|
+
// cheap even though it now answers "which one has room?" as well.
|
|
165
|
+
const usage = await collectBackendUsage(getPoolConfig() ?? undefined);
|
|
166
|
+
const byId = new Map(usage.map((entry) => [entry.id, entry]));
|
|
167
|
+
const backends = available.map((b) => {
|
|
168
|
+
const entry = byId.get(b.id);
|
|
169
|
+
return {
|
|
170
|
+
id: b.id,
|
|
171
|
+
label: b.label,
|
|
172
|
+
current: b.id === currentId,
|
|
173
|
+
headroom: entry
|
|
174
|
+
? Math.round(entry.headroom.headroom * 100) / 100
|
|
175
|
+
: null,
|
|
176
|
+
headroomSource: entry?.headroom.source ?? null,
|
|
177
|
+
};
|
|
178
|
+
});
|
|
179
|
+
const lines = backends.map((b) => {
|
|
180
|
+
const entry = byId.get(b.id);
|
|
181
|
+
const room = entry ? ` — ${formatHeadroom(entry.headroom)} free` : "";
|
|
182
|
+
return `- ${b.id}${b.label && b.label !== b.id ? ` (${b.label})` : ""}${b.current ? " — current" : ""}${room}`;
|
|
183
|
+
});
|
|
147
184
|
return {
|
|
148
185
|
ok: true,
|
|
149
186
|
backends,
|
|
@@ -31,7 +31,7 @@ export const modelTools: ToolDefinition[] = [
|
|
|
31
31
|
{
|
|
32
32
|
name: "plan_usage",
|
|
33
33
|
description:
|
|
34
|
-
"Read
|
|
34
|
+
"Read usage and headroom for EVERY configured backend, not just this chat's: how much of each subscription's rate-limit windows is spent, when they reset, and one comparable headroom figure per backend. A backend with no usage API reports headroom from Talon's own local token ledger against its configured budget (marked 'local budget'), and one with neither says so. Use it before starting long or expensive work, when deciding whether to defer something, or to see which backend background work will be routed to.",
|
|
35
35
|
schema: {},
|
|
36
36
|
execute: (_params, bridge) => bridge("plan_usage", {}),
|
|
37
37
|
tag: "models",
|
|
@@ -40,7 +40,7 @@ export const modelTools: ToolDefinition[] = [
|
|
|
40
40
|
{
|
|
41
41
|
name: "list_backends",
|
|
42
42
|
description:
|
|
43
|
-
"List the available backends (providers)
|
|
43
|
+
"List the available backends (providers), which one this chat is currently using, and how much plan headroom each has left. Useful for understanding the model/provider landscape and for seeing where unpinned background work (sub-agents, cron query jobs, the heartbeat) will be routed. Note: a per-job `model` override must stay on this chat's current backend, so use list_models (no argument) to pick a model for a trigger or cron job.",
|
|
44
44
|
schema: {},
|
|
45
45
|
execute: (_params, bridge) => bridge("list_backends", {}),
|
|
46
46
|
tag: "models",
|
|
@@ -6,15 +6,21 @@
|
|
|
6
6
|
* through whichever provider it fronts, and an API-key install pays per
|
|
7
7
|
* token with no window to be near the end of. Those are listed with a
|
|
8
8
|
* reason rather than omitted, so the answer to "am I close to a limit?" is
|
|
9
|
-
* never silence
|
|
9
|
+
* never silence — and since the plan-aware router landed they carry a
|
|
10
|
+
* headroom figure too, derived from a local token budget where one is
|
|
11
|
+
* configured, so "which backend has room?" is answerable for all of them.
|
|
12
|
+
*
|
|
13
|
+
* The gathering itself lives in core (`engine/backend-router/usage.ts`),
|
|
14
|
+
* because the gateway tools need the same data and core cannot import a
|
|
15
|
+
* frontend. This module is the rendering adapter over it.
|
|
10
16
|
*/
|
|
11
17
|
|
|
12
18
|
import type { TalonConfig } from "../../core/config/index.js";
|
|
13
|
-
import type { PlanUsage } from "../../core/agent-runtime/capabilities.js";
|
|
14
19
|
import {
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
20
|
+
collectBackendUsage,
|
|
21
|
+
formatHeadroom,
|
|
22
|
+
type BackendHeadroom,
|
|
23
|
+
} from "../../core/engine/backend-router/index.js";
|
|
18
24
|
import { buildPlanDisplay, type PlanDisplay } from "./status-context.js";
|
|
19
25
|
|
|
20
26
|
export interface BackendUsageEntry {
|
|
@@ -24,6 +30,10 @@ export interface BackendUsageEntry {
|
|
|
24
30
|
plan: PlanDisplay | null;
|
|
25
31
|
/** Why there is nothing to show. Absent when `plan` is set. */
|
|
26
32
|
note?: string;
|
|
33
|
+
/** Comparable "how much is left", present for every backend. */
|
|
34
|
+
headroom: BackendHeadroom;
|
|
35
|
+
/** One-line rendering of `headroom`, ready to print. */
|
|
36
|
+
headroomLabel: string;
|
|
27
37
|
}
|
|
28
38
|
|
|
29
39
|
/**
|
|
@@ -36,37 +46,18 @@ export interface BackendUsageEntry {
|
|
|
36
46
|
export async function collectPlanUsage(
|
|
37
47
|
config: TalonConfig,
|
|
38
48
|
): Promise<BackendUsageEntry[]> {
|
|
39
|
-
const
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
});
|
|
54
|
-
continue;
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
let usage: PlanUsage | undefined;
|
|
58
|
-
try {
|
|
59
|
-
usage = await backend.usage.getPlanUsage();
|
|
60
|
-
} catch {
|
|
61
|
-
usage = undefined;
|
|
62
|
-
}
|
|
63
|
-
const plan = buildPlanDisplay(usage);
|
|
64
|
-
entries.push(
|
|
65
|
-
plan
|
|
66
|
-
? { id, label, plan }
|
|
67
|
-
: { id, label, plan: null, note: "no usage information available" },
|
|
68
|
-
);
|
|
69
|
-
}
|
|
70
|
-
|
|
71
|
-
return entries;
|
|
49
|
+
const snapshots = await collectBackendUsage(config, { force: true });
|
|
50
|
+
return snapshots.map((snapshot) => {
|
|
51
|
+
const plan = buildPlanDisplay(snapshot.plan);
|
|
52
|
+
return {
|
|
53
|
+
id: snapshot.id,
|
|
54
|
+
label: snapshot.label,
|
|
55
|
+
plan,
|
|
56
|
+
headroom: snapshot.headroom,
|
|
57
|
+
headroomLabel: formatHeadroom(snapshot.headroom),
|
|
58
|
+
...(plan
|
|
59
|
+
? {}
|
|
60
|
+
: { note: snapshot.note ?? "no usage information available" }),
|
|
61
|
+
};
|
|
62
|
+
});
|
|
72
63
|
}
|
|
@@ -443,10 +443,14 @@ export function renderUsageMessage(
|
|
|
443
443
|
|
|
444
444
|
for (const entry of entries) {
|
|
445
445
|
const name = fmt.escape(entry.label || entry.id);
|
|
446
|
+
// Headroom is the one figure every backend can answer, so it goes on
|
|
447
|
+
// every block — including the ones with no plan windows to draw.
|
|
448
|
+
const headroom = ` ${fmt.bold("Headroom:")} ${fmt.escape(entry.headroomLabel)}`;
|
|
446
449
|
if (!entry.plan) {
|
|
447
450
|
lines.push(
|
|
448
451
|
"",
|
|
449
452
|
`${fmt.bold(name)} — ${fmt.italic(fmt.escape(entry.note ?? ""))}`,
|
|
453
|
+
headroom,
|
|
450
454
|
);
|
|
451
455
|
continue;
|
|
452
456
|
}
|
|
@@ -454,7 +458,7 @@ export function renderUsageMessage(
|
|
|
454
458
|
? ` ${fmt.emphasis(`(${entry.plan.ageLabel})`)}`
|
|
455
459
|
: "";
|
|
456
460
|
const plan = entry.plan.plan ? ` · ${fmt.escape(entry.plan.plan)}` : "";
|
|
457
|
-
lines.push("", `${fmt.bold(name)}${plan}${age}
|
|
461
|
+
lines.push("", `${fmt.bold(name)}${plan}${age}`, headroom);
|
|
458
462
|
if (entry.plan.resetsAvailable) {
|
|
459
463
|
const n = entry.plan.resetsAvailable;
|
|
460
464
|
const resets = `usage limit reset${n === 1 ? "" : "s"} available`;
|