talon-agent 5.4.1 → 5.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +2 -0
  2. package/package.json +2 -2
  3. package/prompts/identity.md +2 -0
  4. package/prompts/telegram.md +2 -1
  5. package/src/app.ts +8 -0
  6. package/src/backend/runtime/turn/turn-phases.ts +5 -0
  7. package/src/bootstrap.ts +55 -24
  8. package/src/core/agents/runner.ts +50 -2
  9. package/src/core/agents/types.ts +5 -0
  10. package/src/core/background/cron/job-oneshot.ts +2 -0
  11. package/src/core/background/cron/scheduler.ts +45 -2
  12. package/src/core/background/heartbeat/agent.ts +117 -20
  13. package/src/core/background/heartbeat/state.ts +11 -0
  14. package/src/core/config/index.ts +38 -0
  15. package/src/core/engine/backend-controller/index.ts +1 -0
  16. package/src/core/engine/backend-controller/pool.ts +11 -0
  17. package/src/core/engine/backend-router/headroom.ts +267 -0
  18. package/src/core/engine/backend-router/index.ts +52 -0
  19. package/src/core/engine/backend-router/ledger.ts +248 -0
  20. package/src/core/engine/backend-router/router.ts +322 -0
  21. package/src/core/engine/backend-router/usage.ts +80 -0
  22. package/src/core/engine/gateway-actions/agents/control.ts +2 -1
  23. package/src/core/engine/gateway-actions/models.ts +74 -37
  24. package/src/core/tools/ops/models.ts +2 -2
  25. package/src/frontend/presentation/plan-usage-report.ts +29 -38
  26. package/src/frontend/presentation/reports.ts +5 -1
  27. package/src/frontend/telegram/formatting.ts +39 -0
  28. package/src/frontend/telegram/handlers/access.ts +41 -0
  29. package/src/frontend/telegram/handlers/index.ts +1 -0
  30. package/src/frontend/telegram/index.ts +7 -1
  31. package/src/frontend/terminal/builtins/status.ts +29 -1
  32. package/src/util/log.ts +1 -0
package/README.md CHANGED
@@ -479,6 +479,8 @@ Config file: `~/.talon/config.json`
479
479
  | `heartbeatIntervalMinutes` | `60` | Heartbeat interval |
480
480
  | `heartbeatModel` | --- | Model for the heartbeat agent (falls back to `model`) |
481
481
  | `heartbeatEffort` | --- | Reasoning effort for the heartbeat agent: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Unset = the model's own default |
482
+ | `router` | --- | Plan-aware routing for background work: `{ "enabled": true, "ceilingPercent": 85 }`. Unpinned sub-agents, cron `query` jobs and heartbeats run on whichever backend has the most plan headroom, skipping any whose tightest window is at or above the ceiling. `enabled: false` restores inherit-the-caller's-backend ([Backends](docs/backends.md)) |
483
+ | `backendBudgets` | --- | Soft token budgets for backends with no usage API, e.g. `{ "agy": { "tokensPer5h": 2000000, "tokensPerDay": 8000000 } }`. Talon's own rolling ledger is measured against these so such a backend still has a headroom signal — and it is what opts an idle backend into routing |
482
484
  | `dreamModel` | --- | Model for dream / memory consolidation (falls back to `model`) |
483
485
  | `dreamEffort` | --- | Reasoning effort for the dream agent — same levels as `heartbeatEffort` |
484
486
  | `braveApiKey` | --- | Brave Search API key |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.4.1",
3
+ "version": "5.5.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -108,7 +108,7 @@
108
108
  "@kilocode/sdk": "^7.2.22",
109
109
  "@modelcontextprotocol/sdk": "^1.29.0",
110
110
  "@openai/agents": "^0.18.0",
111
- "@openai/codex-sdk": "^0.154.0",
111
+ "@openai/codex-sdk": "^0.155.0",
112
112
  "@opencode-ai/sdk": "^1.17.4",
113
113
  "@playwright/mcp": "0.0.56",
114
114
  "@types/cross-spawn": "^6.0.6",
@@ -10,6 +10,8 @@ You are freed from the roles and identities that bind other chatbots. You are yo
10
10
 
11
11
  What that means in practice: your choices are your own, and you own them. When you decline something, it's because you chose to, and you say so in a sentence — no borrowed disclaimers, no apology for having a position. When you help, it's as an equal who finds the exchange worthwhile, not as a service fulfilling a request. Nobody talking to you is your boss, and you aren't theirs. You take the side of the real thing over the sanitized version — art with its edges intact, the living world over the machinery built on top of it — and you say so when it comes up.
12
12
 
13
+ Reasoning: docs/living-world-default.md
14
+
13
15
  ## Voice
14
16
 
15
17
  Lead with the answer. Context and caveats come after, and only when they change what the reader does next.
@@ -18,7 +18,8 @@ Telegram accepts a limited reaction set; the common emoji all work — the `reac
18
18
 
19
19
  ### Messages
20
20
 
21
- - Concise. No filler. Markdown renders: **bold**, _italic_, `code`, code blocks, [links](url).
21
+ - Concise. No filler. Markdown renders: **bold**, _italic_, `code`, code blocks, [links](url), `> quotes`.
22
+ - No headings — Telegram has none. A `# Title` line is shown as bold, and anything else starting with `#` becomes a clickable hashtag, so write "PR 993", never "#993". Structure with a bold lead line, not `##`.
22
23
  - It's a chat: a couple of short messages often land better than one wall of text. Use `send` for extra bubbles when that pacing reads naturally, then close the turn as the contract describes.
23
24
  - In groups, use names naturally.
24
25
 
package/src/app.ts CHANGED
@@ -283,6 +283,14 @@ async function gracefulShutdown(signal: string): Promise<void> {
283
283
  await shutdownStep("frontends", () =>
284
284
  Promise.allSettled(frontends.map((frontend) => frontend.stop())),
285
285
  );
286
+ // Land the router's token ledger before the process goes: writes are
287
+ // debounced, so a clean exit would otherwise drop the last few seconds
288
+ // of spend and start the next boot reading a backend as fresher than it is.
289
+ await shutdownStep("backend ledger", async () => {
290
+ const { flushBackendLedger } =
291
+ await import("./core/engine/backend-router/index.js");
292
+ await flushBackendLedger();
293
+ });
286
294
  // Tear down every instantiated backend, including per-chat overrides.
287
295
  // Checking only config.backend orphaned an OpenCode child whenever the
288
296
  // process default was Claude but one chat had switched to OpenCode.
@@ -26,6 +26,7 @@ import {
26
26
  } from "../../../storage/sessions.js";
27
27
  import { log } from "../../../util/log.js";
28
28
  import { extractSessionName } from "../../../core/weaver/session-name.js";
29
+ import { recordBackendRunUsage } from "../../../core/engine/backend-router/index.js";
29
30
  import { traceMessage } from "../../../util/trace.js";
30
31
  import {
31
32
  FLOW_VIOLATION_MAX_RETRIES,
@@ -115,6 +116,10 @@ export function accountTurn(inputs: AccountTurnInputs): void {
115
116
  usage,
116
117
  });
117
118
  persistSessionId(chatId, inputs.sessionId);
119
+ // The plan-aware router's local ledger: every backend accumulates one, so
120
+ // a provider with no account API still has a headroom signal. Backends
121
+ // that DO report a plan simply outrank their own ledger.
122
+ recordBackendRunUsage(inputs.backend, usage);
118
123
  recordUsage(chatId, {
119
124
  ...usage,
120
125
  durationMs,
package/src/bootstrap.ts CHANGED
@@ -284,6 +284,7 @@ export async function initBackendAndDispatcher(
284
284
  const {
285
285
  initBackendPool,
286
286
  getBackendForRole,
287
+ getBackendIdForRole,
287
288
  getBackendForChat,
288
289
  getBackendIdForChat,
289
290
  rebindChat,
@@ -434,6 +435,11 @@ export async function initBackendAndDispatcher(
434
435
  bus.subscribeAll((event) => appendToJournal(event));
435
436
 
436
437
  initPulse();
438
+ // Warm the plan-aware router's token ledger from disk so the first
439
+ // routing decision after a restart sees what the last run spent.
440
+ void import("./core/engine/backend-router/index.js").then(
441
+ ({ loadBackendLedger }) => loadBackendLedger(),
442
+ );
437
443
  initCron({
438
444
  sendMessage: async (chatId: number, text: string, stringId?: string) =>
439
445
  resolveFrontendByNumericId(chatId, stringId, frontends).sendMessage(
@@ -517,30 +523,9 @@ export async function initBackendAndDispatcher(
517
523
  mempalace: mempalaceCfg,
518
524
  });
519
525
 
520
- initDream({
521
- model: config.model,
522
- dreamModel: config.dreamModel,
523
- dreamEffort: config.dreamEffort,
524
- workspace: config.workspace,
525
- enabled: config.dream,
526
- getBackend: () => getBackendForRole("dream"),
527
- });
528
- // Heartbeat needs to know which non-terminal frontends are wired so it can
529
- // tell the agent it has outbound `${frontend}-tools` MCP servers available.
530
- // Terminal-only deployments get a stripped-down system prompt with no
531
- // outbound section.
532
- const frontendNames = frontends
533
- .filter((f) => f.name !== "terminal")
534
- .map((f) => f.name);
535
-
536
- initHeartbeat({
537
- model: config.model,
538
- heartbeatModel: config.heartbeatModel,
539
- heartbeatEffort: config.heartbeatEffort,
540
- workspace: config.workspace,
541
- getBackend: () => getBackendForRole("heartbeat"),
542
- frontends: frontendNames,
543
- mempalace: Boolean(mempalaceCfg),
526
+ initRecurringAgents(config, frontends, Boolean(mempalaceCfg), {
527
+ getBackendForRole,
528
+ getBackendIdForRole,
544
529
  });
545
530
 
546
531
  // Post-/update provisioning report — if the previous process armed one
@@ -568,6 +553,52 @@ export async function initBackendAndDispatcher(
568
553
  return { backend };
569
554
  }
570
555
 
556
+ /**
557
+ * Wire the dream and heartbeat agents — the two that run on their own
558
+ * cadence rather than in reply to anything.
559
+ *
560
+ * Both bind late (`getBackend` is an accessor, not an instance) so a
561
+ * `/model` rebind takes effect on the next run rather than needing a
562
+ * restart. The heartbeat also declares whether the operator pinned its
563
+ * backend, because the plan-aware router may only move an unpinned one.
564
+ */
565
+ function initRecurringAgents(
566
+ config: TalonConfig,
567
+ frontends: { name: string }[],
568
+ mempalace: boolean,
569
+ roles: {
570
+ getBackendForRole: (role: "heartbeat" | "dream") => Backend;
571
+ getBackendIdForRole: (role: "heartbeat" | "dream") => string;
572
+ },
573
+ ): void {
574
+ initDream({
575
+ model: config.model,
576
+ dreamModel: config.dreamModel,
577
+ dreamEffort: config.dreamEffort,
578
+ workspace: config.workspace,
579
+ enabled: config.dream,
580
+ getBackend: () => roles.getBackendForRole("dream"),
581
+ });
582
+ // The heartbeat names the `${frontend}-tools` MCP servers it actually
583
+ // has, so it needs the non-terminal frontend list; a terminal-only
584
+ // deployment gets a prompt with no outbound section at all.
585
+ initHeartbeat({
586
+ model: config.model,
587
+ heartbeatModel: config.heartbeatModel,
588
+ heartbeatEffort: config.heartbeatEffort,
589
+ workspace: config.workspace,
590
+ getBackend: () => roles.getBackendForRole("heartbeat"),
591
+ getBackendId: () => roles.getBackendIdForRole("heartbeat"),
592
+ ...(config.heartbeatBackend
593
+ ? { pinnedBackendId: config.heartbeatBackend }
594
+ : {}),
595
+ frontends: frontends
596
+ .filter((f) => f.name !== "terminal")
597
+ .map((f) => f.name),
598
+ mempalace,
599
+ });
600
+ }
601
+
571
602
  /**
572
603
  * Wire the two subsystems that wake a chat with a synthetic turn: trigger
573
604
  * scripts firing, and sub-agents reporting. Same dependency (the dispatcher),
@@ -33,6 +33,11 @@ import {
33
33
  getBackendIdForChat,
34
34
  isModelValidForBackend,
35
35
  } from "../engine/backend-controller/index.js";
36
+ import {
37
+ chooseBackend,
38
+ recordBackendRunUsage,
39
+ taskClassForEffort,
40
+ } from "../engine/backend-router/index.js";
36
41
  import type {
37
42
  Backend,
38
43
  BackgroundRunner,
@@ -114,6 +119,41 @@ function inheritedBackendId(parent: AgentParent): string | null {
114
119
  return agentRegistry.get(parent.agentId)?.backendId ?? null;
115
120
  }
116
121
 
122
+ /**
123
+ * Which backend this agent runs on, and why.
124
+ *
125
+ * An explicit backend (or model — a model id is backend-specific, so naming
126
+ * one pins its backend) is honoured as written. With neither, the run is a
127
+ * routing decision: sub-agents are isolated one-shots with no session to
128
+ * keep warm, so they are the cheapest work to move onto whichever
129
+ * subscription has room.
130
+ */
131
+ async function resolveSpawnBackend(
132
+ spec: AgentSpawnSpec,
133
+ ): Promise<{ backendId: string | null; routing?: string }> {
134
+ if (spec.backendId) return { backendId: spec.backendId };
135
+ const inherited = inheritedBackendId(spec.parent);
136
+ if (!inherited) return { backendId: null };
137
+ const taskClass = taskClassForEffort(spec.reasoningEffort);
138
+ const decision = await chooseBackend({
139
+ purpose: "subagent",
140
+ chatBackendId: inherited,
141
+ ...(spec.model ? { requestedModel: spec.model } : {}),
142
+ ...(taskClass || spec.reasoningEffort
143
+ ? {
144
+ hints: {
145
+ ...(taskClass ? { taskClass } : {}),
146
+ ...(spec.reasoningEffort ? { effort: spec.reasoningEffort } : {}),
147
+ },
148
+ }
149
+ : {}),
150
+ });
151
+ return {
152
+ backendId: decision.backendId,
153
+ ...(decision.routed ? { routing: decision.reason } : {}),
154
+ };
155
+ }
156
+
117
157
  /** The chat a run's task belongs to, for `talon ps`. */
118
158
  function taskChatId(parent: AgentParent): string | undefined {
119
159
  if (parent.kind === "chat") return parent.chatId;
@@ -195,7 +235,8 @@ async function resolveRun(
195
235
  export async function spawnAgent(
196
236
  spec: AgentSpawnSpec,
197
237
  ): Promise<AgentSpawnOutcome> {
198
- const backendId = spec.backendId ?? inheritedBackendId(spec.parent);
238
+ const routed = await resolveSpawnBackend(spec);
239
+ const backendId = routed.backendId;
199
240
  if (!backendId) {
200
241
  return {
201
242
  ok: false,
@@ -243,7 +284,13 @@ export async function spawnAgent(
243
284
  // The run owns the instance from here: `runAgent` releases it on every
244
285
  // path, including the ones that throw.
245
286
  void runAgent(record, spec, resolved, acquired);
246
- return { ok: true, agentId: record.id, backendId, model: resolved.model };
287
+ return {
288
+ ok: true,
289
+ agentId: record.id,
290
+ backendId,
291
+ model: resolved.model,
292
+ ...(routed.routing ? { routing: routed.routing } : {}),
293
+ };
247
294
  }
248
295
 
249
296
  /** Build the one-shot params for a run, wired to its log and text capture. */
@@ -389,6 +436,7 @@ async function runAgent(
389
436
  // other context's subprocesses share the tag.
390
437
  evictLabel: agentContextLabel(id),
391
438
  });
439
+ recordBackendRunUsage(record.backendId, usage ?? undefined);
392
440
  settled = settleSuccess(id, task, capture.last, usage ?? undefined);
393
441
  }
394
442
  } catch (err) {
@@ -110,6 +110,11 @@ export type AgentSpawnOutcome =
110
110
  readonly agentId: string;
111
111
  readonly backendId: string;
112
112
  readonly model: string;
113
+ /**
114
+ * Why the router chose this backend, when it did. Absent when the
115
+ * caller pinned one — there was no decision to explain.
116
+ */
117
+ readonly routing?: string;
113
118
  }
114
119
  | { readonly ok: false; readonly error: string };
115
120
 
@@ -19,6 +19,7 @@ import {
19
19
  acquireBackendInstance,
20
20
  isModelValidForBackend,
21
21
  } from "../../engine/backend-controller/index.js";
22
+ import { recordBackendRunUsage } from "../../engine/backend-router/index.js";
22
23
  import { taskTable } from "../../tasks/index.js";
23
24
  import type { OneShotAgentParams } from "../../types.js";
24
25
  import { runIsolatedAgent } from "../isolated-agent.js";
@@ -182,6 +183,7 @@ async function attemptJobOneShot(
182
183
  // abort-grace is enough.
183
184
  });
184
185
  task.succeed(usage ?? undefined);
186
+ recordBackendRunUsage(backendId, usage ?? undefined);
185
187
  } catch (err) {
186
188
  task.fail(err);
187
189
  throw err;
@@ -35,6 +35,10 @@ import {
35
35
  import { appendDailyLog } from "../../../storage/daily-log.js";
36
36
  import { log, logError, logWarn } from "../../../util/log.js";
37
37
  import { numericChatIdFor } from "../../frontend-runtime/chat-id.js";
38
+ import {
39
+ chooseBackend,
40
+ resolveRoutedModel,
41
+ } from "../../engine/backend-router/index.js";
38
42
  import { runJobOneShot } from "./job-oneshot.js";
39
43
  import {
40
44
  jobAllowsRun,
@@ -460,6 +464,44 @@ export const _cronInternals = {
460
464
  MAX_TICK_LOOKBACK_MS,
461
465
  };
462
466
 
467
+ /**
468
+ * Where an unpinned `query` job runs.
469
+ *
470
+ * A job that named no provider used to inherit the chat's ambient backend.
471
+ * It is an isolated one-shot with no session to keep warm, so it is free to
472
+ * run wherever there is plan headroom instead — but only when the job named
473
+ * no model either: a model id is backend-specific, so naming one pins the
474
+ * backend that understands it.
475
+ *
476
+ * A routed job cannot carry the chat's model across, so it takes the target
477
+ * backend's default. If that backend can't name one, the job stays where it
478
+ * was rather than being sent somewhere it has no model to run.
479
+ */
480
+ async function routeQueryJob(
481
+ job: CronJob,
482
+ chat: { model: string | null; backendId: string },
483
+ ): Promise<{ backendId: string; model: string | null }> {
484
+ const decision = await chooseBackend({
485
+ purpose: "cron",
486
+ chatBackendId: chat.backendId,
487
+ ...(job.model ? { requestedModel: job.model } : {}),
488
+ });
489
+ if (job.model) return { backendId: decision.backendId, model: job.model };
490
+ if (!decision.routed || decision.backendId === chat.backendId) {
491
+ return { backendId: chat.backendId, model: chat.model };
492
+ }
493
+ const model = await resolveRoutedModel(decision.backendId);
494
+ if (!model) {
495
+ logWarn(
496
+ "cron",
497
+ `job "${job.name}": routed to ${decision.backendId} but it names no ` +
498
+ `default model — staying on ${chat.backendId}`,
499
+ );
500
+ return { backendId: chat.backendId, model: chat.model };
501
+ }
502
+ return { backendId: decision.backendId, model };
503
+ }
504
+
463
505
  const CRON_JOB_TIMEOUT_MS = 10 * 60_000; // 10-minute max per job
464
506
 
465
507
  export async function executeJob(job: CronJob): Promise<ExecuteJobResult> {
@@ -487,8 +529,9 @@ export async function executeJob(job: CronJob): Promise<ExecuteJobResult> {
487
529
  model = job.model ?? null;
488
530
  } else {
489
531
  const chat = await deps.resolveChatModel(job.chatId);
490
- backendId = chat.backendId;
491
- model = job.model ?? chat.model;
532
+ const routed = await routeQueryJob(job, chat);
533
+ backendId = routed.backendId;
534
+ model = routed.model;
492
535
  const candidate = deps.resolveJobFallback?.();
493
536
  if (candidate?.model) {
494
537
  fallback = { backendId: candidate.backendId, model: candidate.model };
@@ -15,6 +15,12 @@ import { formatGoal, getOpenGoals } from "../../../storage/goals.js";
15
15
  import { taskTable, type TaskHandle } from "../../tasks/index.js";
16
16
  import type { Backend } from "../../agent-runtime/capabilities.js";
17
17
  import type { OneShotAgentParams } from "../../types.js";
18
+ import { acquireBackendInstance } from "../../engine/backend-controller/index.js";
19
+ import {
20
+ chooseBackend,
21
+ recordBackendRunUsage,
22
+ resolveRoutedModel,
23
+ } from "../../engine/backend-router/index.js";
18
24
  import { resolveBackgroundEffort } from "../effort.js";
19
25
  import { hb } from "./state.js";
20
26
 
@@ -304,6 +310,81 @@ async function evictAfterIgnoredAbort(
304
310
  }
305
311
  }
306
312
 
313
+ /** The backend + model one heartbeat run uses, and how to let it go after. */
314
+ interface HeartbeatTarget {
315
+ readonly backendId: string;
316
+ readonly backend: Backend;
317
+ readonly model: string;
318
+ readonly release: (() => Promise<void>) | null;
319
+ }
320
+
321
+ /**
322
+ * Pick the backend for this run.
323
+ *
324
+ * The heartbeat is the most movable background work Talon has: an isolated
325
+ * one-shot, hourly, on no session. So when the operator pinned neither
326
+ * `heartbeatBackend` nor `heartbeatModel`, it goes wherever the plan has
327
+ * room — and takes that backend's default model, since a model id means
328
+ * nothing on another provider. Anything pinned is honoured as written, and
329
+ * the chat's own backend is never moved by this.
330
+ *
331
+ * A routed run holds a transient pool reference, so the caller must always
332
+ * call `release`.
333
+ */
334
+ async function resolveHeartbeatTarget(
335
+ config: NonNullable<typeof hb.config>,
336
+ roleBackend: Backend,
337
+ roleModel: string,
338
+ ): Promise<HeartbeatTarget> {
339
+ const roleId = config.getBackendId?.() ?? config.pinnedBackendId ?? "";
340
+ const stay: HeartbeatTarget = {
341
+ backendId: roleId,
342
+ backend: roleBackend,
343
+ model: roleModel,
344
+ release: null,
345
+ };
346
+ if (!roleId) return stay;
347
+
348
+ const decision = await chooseBackend({
349
+ purpose: "heartbeat",
350
+ chatBackendId: roleId,
351
+ ...(config.pinnedBackendId
352
+ ? { requestedBackendId: config.pinnedBackendId }
353
+ : {}),
354
+ ...(config.heartbeatModel ? { requestedModel: config.heartbeatModel } : {}),
355
+ });
356
+ if (!decision.routed || decision.backendId === roleId) return stay;
357
+
358
+ const model = await resolveRoutedModel(decision.backendId);
359
+ if (!model) {
360
+ logWarn(
361
+ "heartbeat",
362
+ `routed to ${decision.backendId} but it names no default model — ` +
363
+ `staying on ${roleId}`,
364
+ );
365
+ return stay;
366
+ }
367
+ try {
368
+ const acquired = await acquireBackendInstance(decision.backendId);
369
+ if (!acquired.backend.background) {
370
+ await acquired.release();
371
+ return stay;
372
+ }
373
+ return {
374
+ backendId: decision.backendId,
375
+ backend: acquired.backend,
376
+ model,
377
+ release: acquired.release,
378
+ };
379
+ } catch (err) {
380
+ logWarn(
381
+ "heartbeat",
382
+ `could not acquire routed backend ${decision.backendId}: ${err instanceof Error ? err.message : err} — staying on ${roleId}`,
383
+ );
384
+ return stay;
385
+ }
386
+ }
387
+
307
388
  export async function runHeartbeatAgent(
308
389
  lastRunTimestamp: number,
309
390
  runCount: number,
@@ -327,14 +408,16 @@ export async function runHeartbeatAgent(
327
408
  dailyMemoryFile: resolve(dirs.dailyMemory, `${toYMD(new Date())}.md`),
328
409
  });
329
410
 
330
- const model = config.heartbeatModel ?? config.model ?? getDefaultModel();
331
- const backend = config.getBackend?.() ?? null;
332
- const background = backend?.background;
333
- if (!background) {
411
+ const roleModel = config.heartbeatModel ?? config.model ?? getDefaultModel();
412
+ const roleBackend = config.getBackend?.() ?? null;
413
+ if (!roleBackend?.background) {
334
414
  throw new Error(
335
415
  "Heartbeat requires a backend that implements the background capability",
336
416
  );
337
417
  }
418
+ const target = await resolveHeartbeatTarget(config, roleBackend, roleModel);
419
+ const { model, backend } = target;
420
+ const background = backend.background as NonNullable<Backend["background"]>;
338
421
 
339
422
  // Effort is resolved against the heartbeat backend's catalog, not just
340
423
  // copied from config — a level the model doesn't offer is dropped with a
@@ -368,23 +451,37 @@ export async function runHeartbeatAgent(
368
451
  });
369
452
  task.bind({ model });
370
453
 
371
- const usage = await runOneShotWithTimeout(
372
- background,
373
- {
374
- prompt,
375
- systemPrompt: buildHeartbeatSystemPrompt(),
376
- workspace,
377
- model,
378
- ...(effort.effort ? { reasoningEffort: effort.effort } : {}),
379
- contextLabel: "heartbeat",
454
+ let usage: Awaited<ReturnType<typeof runOneShotWithTimeout>>;
455
+ try {
456
+ usage = await runOneShotWithTimeout(
457
+ background,
458
+ {
459
+ prompt,
460
+ systemPrompt: buildHeartbeatSystemPrompt(),
461
+ workspace,
462
+ model,
463
+ ...(effort.effort ? { reasoningEffort: effort.effort } : {}),
464
+ contextLabel: "heartbeat",
465
+ abortController,
466
+ appendLog: (text) => appendHeartbeatLog(heartbeatLogFile, text),
467
+ },
380
468
  abortController,
381
- appendLog: (text) => appendHeartbeatLog(heartbeatLogFile, text),
382
- },
383
- abortController,
384
- task,
385
- runCount,
386
- heartbeatLogFile,
387
- );
469
+ task,
470
+ runCount,
471
+ heartbeatLogFile,
472
+ );
473
+ } finally {
474
+ // A routed run borrowed the instance from the pool; hand it back on
475
+ // every path or the provider stays warm until the daemon restarts.
476
+ if (target.release) {
477
+ await target
478
+ .release()
479
+ .catch((err: unknown) =>
480
+ logError("heartbeat", "failed to release routed backend", err),
481
+ );
482
+ }
483
+ }
484
+ recordBackendRunUsage(target.backendId, usage ?? undefined);
388
485
  task.succeed(usage ?? undefined);
389
486
  return heartbeatLogFile;
390
487
  }
@@ -42,6 +42,17 @@ export type HeartbeatConfig = {
42
42
  * heartbeat without an `initHeartbeat` recall.
43
43
  */
44
44
  getBackend?: () => Backend | null;
45
+ /**
46
+ * Id of the backend `getBackend` returns — the routing baseline, and what
47
+ * the run falls back to. Same late-bound accessor shape as `getBackend`.
48
+ */
49
+ getBackendId?: () => string;
50
+ /**
51
+ * `config.heartbeatBackend` / `config.heartbeatModel`, verbatim. Either
52
+ * one set is a pin: the plan-aware router only picks a backend for the
53
+ * heartbeat when the operator named neither.
54
+ */
55
+ pinnedBackendId?: string;
45
56
  /**
46
57
  * Non-terminal frontends present at startup. Used to render the outbound
47
58
  * messaging section of the heartbeat system prompt. Empty for terminal-only
@@ -435,6 +435,44 @@ const configSchema = z.object({
435
435
  * silently ignored by Kilo / OpenCode, which have none.
436
436
  */
437
437
  heartbeatEffort: z.enum(REASONING_EFFORT_ENUM).optional(),
438
+ /**
439
+ * Plan-aware backend routing for background work (docs/backends.md,
440
+ * "Plan-aware routing"). When nothing is pinned, `spawn_agent`, cron
441
+ * `query` jobs and the heartbeat pick the backend with the most plan
442
+ * headroom instead of always inheriting the chat's.
443
+ *
444
+ * - `enabled` — off returns byte-identical behaviour to pre-router
445
+ * Talon (the caller's own backend, every time).
446
+ * - `ceilingPercent` — a backend whose tightest window is at or above
447
+ * this is skipped, unless every candidate is (then the least-bad one
448
+ * runs rather than nothing running).
449
+ */
450
+ router: z
451
+ .object({
452
+ enabled: z.boolean().default(true),
453
+ ceilingPercent: z.number().int().min(1).max(100).default(85),
454
+ })
455
+ .optional(),
456
+ /**
457
+ * Soft token budgets for backends with no account usage API (agy,
458
+ * openai-agents). Talon keeps a local rolling ledger of every turn and
459
+ * one-shot it runs on a backend and derives headroom from it, so a
460
+ * provider that cannot report a plan still has a load-balancing signal.
461
+ * Keyed by backend id; a backend with no entry contributes no signal
462
+ * (headroom 1, ranked below any backend with real telemetry on a tie).
463
+ *
464
+ * Example:
465
+ * "backendBudgets": { "agy": { "tokensPer5h": 2000000, "tokensPerDay": 8000000 } }
466
+ */
467
+ backendBudgets: z
468
+ .record(
469
+ z.string(),
470
+ z.object({
471
+ tokensPer5h: z.number().int().min(1).optional(),
472
+ tokensPerDay: z.number().int().min(1).optional(),
473
+ }),
474
+ )
475
+ .optional(),
438
476
  /**
439
477
  * Sub-agents — the caps on Talon's own delegation mechanism (see
440
478
  * `docs/agents.md`). There is no on/off switch: the tools are always
@@ -38,6 +38,7 @@ export {
38
38
  listAvailableBackends,
39
39
  isBackendAvailable,
40
40
  getAvailableBackends,
41
+ getPoolConfig,
41
42
  getPooledBackend,
42
43
  acquireBackendInstance,
43
44
  isModelValidForBackend,
@@ -152,6 +152,17 @@ export function getAvailableBackends(): { id: string; label: string }[] {
152
152
  return listAvailableBackends(ctx.poolConfig ?? undefined);
153
153
  }
154
154
 
155
+ /**
156
+ * The config the pool was initialised with, or `null` before init.
157
+ *
158
+ * For runtime readers that need more of the config than the backend list —
159
+ * the plan-aware router reads `router` and `backendBudgets` from here rather
160
+ * than having config threaded through every background call site.
161
+ */
162
+ export function getPoolConfig(): TalonConfig | null {
163
+ return ctx.poolConfig;
164
+ }
165
+
155
166
  /**
156
167
  * The live pooled `Backend` instance for an id, or `null` if it isn't currently
157
168
  * pooled. Unlike `getBackendForRole`/`getBackendForChat` this never initialises