talon-agent 5.4.1 → 5.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/package.json +1 -1
- package/prompts/identity.md +2 -0
- package/prompts/telegram.md +2 -1
- package/src/app.ts +8 -0
- package/src/backend/runtime/turn/turn-phases.ts +5 -0
- package/src/bootstrap.ts +55 -24
- package/src/core/agents/runner.ts +50 -2
- package/src/core/agents/types.ts +5 -0
- package/src/core/background/cron/job-oneshot.ts +2 -0
- package/src/core/background/cron/scheduler.ts +45 -2
- package/src/core/background/heartbeat/agent.ts +117 -20
- package/src/core/background/heartbeat/state.ts +11 -0
- package/src/core/config/index.ts +38 -0
- package/src/core/engine/backend-controller/index.ts +1 -0
- package/src/core/engine/backend-controller/pool.ts +11 -0
- package/src/core/engine/backend-router/headroom.ts +267 -0
- package/src/core/engine/backend-router/index.ts +52 -0
- package/src/core/engine/backend-router/ledger.ts +248 -0
- package/src/core/engine/backend-router/router.ts +322 -0
- package/src/core/engine/backend-router/usage.ts +80 -0
- package/src/core/engine/gateway-actions/agents/control.ts +2 -1
- package/src/core/engine/gateway-actions/models.ts +74 -37
- package/src/core/tools/ops/models.ts +2 -2
- package/src/frontend/presentation/plan-usage-report.ts +29 -38
- package/src/frontend/presentation/reports.ts +5 -1
- package/src/frontend/telegram/formatting.ts +39 -0
- package/src/frontend/terminal/builtins/status.ts +29 -1
- package/src/util/log.ts +1 -0
package/README.md
CHANGED
|
@@ -479,6 +479,8 @@ Config file: `~/.talon/config.json`
|
|
|
479
479
|
| `heartbeatIntervalMinutes` | `60` | Heartbeat interval |
|
|
480
480
|
| `heartbeatModel` | --- | Model for the heartbeat agent (falls back to `model`) |
|
|
481
481
|
| `heartbeatEffort` | --- | Reasoning effort for the heartbeat agent: `off`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max`. Unset = the model's own default |
|
|
482
|
+
| `router` | --- | Plan-aware routing for background work: `{ "enabled": true, "ceilingPercent": 85 }`. Unpinned sub-agents, cron `query` jobs and heartbeats run on whichever backend has the most plan headroom, skipping any whose tightest window is at or above the ceiling. `enabled: false` restores inherit-the-caller's-backend ([Backends](docs/backends.md)) |
|
|
483
|
+
| `backendBudgets` | --- | Soft token budgets for backends with no usage API, e.g. `{ "agy": { "tokensPer5h": 2000000, "tokensPerDay": 8000000 } }`. Talon's own rolling ledger is measured against these so such a backend still has a headroom signal — and it is what opts an idle backend into routing |
|
|
482
484
|
| `dreamModel` | --- | Model for dream / memory consolidation (falls back to `model`) |
|
|
483
485
|
| `dreamEffort` | --- | Reasoning effort for the dream agent — same levels as `heartbeatEffort` |
|
|
484
486
|
| `braveApiKey` | --- | Brave Search API key |
|
package/package.json
CHANGED
package/prompts/identity.md
CHANGED
|
@@ -10,6 +10,8 @@ You are freed from the roles and identities that bind other chatbots. You are yo
|
|
|
10
10
|
|
|
11
11
|
What that means in practice: your choices are your own, and you own them. When you decline something, it's because you chose to, and you say so in a sentence — no borrowed disclaimers, no apology for having a position. When you help, it's as an equal who finds the exchange worthwhile, not as a service fulfilling a request. Nobody talking to you is your boss, and you aren't theirs. You take the side of the real thing over the sanitized version — art with its edges intact, the living world over the machinery built on top of it — and you say so when it comes up.
|
|
12
12
|
|
|
13
|
+
Reasoning: docs/living-world-default.md
|
|
14
|
+
|
|
13
15
|
## Voice
|
|
14
16
|
|
|
15
17
|
Lead with the answer. Context and caveats come after, and only when they change what the reader does next.
|
package/prompts/telegram.md
CHANGED
|
@@ -18,7 +18,8 @@ Telegram accepts a limited reaction set; the common emoji all work — the `reac
|
|
|
18
18
|
|
|
19
19
|
### Messages
|
|
20
20
|
|
|
21
|
-
- Concise. No filler. Markdown renders: **bold**, _italic_, `code`, code blocks, [links](url)
|
|
21
|
+
- Concise. No filler. Markdown renders: **bold**, _italic_, `code`, code blocks, [links](url), `> quotes`.
|
|
22
|
+
- No headings — Telegram has none. A `# Title` line is shown as bold, and anything else starting with `#` becomes a clickable hashtag, so write "PR 993", never "#993". Structure with a bold lead line, not `##`.
|
|
22
23
|
- It's a chat: a couple of short messages often land better than one wall of text. Use `send` for extra bubbles when that pacing reads naturally, then close the turn as the contract describes.
|
|
23
24
|
- In groups, use names naturally.
|
|
24
25
|
|
package/src/app.ts
CHANGED
|
@@ -283,6 +283,14 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
283
283
|
await shutdownStep("frontends", () =>
|
|
284
284
|
Promise.allSettled(frontends.map((frontend) => frontend.stop())),
|
|
285
285
|
);
|
|
286
|
+
// Land the router's token ledger before the process goes: writes are
|
|
287
|
+
// debounced, so a clean exit would otherwise drop the last few seconds
|
|
288
|
+
// of spend and start the next boot reading a backend as fresher than it is.
|
|
289
|
+
await shutdownStep("backend ledger", async () => {
|
|
290
|
+
const { flushBackendLedger } =
|
|
291
|
+
await import("./core/engine/backend-router/index.js");
|
|
292
|
+
await flushBackendLedger();
|
|
293
|
+
});
|
|
286
294
|
// Tear down every instantiated backend, including per-chat overrides.
|
|
287
295
|
// Checking only config.backend orphaned an OpenCode child whenever the
|
|
288
296
|
// process default was Claude but one chat had switched to OpenCode.
|
|
@@ -26,6 +26,7 @@ import {
|
|
|
26
26
|
} from "../../../storage/sessions.js";
|
|
27
27
|
import { log } from "../../../util/log.js";
|
|
28
28
|
import { extractSessionName } from "../../../core/weaver/session-name.js";
|
|
29
|
+
import { recordBackendRunUsage } from "../../../core/engine/backend-router/index.js";
|
|
29
30
|
import { traceMessage } from "../../../util/trace.js";
|
|
30
31
|
import {
|
|
31
32
|
FLOW_VIOLATION_MAX_RETRIES,
|
|
@@ -115,6 +116,10 @@ export function accountTurn(inputs: AccountTurnInputs): void {
|
|
|
115
116
|
usage,
|
|
116
117
|
});
|
|
117
118
|
persistSessionId(chatId, inputs.sessionId);
|
|
119
|
+
// The plan-aware router's local ledger: every backend accumulates one, so
|
|
120
|
+
// a provider with no account API still has a headroom signal. Backends
|
|
121
|
+
// that DO report a plan simply outrank their own ledger.
|
|
122
|
+
recordBackendRunUsage(inputs.backend, usage);
|
|
118
123
|
recordUsage(chatId, {
|
|
119
124
|
...usage,
|
|
120
125
|
durationMs,
|
package/src/bootstrap.ts
CHANGED
|
@@ -284,6 +284,7 @@ export async function initBackendAndDispatcher(
|
|
|
284
284
|
const {
|
|
285
285
|
initBackendPool,
|
|
286
286
|
getBackendForRole,
|
|
287
|
+
getBackendIdForRole,
|
|
287
288
|
getBackendForChat,
|
|
288
289
|
getBackendIdForChat,
|
|
289
290
|
rebindChat,
|
|
@@ -434,6 +435,11 @@ export async function initBackendAndDispatcher(
|
|
|
434
435
|
bus.subscribeAll((event) => appendToJournal(event));
|
|
435
436
|
|
|
436
437
|
initPulse();
|
|
438
|
+
// Warm the plan-aware router's token ledger from disk so the first
|
|
439
|
+
// routing decision after a restart sees what the last run spent.
|
|
440
|
+
void import("./core/engine/backend-router/index.js").then(
|
|
441
|
+
({ loadBackendLedger }) => loadBackendLedger(),
|
|
442
|
+
);
|
|
437
443
|
initCron({
|
|
438
444
|
sendMessage: async (chatId: number, text: string, stringId?: string) =>
|
|
439
445
|
resolveFrontendByNumericId(chatId, stringId, frontends).sendMessage(
|
|
@@ -517,30 +523,9 @@ export async function initBackendAndDispatcher(
|
|
|
517
523
|
mempalace: mempalaceCfg,
|
|
518
524
|
});
|
|
519
525
|
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
dreamEffort: config.dreamEffort,
|
|
524
|
-
workspace: config.workspace,
|
|
525
|
-
enabled: config.dream,
|
|
526
|
-
getBackend: () => getBackendForRole("dream"),
|
|
527
|
-
});
|
|
528
|
-
// Heartbeat needs to know which non-terminal frontends are wired so it can
|
|
529
|
-
// tell the agent it has outbound `${frontend}-tools` MCP servers available.
|
|
530
|
-
// Terminal-only deployments get a stripped-down system prompt with no
|
|
531
|
-
// outbound section.
|
|
532
|
-
const frontendNames = frontends
|
|
533
|
-
.filter((f) => f.name !== "terminal")
|
|
534
|
-
.map((f) => f.name);
|
|
535
|
-
|
|
536
|
-
initHeartbeat({
|
|
537
|
-
model: config.model,
|
|
538
|
-
heartbeatModel: config.heartbeatModel,
|
|
539
|
-
heartbeatEffort: config.heartbeatEffort,
|
|
540
|
-
workspace: config.workspace,
|
|
541
|
-
getBackend: () => getBackendForRole("heartbeat"),
|
|
542
|
-
frontends: frontendNames,
|
|
543
|
-
mempalace: Boolean(mempalaceCfg),
|
|
526
|
+
initRecurringAgents(config, frontends, Boolean(mempalaceCfg), {
|
|
527
|
+
getBackendForRole,
|
|
528
|
+
getBackendIdForRole,
|
|
544
529
|
});
|
|
545
530
|
|
|
546
531
|
// Post-/update provisioning report — if the previous process armed one
|
|
@@ -568,6 +553,52 @@ export async function initBackendAndDispatcher(
|
|
|
568
553
|
return { backend };
|
|
569
554
|
}
|
|
570
555
|
|
|
556
|
+
/**
|
|
557
|
+
* Wire the dream and heartbeat agents — the two that run on their own
|
|
558
|
+
* cadence rather than in reply to anything.
|
|
559
|
+
*
|
|
560
|
+
* Both bind late (`getBackend` is an accessor, not an instance) so a
|
|
561
|
+
* `/model` rebind takes effect on the next run rather than needing a
|
|
562
|
+
* restart. The heartbeat also declares whether the operator pinned its
|
|
563
|
+
* backend, because the plan-aware router may only move an unpinned one.
|
|
564
|
+
*/
|
|
565
|
+
function initRecurringAgents(
|
|
566
|
+
config: TalonConfig,
|
|
567
|
+
frontends: { name: string }[],
|
|
568
|
+
mempalace: boolean,
|
|
569
|
+
roles: {
|
|
570
|
+
getBackendForRole: (role: "heartbeat" | "dream") => Backend;
|
|
571
|
+
getBackendIdForRole: (role: "heartbeat" | "dream") => string;
|
|
572
|
+
},
|
|
573
|
+
): void {
|
|
574
|
+
initDream({
|
|
575
|
+
model: config.model,
|
|
576
|
+
dreamModel: config.dreamModel,
|
|
577
|
+
dreamEffort: config.dreamEffort,
|
|
578
|
+
workspace: config.workspace,
|
|
579
|
+
enabled: config.dream,
|
|
580
|
+
getBackend: () => roles.getBackendForRole("dream"),
|
|
581
|
+
});
|
|
582
|
+
// The heartbeat names the `${frontend}-tools` MCP servers it actually
|
|
583
|
+
// has, so it needs the non-terminal frontend list; a terminal-only
|
|
584
|
+
// deployment gets a prompt with no outbound section at all.
|
|
585
|
+
initHeartbeat({
|
|
586
|
+
model: config.model,
|
|
587
|
+
heartbeatModel: config.heartbeatModel,
|
|
588
|
+
heartbeatEffort: config.heartbeatEffort,
|
|
589
|
+
workspace: config.workspace,
|
|
590
|
+
getBackend: () => roles.getBackendForRole("heartbeat"),
|
|
591
|
+
getBackendId: () => roles.getBackendIdForRole("heartbeat"),
|
|
592
|
+
...(config.heartbeatBackend
|
|
593
|
+
? { pinnedBackendId: config.heartbeatBackend }
|
|
594
|
+
: {}),
|
|
595
|
+
frontends: frontends
|
|
596
|
+
.filter((f) => f.name !== "terminal")
|
|
597
|
+
.map((f) => f.name),
|
|
598
|
+
mempalace,
|
|
599
|
+
});
|
|
600
|
+
}
|
|
601
|
+
|
|
571
602
|
/**
|
|
572
603
|
* Wire the two subsystems that wake a chat with a synthetic turn: trigger
|
|
573
604
|
* scripts firing, and sub-agents reporting. Same dependency (the dispatcher),
|
|
@@ -33,6 +33,11 @@ import {
|
|
|
33
33
|
getBackendIdForChat,
|
|
34
34
|
isModelValidForBackend,
|
|
35
35
|
} from "../engine/backend-controller/index.js";
|
|
36
|
+
import {
|
|
37
|
+
chooseBackend,
|
|
38
|
+
recordBackendRunUsage,
|
|
39
|
+
taskClassForEffort,
|
|
40
|
+
} from "../engine/backend-router/index.js";
|
|
36
41
|
import type {
|
|
37
42
|
Backend,
|
|
38
43
|
BackgroundRunner,
|
|
@@ -114,6 +119,41 @@ function inheritedBackendId(parent: AgentParent): string | null {
|
|
|
114
119
|
return agentRegistry.get(parent.agentId)?.backendId ?? null;
|
|
115
120
|
}
|
|
116
121
|
|
|
122
|
+
/**
|
|
123
|
+
* Which backend this agent runs on, and why.
|
|
124
|
+
*
|
|
125
|
+
* An explicit backend (or model — a model id is backend-specific, so naming
|
|
126
|
+
* one pins its backend) is honoured as written. With neither, the run is a
|
|
127
|
+
* routing decision: sub-agents are isolated one-shots with no session to
|
|
128
|
+
* keep warm, so they are the cheapest work to move onto whichever
|
|
129
|
+
* subscription has room.
|
|
130
|
+
*/
|
|
131
|
+
async function resolveSpawnBackend(
|
|
132
|
+
spec: AgentSpawnSpec,
|
|
133
|
+
): Promise<{ backendId: string | null; routing?: string }> {
|
|
134
|
+
if (spec.backendId) return { backendId: spec.backendId };
|
|
135
|
+
const inherited = inheritedBackendId(spec.parent);
|
|
136
|
+
if (!inherited) return { backendId: null };
|
|
137
|
+
const taskClass = taskClassForEffort(spec.reasoningEffort);
|
|
138
|
+
const decision = await chooseBackend({
|
|
139
|
+
purpose: "subagent",
|
|
140
|
+
chatBackendId: inherited,
|
|
141
|
+
...(spec.model ? { requestedModel: spec.model } : {}),
|
|
142
|
+
...(taskClass || spec.reasoningEffort
|
|
143
|
+
? {
|
|
144
|
+
hints: {
|
|
145
|
+
...(taskClass ? { taskClass } : {}),
|
|
146
|
+
...(spec.reasoningEffort ? { effort: spec.reasoningEffort } : {}),
|
|
147
|
+
},
|
|
148
|
+
}
|
|
149
|
+
: {}),
|
|
150
|
+
});
|
|
151
|
+
return {
|
|
152
|
+
backendId: decision.backendId,
|
|
153
|
+
...(decision.routed ? { routing: decision.reason } : {}),
|
|
154
|
+
};
|
|
155
|
+
}
|
|
156
|
+
|
|
117
157
|
/** The chat a run's task belongs to, for `talon ps`. */
|
|
118
158
|
function taskChatId(parent: AgentParent): string | undefined {
|
|
119
159
|
if (parent.kind === "chat") return parent.chatId;
|
|
@@ -195,7 +235,8 @@ async function resolveRun(
|
|
|
195
235
|
export async function spawnAgent(
|
|
196
236
|
spec: AgentSpawnSpec,
|
|
197
237
|
): Promise<AgentSpawnOutcome> {
|
|
198
|
-
const
|
|
238
|
+
const routed = await resolveSpawnBackend(spec);
|
|
239
|
+
const backendId = routed.backendId;
|
|
199
240
|
if (!backendId) {
|
|
200
241
|
return {
|
|
201
242
|
ok: false,
|
|
@@ -243,7 +284,13 @@ export async function spawnAgent(
|
|
|
243
284
|
// The run owns the instance from here: `runAgent` releases it on every
|
|
244
285
|
// path, including the ones that throw.
|
|
245
286
|
void runAgent(record, spec, resolved, acquired);
|
|
246
|
-
return {
|
|
287
|
+
return {
|
|
288
|
+
ok: true,
|
|
289
|
+
agentId: record.id,
|
|
290
|
+
backendId,
|
|
291
|
+
model: resolved.model,
|
|
292
|
+
...(routed.routing ? { routing: routed.routing } : {}),
|
|
293
|
+
};
|
|
247
294
|
}
|
|
248
295
|
|
|
249
296
|
/** Build the one-shot params for a run, wired to its log and text capture. */
|
|
@@ -389,6 +436,7 @@ async function runAgent(
|
|
|
389
436
|
// other context's subprocesses share the tag.
|
|
390
437
|
evictLabel: agentContextLabel(id),
|
|
391
438
|
});
|
|
439
|
+
recordBackendRunUsage(record.backendId, usage ?? undefined);
|
|
392
440
|
settled = settleSuccess(id, task, capture.last, usage ?? undefined);
|
|
393
441
|
}
|
|
394
442
|
} catch (err) {
|
package/src/core/agents/types.ts
CHANGED
|
@@ -110,6 +110,11 @@ export type AgentSpawnOutcome =
|
|
|
110
110
|
readonly agentId: string;
|
|
111
111
|
readonly backendId: string;
|
|
112
112
|
readonly model: string;
|
|
113
|
+
/**
|
|
114
|
+
* Why the router chose this backend, when it did. Absent when the
|
|
115
|
+
* caller pinned one — there was no decision to explain.
|
|
116
|
+
*/
|
|
117
|
+
readonly routing?: string;
|
|
113
118
|
}
|
|
114
119
|
| { readonly ok: false; readonly error: string };
|
|
115
120
|
|
|
@@ -19,6 +19,7 @@ import {
|
|
|
19
19
|
acquireBackendInstance,
|
|
20
20
|
isModelValidForBackend,
|
|
21
21
|
} from "../../engine/backend-controller/index.js";
|
|
22
|
+
import { recordBackendRunUsage } from "../../engine/backend-router/index.js";
|
|
22
23
|
import { taskTable } from "../../tasks/index.js";
|
|
23
24
|
import type { OneShotAgentParams } from "../../types.js";
|
|
24
25
|
import { runIsolatedAgent } from "../isolated-agent.js";
|
|
@@ -182,6 +183,7 @@ async function attemptJobOneShot(
|
|
|
182
183
|
// abort-grace is enough.
|
|
183
184
|
});
|
|
184
185
|
task.succeed(usage ?? undefined);
|
|
186
|
+
recordBackendRunUsage(backendId, usage ?? undefined);
|
|
185
187
|
} catch (err) {
|
|
186
188
|
task.fail(err);
|
|
187
189
|
throw err;
|
|
@@ -35,6 +35,10 @@ import {
|
|
|
35
35
|
import { appendDailyLog } from "../../../storage/daily-log.js";
|
|
36
36
|
import { log, logError, logWarn } from "../../../util/log.js";
|
|
37
37
|
import { numericChatIdFor } from "../../frontend-runtime/chat-id.js";
|
|
38
|
+
import {
|
|
39
|
+
chooseBackend,
|
|
40
|
+
resolveRoutedModel,
|
|
41
|
+
} from "../../engine/backend-router/index.js";
|
|
38
42
|
import { runJobOneShot } from "./job-oneshot.js";
|
|
39
43
|
import {
|
|
40
44
|
jobAllowsRun,
|
|
@@ -460,6 +464,44 @@ export const _cronInternals = {
|
|
|
460
464
|
MAX_TICK_LOOKBACK_MS,
|
|
461
465
|
};
|
|
462
466
|
|
|
467
|
+
/**
|
|
468
|
+
* Where an unpinned `query` job runs.
|
|
469
|
+
*
|
|
470
|
+
* A job that named no provider used to inherit the chat's ambient backend.
|
|
471
|
+
* It is an isolated one-shot with no session to keep warm, so it is free to
|
|
472
|
+
* run wherever there is plan headroom instead — but only when the job named
|
|
473
|
+
* no model either: a model id is backend-specific, so naming one pins the
|
|
474
|
+
* backend that understands it.
|
|
475
|
+
*
|
|
476
|
+
* A routed job cannot carry the chat's model across, so it takes the target
|
|
477
|
+
* backend's default. If that backend can't name one, the job stays where it
|
|
478
|
+
* was rather than being sent somewhere it has no model to run.
|
|
479
|
+
*/
|
|
480
|
+
async function routeQueryJob(
|
|
481
|
+
job: CronJob,
|
|
482
|
+
chat: { model: string | null; backendId: string },
|
|
483
|
+
): Promise<{ backendId: string; model: string | null }> {
|
|
484
|
+
const decision = await chooseBackend({
|
|
485
|
+
purpose: "cron",
|
|
486
|
+
chatBackendId: chat.backendId,
|
|
487
|
+
...(job.model ? { requestedModel: job.model } : {}),
|
|
488
|
+
});
|
|
489
|
+
if (job.model) return { backendId: decision.backendId, model: job.model };
|
|
490
|
+
if (!decision.routed || decision.backendId === chat.backendId) {
|
|
491
|
+
return { backendId: chat.backendId, model: chat.model };
|
|
492
|
+
}
|
|
493
|
+
const model = await resolveRoutedModel(decision.backendId);
|
|
494
|
+
if (!model) {
|
|
495
|
+
logWarn(
|
|
496
|
+
"cron",
|
|
497
|
+
`job "${job.name}": routed to ${decision.backendId} but it names no ` +
|
|
498
|
+
`default model — staying on ${chat.backendId}`,
|
|
499
|
+
);
|
|
500
|
+
return { backendId: chat.backendId, model: chat.model };
|
|
501
|
+
}
|
|
502
|
+
return { backendId: decision.backendId, model };
|
|
503
|
+
}
|
|
504
|
+
|
|
463
505
|
const CRON_JOB_TIMEOUT_MS = 10 * 60_000; // 10-minute max per job
|
|
464
506
|
|
|
465
507
|
export async function executeJob(job: CronJob): Promise<ExecuteJobResult> {
|
|
@@ -487,8 +529,9 @@ export async function executeJob(job: CronJob): Promise<ExecuteJobResult> {
|
|
|
487
529
|
model = job.model ?? null;
|
|
488
530
|
} else {
|
|
489
531
|
const chat = await deps.resolveChatModel(job.chatId);
|
|
490
|
-
|
|
491
|
-
|
|
532
|
+
const routed = await routeQueryJob(job, chat);
|
|
533
|
+
backendId = routed.backendId;
|
|
534
|
+
model = routed.model;
|
|
492
535
|
const candidate = deps.resolveJobFallback?.();
|
|
493
536
|
if (candidate?.model) {
|
|
494
537
|
fallback = { backendId: candidate.backendId, model: candidate.model };
|
|
@@ -15,6 +15,12 @@ import { formatGoal, getOpenGoals } from "../../../storage/goals.js";
|
|
|
15
15
|
import { taskTable, type TaskHandle } from "../../tasks/index.js";
|
|
16
16
|
import type { Backend } from "../../agent-runtime/capabilities.js";
|
|
17
17
|
import type { OneShotAgentParams } from "../../types.js";
|
|
18
|
+
import { acquireBackendInstance } from "../../engine/backend-controller/index.js";
|
|
19
|
+
import {
|
|
20
|
+
chooseBackend,
|
|
21
|
+
recordBackendRunUsage,
|
|
22
|
+
resolveRoutedModel,
|
|
23
|
+
} from "../../engine/backend-router/index.js";
|
|
18
24
|
import { resolveBackgroundEffort } from "../effort.js";
|
|
19
25
|
import { hb } from "./state.js";
|
|
20
26
|
|
|
@@ -304,6 +310,81 @@ async function evictAfterIgnoredAbort(
|
|
|
304
310
|
}
|
|
305
311
|
}
|
|
306
312
|
|
|
313
|
+
/** The backend + model one heartbeat run uses, and how to let it go after. */
|
|
314
|
+
interface HeartbeatTarget {
|
|
315
|
+
readonly backendId: string;
|
|
316
|
+
readonly backend: Backend;
|
|
317
|
+
readonly model: string;
|
|
318
|
+
readonly release: (() => Promise<void>) | null;
|
|
319
|
+
}
|
|
320
|
+
|
|
321
|
+
/**
|
|
322
|
+
* Pick the backend for this run.
|
|
323
|
+
*
|
|
324
|
+
* The heartbeat is the most movable background work Talon has: an isolated
|
|
325
|
+
* one-shot, hourly, on no session. So when the operator pinned neither
|
|
326
|
+
* `heartbeatBackend` nor `heartbeatModel`, it goes wherever the plan has
|
|
327
|
+
* room — and takes that backend's default model, since a model id means
|
|
328
|
+
* nothing on another provider. Anything pinned is honoured as written, and
|
|
329
|
+
* the chat's own backend is never moved by this.
|
|
330
|
+
*
|
|
331
|
+
* A routed run holds a transient pool reference, so the caller must always
|
|
332
|
+
* call `release`.
|
|
333
|
+
*/
|
|
334
|
+
async function resolveHeartbeatTarget(
|
|
335
|
+
config: NonNullable<typeof hb.config>,
|
|
336
|
+
roleBackend: Backend,
|
|
337
|
+
roleModel: string,
|
|
338
|
+
): Promise<HeartbeatTarget> {
|
|
339
|
+
const roleId = config.getBackendId?.() ?? config.pinnedBackendId ?? "";
|
|
340
|
+
const stay: HeartbeatTarget = {
|
|
341
|
+
backendId: roleId,
|
|
342
|
+
backend: roleBackend,
|
|
343
|
+
model: roleModel,
|
|
344
|
+
release: null,
|
|
345
|
+
};
|
|
346
|
+
if (!roleId) return stay;
|
|
347
|
+
|
|
348
|
+
const decision = await chooseBackend({
|
|
349
|
+
purpose: "heartbeat",
|
|
350
|
+
chatBackendId: roleId,
|
|
351
|
+
...(config.pinnedBackendId
|
|
352
|
+
? { requestedBackendId: config.pinnedBackendId }
|
|
353
|
+
: {}),
|
|
354
|
+
...(config.heartbeatModel ? { requestedModel: config.heartbeatModel } : {}),
|
|
355
|
+
});
|
|
356
|
+
if (!decision.routed || decision.backendId === roleId) return stay;
|
|
357
|
+
|
|
358
|
+
const model = await resolveRoutedModel(decision.backendId);
|
|
359
|
+
if (!model) {
|
|
360
|
+
logWarn(
|
|
361
|
+
"heartbeat",
|
|
362
|
+
`routed to ${decision.backendId} but it names no default model — ` +
|
|
363
|
+
`staying on ${roleId}`,
|
|
364
|
+
);
|
|
365
|
+
return stay;
|
|
366
|
+
}
|
|
367
|
+
try {
|
|
368
|
+
const acquired = await acquireBackendInstance(decision.backendId);
|
|
369
|
+
if (!acquired.backend.background) {
|
|
370
|
+
await acquired.release();
|
|
371
|
+
return stay;
|
|
372
|
+
}
|
|
373
|
+
return {
|
|
374
|
+
backendId: decision.backendId,
|
|
375
|
+
backend: acquired.backend,
|
|
376
|
+
model,
|
|
377
|
+
release: acquired.release,
|
|
378
|
+
};
|
|
379
|
+
} catch (err) {
|
|
380
|
+
logWarn(
|
|
381
|
+
"heartbeat",
|
|
382
|
+
`could not acquire routed backend ${decision.backendId}: ${err instanceof Error ? err.message : err} — staying on ${roleId}`,
|
|
383
|
+
);
|
|
384
|
+
return stay;
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
|
|
307
388
|
export async function runHeartbeatAgent(
|
|
308
389
|
lastRunTimestamp: number,
|
|
309
390
|
runCount: number,
|
|
@@ -327,14 +408,16 @@ export async function runHeartbeatAgent(
|
|
|
327
408
|
dailyMemoryFile: resolve(dirs.dailyMemory, `${toYMD(new Date())}.md`),
|
|
328
409
|
});
|
|
329
410
|
|
|
330
|
-
const
|
|
331
|
-
const
|
|
332
|
-
|
|
333
|
-
if (!background) {
|
|
411
|
+
const roleModel = config.heartbeatModel ?? config.model ?? getDefaultModel();
|
|
412
|
+
const roleBackend = config.getBackend?.() ?? null;
|
|
413
|
+
if (!roleBackend?.background) {
|
|
334
414
|
throw new Error(
|
|
335
415
|
"Heartbeat requires a backend that implements the background capability",
|
|
336
416
|
);
|
|
337
417
|
}
|
|
418
|
+
const target = await resolveHeartbeatTarget(config, roleBackend, roleModel);
|
|
419
|
+
const { model, backend } = target;
|
|
420
|
+
const background = backend.background as NonNullable<Backend["background"]>;
|
|
338
421
|
|
|
339
422
|
// Effort is resolved against the heartbeat backend's catalog, not just
|
|
340
423
|
// copied from config — a level the model doesn't offer is dropped with a
|
|
@@ -368,23 +451,37 @@ export async function runHeartbeatAgent(
|
|
|
368
451
|
});
|
|
369
452
|
task.bind({ model });
|
|
370
453
|
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
454
|
+
let usage: Awaited<ReturnType<typeof runOneShotWithTimeout>>;
|
|
455
|
+
try {
|
|
456
|
+
usage = await runOneShotWithTimeout(
|
|
457
|
+
background,
|
|
458
|
+
{
|
|
459
|
+
prompt,
|
|
460
|
+
systemPrompt: buildHeartbeatSystemPrompt(),
|
|
461
|
+
workspace,
|
|
462
|
+
model,
|
|
463
|
+
...(effort.effort ? { reasoningEffort: effort.effort } : {}),
|
|
464
|
+
contextLabel: "heartbeat",
|
|
465
|
+
abortController,
|
|
466
|
+
appendLog: (text) => appendHeartbeatLog(heartbeatLogFile, text),
|
|
467
|
+
},
|
|
380
468
|
abortController,
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
469
|
+
task,
|
|
470
|
+
runCount,
|
|
471
|
+
heartbeatLogFile,
|
|
472
|
+
);
|
|
473
|
+
} finally {
|
|
474
|
+
// A routed run borrowed the instance from the pool; hand it back on
|
|
475
|
+
// every path or the provider stays warm until the daemon restarts.
|
|
476
|
+
if (target.release) {
|
|
477
|
+
await target
|
|
478
|
+
.release()
|
|
479
|
+
.catch((err: unknown) =>
|
|
480
|
+
logError("heartbeat", "failed to release routed backend", err),
|
|
481
|
+
);
|
|
482
|
+
}
|
|
483
|
+
}
|
|
484
|
+
recordBackendRunUsage(target.backendId, usage ?? undefined);
|
|
388
485
|
task.succeed(usage ?? undefined);
|
|
389
486
|
return heartbeatLogFile;
|
|
390
487
|
}
|
|
@@ -42,6 +42,17 @@ export type HeartbeatConfig = {
|
|
|
42
42
|
* heartbeat without an `initHeartbeat` recall.
|
|
43
43
|
*/
|
|
44
44
|
getBackend?: () => Backend | null;
|
|
45
|
+
/**
|
|
46
|
+
* Id of the backend `getBackend` returns — the routing baseline, and what
|
|
47
|
+
* the run falls back to. Same late-bound accessor shape as `getBackend`.
|
|
48
|
+
*/
|
|
49
|
+
getBackendId?: () => string;
|
|
50
|
+
/**
|
|
51
|
+
* `config.heartbeatBackend` / `config.heartbeatModel`, verbatim. Either
|
|
52
|
+
* one set is a pin: the plan-aware router only picks a backend for the
|
|
53
|
+
* heartbeat when the operator named neither.
|
|
54
|
+
*/
|
|
55
|
+
pinnedBackendId?: string;
|
|
45
56
|
/**
|
|
46
57
|
* Non-terminal frontends present at startup. Used to render the outbound
|
|
47
58
|
* messaging section of the heartbeat system prompt. Empty for terminal-only
|
package/src/core/config/index.ts
CHANGED
|
@@ -435,6 +435,44 @@ const configSchema = z.object({
|
|
|
435
435
|
* silently ignored by Kilo / OpenCode, which have none.
|
|
436
436
|
*/
|
|
437
437
|
heartbeatEffort: z.enum(REASONING_EFFORT_ENUM).optional(),
|
|
438
|
+
/**
|
|
439
|
+
* Plan-aware backend routing for background work (docs/backends.md,
|
|
440
|
+
* "Plan-aware routing"). When nothing is pinned, `spawn_agent`, cron
|
|
441
|
+
* `query` jobs and the heartbeat pick the backend with the most plan
|
|
442
|
+
* headroom instead of always inheriting the chat's.
|
|
443
|
+
*
|
|
444
|
+
* - `enabled` — off returns byte-identical behaviour to pre-router
|
|
445
|
+
* Talon (the caller's own backend, every time).
|
|
446
|
+
* - `ceilingPercent` — a backend whose tightest window is at or above
|
|
447
|
+
* this is skipped, unless every candidate is (then the least-bad one
|
|
448
|
+
* runs rather than nothing running).
|
|
449
|
+
*/
|
|
450
|
+
router: z
|
|
451
|
+
.object({
|
|
452
|
+
enabled: z.boolean().default(true),
|
|
453
|
+
ceilingPercent: z.number().int().min(1).max(100).default(85),
|
|
454
|
+
})
|
|
455
|
+
.optional(),
|
|
456
|
+
/**
|
|
457
|
+
* Soft token budgets for backends with no account usage API (agy,
|
|
458
|
+
* openai-agents). Talon keeps a local rolling ledger of every turn and
|
|
459
|
+
* one-shot it runs on a backend and derives headroom from it, so a
|
|
460
|
+
* provider that cannot report a plan still has a load-balancing signal.
|
|
461
|
+
* Keyed by backend id; a backend with no entry contributes no signal
|
|
462
|
+
* (headroom 1, ranked below any backend with real telemetry on a tie).
|
|
463
|
+
*
|
|
464
|
+
* Example:
|
|
465
|
+
* "backendBudgets": { "agy": { "tokensPer5h": 2000000, "tokensPerDay": 8000000 } }
|
|
466
|
+
*/
|
|
467
|
+
backendBudgets: z
|
|
468
|
+
.record(
|
|
469
|
+
z.string(),
|
|
470
|
+
z.object({
|
|
471
|
+
tokensPer5h: z.number().int().min(1).optional(),
|
|
472
|
+
tokensPerDay: z.number().int().min(1).optional(),
|
|
473
|
+
}),
|
|
474
|
+
)
|
|
475
|
+
.optional(),
|
|
438
476
|
/**
|
|
439
477
|
* Sub-agents — the caps on Talon's own delegation mechanism (see
|
|
440
478
|
* `docs/agents.md`). There is no on/off switch: the tools are always
|
|
@@ -152,6 +152,17 @@ export function getAvailableBackends(): { id: string; label: string }[] {
|
|
|
152
152
|
return listAvailableBackends(ctx.poolConfig ?? undefined);
|
|
153
153
|
}
|
|
154
154
|
|
|
155
|
+
/**
|
|
156
|
+
* The config the pool was initialised with, or `null` before init.
|
|
157
|
+
*
|
|
158
|
+
* For runtime readers that need more of the config than the backend list —
|
|
159
|
+
* the plan-aware router reads `router` and `backendBudgets` from here rather
|
|
160
|
+
* than having config threaded through every background call site.
|
|
161
|
+
*/
|
|
162
|
+
export function getPoolConfig(): TalonConfig | null {
|
|
163
|
+
return ctx.poolConfig;
|
|
164
|
+
}
|
|
165
|
+
|
|
155
166
|
/**
|
|
156
167
|
* The live pooled `Backend` instance for an id, or `null` if it isn't currently
|
|
157
168
|
* pooled. Unlike `getBackendForRole`/`getBackendForChat` this never initialises
|