talon-agent 5.4.0 → 5.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/package.json +1 -1
- package/prompts/identity.md +2 -0
- package/prompts/telegram.md +2 -1
- package/src/app.ts +12 -0
- package/src/backend/runtime/turn/turn-phases.ts +5 -0
- package/src/bootstrap.ts +55 -24
- package/src/core/agents/runner.ts +50 -2
- package/src/core/agents/types.ts +5 -0
- package/src/core/background/cron/job-oneshot.ts +2 -0
- package/src/core/background/cron/scheduler.ts +45 -2
- package/src/core/background/heartbeat/agent.ts +117 -20
- package/src/core/background/heartbeat/state.ts +11 -0
- package/src/core/config/index.ts +38 -0
- package/src/core/daemon/respawn.ts +34 -10
- package/src/core/daemon/signals.ts +111 -0
- package/src/core/engine/backend-controller/index.ts +1 -0
- package/src/core/engine/backend-controller/pool.ts +11 -0
- package/src/core/engine/backend-router/headroom.ts +267 -0
- package/src/core/engine/backend-router/index.ts +52 -0
- package/src/core/engine/backend-router/ledger.ts +248 -0
- package/src/core/engine/backend-router/router.ts +322 -0
- package/src/core/engine/backend-router/usage.ts +80 -0
- package/src/core/engine/gateway-actions/agents/control.ts +2 -1
- package/src/core/engine/gateway-actions/models.ts +74 -37
- package/src/core/tools/ops/models.ts +2 -2
- package/src/frontend/presentation/plan-usage-report.ts +29 -38
- package/src/frontend/presentation/reports.ts +5 -1
- package/src/frontend/telegram/formatting.ts +39 -0
- package/src/frontend/terminal/builtins/status.ts +29 -1
- package/src/index.ts +7 -0
- package/src/util/log.ts +1 -0
|
@@ -10,11 +10,33 @@ import {
|
|
|
10
10
|
getBackendForChat,
|
|
11
11
|
getBackendIdForChat,
|
|
12
12
|
getAvailableBackends,
|
|
13
|
+
getPoolConfig,
|
|
13
14
|
getPooledBackend,
|
|
14
15
|
acquireBackendInstance,
|
|
15
16
|
} from "../backend-controller/index.js";
|
|
17
|
+
import {
|
|
18
|
+
collectBackendUsage,
|
|
19
|
+
formatHeadroom,
|
|
20
|
+
leadWith,
|
|
21
|
+
type BackendUsageSnapshot,
|
|
22
|
+
} from "../backend-router/index.js";
|
|
16
23
|
import type { SharedActionHandlers } from "./types.js";
|
|
17
24
|
|
|
25
|
+
/** One `plan_usage` block: the headroom line, then any plan windows. */
|
|
26
|
+
function usageLines(entry: BackendUsageSnapshot): string[] {
|
|
27
|
+
const head = `- ${entry.label || entry.id}: ${formatHeadroom(entry.headroom)}`;
|
|
28
|
+
if (!entry.plan) {
|
|
29
|
+
return [`${head}${entry.note ? ` (${entry.note})` : ""}`];
|
|
30
|
+
}
|
|
31
|
+
return [
|
|
32
|
+
`${head}${entry.plan.plan ? ` · ${entry.plan.plan}` : ""}`,
|
|
33
|
+
...entry.plan.windows.map(
|
|
34
|
+
(w) =>
|
|
35
|
+
` ${w.label}: ${w.percent}% used${w.resetsAt ? `, resets ${w.resetsAt}` : ""}`,
|
|
36
|
+
),
|
|
37
|
+
];
|
|
38
|
+
}
|
|
39
|
+
|
|
18
40
|
export const modelHandlers: SharedActionHandlers = {
|
|
19
41
|
list_models: async (body, chatId, _backend, chatKey) => {
|
|
20
42
|
const chatIdStr = chatKey;
|
|
@@ -98,52 +120,67 @@ export const modelHandlers: SharedActionHandlers = {
|
|
|
98
120
|
}
|
|
99
121
|
},
|
|
100
122
|
|
|
101
|
-
// Account-level
|
|
102
|
-
//
|
|
123
|
+
// Account-level and fleet-wide: every exposed backend, with the headroom
|
|
124
|
+
// figure the plan-aware router ranks on — a backend with no usage API
|
|
125
|
+
// still answers, from its local budget ledger. The chat's own backend
|
|
126
|
+
// leads so a caller reading only the first entry sees what it used to.
|
|
103
127
|
plan_usage: async (_body, chatId, _backend, chatKey) => {
|
|
104
|
-
const
|
|
105
|
-
const
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
if (!source?.usage?.getPlanUsage)
|
|
109
|
-
return {
|
|
110
|
-
ok: false,
|
|
111
|
-
error: "No configured backend reports subscription rate limits.",
|
|
112
|
-
};
|
|
113
|
-
|
|
114
|
-
const usage = await source.usage.getPlanUsage();
|
|
115
|
-
if (!usage)
|
|
116
|
-
return {
|
|
117
|
-
ok: false,
|
|
118
|
-
error:
|
|
119
|
-
"Plan usage is unavailable — no subscription credentials, or this session authenticates with an API key.",
|
|
120
|
-
};
|
|
121
|
-
|
|
122
|
-
const lines = usage.windows.map(
|
|
123
|
-
(w) =>
|
|
124
|
-
`- ${w.label}: ${w.percent}% used${w.resetsAt ? `, resets ${w.resetsAt}` : ""}`,
|
|
128
|
+
const currentId = getBackendIdForChat(chatKey);
|
|
129
|
+
const entries = leadWith(
|
|
130
|
+
await collectBackendUsage(getPoolConfig() ?? undefined, { force: true }),
|
|
131
|
+
currentId,
|
|
125
132
|
);
|
|
133
|
+
if (entries.length === 0)
|
|
134
|
+
return { ok: false, error: "No backends are available." };
|
|
135
|
+
|
|
136
|
+
const lead = entries[0] as BackendUsageSnapshot;
|
|
126
137
|
return {
|
|
127
138
|
ok: true,
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
139
|
+
// Kept for callers written against the single-backend shape.
|
|
140
|
+
plan: lead.plan?.plan ?? null,
|
|
141
|
+
windows: lead.plan?.windows ?? [],
|
|
142
|
+
backends: entries.map((entry) => ({
|
|
143
|
+
id: entry.id,
|
|
144
|
+
label: entry.label,
|
|
145
|
+
current: entry.id === currentId,
|
|
146
|
+
headroom: Math.round(entry.headroom.headroom * 100) / 100,
|
|
147
|
+
source: entry.headroom.source,
|
|
148
|
+
limiting: entry.headroom.limiting ?? null,
|
|
149
|
+
stale: entry.headroom.stale ?? false,
|
|
150
|
+
plan: entry.plan?.plan ?? null,
|
|
151
|
+
windows: entry.plan?.windows ?? [],
|
|
152
|
+
...(entry.note ? { note: entry.note } : {}),
|
|
153
|
+
})),
|
|
154
|
+
text: `Plan usage and headroom by backend:\n${entries.flatMap(usageLines).join("\n")}`,
|
|
131
155
|
};
|
|
132
156
|
},
|
|
133
157
|
|
|
134
|
-
list_backends: (body, chatId, _backend, chatKey) => {
|
|
158
|
+
list_backends: async (body, chatId, _backend, chatKey) => {
|
|
135
159
|
const currentId = getBackendIdForChat(chatKey);
|
|
136
|
-
const
|
|
137
|
-
|
|
138
|
-
label: b.label,
|
|
139
|
-
current: b.id === currentId,
|
|
140
|
-
}));
|
|
141
|
-
if (backends.length === 0)
|
|
160
|
+
const available = getAvailableBackends();
|
|
161
|
+
if (available.length === 0)
|
|
142
162
|
return { ok: true, backends: [], text: "No backends are available." };
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
);
|
|
163
|
+
// Headroom comes off the router's 60s cache, so listing backends is
|
|
164
|
+
// cheap even though it now answers "which one has room?" as well.
|
|
165
|
+
const usage = await collectBackendUsage(getPoolConfig() ?? undefined);
|
|
166
|
+
const byId = new Map(usage.map((entry) => [entry.id, entry]));
|
|
167
|
+
const backends = available.map((b) => {
|
|
168
|
+
const entry = byId.get(b.id);
|
|
169
|
+
return {
|
|
170
|
+
id: b.id,
|
|
171
|
+
label: b.label,
|
|
172
|
+
current: b.id === currentId,
|
|
173
|
+
headroom: entry
|
|
174
|
+
? Math.round(entry.headroom.headroom * 100) / 100
|
|
175
|
+
: null,
|
|
176
|
+
headroomSource: entry?.headroom.source ?? null,
|
|
177
|
+
};
|
|
178
|
+
});
|
|
179
|
+
const lines = backends.map((b) => {
|
|
180
|
+
const entry = byId.get(b.id);
|
|
181
|
+
const room = entry ? ` — ${formatHeadroom(entry.headroom)} free` : "";
|
|
182
|
+
return `- ${b.id}${b.label && b.label !== b.id ? ` (${b.label})` : ""}${b.current ? " — current" : ""}${room}`;
|
|
183
|
+
});
|
|
147
184
|
return {
|
|
148
185
|
ok: true,
|
|
149
186
|
backends,
|
|
@@ -31,7 +31,7 @@ export const modelTools: ToolDefinition[] = [
|
|
|
31
31
|
{
|
|
32
32
|
name: "plan_usage",
|
|
33
33
|
description:
|
|
34
|
-
"Read
|
|
34
|
+
"Read usage and headroom for EVERY configured backend, not just this chat's: how much of each subscription's rate-limit windows is spent, when they reset, and one comparable headroom figure per backend. A backend with no usage API reports headroom from Talon's own local token ledger against its configured budget (marked 'local budget'), and one with neither says so. Use it before starting long or expensive work, when deciding whether to defer something, or to see which backend background work will be routed to.",
|
|
35
35
|
schema: {},
|
|
36
36
|
execute: (_params, bridge) => bridge("plan_usage", {}),
|
|
37
37
|
tag: "models",
|
|
@@ -40,7 +40,7 @@ export const modelTools: ToolDefinition[] = [
|
|
|
40
40
|
{
|
|
41
41
|
name: "list_backends",
|
|
42
42
|
description:
|
|
43
|
-
"List the available backends (providers)
|
|
43
|
+
"List the available backends (providers), which one this chat is currently using, and how much plan headroom each has left. Useful for understanding the model/provider landscape and for seeing where unpinned background work (sub-agents, cron query jobs, the heartbeat) will be routed. Note: a per-job `model` override must stay on this chat's current backend, so use list_models (no argument) to pick a model for a trigger or cron job.",
|
|
44
44
|
schema: {},
|
|
45
45
|
execute: (_params, bridge) => bridge("list_backends", {}),
|
|
46
46
|
tag: "models",
|
|
@@ -6,15 +6,21 @@
|
|
|
6
6
|
* through whichever provider it fronts, and an API-key install pays per
|
|
7
7
|
* token with no window to be near the end of. Those are listed with a
|
|
8
8
|
* reason rather than omitted, so the answer to "am I close to a limit?" is
|
|
9
|
-
* never silence
|
|
9
|
+
* never silence — and since the plan-aware router landed they carry a
|
|
10
|
+
* headroom figure too, derived from a local token budget where one is
|
|
11
|
+
* configured, so "which backend has room?" is answerable for all of them.
|
|
12
|
+
*
|
|
13
|
+
* The gathering itself lives in core (`engine/backend-router/usage.ts`),
|
|
14
|
+
* because the gateway tools need the same data and core cannot import a
|
|
15
|
+
* frontend. This module is the rendering adapter over it.
|
|
10
16
|
*/
|
|
11
17
|
|
|
12
18
|
import type { TalonConfig } from "../../core/config/index.js";
|
|
13
|
-
import type { PlanUsage } from "../../core/agent-runtime/capabilities.js";
|
|
14
19
|
import {
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
20
|
+
collectBackendUsage,
|
|
21
|
+
formatHeadroom,
|
|
22
|
+
type BackendHeadroom,
|
|
23
|
+
} from "../../core/engine/backend-router/index.js";
|
|
18
24
|
import { buildPlanDisplay, type PlanDisplay } from "./status-context.js";
|
|
19
25
|
|
|
20
26
|
export interface BackendUsageEntry {
|
|
@@ -24,6 +30,10 @@ export interface BackendUsageEntry {
|
|
|
24
30
|
plan: PlanDisplay | null;
|
|
25
31
|
/** Why there is nothing to show. Absent when `plan` is set. */
|
|
26
32
|
note?: string;
|
|
33
|
+
/** Comparable "how much is left", present for every backend. */
|
|
34
|
+
headroom: BackendHeadroom;
|
|
35
|
+
/** One-line rendering of `headroom`, ready to print. */
|
|
36
|
+
headroomLabel: string;
|
|
27
37
|
}
|
|
28
38
|
|
|
29
39
|
/**
|
|
@@ -36,37 +46,18 @@ export interface BackendUsageEntry {
|
|
|
36
46
|
export async function collectPlanUsage(
|
|
37
47
|
config: TalonConfig,
|
|
38
48
|
): Promise<BackendUsageEntry[]> {
|
|
39
|
-
const
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
});
|
|
54
|
-
continue;
|
|
55
|
-
}
|
|
56
|
-
|
|
57
|
-
let usage: PlanUsage | undefined;
|
|
58
|
-
try {
|
|
59
|
-
usage = await backend.usage.getPlanUsage();
|
|
60
|
-
} catch {
|
|
61
|
-
usage = undefined;
|
|
62
|
-
}
|
|
63
|
-
const plan = buildPlanDisplay(usage);
|
|
64
|
-
entries.push(
|
|
65
|
-
plan
|
|
66
|
-
? { id, label, plan }
|
|
67
|
-
: { id, label, plan: null, note: "no usage information available" },
|
|
68
|
-
);
|
|
69
|
-
}
|
|
70
|
-
|
|
71
|
-
return entries;
|
|
49
|
+
const snapshots = await collectBackendUsage(config, { force: true });
|
|
50
|
+
return snapshots.map((snapshot) => {
|
|
51
|
+
const plan = buildPlanDisplay(snapshot.plan);
|
|
52
|
+
return {
|
|
53
|
+
id: snapshot.id,
|
|
54
|
+
label: snapshot.label,
|
|
55
|
+
plan,
|
|
56
|
+
headroom: snapshot.headroom,
|
|
57
|
+
headroomLabel: formatHeadroom(snapshot.headroom),
|
|
58
|
+
...(plan
|
|
59
|
+
? {}
|
|
60
|
+
: { note: snapshot.note ?? "no usage information available" }),
|
|
61
|
+
};
|
|
62
|
+
});
|
|
72
63
|
}
|
|
@@ -443,10 +443,14 @@ export function renderUsageMessage(
|
|
|
443
443
|
|
|
444
444
|
for (const entry of entries) {
|
|
445
445
|
const name = fmt.escape(entry.label || entry.id);
|
|
446
|
+
// Headroom is the one figure every backend can answer, so it goes on
|
|
447
|
+
// every block — including the ones with no plan windows to draw.
|
|
448
|
+
const headroom = ` ${fmt.bold("Headroom:")} ${fmt.escape(entry.headroomLabel)}`;
|
|
446
449
|
if (!entry.plan) {
|
|
447
450
|
lines.push(
|
|
448
451
|
"",
|
|
449
452
|
`${fmt.bold(name)} — ${fmt.italic(fmt.escape(entry.note ?? ""))}`,
|
|
453
|
+
headroom,
|
|
450
454
|
);
|
|
451
455
|
continue;
|
|
452
456
|
}
|
|
@@ -454,7 +458,7 @@ export function renderUsageMessage(
|
|
|
454
458
|
? ` ${fmt.emphasis(`(${entry.plan.ageLabel})`)}`
|
|
455
459
|
: "";
|
|
456
460
|
const plan = entry.plan.plan ? ` · ${fmt.escape(entry.plan.plan)}` : "";
|
|
457
|
-
lines.push("", `${fmt.bold(name)}${plan}${age}
|
|
461
|
+
lines.push("", `${fmt.bold(name)}${plan}${age}`, headroom);
|
|
458
462
|
if (entry.plan.resetsAvailable) {
|
|
459
463
|
const n = entry.plan.resetsAvailable;
|
|
460
464
|
const resets = `usage limit reset${n === 1 ? "" : "s"} available`;
|
|
@@ -89,6 +89,36 @@ function applyInlineFormatting(input: string, urls: string[]): string {
|
|
|
89
89
|
return out;
|
|
90
90
|
}
|
|
91
91
|
|
|
92
|
+
/**
|
|
93
|
+
* Block-level passes over already-escaped text: ATX headings → bold lines,
|
|
94
|
+
* runs of `> ` lines → one `<blockquote>`. Placeholders for code are opaque
|
|
95
|
+
* `\x00…\x00` tokens, so a `#` inside a fenced block is never seen here.
|
|
96
|
+
*/
|
|
97
|
+
function applyBlockFormatting(input: string): string {
|
|
98
|
+
const lines = input.split("\n");
|
|
99
|
+
const out: string[] = [];
|
|
100
|
+
let quote: string[] | null = null;
|
|
101
|
+
const flushQuote = () => {
|
|
102
|
+
if (quote) out.push(`<blockquote>${quote.join("\n")}</blockquote>`);
|
|
103
|
+
quote = null;
|
|
104
|
+
};
|
|
105
|
+
for (const line of lines) {
|
|
106
|
+
const quoted = /^>[ \t]?(.*)$/.exec(line);
|
|
107
|
+
if (quoted) {
|
|
108
|
+
(quote ??= []).push(quoted[1] ?? "");
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
flushQuote();
|
|
112
|
+
// `#` must be followed by whitespace to be a heading: `#993` is a
|
|
113
|
+
// reference, not a title. Trailing closing hashes (`## Title ##`) are
|
|
114
|
+
// markdown too and are dropped.
|
|
115
|
+
const heading = /^#{1,6}[ \t]+(.+?)(?:[ \t]+#+)?[ \t]*$/.exec(line);
|
|
116
|
+
out.push(heading ? `<b>${heading[1]}</b>` : line);
|
|
117
|
+
}
|
|
118
|
+
flushQuote();
|
|
119
|
+
return out.join("\n");
|
|
120
|
+
}
|
|
121
|
+
|
|
92
122
|
/**
|
|
93
123
|
* Convert Markdown output to Telegram-safe HTML.
|
|
94
124
|
*
|
|
@@ -145,6 +175,15 @@ export function markdownToTelegramHtml(text: string): string {
|
|
|
145
175
|
// oxlint-disable-next-line no-control-regex
|
|
146
176
|
processed = processed.replace(/[^`\x00]+/g, (segment) => escapeHtml(segment));
|
|
147
177
|
|
|
178
|
+
// Step 3b: Block-level markdown. Telegram HTML has no headings, so an
|
|
179
|
+
// ATX heading becomes a bold line — the alternative is the `###` going
|
|
180
|
+
// out raw, and `# Title` alone turns `#Title` into a hashtag entity.
|
|
181
|
+
// Only `#` followed by whitespace is a heading: `#993` is a PR/hashtag
|
|
182
|
+
// reference and stays literal. Runs of `> ` lines become a blockquote,
|
|
183
|
+
// which Telegram does render. The text is already escaped here, so the
|
|
184
|
+
// quote marker is `>`.
|
|
185
|
+
processed = applyBlockFormatting(processed);
|
|
186
|
+
|
|
148
187
|
// Step 4: Apply inline formatting, but only keep it if the result is
|
|
149
188
|
// actually parseable. `processed` at this point holds escaped text plus
|
|
150
189
|
// tag-free placeholders, so it is a safe unformatted fallback.
|
|
@@ -14,7 +14,14 @@ import {
|
|
|
14
14
|
import { getChatSettings } from "../../../storage/chat-settings.js";
|
|
15
15
|
import { getSessionInfo } from "../../../storage/sessions.js";
|
|
16
16
|
import { getLoadedPlugins } from "../../../core/plugin/index.js";
|
|
17
|
-
import {
|
|
17
|
+
import {
|
|
18
|
+
getPoolConfig,
|
|
19
|
+
getPooledBackend,
|
|
20
|
+
} from "../../../core/engine/backend-controller/index.js";
|
|
21
|
+
import {
|
|
22
|
+
collectBackendUsage,
|
|
23
|
+
formatHeadroom,
|
|
24
|
+
} from "../../../core/engine/backend-router/index.js";
|
|
18
25
|
import type { Command, CommandContext } from "../command-registry.js";
|
|
19
26
|
|
|
20
27
|
type BackendRef = CommandContext["backend"];
|
|
@@ -146,6 +153,27 @@ async function writePlan(ctx: CommandContext, be: BackendRef): Promise<void> {
|
|
|
146
153
|
);
|
|
147
154
|
}
|
|
148
155
|
}
|
|
156
|
+
await writeHeadroom(ctx);
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/**
|
|
160
|
+
* What the plan-aware router sees: one comparable figure per backend, so
|
|
161
|
+
* "why did that sub-agent run on agy?" is answerable from `/status`.
|
|
162
|
+
* Ledger-derived rows say so — they are Talon's own count, not the
|
|
163
|
+
* provider's.
|
|
164
|
+
*/
|
|
165
|
+
async function writeHeadroom(ctx: CommandContext): Promise<void> {
|
|
166
|
+
const entries = await collectBackendUsage(getPoolConfig() ?? undefined).catch(
|
|
167
|
+
() => [],
|
|
168
|
+
);
|
|
169
|
+
if (entries.length === 0) return;
|
|
170
|
+
ctx.renderer.writeln();
|
|
171
|
+
ctx.renderer.writeln(` ${pc.bold("Headroom")}`);
|
|
172
|
+
for (const entry of entries) {
|
|
173
|
+
ctx.renderer.writeln(
|
|
174
|
+
` ${(entry.label || entry.id).padEnd(14)}${pc.dim(formatHeadroom(entry.headroom))}`,
|
|
175
|
+
);
|
|
176
|
+
}
|
|
149
177
|
}
|
|
150
178
|
|
|
151
179
|
function writePlugins(ctx: CommandContext): void {
|
package/src/index.ts
CHANGED
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
* backends/frontends/plugins.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
|
+
import { installSignalListenerGuard } from "./core/daemon/signals.js";
|
|
13
14
|
import {
|
|
14
15
|
MCP_LAUNCH_SUBCOMMAND,
|
|
15
16
|
runSupervisor,
|
|
@@ -20,6 +21,12 @@ import {
|
|
|
20
21
|
runHandoffWatch,
|
|
21
22
|
} from "./core/daemon/handoff.js";
|
|
22
23
|
|
|
24
|
+
// Before anything can remove a signal listener: under Bun that would
|
|
25
|
+
// silently disarm SIGTERM/SIGINT for the whole process (core/daemon/
|
|
26
|
+
// signals.ts). Every Talon process passes through here, so every one
|
|
27
|
+
// is covered.
|
|
28
|
+
installSignalListenerGuard();
|
|
29
|
+
|
|
23
30
|
if (process.argv[2] === MCP_LAUNCH_SUBCOMMAND) {
|
|
24
31
|
await runSupervisor(process.argv.slice(3));
|
|
25
32
|
} else if (process.argv[2] === LUA_RUN_SUBCOMMAND) {
|