talon-agent 5.4.0 → 5.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/package.json +1 -1
- package/prompts/identity.md +2 -0
- package/prompts/telegram.md +2 -1
- package/src/app.ts +12 -0
- package/src/backend/runtime/turn/turn-phases.ts +5 -0
- package/src/bootstrap.ts +55 -24
- package/src/core/agents/runner.ts +50 -2
- package/src/core/agents/types.ts +5 -0
- package/src/core/background/cron/job-oneshot.ts +2 -0
- package/src/core/background/cron/scheduler.ts +45 -2
- package/src/core/background/heartbeat/agent.ts +117 -20
- package/src/core/background/heartbeat/state.ts +11 -0
- package/src/core/config/index.ts +38 -0
- package/src/core/daemon/respawn.ts +34 -10
- package/src/core/daemon/signals.ts +111 -0
- package/src/core/engine/backend-controller/index.ts +1 -0
- package/src/core/engine/backend-controller/pool.ts +11 -0
- package/src/core/engine/backend-router/headroom.ts +267 -0
- package/src/core/engine/backend-router/index.ts +52 -0
- package/src/core/engine/backend-router/ledger.ts +248 -0
- package/src/core/engine/backend-router/router.ts +322 -0
- package/src/core/engine/backend-router/usage.ts +80 -0
- package/src/core/engine/gateway-actions/agents/control.ts +2 -1
- package/src/core/engine/gateway-actions/models.ts +74 -37
- package/src/core/tools/ops/models.ts +2 -2
- package/src/frontend/presentation/plan-usage-report.ts +29 -38
- package/src/frontend/presentation/reports.ts +5 -1
- package/src/frontend/telegram/formatting.ts +39 -0
- package/src/frontend/terminal/builtins/status.ts +29 -1
- package/src/index.ts +7 -0
- package/src/util/log.ts +1 -0
|
@@ -13,9 +13,9 @@
|
|
|
13
13
|
* systemd, foreman, pm2, or running under a debugger. Respawning from
|
|
14
14
|
* our own `process.argv` works regardless of launch method.
|
|
15
15
|
*
|
|
16
|
-
* Ordering matters. `respawnSelf()` only *arms* the handoff and
|
|
17
|
-
*
|
|
18
|
-
*
|
|
16
|
+
* Ordering matters. `respawnSelf()` only *arms* the handoff and enters
|
|
17
|
+
* graceful shutdown; the successor is spawned by `spawnSuccessor()` at
|
|
18
|
+
* the tail of it, once the frontends have stopped. Spawning up-front
|
|
19
19
|
* (the original behaviour) left the successor long-polling `getUpdates`
|
|
20
20
|
* while the outgoing process was still draining in-flight queries — up
|
|
21
21
|
* to DRAIN_TIMEOUT_MS of two live pollers. Telegram answers only one of
|
|
@@ -40,6 +40,16 @@
|
|
|
40
40
|
* within a bounded window, and starts the daemon the way `talon
|
|
41
41
|
* start` does if the successor never comes up. Nothing in the
|
|
42
42
|
* handoff depends on a process that is about to call process.exit().
|
|
43
|
+
*
|
|
44
|
+
* And one thing 2026-09-20 added: the shutdown is entered directly,
|
|
45
|
+
* through the function app.ts registers with `setRespawnShutdown()`,
|
|
46
|
+
* not by sending ourselves a SIGTERM. The signal round trip bought
|
|
47
|
+
* nothing, and under Bun it lost the whole restart: by the time
|
|
48
|
+
* `/restart` ran, the process had silently lost its OS-level SIGTERM
|
|
49
|
+
* handler (see ./signals.ts), so the signal terminated it on the spot —
|
|
50
|
+
* no shutdown, no successor, no pidfile cleanup, and nothing in any log
|
|
51
|
+
* after "Respawn requested". Only a process that never registered — an
|
|
52
|
+
* embedder, a test — still falls back to the signal.
|
|
43
53
|
*/
|
|
44
54
|
|
|
45
55
|
import { spawn } from "node:child_process";
|
|
@@ -47,6 +57,18 @@ import { log, logError, openRespawnLog } from "../../util/log.js";
|
|
|
47
57
|
import { HANDOFF_WATCH_SUBCOMMAND } from "./handoff.js";
|
|
48
58
|
|
|
49
59
|
let pendingReason: string | null = null;
|
|
60
|
+
let shutdown: ((reason: string) => void) | null = null;
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Register the graceful-shutdown entry a respawn should run. app.ts
|
|
64
|
+
* hands over `gracefulShutdown`; `respawnSelf()` then calls it in-process
|
|
65
|
+
* instead of round-tripping a SIGTERM through the OS. `null` clears it.
|
|
66
|
+
*/
|
|
67
|
+
export function setRespawnShutdown(
|
|
68
|
+
fn: ((reason: string) => void) | null,
|
|
69
|
+
): void {
|
|
70
|
+
shutdown = fn;
|
|
71
|
+
}
|
|
50
72
|
|
|
51
73
|
/**
|
|
52
74
|
* Argv flag that makes the daemon entry resolve its whole import graph
|
|
@@ -84,9 +106,8 @@ function selfInvocation(extra: string[]): { cmd: string; args: string[] } {
|
|
|
84
106
|
}
|
|
85
107
|
|
|
86
108
|
/**
|
|
87
|
-
* Arm a respawn and
|
|
88
|
-
*
|
|
89
|
-
* and hands off via `spawnSuccessor()`.
|
|
109
|
+
* Arm a respawn and enter graceful shutdown, which stops the frontends,
|
|
110
|
+
* flushes state, and hands off via `spawnSuccessor()`.
|
|
90
111
|
*
|
|
91
112
|
* `reason` is logged for operator visibility (e.g. "telegram
|
|
92
113
|
* /restart"). The function returns immediately; the successor starts
|
|
@@ -95,10 +116,13 @@ function selfInvocation(extra: string[]): { cmd: string; args: string[] } {
|
|
|
95
116
|
export function respawnSelf(reason: string): void {
|
|
96
117
|
log("shutdown", `Respawn requested (${reason})`);
|
|
97
118
|
pendingReason = reason;
|
|
98
|
-
//
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
|
|
119
|
+
// Don't exit here directly — that would skip the flush and leave the
|
|
120
|
+
// PID file dangling. The registered shutdown is the same path a
|
|
121
|
+
// SIGTERM takes, entered without the signal.
|
|
122
|
+
if (shutdown) {
|
|
123
|
+
shutdown(reason);
|
|
124
|
+
return;
|
|
125
|
+
}
|
|
102
126
|
process.kill(process.pid, "SIGTERM");
|
|
103
127
|
}
|
|
104
128
|
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Keep OS-level signal handlers armed under Bun.
|
|
3
|
+
*
|
|
4
|
+
* Bun (observed on 1.3.9) uninstalls the process-wide handler for a
|
|
5
|
+
* signal as soon as ANY listener for it is removed — even while other
|
|
6
|
+
* listeners remain:
|
|
7
|
+
*
|
|
8
|
+
* process.on("SIGTERM", a);
|
|
9
|
+
* process.on("SIGTERM", b);
|
|
10
|
+
* process.removeListener("SIGTERM", b);
|
|
11
|
+
* // `a` is still registered in JS, yet SigCgt in /proc/<pid>/status
|
|
12
|
+
* // no longer lists SIGTERM: the next one terminates the process
|
|
13
|
+
* // with the default action, running no JS at all.
|
|
14
|
+
*
|
|
15
|
+
* Node keeps the handler until the last listener goes. Adding any
|
|
16
|
+
* listener makes Bun install the handler again, which is what this
|
|
17
|
+
* guard relies on.
|
|
18
|
+
*
|
|
19
|
+
* Talon hit it through `write-file-atomic`: each call loads signal-exit,
|
|
20
|
+
* which puts its own listeners on the termination signals, and unloads
|
|
21
|
+
* it afterwards, removing them. The daemon writes its pidfile that way
|
|
22
|
+
* while booting — so from that moment every SIGTERM it sent itself
|
|
23
|
+
* (`/restart` and `/update`, before respawn.ts entered the shutdown
|
|
24
|
+
* directly) or received from outside (`talon stop`'s fallback, systemd,
|
|
25
|
+
* docker) killed it outright: no graceful shutdown, no successor, no
|
|
26
|
+
* pidfile cleanup. Every `/update` between 2026-09-18 (the move to Bun)
|
|
27
|
+
* and 2026-09-20 ended this way.
|
|
28
|
+
*
|
|
29
|
+
* The guard wraps `process.removeListener` / `process.off`. After a
|
|
30
|
+
* removal that still leaves real listeners on a signal, it re-adds a
|
|
31
|
+
* no-op sentinel listener (removing any earlier copy first), so the
|
|
32
|
+
* handler is installed again before the caller gets control back. Once
|
|
33
|
+
* the last real listener is gone the sentinel goes too, and the signal
|
|
34
|
+
* falls back to its default action exactly as it would on Node — a
|
|
35
|
+
* lingering no-op listener would otherwise make the process ignore it.
|
|
36
|
+
*
|
|
37
|
+
* Installed by the entry shim (src/index.ts) so every Talon process —
|
|
38
|
+
* daemon, MCP supervisor, handoff watcher — is covered before any
|
|
39
|
+
* library gets a chance to remove a listener. A no-op on other runtimes.
|
|
40
|
+
*/
|
|
41
|
+
|
|
42
|
+
import { isBunRuntime } from "../../util/runtime.js";
|
|
43
|
+
|
|
44
|
+
type Listener = (...args: unknown[]) => void;
|
|
45
|
+
|
|
46
|
+
/** The slice of `process` the guard touches; tests pass an EventEmitter. */
|
|
47
|
+
export interface SignalEmitter {
|
|
48
|
+
on(event: string | symbol, listener: Listener): unknown;
|
|
49
|
+
removeListener(event: string | symbol, listener: Listener): unknown;
|
|
50
|
+
off?(event: string | symbol, listener: Listener): unknown;
|
|
51
|
+
listeners(event: string | symbol): readonly unknown[];
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
export interface SignalGuardOptions {
|
|
55
|
+
/** Install even off Bun (tests exercise the wrapper on Node). */
|
|
56
|
+
force?: boolean;
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
const SIGNAL_EVENT = /^SIG[A-Z0-9]+$/;
|
|
60
|
+
|
|
61
|
+
const INSTALLED = Symbol.for("talon.signalListenerGuard");
|
|
62
|
+
|
|
63
|
+
type Guarded = SignalEmitter & { [INSTALLED]?: () => void };
|
|
64
|
+
|
|
65
|
+
/** The sentinel. Its only job is to exist, so Bun keeps the handler. */
|
|
66
|
+
function talonSignalRearm(): void {}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Wrap the emitter's listener removal so a signal never silently loses
|
|
70
|
+
* its OS handler. Idempotent: a second install returns the first one's
|
|
71
|
+
* uninstall. Production never uninstalls; tests do.
|
|
72
|
+
*/
|
|
73
|
+
export function installSignalListenerGuard(
|
|
74
|
+
target: SignalEmitter = process,
|
|
75
|
+
opts: SignalGuardOptions = {},
|
|
76
|
+
): () => void {
|
|
77
|
+
const emitter = target as Guarded;
|
|
78
|
+
if (!opts.force && !isBunRuntime()) return () => {};
|
|
79
|
+
const existing = emitter[INSTALLED];
|
|
80
|
+
if (existing) return existing;
|
|
81
|
+
|
|
82
|
+
const originalRemove = emitter.removeListener;
|
|
83
|
+
const originalOff = emitter.off;
|
|
84
|
+
|
|
85
|
+
const guardedRemove = function (
|
|
86
|
+
this: Guarded,
|
|
87
|
+
event: string | symbol,
|
|
88
|
+
listener: Listener,
|
|
89
|
+
): unknown {
|
|
90
|
+
const result = originalRemove.call(this, event, listener);
|
|
91
|
+
if (typeof event !== "string" || !SIGNAL_EVENT.test(event)) return result;
|
|
92
|
+
// Bun has just dropped the handler. What is left decides whether the
|
|
93
|
+
// signal should still be handled at all.
|
|
94
|
+
const remaining = this.listeners(event).filter(
|
|
95
|
+
(fn) => fn !== talonSignalRearm,
|
|
96
|
+
);
|
|
97
|
+
originalRemove.call(this, event, talonSignalRearm);
|
|
98
|
+
if (remaining.length > 0) this.on(event, talonSignalRearm);
|
|
99
|
+
return result;
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
emitter.removeListener = guardedRemove;
|
|
103
|
+
if (originalOff) emitter.off = guardedRemove;
|
|
104
|
+
const uninstall = (): void => {
|
|
105
|
+
emitter.removeListener = originalRemove;
|
|
106
|
+
if (originalOff) emitter.off = originalOff;
|
|
107
|
+
delete emitter[INSTALLED];
|
|
108
|
+
};
|
|
109
|
+
emitter[INSTALLED] = uninstall;
|
|
110
|
+
return uninstall;
|
|
111
|
+
}
|
|
@@ -152,6 +152,17 @@ export function getAvailableBackends(): { id: string; label: string }[] {
|
|
|
152
152
|
return listAvailableBackends(ctx.poolConfig ?? undefined);
|
|
153
153
|
}
|
|
154
154
|
|
|
155
|
+
/**
|
|
156
|
+
* The config the pool was initialised with, or `null` before init.
|
|
157
|
+
*
|
|
158
|
+
* For runtime readers that need more of the config than the backend list —
|
|
159
|
+
* the plan-aware router reads `router` and `backendBudgets` from here rather
|
|
160
|
+
* than having config threaded through every background call site.
|
|
161
|
+
*/
|
|
162
|
+
export function getPoolConfig(): TalonConfig | null {
|
|
163
|
+
return ctx.poolConfig;
|
|
164
|
+
}
|
|
165
|
+
|
|
155
166
|
/**
|
|
156
167
|
* The live pooled `Backend` instance for an id, or `null` if it isn't currently
|
|
157
168
|
* pooled. Unlike `getBackendForRole`/`getBackendForChat` this never initialises
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Headroom — "how much of this backend is left?", answered the same way for
|
|
3
|
+
* every backend regardless of what it can tell us about itself.
|
|
4
|
+
*
|
|
5
|
+
* Two sources, in precedence order:
|
|
6
|
+
*
|
|
7
|
+
* - `plan` — the backend's own subscription windows
|
|
8
|
+
* (`UsageTelemetry.getPlanUsage`). Authoritative: it is the
|
|
9
|
+
* provider's own count, including spend from outside Talon.
|
|
10
|
+
* - `ledger` — Talon's local rolling token count (see `ledger.ts`) against
|
|
11
|
+
* the operator's soft budget (`config.backendBudgets`). The
|
|
12
|
+
* fallback for backends with no account API.
|
|
13
|
+
*
|
|
14
|
+
* A backend with neither reports `source: "none"` and headroom 1. That is a
|
|
15
|
+
* deliberate "no evidence of pressure", not a claim of capacity — the
|
|
16
|
+
* router's comparator ranks it *below* any backend with real telemetry at
|
|
17
|
+
* the same headroom, so an unmeasured backend never outranks a measured one
|
|
18
|
+
* it is tied with.
|
|
19
|
+
*
|
|
20
|
+
* Reads are cached for 60s per backend: `/usage`, the router and the
|
|
21
|
+
* `plan_usage` tool all ask, and a plan lookup can be a subprocess spawn. A
|
|
22
|
+
* failed refresh keeps the last good value and flags it `stale` rather than
|
|
23
|
+
* pretending the backend emptied.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { PlanUsage } from "../../agent-runtime/capabilities.js";
|
|
27
|
+
import type { TalonConfig } from "../../config/index.js";
|
|
28
|
+
import {
|
|
29
|
+
getPooledBackend,
|
|
30
|
+
listAvailableBackends,
|
|
31
|
+
} from "../backend-controller/index.js";
|
|
32
|
+
import { ledgerUsage } from "./ledger.js";
|
|
33
|
+
|
|
34
|
+
/** How long a headroom reading is reused before the source is asked again. */
|
|
35
|
+
export const HEADROOM_CACHE_MS = 60_000;
|
|
36
|
+
|
|
37
|
+
/** Where a headroom figure came from. Also its ranking priority. */
|
|
38
|
+
export type HeadroomSource = "plan" | "ledger" | "none";
|
|
39
|
+
|
|
40
|
+
/** The window that is closest to its limit — what the ceiling is judged on. */
|
|
41
|
+
export interface LimitingWindow {
|
|
42
|
+
readonly label: string;
|
|
43
|
+
/** 0-100. */
|
|
44
|
+
readonly percent: number;
|
|
45
|
+
readonly resetsAt?: string;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export interface BackendHeadroom {
|
|
49
|
+
readonly id: string;
|
|
50
|
+
readonly label: string;
|
|
51
|
+
/** 0..1 — 1 is empty, 0 is at the limit. */
|
|
52
|
+
readonly headroom: number;
|
|
53
|
+
readonly limiting?: LimitingWindow;
|
|
54
|
+
readonly source: HeadroomSource;
|
|
55
|
+
/** Epoch ms of the underlying read. */
|
|
56
|
+
readonly fetchedAt: number;
|
|
57
|
+
/** True when the last refresh failed and this is the previous value. */
|
|
58
|
+
readonly stale?: boolean;
|
|
59
|
+
/**
|
|
60
|
+
* The raw plan reading this came from, when `source` is `"plan"`. Carried
|
|
61
|
+
* so `/usage` and the `plan_usage` tool render the real windows off the
|
|
62
|
+
* same (cached) fetch the router ranked on, instead of asking twice.
|
|
63
|
+
*/
|
|
64
|
+
readonly plan?: PlanUsage;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
interface CacheEntry {
|
|
68
|
+
value: BackendHeadroom;
|
|
69
|
+
/** Epoch ms the value was computed (not the same as a stale `fetchedAt`). */
|
|
70
|
+
cachedAt: number;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const cache = new Map<string, CacheEntry>();
|
|
74
|
+
|
|
75
|
+
function clampPercent(percent: number): number {
|
|
76
|
+
if (!Number.isFinite(percent)) return 0;
|
|
77
|
+
return Math.max(0, Math.min(100, percent));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** headroom = 1 − (tightest window) / 100. */
|
|
81
|
+
function headroomFor(percent: number): number {
|
|
82
|
+
return Math.max(0, Math.min(1, 1 - clampPercent(percent) / 100));
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* The tightest of a plan's windows. `undefined` when the plan reports none,
|
|
87
|
+
* which is how a backend that answers but has nothing to say is told apart
|
|
88
|
+
* from one that never answered.
|
|
89
|
+
*/
|
|
90
|
+
export function limitingWindowOf(
|
|
91
|
+
usage: PlanUsage | undefined,
|
|
92
|
+
): LimitingWindow | undefined {
|
|
93
|
+
if (!usage || usage.windows.length === 0) return undefined;
|
|
94
|
+
let worst = usage.windows[0] as NonNullable<(typeof usage.windows)[0]>;
|
|
95
|
+
for (const window of usage.windows) {
|
|
96
|
+
if (window.percent > worst.percent) worst = window;
|
|
97
|
+
}
|
|
98
|
+
return {
|
|
99
|
+
label: worst.label,
|
|
100
|
+
percent: clampPercent(worst.percent),
|
|
101
|
+
...(worst.resetsAt ? { resetsAt: worst.resetsAt } : {}),
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Headroom from a `PlanUsage`, or `undefined` when it carries no windows. */
|
|
106
|
+
export function headroomFromPlan(
|
|
107
|
+
id: string,
|
|
108
|
+
label: string,
|
|
109
|
+
usage: PlanUsage | undefined,
|
|
110
|
+
): BackendHeadroom | undefined {
|
|
111
|
+
const limiting = limitingWindowOf(usage);
|
|
112
|
+
if (!limiting || !usage) return undefined;
|
|
113
|
+
return {
|
|
114
|
+
id,
|
|
115
|
+
label,
|
|
116
|
+
headroom: headroomFor(limiting.percent),
|
|
117
|
+
limiting,
|
|
118
|
+
source: "plan",
|
|
119
|
+
fetchedAt: usage.fetchedAt,
|
|
120
|
+
plan: usage,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** The soft budget an operator declared for a backend, if any. */
|
|
125
|
+
function budgetFor(
|
|
126
|
+
config: TalonConfig | undefined,
|
|
127
|
+
id: string,
|
|
128
|
+
): { tokensPer5h?: number; tokensPerDay?: number } | undefined {
|
|
129
|
+
const budget = config?.backendBudgets?.[id];
|
|
130
|
+
if (!budget) return undefined;
|
|
131
|
+
if (budget.tokensPer5h === undefined && budget.tokensPerDay === undefined) {
|
|
132
|
+
return undefined;
|
|
133
|
+
}
|
|
134
|
+
return budget;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/** Whether the operator gave this backend a local budget to measure against. */
|
|
138
|
+
export function hasBudget(
|
|
139
|
+
config: TalonConfig | undefined,
|
|
140
|
+
id: string,
|
|
141
|
+
): boolean {
|
|
142
|
+
return budgetFor(config, id) !== undefined;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Headroom from the local ledger. The tighter of the two configured windows
|
|
147
|
+
* wins, so a backend that is fine on the day but has just burned its 5h
|
|
148
|
+
* allowance still reads as full.
|
|
149
|
+
*/
|
|
150
|
+
export function headroomFromLedger(
|
|
151
|
+
id: string,
|
|
152
|
+
label: string,
|
|
153
|
+
config: TalonConfig | undefined,
|
|
154
|
+
now = Date.now(),
|
|
155
|
+
): BackendHeadroom | undefined {
|
|
156
|
+
const budget = budgetFor(config, id);
|
|
157
|
+
if (!budget) return undefined;
|
|
158
|
+
const used = ledgerUsage(id, now);
|
|
159
|
+
const windows: LimitingWindow[] = [];
|
|
160
|
+
if (budget.tokensPer5h !== undefined) {
|
|
161
|
+
windows.push({
|
|
162
|
+
label: "5h (local budget)",
|
|
163
|
+
percent: clampPercent((used.tokens5h / budget.tokensPer5h) * 100),
|
|
164
|
+
});
|
|
165
|
+
}
|
|
166
|
+
if (budget.tokensPerDay !== undefined) {
|
|
167
|
+
windows.push({
|
|
168
|
+
label: "24h (local budget)",
|
|
169
|
+
percent: clampPercent((used.tokensDay / budget.tokensPerDay) * 100),
|
|
170
|
+
});
|
|
171
|
+
}
|
|
172
|
+
let worst = windows[0] as LimitingWindow;
|
|
173
|
+
for (const window of windows)
|
|
174
|
+
if (window.percent > worst.percent) worst = window;
|
|
175
|
+
return {
|
|
176
|
+
id,
|
|
177
|
+
label,
|
|
178
|
+
headroom: headroomFor(worst.percent),
|
|
179
|
+
limiting: worst,
|
|
180
|
+
source: "ledger",
|
|
181
|
+
fetchedAt: now,
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** The "nothing to measure" reading. Headroom 1, but lowest ranking source. */
|
|
186
|
+
function unknownHeadroom(
|
|
187
|
+
id: string,
|
|
188
|
+
label: string,
|
|
189
|
+
now: number,
|
|
190
|
+
): BackendHeadroom {
|
|
191
|
+
return { id, label, headroom: 1, source: "none", fetchedAt: now };
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Ask a pooled backend for its plan windows. Rejects like the backend does. */
|
|
195
|
+
async function readPlanUsage(id: string): Promise<PlanUsage | undefined> {
|
|
196
|
+
const backend = getPooledBackend(id);
|
|
197
|
+
const read = backend?.usage?.getPlanUsage;
|
|
198
|
+
if (!read || !backend?.usage) return undefined;
|
|
199
|
+
return read.call(backend.usage);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Headroom for one backend, cached for {@link HEADROOM_CACHE_MS}.
|
|
204
|
+
*
|
|
205
|
+
* `force` skips the cache — the `plan_usage` tool asks for a fresh read
|
|
206
|
+
* because the operator is looking at the number right now.
|
|
207
|
+
*/
|
|
208
|
+
export async function getBackendHeadroom(
|
|
209
|
+
id: string,
|
|
210
|
+
label: string,
|
|
211
|
+
config: TalonConfig | undefined,
|
|
212
|
+
options?: { force?: boolean; now?: number },
|
|
213
|
+
): Promise<BackendHeadroom> {
|
|
214
|
+
const now = options?.now ?? Date.now();
|
|
215
|
+
const cached = cache.get(id);
|
|
216
|
+
if (!options?.force && cached && now - cached.cachedAt < HEADROOM_CACHE_MS) {
|
|
217
|
+
return cached.value;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
let value: BackendHeadroom;
|
|
221
|
+
try {
|
|
222
|
+
const plan = headroomFromPlan(id, label, await readPlanUsage(id));
|
|
223
|
+
value =
|
|
224
|
+
plan ??
|
|
225
|
+
headroomFromLedger(id, label, config, now) ??
|
|
226
|
+
unknownHeadroom(id, label, now);
|
|
227
|
+
} catch {
|
|
228
|
+
// The source is unreachable this minute. Keeping the last good reading
|
|
229
|
+
// is the conservative answer: forgetting it would read as "empty" and
|
|
230
|
+
// send the next background run straight at a backend near its ceiling.
|
|
231
|
+
value = cached
|
|
232
|
+
? { ...cached.value, stale: true }
|
|
233
|
+
: unknownHeadroom(id, label, now);
|
|
234
|
+
}
|
|
235
|
+
cache.set(id, { value, cachedAt: now });
|
|
236
|
+
return value;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/** Headroom for every backend the config exposes, in config order. */
|
|
240
|
+
export async function collectBackendHeadroom(
|
|
241
|
+
config: TalonConfig | undefined,
|
|
242
|
+
options?: { force?: boolean; now?: number },
|
|
243
|
+
): Promise<BackendHeadroom[]> {
|
|
244
|
+
const backends = listAvailableBackends(config);
|
|
245
|
+
return Promise.all(
|
|
246
|
+
backends.map(({ id, label }) =>
|
|
247
|
+
getBackendHeadroom(id, label, config, options),
|
|
248
|
+
),
|
|
249
|
+
);
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
/** One-line rendering shared by `/usage`, `plan_usage` and the router log. */
|
|
253
|
+
export function formatHeadroom(entry: BackendHeadroom): string {
|
|
254
|
+
const pct = `${Math.round(entry.headroom * 100)}%`;
|
|
255
|
+
const detail =
|
|
256
|
+
entry.source === "none"
|
|
257
|
+
? "no usage signal"
|
|
258
|
+
: `${entry.limiting?.label ?? "window"} ${Math.round(entry.limiting?.percent ?? 0)}% used`;
|
|
259
|
+
const tag = entry.source === "ledger" ? " (local budget)" : "";
|
|
260
|
+
const stale = entry.stale ? " (stale)" : "";
|
|
261
|
+
return `${pct} — ${detail}${tag}${stale}`;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/** Test seam — drop every cached reading. */
|
|
265
|
+
export function resetHeadroomCacheForTest(): void {
|
|
266
|
+
cache.clear();
|
|
267
|
+
}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Plan-aware backend router.
|
|
3
|
+
*
|
|
4
|
+
* - `ledger` — the local rolling token count that gives budget-only
|
|
5
|
+
* backends a headroom signal.
|
|
6
|
+
* - `headroom` — one comparable "how much is left" per backend, from the
|
|
7
|
+
* plan API where there is one and the ledger where there
|
|
8
|
+
* is not.
|
|
9
|
+
* - `router` — the decision: who runs this background job.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
export {
|
|
13
|
+
flushBackendLedger,
|
|
14
|
+
ledgerUsage,
|
|
15
|
+
loadBackendLedger,
|
|
16
|
+
recordBackendRunUsage,
|
|
17
|
+
recordBackendUsage,
|
|
18
|
+
resetBackendLedgerForTest,
|
|
19
|
+
tokensInWindow,
|
|
20
|
+
LEDGER_RETENTION_MS,
|
|
21
|
+
LEDGER_SHORT_WINDOW_MS,
|
|
22
|
+
} from "./ledger.js";
|
|
23
|
+
export {
|
|
24
|
+
collectBackendHeadroom,
|
|
25
|
+
formatHeadroom,
|
|
26
|
+
getBackendHeadroom,
|
|
27
|
+
hasBudget,
|
|
28
|
+
headroomFromLedger,
|
|
29
|
+
headroomFromPlan,
|
|
30
|
+
limitingWindowOf,
|
|
31
|
+
resetHeadroomCacheForTest,
|
|
32
|
+
HEADROOM_CACHE_MS,
|
|
33
|
+
type BackendHeadroom,
|
|
34
|
+
type HeadroomSource,
|
|
35
|
+
type LimitingWindow,
|
|
36
|
+
} from "./headroom.js";
|
|
37
|
+
export {
|
|
38
|
+
collectBackendUsage,
|
|
39
|
+
leadWith,
|
|
40
|
+
type BackendUsageSnapshot,
|
|
41
|
+
} from "./usage.js";
|
|
42
|
+
export {
|
|
43
|
+
chooseBackend,
|
|
44
|
+
resolveRoutedModel,
|
|
45
|
+
taskClassForEffort,
|
|
46
|
+
DEFAULT_CEILING_PERCENT,
|
|
47
|
+
type RouteDecision,
|
|
48
|
+
type RouteHints,
|
|
49
|
+
type RoutePurpose,
|
|
50
|
+
type RouteRequest,
|
|
51
|
+
type TaskClass,
|
|
52
|
+
} from "./router.js";
|