talon-agent 5.4.1 → 5.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/package.json +2 -2
- package/prompts/identity.md +2 -0
- package/prompts/telegram.md +2 -1
- package/src/app.ts +8 -0
- package/src/backend/runtime/turn/turn-phases.ts +5 -0
- package/src/bootstrap.ts +55 -24
- package/src/core/agents/runner.ts +50 -2
- package/src/core/agents/types.ts +5 -0
- package/src/core/background/cron/job-oneshot.ts +2 -0
- package/src/core/background/cron/scheduler.ts +45 -2
- package/src/core/background/heartbeat/agent.ts +117 -20
- package/src/core/background/heartbeat/state.ts +11 -0
- package/src/core/config/index.ts +38 -0
- package/src/core/engine/backend-controller/index.ts +1 -0
- package/src/core/engine/backend-controller/pool.ts +11 -0
- package/src/core/engine/backend-router/headroom.ts +267 -0
- package/src/core/engine/backend-router/index.ts +52 -0
- package/src/core/engine/backend-router/ledger.ts +248 -0
- package/src/core/engine/backend-router/router.ts +322 -0
- package/src/core/engine/backend-router/usage.ts +80 -0
- package/src/core/engine/gateway-actions/agents/control.ts +2 -1
- package/src/core/engine/gateway-actions/models.ts +74 -37
- package/src/core/tools/ops/models.ts +2 -2
- package/src/frontend/presentation/plan-usage-report.ts +29 -38
- package/src/frontend/presentation/reports.ts +5 -1
- package/src/frontend/telegram/formatting.ts +39 -0
- package/src/frontend/telegram/handlers/access.ts +41 -0
- package/src/frontend/telegram/handlers/index.ts +1 -0
- package/src/frontend/telegram/index.ts +7 -1
- package/src/frontend/terminal/builtins/status.ts +29 -1
- package/src/util/log.ts +1 -0
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Headroom — "how much of this backend is left?", answered the same way for
|
|
3
|
+
* every backend regardless of what it can tell us about itself.
|
|
4
|
+
*
|
|
5
|
+
* Two sources, in precedence order:
|
|
6
|
+
*
|
|
7
|
+
* - `plan` — the backend's own subscription windows
|
|
8
|
+
* (`UsageTelemetry.getPlanUsage`). Authoritative: it is the
|
|
9
|
+
* provider's own count, including spend from outside Talon.
|
|
10
|
+
* - `ledger` — Talon's local rolling token count (see `ledger.ts`) against
|
|
11
|
+
* the operator's soft budget (`config.backendBudgets`). The
|
|
12
|
+
* fallback for backends with no account API.
|
|
13
|
+
*
|
|
14
|
+
* A backend with neither reports `source: "none"` and headroom 1. That is a
|
|
15
|
+
* deliberate "no evidence of pressure", not a claim of capacity — the
|
|
16
|
+
* router's comparator ranks it *below* any backend with real telemetry at
|
|
17
|
+
* the same headroom, so an unmeasured backend never outranks a measured one
|
|
18
|
+
* it is tied with.
|
|
19
|
+
*
|
|
20
|
+
* Reads are cached for 60s per backend: `/usage`, the router and the
|
|
21
|
+
* `plan_usage` tool all ask, and a plan lookup can be a subprocess spawn. A
|
|
22
|
+
* failed refresh keeps the last good value and flags it `stale` rather than
|
|
23
|
+
* pretending the backend emptied.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { PlanUsage } from "../../agent-runtime/capabilities.js";
|
|
27
|
+
import type { TalonConfig } from "../../config/index.js";
|
|
28
|
+
import {
|
|
29
|
+
getPooledBackend,
|
|
30
|
+
listAvailableBackends,
|
|
31
|
+
} from "../backend-controller/index.js";
|
|
32
|
+
import { ledgerUsage } from "./ledger.js";
|
|
33
|
+
|
|
34
|
+
/** How long a headroom reading is reused before the source is asked again. */
|
|
35
|
+
export const HEADROOM_CACHE_MS = 60_000;
|
|
36
|
+
|
|
37
|
+
/** Where a headroom figure came from. Also its ranking priority. */
|
|
38
|
+
export type HeadroomSource = "plan" | "ledger" | "none";
|
|
39
|
+
|
|
40
|
+
/** The window that is closest to its limit — what the ceiling is judged on. */
|
|
41
|
+
export interface LimitingWindow {
|
|
42
|
+
readonly label: string;
|
|
43
|
+
/** 0-100. */
|
|
44
|
+
readonly percent: number;
|
|
45
|
+
readonly resetsAt?: string;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export interface BackendHeadroom {
|
|
49
|
+
readonly id: string;
|
|
50
|
+
readonly label: string;
|
|
51
|
+
/** 0..1 — 1 is empty, 0 is at the limit. */
|
|
52
|
+
readonly headroom: number;
|
|
53
|
+
readonly limiting?: LimitingWindow;
|
|
54
|
+
readonly source: HeadroomSource;
|
|
55
|
+
/** Epoch ms of the underlying read. */
|
|
56
|
+
readonly fetchedAt: number;
|
|
57
|
+
/** True when the last refresh failed and this is the previous value. */
|
|
58
|
+
readonly stale?: boolean;
|
|
59
|
+
/**
|
|
60
|
+
* The raw plan reading this came from, when `source` is `"plan"`. Carried
|
|
61
|
+
* so `/usage` and the `plan_usage` tool render the real windows off the
|
|
62
|
+
* same (cached) fetch the router ranked on, instead of asking twice.
|
|
63
|
+
*/
|
|
64
|
+
readonly plan?: PlanUsage;
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
interface CacheEntry {
|
|
68
|
+
value: BackendHeadroom;
|
|
69
|
+
/** Epoch ms the value was computed (not the same as a stale `fetchedAt`). */
|
|
70
|
+
cachedAt: number;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const cache = new Map<string, CacheEntry>();
|
|
74
|
+
|
|
75
|
+
function clampPercent(percent: number): number {
|
|
76
|
+
if (!Number.isFinite(percent)) return 0;
|
|
77
|
+
return Math.max(0, Math.min(100, percent));
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** headroom = 1 − (tightest window) / 100. */
|
|
81
|
+
function headroomFor(percent: number): number {
|
|
82
|
+
return Math.max(0, Math.min(1, 1 - clampPercent(percent) / 100));
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* The tightest of a plan's windows. `undefined` when the plan reports none,
|
|
87
|
+
* which is how a backend that answers but has nothing to say is told apart
|
|
88
|
+
* from one that never answered.
|
|
89
|
+
*/
|
|
90
|
+
export function limitingWindowOf(
|
|
91
|
+
usage: PlanUsage | undefined,
|
|
92
|
+
): LimitingWindow | undefined {
|
|
93
|
+
if (!usage || usage.windows.length === 0) return undefined;
|
|
94
|
+
let worst = usage.windows[0] as NonNullable<(typeof usage.windows)[0]>;
|
|
95
|
+
for (const window of usage.windows) {
|
|
96
|
+
if (window.percent > worst.percent) worst = window;
|
|
97
|
+
}
|
|
98
|
+
return {
|
|
99
|
+
label: worst.label,
|
|
100
|
+
percent: clampPercent(worst.percent),
|
|
101
|
+
...(worst.resetsAt ? { resetsAt: worst.resetsAt } : {}),
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/** Headroom from a `PlanUsage`, or `undefined` when it carries no windows. */
|
|
106
|
+
export function headroomFromPlan(
|
|
107
|
+
id: string,
|
|
108
|
+
label: string,
|
|
109
|
+
usage: PlanUsage | undefined,
|
|
110
|
+
): BackendHeadroom | undefined {
|
|
111
|
+
const limiting = limitingWindowOf(usage);
|
|
112
|
+
if (!limiting || !usage) return undefined;
|
|
113
|
+
return {
|
|
114
|
+
id,
|
|
115
|
+
label,
|
|
116
|
+
headroom: headroomFor(limiting.percent),
|
|
117
|
+
limiting,
|
|
118
|
+
source: "plan",
|
|
119
|
+
fetchedAt: usage.fetchedAt,
|
|
120
|
+
plan: usage,
|
|
121
|
+
};
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
/** The soft budget an operator declared for a backend, if any. */
|
|
125
|
+
function budgetFor(
|
|
126
|
+
config: TalonConfig | undefined,
|
|
127
|
+
id: string,
|
|
128
|
+
): { tokensPer5h?: number; tokensPerDay?: number } | undefined {
|
|
129
|
+
const budget = config?.backendBudgets?.[id];
|
|
130
|
+
if (!budget) return undefined;
|
|
131
|
+
if (budget.tokensPer5h === undefined && budget.tokensPerDay === undefined) {
|
|
132
|
+
return undefined;
|
|
133
|
+
}
|
|
134
|
+
return budget;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/** Whether the operator gave this backend a local budget to measure against. */
|
|
138
|
+
export function hasBudget(
|
|
139
|
+
config: TalonConfig | undefined,
|
|
140
|
+
id: string,
|
|
141
|
+
): boolean {
|
|
142
|
+
return budgetFor(config, id) !== undefined;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Headroom from the local ledger. The tighter of the two configured windows
|
|
147
|
+
* wins, so a backend that is fine on the day but has just burned its 5h
|
|
148
|
+
* allowance still reads as full.
|
|
149
|
+
*/
|
|
150
|
+
export function headroomFromLedger(
|
|
151
|
+
id: string,
|
|
152
|
+
label: string,
|
|
153
|
+
config: TalonConfig | undefined,
|
|
154
|
+
now = Date.now(),
|
|
155
|
+
): BackendHeadroom | undefined {
|
|
156
|
+
const budget = budgetFor(config, id);
|
|
157
|
+
if (!budget) return undefined;
|
|
158
|
+
const used = ledgerUsage(id, now);
|
|
159
|
+
const windows: LimitingWindow[] = [];
|
|
160
|
+
if (budget.tokensPer5h !== undefined) {
|
|
161
|
+
windows.push({
|
|
162
|
+
label: "5h (local budget)",
|
|
163
|
+
percent: clampPercent((used.tokens5h / budget.tokensPer5h) * 100),
|
|
164
|
+
});
|
|
165
|
+
}
|
|
166
|
+
if (budget.tokensPerDay !== undefined) {
|
|
167
|
+
windows.push({
|
|
168
|
+
label: "24h (local budget)",
|
|
169
|
+
percent: clampPercent((used.tokensDay / budget.tokensPerDay) * 100),
|
|
170
|
+
});
|
|
171
|
+
}
|
|
172
|
+
let worst = windows[0] as LimitingWindow;
|
|
173
|
+
for (const window of windows)
|
|
174
|
+
if (window.percent > worst.percent) worst = window;
|
|
175
|
+
return {
|
|
176
|
+
id,
|
|
177
|
+
label,
|
|
178
|
+
headroom: headroomFor(worst.percent),
|
|
179
|
+
limiting: worst,
|
|
180
|
+
source: "ledger",
|
|
181
|
+
fetchedAt: now,
|
|
182
|
+
};
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/** The "nothing to measure" reading. Headroom 1, but lowest ranking source. */
|
|
186
|
+
function unknownHeadroom(
|
|
187
|
+
id: string,
|
|
188
|
+
label: string,
|
|
189
|
+
now: number,
|
|
190
|
+
): BackendHeadroom {
|
|
191
|
+
return { id, label, headroom: 1, source: "none", fetchedAt: now };
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Ask a pooled backend for its plan windows. Rejects like the backend does. */
|
|
195
|
+
async function readPlanUsage(id: string): Promise<PlanUsage | undefined> {
|
|
196
|
+
const backend = getPooledBackend(id);
|
|
197
|
+
const read = backend?.usage?.getPlanUsage;
|
|
198
|
+
if (!read || !backend?.usage) return undefined;
|
|
199
|
+
return read.call(backend.usage);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
/**
|
|
203
|
+
* Headroom for one backend, cached for {@link HEADROOM_CACHE_MS}.
|
|
204
|
+
*
|
|
205
|
+
* `force` skips the cache — the `plan_usage` tool asks for a fresh read
|
|
206
|
+
* because the operator is looking at the number right now.
|
|
207
|
+
*/
|
|
208
|
+
export async function getBackendHeadroom(
|
|
209
|
+
id: string,
|
|
210
|
+
label: string,
|
|
211
|
+
config: TalonConfig | undefined,
|
|
212
|
+
options?: { force?: boolean; now?: number },
|
|
213
|
+
): Promise<BackendHeadroom> {
|
|
214
|
+
const now = options?.now ?? Date.now();
|
|
215
|
+
const cached = cache.get(id);
|
|
216
|
+
if (!options?.force && cached && now - cached.cachedAt < HEADROOM_CACHE_MS) {
|
|
217
|
+
return cached.value;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
let value: BackendHeadroom;
|
|
221
|
+
try {
|
|
222
|
+
const plan = headroomFromPlan(id, label, await readPlanUsage(id));
|
|
223
|
+
value =
|
|
224
|
+
plan ??
|
|
225
|
+
headroomFromLedger(id, label, config, now) ??
|
|
226
|
+
unknownHeadroom(id, label, now);
|
|
227
|
+
} catch {
|
|
228
|
+
// The source is unreachable this minute. Keeping the last good reading
|
|
229
|
+
// is the conservative answer: forgetting it would read as "empty" and
|
|
230
|
+
// send the next background run straight at a backend near its ceiling.
|
|
231
|
+
value = cached
|
|
232
|
+
? { ...cached.value, stale: true }
|
|
233
|
+
: unknownHeadroom(id, label, now);
|
|
234
|
+
}
|
|
235
|
+
cache.set(id, { value, cachedAt: now });
|
|
236
|
+
return value;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/** Headroom for every backend the config exposes, in config order. */
|
|
240
|
+
export async function collectBackendHeadroom(
|
|
241
|
+
config: TalonConfig | undefined,
|
|
242
|
+
options?: { force?: boolean; now?: number },
|
|
243
|
+
): Promise<BackendHeadroom[]> {
|
|
244
|
+
const backends = listAvailableBackends(config);
|
|
245
|
+
return Promise.all(
|
|
246
|
+
backends.map(({ id, label }) =>
|
|
247
|
+
getBackendHeadroom(id, label, config, options),
|
|
248
|
+
),
|
|
249
|
+
);
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
/** One-line rendering shared by `/usage`, `plan_usage` and the router log. */
|
|
253
|
+
export function formatHeadroom(entry: BackendHeadroom): string {
|
|
254
|
+
const pct = `${Math.round(entry.headroom * 100)}%`;
|
|
255
|
+
const detail =
|
|
256
|
+
entry.source === "none"
|
|
257
|
+
? "no usage signal"
|
|
258
|
+
: `${entry.limiting?.label ?? "window"} ${Math.round(entry.limiting?.percent ?? 0)}% used`;
|
|
259
|
+
const tag = entry.source === "ledger" ? " (local budget)" : "";
|
|
260
|
+
const stale = entry.stale ? " (stale)" : "";
|
|
261
|
+
return `${pct} — ${detail}${tag}${stale}`;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
/** Test seam — drop every cached reading. */
|
|
265
|
+
export function resetHeadroomCacheForTest(): void {
|
|
266
|
+
cache.clear();
|
|
267
|
+
}
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Plan-aware backend router.
|
|
3
|
+
*
|
|
4
|
+
* - `ledger` — the local rolling token count that gives budget-only
|
|
5
|
+
* backends a headroom signal.
|
|
6
|
+
* - `headroom` — one comparable "how much is left" per backend, from the
|
|
7
|
+
* plan API where there is one and the ledger where there
|
|
8
|
+
* is not.
|
|
9
|
+
* - `router` — the decision: who runs this background job.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
export {
|
|
13
|
+
flushBackendLedger,
|
|
14
|
+
ledgerUsage,
|
|
15
|
+
loadBackendLedger,
|
|
16
|
+
recordBackendRunUsage,
|
|
17
|
+
recordBackendUsage,
|
|
18
|
+
resetBackendLedgerForTest,
|
|
19
|
+
tokensInWindow,
|
|
20
|
+
LEDGER_RETENTION_MS,
|
|
21
|
+
LEDGER_SHORT_WINDOW_MS,
|
|
22
|
+
} from "./ledger.js";
|
|
23
|
+
export {
|
|
24
|
+
collectBackendHeadroom,
|
|
25
|
+
formatHeadroom,
|
|
26
|
+
getBackendHeadroom,
|
|
27
|
+
hasBudget,
|
|
28
|
+
headroomFromLedger,
|
|
29
|
+
headroomFromPlan,
|
|
30
|
+
limitingWindowOf,
|
|
31
|
+
resetHeadroomCacheForTest,
|
|
32
|
+
HEADROOM_CACHE_MS,
|
|
33
|
+
type BackendHeadroom,
|
|
34
|
+
type HeadroomSource,
|
|
35
|
+
type LimitingWindow,
|
|
36
|
+
} from "./headroom.js";
|
|
37
|
+
export {
|
|
38
|
+
collectBackendUsage,
|
|
39
|
+
leadWith,
|
|
40
|
+
type BackendUsageSnapshot,
|
|
41
|
+
} from "./usage.js";
|
|
42
|
+
export {
|
|
43
|
+
chooseBackend,
|
|
44
|
+
resolveRoutedModel,
|
|
45
|
+
taskClassForEffort,
|
|
46
|
+
DEFAULT_CEILING_PERCENT,
|
|
47
|
+
type RouteDecision,
|
|
48
|
+
type RouteHints,
|
|
49
|
+
type RoutePurpose,
|
|
50
|
+
type RouteRequest,
|
|
51
|
+
type TaskClass,
|
|
52
|
+
} from "./router.js";
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local rolling token ledger — the headroom signal for backends with no
|
|
3
|
+
* account usage API.
|
|
4
|
+
*
|
|
5
|
+
* Claude and Codex report subscription windows; `agy` has no account
|
|
6
|
+
* endpoint and `openai-agents` has no plan at all. Without a second signal
|
|
7
|
+
* the router would treat those as infinitely fresh and pile every background
|
|
8
|
+
* run onto them. So Talon counts what it spends itself: every chat turn,
|
|
9
|
+
* one-shot and sub-agent folds its token total into a per-backend ledger,
|
|
10
|
+
* and `headroom.ts` reads that against the operator's soft budget
|
|
11
|
+
* (`config.backendBudgets`).
|
|
12
|
+
*
|
|
13
|
+
* The ledger is deliberately a *local estimate*, not accounting: it only
|
|
14
|
+
* sees what this daemon ran, and a plan API always takes precedence when
|
|
15
|
+
* one exists. Entries older than the widest window (24h) are pruned on
|
|
16
|
+
* every read and write, which also bounds the file.
|
|
17
|
+
*
|
|
18
|
+
* Persistence is `~/.talon/data/backend-ledger.json`, written atomically so
|
|
19
|
+
* a restart mid-write can't leave a truncated file — a zeroed ledger would
|
|
20
|
+
* silently hand a spent backend a full headroom score.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { readFile } from "node:fs/promises";
|
|
24
|
+
import { resolve } from "node:path";
|
|
25
|
+
import { dirs } from "../../../util/paths.js";
|
|
26
|
+
import { logWarn } from "../../../util/log.js";
|
|
27
|
+
import { writePrivateJson } from "../../mesh/persist.js";
|
|
28
|
+
|
|
29
|
+
/** Widest window the ledger answers for; everything older is dropped. */
|
|
30
|
+
export const LEDGER_RETENTION_MS = 24 * 60 * 60_000;
|
|
31
|
+
/** The short window, as `backendBudgets.tokensPer5h` measures it. */
|
|
32
|
+
export const LEDGER_SHORT_WINDOW_MS = 5 * 60 * 60_000;
|
|
33
|
+
|
|
34
|
+
/** How long a write is coalesced for, so a burst of turns is one fsync. */
|
|
35
|
+
const FLUSH_DEBOUNCE_MS = 2_000;
|
|
36
|
+
|
|
37
|
+
/** One recorded spend: when it happened and how many tokens it cost. */
|
|
38
|
+
interface LedgerEntry {
|
|
39
|
+
/** Epoch ms. */
|
|
40
|
+
readonly t: number;
|
|
41
|
+
/** Total tokens (input + output + cache + thinking, as the backend counts). */
|
|
42
|
+
readonly n: number;
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
interface LedgerFile {
|
|
46
|
+
readonly version: 1;
|
|
47
|
+
readonly backends: Record<string, LedgerEntry[]>;
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
interface LedgerState {
|
|
51
|
+
entries: Map<string, LedgerEntry[]>;
|
|
52
|
+
loaded: boolean;
|
|
53
|
+
loading: Promise<void> | null;
|
|
54
|
+
flushTimer: ReturnType<typeof setTimeout> | null;
|
|
55
|
+
dirty: boolean;
|
|
56
|
+
/** Overridable for tests; resolved lazily so TALON_HOME changes are seen. */
|
|
57
|
+
path: string | null;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const state: LedgerState = {
|
|
61
|
+
entries: new Map(),
|
|
62
|
+
loaded: false,
|
|
63
|
+
loading: null,
|
|
64
|
+
flushTimer: null,
|
|
65
|
+
dirty: false,
|
|
66
|
+
path: null,
|
|
67
|
+
};
|
|
68
|
+
|
|
69
|
+
function ledgerPath(): string {
|
|
70
|
+
return state.path ?? resolve(dirs.data, "backend-ledger.json");
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Drop entries that have aged out of the widest window. */
|
|
74
|
+
function prune(list: LedgerEntry[], now: number): LedgerEntry[] {
|
|
75
|
+
const floor = now - LEDGER_RETENTION_MS;
|
|
76
|
+
// Entries are appended in time order, so the survivors are a suffix —
|
|
77
|
+
// but a clock step backwards can break that, hence a filter not a slice.
|
|
78
|
+
return list.filter((e) => e.t > floor);
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
function parseLedger(raw: unknown): Map<string, LedgerEntry[]> {
|
|
82
|
+
const out = new Map<string, LedgerEntry[]>();
|
|
83
|
+
if (!raw || typeof raw !== "object") return out;
|
|
84
|
+
const file = raw as Partial<LedgerFile>;
|
|
85
|
+
if (file.version !== 1 || !file.backends) return out;
|
|
86
|
+
const now = Date.now();
|
|
87
|
+
for (const [id, list] of Object.entries(file.backends)) {
|
|
88
|
+
if (!Array.isArray(list)) continue;
|
|
89
|
+
const clean = list.filter(
|
|
90
|
+
(e): e is LedgerEntry =>
|
|
91
|
+
Boolean(e) &&
|
|
92
|
+
typeof (e as LedgerEntry).t === "number" &&
|
|
93
|
+
typeof (e as LedgerEntry).n === "number" &&
|
|
94
|
+
Number.isFinite((e as LedgerEntry).t) &&
|
|
95
|
+
Number.isFinite((e as LedgerEntry).n),
|
|
96
|
+
);
|
|
97
|
+
const pruned = prune(clean, now);
|
|
98
|
+
if (pruned.length > 0) out.set(id, pruned);
|
|
99
|
+
}
|
|
100
|
+
return out;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
/**
|
|
104
|
+
* Load the persisted ledger once per process. Idempotent and safe to call
|
|
105
|
+
* concurrently — every caller awaits the same read. A missing or corrupt
|
|
106
|
+
* file starts an empty ledger rather than failing a routing decision.
|
|
107
|
+
*/
|
|
108
|
+
export async function loadBackendLedger(): Promise<void> {
|
|
109
|
+
if (state.loaded) return;
|
|
110
|
+
if (state.loading) return state.loading;
|
|
111
|
+
state.loading = (async () => {
|
|
112
|
+
try {
|
|
113
|
+
const raw = await readFile(ledgerPath(), "utf8");
|
|
114
|
+
const parsed = parseLedger(JSON.parse(raw));
|
|
115
|
+
// In-process records taken while the read was in flight win: merge
|
|
116
|
+
// rather than replace, so a turn that landed during boot isn't lost.
|
|
117
|
+
for (const [id, list] of parsed) {
|
|
118
|
+
state.entries.set(id, [...list, ...(state.entries.get(id) ?? [])]);
|
|
119
|
+
}
|
|
120
|
+
} catch {
|
|
121
|
+
/* no ledger yet, or unreadable — start empty */
|
|
122
|
+
}
|
|
123
|
+
state.loaded = true;
|
|
124
|
+
state.loading = null;
|
|
125
|
+
})();
|
|
126
|
+
return state.loading;
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
function scheduleFlush(): void {
|
|
130
|
+
state.dirty = true;
|
|
131
|
+
if (state.flushTimer) return;
|
|
132
|
+
const timer = setTimeout(() => {
|
|
133
|
+
state.flushTimer = null;
|
|
134
|
+
void flushBackendLedger();
|
|
135
|
+
}, FLUSH_DEBOUNCE_MS);
|
|
136
|
+
timer.unref?.();
|
|
137
|
+
state.flushTimer = timer;
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** Write the ledger out now. Exported so shutdown and tests can force it. */
|
|
141
|
+
export async function flushBackendLedger(): Promise<void> {
|
|
142
|
+
if (!state.dirty) return;
|
|
143
|
+
state.dirty = false;
|
|
144
|
+
const now = Date.now();
|
|
145
|
+
const backends: Record<string, LedgerEntry[]> = {};
|
|
146
|
+
for (const [id, list] of state.entries) {
|
|
147
|
+
const pruned = prune(list, now);
|
|
148
|
+
state.entries.set(id, pruned);
|
|
149
|
+
if (pruned.length > 0) backends[id] = pruned;
|
|
150
|
+
}
|
|
151
|
+
try {
|
|
152
|
+
await writePrivateJson(ledgerPath(), {
|
|
153
|
+
version: 1,
|
|
154
|
+
backends,
|
|
155
|
+
} satisfies LedgerFile);
|
|
156
|
+
} catch (err) {
|
|
157
|
+
logWarn(
|
|
158
|
+
"router",
|
|
159
|
+
`Could not persist the backend ledger: ${err instanceof Error ? err.message : String(err)}`,
|
|
160
|
+
);
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* Fold one run's token spend into a backend's ledger.
|
|
166
|
+
*
|
|
167
|
+
* Called from the shared turn accounting and from every one-shot completion
|
|
168
|
+
* path, so *all* backends accumulate a ledger — the plan API simply wins over
|
|
169
|
+
* it where one exists. Cheap and synchronous: the disk write is debounced.
|
|
170
|
+
*/
|
|
171
|
+
export function recordBackendUsage(
|
|
172
|
+
backendId: string,
|
|
173
|
+
tokens: number,
|
|
174
|
+
at = Date.now(),
|
|
175
|
+
): void {
|
|
176
|
+
if (!backendId) return;
|
|
177
|
+
if (!Number.isFinite(tokens) || tokens <= 0) return;
|
|
178
|
+
// A first record before the file has been read would be overwritten by the
|
|
179
|
+
// load; kick the read off here so the merge in loadBackendLedger keeps it.
|
|
180
|
+
if (!state.loaded && !state.loading) void loadBackendLedger();
|
|
181
|
+
const list = state.entries.get(backendId) ?? [];
|
|
182
|
+
list.push({ t: at, n: Math.round(tokens) });
|
|
183
|
+
state.entries.set(backendId, prune(list, at));
|
|
184
|
+
scheduleFlush();
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Fold a completed run's usage into the ledger. The same four fields every
|
|
189
|
+
* background path reports (`OneShotUsage`, `TaskUsage`), summed — cache
|
|
190
|
+
* reads included, because they still count against a subscription window.
|
|
191
|
+
*/
|
|
192
|
+
export function recordBackendRunUsage(
|
|
193
|
+
backendId: string,
|
|
194
|
+
usage:
|
|
195
|
+
| {
|
|
196
|
+
inputTokens?: number;
|
|
197
|
+
outputTokens?: number;
|
|
198
|
+
cacheRead?: number;
|
|
199
|
+
cacheWrite?: number;
|
|
200
|
+
}
|
|
201
|
+
| undefined
|
|
202
|
+
| null,
|
|
203
|
+
at = Date.now(),
|
|
204
|
+
): void {
|
|
205
|
+
if (!usage) return;
|
|
206
|
+
const total =
|
|
207
|
+
(usage.inputTokens ?? 0) +
|
|
208
|
+
(usage.outputTokens ?? 0) +
|
|
209
|
+
(usage.cacheRead ?? 0) +
|
|
210
|
+
(usage.cacheWrite ?? 0);
|
|
211
|
+
recordBackendUsage(backendId, total, at);
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** Tokens a backend spent inside a window ending now. */
|
|
215
|
+
export function tokensInWindow(
|
|
216
|
+
backendId: string,
|
|
217
|
+
windowMs: number,
|
|
218
|
+
now = Date.now(),
|
|
219
|
+
): number {
|
|
220
|
+
const list = state.entries.get(backendId);
|
|
221
|
+
if (!list || list.length === 0) return 0;
|
|
222
|
+
const floor = now - windowMs;
|
|
223
|
+
let total = 0;
|
|
224
|
+
for (const entry of list) if (entry.t > floor) total += entry.n;
|
|
225
|
+
return total;
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
/** Both windows the budget schema knows about, for one backend. */
|
|
229
|
+
export function ledgerUsage(
|
|
230
|
+
backendId: string,
|
|
231
|
+
now = Date.now(),
|
|
232
|
+
): { tokens5h: number; tokensDay: number } {
|
|
233
|
+
return {
|
|
234
|
+
tokens5h: tokensInWindow(backendId, LEDGER_SHORT_WINDOW_MS, now),
|
|
235
|
+
tokensDay: tokensInWindow(backendId, LEDGER_RETENTION_MS, now),
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/** Test seam — point the ledger at a temp file and start from empty. */
|
|
240
|
+
export function resetBackendLedgerForTest(path?: string): void {
|
|
241
|
+
if (state.flushTimer) clearTimeout(state.flushTimer);
|
|
242
|
+
state.entries = new Map();
|
|
243
|
+
state.loaded = false;
|
|
244
|
+
state.loading = null;
|
|
245
|
+
state.flushTimer = null;
|
|
246
|
+
state.dirty = false;
|
|
247
|
+
state.path = path ?? null;
|
|
248
|
+
}
|