talon-agent 5.24.0 → 5.24.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app.ts +30 -0
- package/src/backend/agy/process/orphans.ts +5 -1
- package/src/backend/claude-sdk/one-shot.ts +8 -1
- package/src/backend/codex/factory.ts +5 -1
- package/src/backend/codex/plan-usage.ts +12 -0
- package/src/backend/runtime/turn/turn-phases.ts +7 -1
- package/src/bootstrap.ts +1 -0
- package/src/cli.ts +18 -12
- package/src/core/agent-runtime/capabilities.ts +7 -0
- package/src/core/agents/runner.ts +4 -0
- package/src/core/background/cron/job-oneshot.ts +7 -1
- package/src/core/background/cron/scheduler.ts +28 -14
- package/src/core/background/cron/spec.ts +23 -3
- package/src/core/background/heartbeat/agent.ts +6 -0
- package/src/core/background/triggers/resume.ts +19 -4
- package/src/core/background/triggers/spawn.ts +1 -1
- package/src/core/config/index.ts +82 -0
- package/src/core/daemon/discovery.ts +160 -0
- package/src/core/daemon/pidfile.ts +107 -0
- package/src/core/daemon/respawn.ts +4 -1
- package/src/core/engine/backend-router/breaker.ts +158 -0
- package/src/core/engine/backend-router/headroom.ts +90 -22
- package/src/core/engine/backend-router/index.ts +11 -0
- package/src/core/engine/gateway-actions/cron.ts +5 -0
- package/src/core/engine/gateway-actions/fetch-url/guard.ts +13 -44
- package/src/core/engine/gateway-actions/fetch-url/index.ts +23 -43
- package/src/core/engine/gateway-actions/mesh.ts +4 -0
- package/src/core/engine/gateway-actions/models.ts +19 -4
- package/src/core/engine/gateway-routes.ts +31 -2
- package/src/core/engine/gateway.ts +11 -1
- package/src/core/fetch/classify.ts +73 -0
- package/src/core/fetch/curl-impersonate.ts +447 -0
- package/src/core/fetch/errors.ts +14 -0
- package/src/core/fetch/index.ts +133 -0
- package/src/core/fetch/ladder.ts +401 -0
- package/src/core/fetch/rungs.ts +319 -0
- package/src/core/fetch/types.ts +105 -0
- package/src/core/frontend-runtime/admin-notify.ts +100 -12
- package/src/core/mcp-hub/child-guard.ts +215 -0
- package/src/core/mcp-hub/child-transport.ts +89 -39
- package/src/core/mcp-hub/children.ts +65 -9
- package/src/core/mcp-hub/guest-scope.ts +3 -2
- package/src/core/mcp-hub/index.ts +36 -16
- package/src/core/mcp-hub/launcher.ts +81 -37
- package/src/core/mcp-hub/proxy-server.ts +12 -8
- package/src/core/mcp-hub/reaper.ts +120 -0
- package/src/core/mesh/credentials/index.ts +1 -1
- package/src/core/mesh/credentials/store.ts +1 -1
- package/src/core/mesh/devices/service.ts +8 -0
- package/src/core/mesh/links/bridge-links.ts +37 -0
- package/src/core/mesh/links/companion-pairing.ts +2 -2
- package/src/core/plugin/mcp.ts +8 -8
- package/src/core/tools/bridge.ts +3 -0
- package/src/core/tools/content/web.ts +1 -1
- package/src/core/tools/ops/mesh.ts +21 -0
- package/src/core/tools/ops/scheduling.ts +15 -0
- package/src/frontend/native/bridge/credentials/claims.ts +11 -1
- package/src/frontend/telegram/actions/media.ts +164 -12
- package/src/index.ts +22 -19
- package/src/plugins/playwright/version-coupling.ts +172 -62
- package/src/storage/cron.ts +34 -2
- package/src/storage/db.ts +1 -0
- package/src/storage/repositories/cron-repo.ts +9 -0
- package/src/storage/sql/cron.sql +5 -5
- package/src/storage/sql/db.sql +5 -0
- package/src/storage/sql/schema.sql +2 -1
- package/src/storage/sql/statements.generated.ts +10 -6
- package/src/util/log.ts +2 -1
- package/src/util/paths.ts +2 -0
- package/src/core/background/triggers/pid.ts +0 -28
|
@@ -17,6 +17,7 @@
|
|
|
17
17
|
*/
|
|
18
18
|
|
|
19
19
|
import { readPidRecord, isProcessAlive } from "./pidfile.js";
|
|
20
|
+
import { log, logWarn } from "../../util/log.js";
|
|
20
21
|
import {
|
|
21
22
|
gatewayAuthHeaders,
|
|
22
23
|
readGatewayToken,
|
|
@@ -174,3 +175,162 @@ export async function findRunningInstance(
|
|
|
174
175
|
|
|
175
176
|
return scanForDaemon();
|
|
176
177
|
}
|
|
178
|
+
|
|
179
|
+
// ── Single-instance guard ───────────────────────────────────────────────────
|
|
180
|
+
|
|
181
|
+
/**
|
|
182
|
+
* Single-instance guard — a daemon refuses to boot while another one runs.
|
|
183
|
+
*
|
|
184
|
+
* `talon start` already checked for a running instance, but only the CLI
|
|
185
|
+
* did. The daemon entry wrote its pidfile unconditionally. Anything that
|
|
186
|
+
* started the entry directly (systemd, a stray `bun src/index.ts`, the
|
|
187
|
+
* handoff watcher's fallback racing a slow successor) got a second daemon.
|
|
188
|
+
* On 2026-09-27 that ran for 13 minutes:
|
|
189
|
+
* - the newcomer overwrote the pidfile;
|
|
190
|
+
* - its gateway and bridge fell back to 19877/19881;
|
|
191
|
+
* - both polled Telegram, and every getUpdates answered 409;
|
|
192
|
+
* - the newcomer's trigger resume killed the running daemon's watchers
|
|
193
|
+
* as "orphans".
|
|
194
|
+
*
|
|
195
|
+
* This guard runs first thing in `app.ts`, before any side effect. It
|
|
196
|
+
* refuses when an identity-checked `/health` answers as a daemon (or a
|
|
197
|
+
* live pidfile pid turns into one while we wait for it to finish booting).
|
|
198
|
+
*
|
|
199
|
+
* The one daemon allowed to overlap is our own predecessor. A `/restart`
|
|
200
|
+
* successor is spawned in the last moments of the old process's graceful
|
|
201
|
+
* shutdown, so the guard waits for that pid to exit and then carries on.
|
|
202
|
+
*/
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Set by `spawnSuccessor()` on the child it spawns: the pid of the daemon
|
|
206
|
+
* handing over. The guard waits that process out instead of refusing.
|
|
207
|
+
*/
|
|
208
|
+
export const PREDECESSOR_PID_ENV = "TALON_PREDECESSOR_PID";
|
|
209
|
+
|
|
210
|
+
/** How long a successor waits for its predecessor to exit. */
|
|
211
|
+
const PREDECESSOR_WAIT_MS = 20_000;
|
|
212
|
+
/** How long a live-but-silent pidfile pid gets to answer /health. */
|
|
213
|
+
const UNVERIFIED_WAIT_MS = 15_000;
|
|
214
|
+
const GUARD_POLL_MS = 250;
|
|
215
|
+
|
|
216
|
+
export type InstanceCheck =
|
|
217
|
+
{ ok: true } | { ok: false; instance: RunningInstance };
|
|
218
|
+
|
|
219
|
+
export interface InstanceCheckDeps {
|
|
220
|
+
find?: (pidfilePath?: string) => Promise<RunningInstance | null>;
|
|
221
|
+
isAlive?: (pid: number) => boolean;
|
|
222
|
+
sleep?: (ms: number) => Promise<void>;
|
|
223
|
+
now?: () => number;
|
|
224
|
+
env?: NodeJS.ProcessEnv;
|
|
225
|
+
pidfilePath?: string;
|
|
226
|
+
predecessorWaitMs?: number;
|
|
227
|
+
unverifiedWaitMs?: number;
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
const defaultSleep = (ms: number): Promise<void> =>
|
|
231
|
+
new Promise((resolve) => setTimeout(resolve, ms));
|
|
232
|
+
|
|
233
|
+
function predecessorPid(env: NodeJS.ProcessEnv): number | undefined {
|
|
234
|
+
const pid = Number(env[PREDECESSOR_PID_ENV]);
|
|
235
|
+
return Number.isInteger(pid) && pid > 0 && pid !== process.pid
|
|
236
|
+
? pid
|
|
237
|
+
: undefined;
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/** Poll until `done()` or the deadline; resolves whether `done()` held. */
|
|
241
|
+
async function waitFor(
|
|
242
|
+
done: () => boolean | Promise<boolean>,
|
|
243
|
+
ms: number,
|
|
244
|
+
sleep: (ms: number) => Promise<void>,
|
|
245
|
+
now: () => number,
|
|
246
|
+
): Promise<boolean> {
|
|
247
|
+
const deadline = now() + ms;
|
|
248
|
+
for (;;) {
|
|
249
|
+
if (await done()) return true;
|
|
250
|
+
if (now() >= deadline) return false;
|
|
251
|
+
await sleep(GUARD_POLL_MS);
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
/**
|
|
256
|
+
* Decide whether this process may boot as the daemon. Never throws: a
|
|
257
|
+
* discovery failure answers "ok" (a guard that keeps the only daemon down
|
|
258
|
+
* is worse than the duplicate it prevents).
|
|
259
|
+
*/
|
|
260
|
+
export async function checkSingleInstance(
|
|
261
|
+
deps: InstanceCheckDeps = {},
|
|
262
|
+
): Promise<InstanceCheck> {
|
|
263
|
+
const find = deps.find ?? findRunningInstance;
|
|
264
|
+
const isAlive = deps.isAlive ?? isProcessAlive;
|
|
265
|
+
const sleep = deps.sleep ?? defaultSleep;
|
|
266
|
+
const now = deps.now ?? Date.now;
|
|
267
|
+
const env = deps.env ?? process.env;
|
|
268
|
+
|
|
269
|
+
try {
|
|
270
|
+
const predecessor = predecessorPid(env);
|
|
271
|
+
// Consumed: children of this daemon must not inherit it.
|
|
272
|
+
delete env[PREDECESSOR_PID_ENV];
|
|
273
|
+
if (predecessor !== undefined && isAlive(predecessor)) {
|
|
274
|
+
log("bot", `Waiting for predecessor daemon ${predecessor} to exit`);
|
|
275
|
+
const gone = await waitFor(
|
|
276
|
+
() => !isAlive(predecessor),
|
|
277
|
+
deps.predecessorWaitMs ?? PREDECESSOR_WAIT_MS,
|
|
278
|
+
sleep,
|
|
279
|
+
now,
|
|
280
|
+
);
|
|
281
|
+
if (!gone) {
|
|
282
|
+
// It is past its own 15s force-exit timer and has already let go of
|
|
283
|
+
// the frontends. Staying down would leave nothing running.
|
|
284
|
+
logWarn(
|
|
285
|
+
"bot",
|
|
286
|
+
`Predecessor daemon ${predecessor} is still alive — booting anyway`,
|
|
287
|
+
);
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
let instance = await find(deps.pidfilePath);
|
|
292
|
+
if (instance?.source === "pidfile-unverified") {
|
|
293
|
+
// A live pid with no /health: a daemon still booting, or a recycled
|
|
294
|
+
// pid. Give it time to answer before deciding.
|
|
295
|
+
const pid = instance.pid;
|
|
296
|
+
await waitFor(
|
|
297
|
+
async () => {
|
|
298
|
+
if (!isAlive(pid)) return true;
|
|
299
|
+
instance = await find(deps.pidfilePath);
|
|
300
|
+
return instance?.source !== "pidfile-unverified";
|
|
301
|
+
},
|
|
302
|
+
deps.unverifiedWaitMs ?? UNVERIFIED_WAIT_MS,
|
|
303
|
+
sleep,
|
|
304
|
+
now,
|
|
305
|
+
);
|
|
306
|
+
if (instance?.source === "pidfile-unverified") {
|
|
307
|
+
logWarn(
|
|
308
|
+
"bot",
|
|
309
|
+
`pidfile names live pid ${instance.pid} but no daemon answers — ` +
|
|
310
|
+
`treating it as a recycled pid`,
|
|
311
|
+
);
|
|
312
|
+
return { ok: true };
|
|
313
|
+
}
|
|
314
|
+
}
|
|
315
|
+
if (!instance) return { ok: true };
|
|
316
|
+
if (instance.pid === process.pid || instance.pid === predecessor) {
|
|
317
|
+
return { ok: true };
|
|
318
|
+
}
|
|
319
|
+
return { ok: false, instance };
|
|
320
|
+
} catch (err) {
|
|
321
|
+
logWarn(
|
|
322
|
+
"bot",
|
|
323
|
+
`single-instance check failed (${err instanceof Error ? err.message : String(err)}) — booting`,
|
|
324
|
+
);
|
|
325
|
+
return { ok: true };
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
|
|
329
|
+
/** One line for the log and stderr when the guard refuses. */
|
|
330
|
+
export function describeRefusal(instance: RunningInstance): string {
|
|
331
|
+
const where = instance.port ? ` on :${instance.port}` : "";
|
|
332
|
+
return (
|
|
333
|
+
`another Talon daemon is already running (pid ${instance.pid}${where}). ` +
|
|
334
|
+
"Refusing to start a second one — use `talon restart` to replace it."
|
|
335
|
+
);
|
|
336
|
+
}
|
|
@@ -96,3 +96,110 @@ export function isProcessAlive(pid: number): boolean {
|
|
|
96
96
|
return (err as NodeJS.ErrnoException).code === "EPERM";
|
|
97
97
|
}
|
|
98
98
|
}
|
|
99
|
+
|
|
100
|
+
// ── Child ownership ─────────────────────────────────────────────────────────
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Daemon ownership of child processes.
|
|
104
|
+
*
|
|
105
|
+
* Every process the daemon spawns (backend CLIs, MCP children, trigger
|
|
106
|
+
* scripts) inherits its environment. At boot the daemon stamps two
|
|
107
|
+
* variables into that environment: its own pid and its /proc start time.
|
|
108
|
+
* Any orphan sweep can then tell a child whose daemon is gone (safe to
|
|
109
|
+
* reap) from a child of a daemon that is still running.
|
|
110
|
+
*
|
|
111
|
+
* The distinction matters because "orphan" used to mean "alive and
|
|
112
|
+
* tagged with our chat or trigger id". On 2026-09-27 two daemons ran at
|
|
113
|
+
* once for 13 minutes. Each one's sweeps saw the other's live children as
|
|
114
|
+
* leftovers from a previous run, and the newcomer's trigger resume
|
|
115
|
+
* SIGKILLed the running daemon's watchers.
|
|
116
|
+
*
|
|
117
|
+
* Linux-only in practice: the reads go through /proc. Where /proc is
|
|
118
|
+
* absent, {@link childBelongsToLiveDaemon} answers false and the sweeps
|
|
119
|
+
* behave exactly as before.
|
|
120
|
+
*/
|
|
121
|
+
|
|
122
|
+
/** Pid of the daemon that spawned this process (inherited env). */
|
|
123
|
+
export const DAEMON_PID_ENV = "TALON_DAEMON_PID";
|
|
124
|
+
/** That daemon's /proc start time, so a recycled pid cannot pass for it. */
|
|
125
|
+
export const DAEMON_STARTTIME_ENV = "TALON_DAEMON_STARTTIME";
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Field 22 of /proc/<pid>/stat: start time in jiffies since boot.
|
|
129
|
+
* Monotonic per boot and unchanged by exec(), so it pins a pid to one
|
|
130
|
+
* process. `undefined` without /proc or when the read fails.
|
|
131
|
+
*
|
|
132
|
+
* The `comm` field (2nd) is wrapped in parens and may itself contain ')'
|
|
133
|
+
* — split after the LAST ')'; index 19 of the rest is field 22.
|
|
134
|
+
*/
|
|
135
|
+
export function readPidStarttimeSync(pid: number): number | undefined {
|
|
136
|
+
try {
|
|
137
|
+
const stat = readFileSync(`/proc/${pid}/stat`, "utf-8");
|
|
138
|
+
const lastParen = stat.lastIndexOf(")");
|
|
139
|
+
if (lastParen < 0) return undefined;
|
|
140
|
+
const tail = stat.slice(lastParen + 2).split(" ");
|
|
141
|
+
const starttime = Number(tail[19]);
|
|
142
|
+
return Number.isFinite(starttime) ? starttime : undefined;
|
|
143
|
+
} catch {
|
|
144
|
+
return undefined;
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Stamp this process as the daemon in `env` (default: our own), so every
|
|
150
|
+
* child spawned from here on names us. Overwrites any inherited stamp: a
|
|
151
|
+
* `/restart` successor inherits its predecessor's environment.
|
|
152
|
+
*/
|
|
153
|
+
export function stampDaemonOwner(env: NodeJS.ProcessEnv = process.env): void {
|
|
154
|
+
env[DAEMON_PID_ENV] = String(process.pid);
|
|
155
|
+
const starttime = readPidStarttimeSync(process.pid);
|
|
156
|
+
if (starttime !== undefined) env[DAEMON_STARTTIME_ENV] = String(starttime);
|
|
157
|
+
else delete env[DAEMON_STARTTIME_ENV];
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
export interface DaemonOwner {
|
|
161
|
+
pid: number;
|
|
162
|
+
starttime?: number;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
/** The owner stamp carried in a NUL-split `/proc/<pid>/environ`. */
|
|
166
|
+
export function ownerFromEnviron(entries: string[]): DaemonOwner | undefined {
|
|
167
|
+
const value = (key: string): string | undefined => {
|
|
168
|
+
const prefix = `${key}=`;
|
|
169
|
+
return entries.find((e) => e.startsWith(prefix))?.slice(prefix.length);
|
|
170
|
+
};
|
|
171
|
+
const pid = Number(value(DAEMON_PID_ENV));
|
|
172
|
+
if (!Number.isInteger(pid) || pid <= 0) return undefined;
|
|
173
|
+
const starttime = Number(value(DAEMON_STARTTIME_ENV));
|
|
174
|
+
return Number.isFinite(starttime) && value(DAEMON_STARTTIME_ENV)
|
|
175
|
+
? { pid, starttime }
|
|
176
|
+
: { pid };
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* Whether `owner` is a daemon other than this one that is still running.
|
|
181
|
+
* A matching pid with a different start time is a recycled pid, so the
|
|
182
|
+
* owner is dead.
|
|
183
|
+
*/
|
|
184
|
+
export function isOtherLiveDaemon(owner: DaemonOwner | undefined): boolean {
|
|
185
|
+
if (!owner || owner.pid === process.pid) return false;
|
|
186
|
+
if (!isProcessAlive(owner.pid)) return false;
|
|
187
|
+
if (owner.starttime === undefined) return true;
|
|
188
|
+
const current = readPidStarttimeSync(owner.pid);
|
|
189
|
+
return current === undefined || current === owner.starttime;
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Whether process `pid` was spawned by a daemon other than this one that
|
|
194
|
+
* is still alive. Orphan sweeps must leave such a process alone. Reads
|
|
195
|
+
* `/proc/<pid>/environ`; false when it can't be read.
|
|
196
|
+
*/
|
|
197
|
+
export function childBelongsToLiveDaemon(pid: number): boolean {
|
|
198
|
+
let raw: string;
|
|
199
|
+
try {
|
|
200
|
+
raw = readFileSync(`/proc/${pid}/environ`, "utf-8");
|
|
201
|
+
} catch {
|
|
202
|
+
return false;
|
|
203
|
+
}
|
|
204
|
+
return isOtherLiveDaemon(ownerFromEnviron(raw.split("\0")));
|
|
205
|
+
}
|
|
@@ -55,6 +55,7 @@
|
|
|
55
55
|
import { spawn } from "node:child_process";
|
|
56
56
|
import { log, logError, openRespawnLog } from "../../util/log.js";
|
|
57
57
|
import { HANDOFF_WATCH_SUBCOMMAND } from "./handoff.js";
|
|
58
|
+
import { PREDECESSOR_PID_ENV } from "./discovery.js";
|
|
58
59
|
|
|
59
60
|
let pendingReason: string | null = null;
|
|
60
61
|
let shutdown: ((reason: string) => void) | null = null;
|
|
@@ -187,7 +188,9 @@ export function spawnSuccessor(spawnFn: SpawnFn = spawn): void {
|
|
|
187
188
|
cwd: process.cwd(),
|
|
188
189
|
detached: true,
|
|
189
190
|
stdio: ["ignore", fd ?? "ignore", fd ?? "ignore"],
|
|
190
|
-
|
|
191
|
+
// Tells the successor's single-instance guard that the daemon it
|
|
192
|
+
// can still see is us, on our way out — wait, don't refuse.
|
|
193
|
+
env: { ...process.env, [PREDECESSOR_PID_ENV]: String(process.pid) },
|
|
191
194
|
});
|
|
192
195
|
child.once("error", (err) => {
|
|
193
196
|
logError("shutdown", `Respawn failed (${reason}) — see respawn.log`, err);
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Backend circuit breaker — "is this backend actually working right now?"
|
|
3
|
+
*
|
|
4
|
+
* Headroom answers "how much plan is left". It says nothing about whether
|
|
5
|
+
* runs on the backend succeed. In Sep 2026 Codex's login expired and its
|
|
6
|
+
* default model was retired, so every run on it failed. Codex still had no
|
|
7
|
+
* usage signal and read as 100% headroom, and the router kept sending
|
|
8
|
+
* background work there for 36 hours. This module holds the missing
|
|
9
|
+
* signal, fed by the outcome of every routed run:
|
|
10
|
+
*
|
|
11
|
+
* - an auth failure (401, expired login, "run `… login`") opens the
|
|
12
|
+
* breaker at once: the next run is not going to fix a credential;
|
|
13
|
+
* - {@link BREAKER_FAILURE_THRESHOLD} consecutive failures of any other
|
|
14
|
+
* kind open it too;
|
|
15
|
+
* - a success closes it and clears the count.
|
|
16
|
+
*
|
|
17
|
+
* An open breaker reads as zero headroom (see `headroom.ts`) until its
|
|
18
|
+
* cool-off ends. After that the backend is half-open: it can win again,
|
|
19
|
+
* and one more failure without a success in between re-opens it straight
|
|
20
|
+
* away with twice the cool-off, capped at {@link BREAKER_MAX_COOLOFF_MS}.
|
|
21
|
+
*
|
|
22
|
+
* State lives in memory only. A restart gives every backend a fresh
|
|
23
|
+
* chance, which is what an operator who just ran `codex login` and
|
|
24
|
+
* restarted expects.
|
|
25
|
+
*/
|
|
26
|
+
|
|
27
|
+
import { logWarn } from "../../../util/log.js";
|
|
28
|
+
|
|
29
|
+
/** Consecutive non-auth failures that open the breaker. */
|
|
30
|
+
export const BREAKER_FAILURE_THRESHOLD = 3;
|
|
31
|
+
/** First cool-off. Each re-trip without a success in between doubles it. */
|
|
32
|
+
export const BREAKER_BASE_COOLOFF_MS = 15 * 60_000;
|
|
33
|
+
/** Ceiling on the cool-off. */
|
|
34
|
+
export const BREAKER_MAX_COOLOFF_MS = 4 * 60 * 60_000;
|
|
35
|
+
|
|
36
|
+
interface BreakerState {
|
|
37
|
+
/** Failures since the last success. */
|
|
38
|
+
consecutive: number;
|
|
39
|
+
/** Times the breaker has opened since the last success. */
|
|
40
|
+
trips: number;
|
|
41
|
+
/** Epoch ms the current cool-off ends; undefined when never opened. */
|
|
42
|
+
openUntil?: number;
|
|
43
|
+
/** Why it last opened. */
|
|
44
|
+
reason?: string;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** An open breaker, as `headroom.ts` and the router log see it. */
|
|
48
|
+
export interface OpenBreaker {
|
|
49
|
+
readonly reason: string;
|
|
50
|
+
/** Epoch ms the cool-off ends. */
|
|
51
|
+
readonly until: number;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const breakers = new Map<string, BreakerState>();
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Credential failures, not model or transport ones. Covers what the
|
|
58
|
+
* backends actually say: HTTP 401, Codex's "refresh token" / "login
|
|
59
|
+
* expired" wording, agy's "authentication required", and any "run
|
|
60
|
+
* `x login`" remedy.
|
|
61
|
+
*/
|
|
62
|
+
const AUTH_FAILURE_RE =
|
|
63
|
+
/\b401\b|unauthori[sz]ed|authentication (?:required|failed)|not (?:logged|signed) in|log(?:in|ged in) (?:has )?expired|refresh[_ ]token|invalid[_ ](?:api[_ ])?key|run [`'"]?[\w-]+ login/i;
|
|
64
|
+
|
|
65
|
+
/** Whether an error message describes a credential problem. */
|
|
66
|
+
export function isAuthFailureMessage(message: string): boolean {
|
|
67
|
+
return AUTH_FAILURE_RE.test(message);
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function messageOf(err: unknown): string {
|
|
71
|
+
return err instanceof Error ? err.message : String(err);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* A run the caller cancelled says nothing about the backend. A timeout
|
|
76
|
+
* does count: a backend that hangs is as unusable as one that errors.
|
|
77
|
+
*/
|
|
78
|
+
function isCallerAbort(err: unknown): boolean {
|
|
79
|
+
if (err instanceof Error && err.name === "AbortError") return true;
|
|
80
|
+
return /aborted before the run started/i.test(messageOf(err));
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
function stateFor(id: string): BreakerState {
|
|
84
|
+
let state = breakers.get(id);
|
|
85
|
+
if (!state) {
|
|
86
|
+
state = { consecutive: 0, trips: 0 };
|
|
87
|
+
breakers.set(id, state);
|
|
88
|
+
}
|
|
89
|
+
return state;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function open(id: string, state: BreakerState, reason: string, now: number) {
|
|
93
|
+
state.trips += 1;
|
|
94
|
+
const cooloff = Math.min(
|
|
95
|
+
BREAKER_BASE_COOLOFF_MS * 2 ** (state.trips - 1),
|
|
96
|
+
BREAKER_MAX_COOLOFF_MS,
|
|
97
|
+
);
|
|
98
|
+
state.openUntil = now + cooloff;
|
|
99
|
+
state.reason = reason;
|
|
100
|
+
logWarn(
|
|
101
|
+
"router",
|
|
102
|
+
`breaker open for ${id} (${reason}) — no routed work for ` +
|
|
103
|
+
`${Math.round(cooloff / 60_000)}m`,
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** Record a failed run on a backend. */
|
|
108
|
+
export function recordBackendRunFailure(
|
|
109
|
+
id: string,
|
|
110
|
+
err: unknown,
|
|
111
|
+
now = Date.now(),
|
|
112
|
+
): void {
|
|
113
|
+
if (isCallerAbort(err)) return;
|
|
114
|
+
const state = stateFor(id);
|
|
115
|
+
state.consecutive += 1;
|
|
116
|
+
const message = messageOf(err).split("\n")[0]?.trim().slice(0, 160) ?? "";
|
|
117
|
+
// Already open: a failure in the cool-off (a pinned run, say) must not
|
|
118
|
+
// stretch it.
|
|
119
|
+
if (state.openUntil !== undefined && now < state.openUntil) return;
|
|
120
|
+
if (isAuthFailureMessage(message)) {
|
|
121
|
+
open(id, state, `auth failure: ${message}`, now);
|
|
122
|
+
return;
|
|
123
|
+
}
|
|
124
|
+
// Half-open (tripped before, no success since): the probe failed, so
|
|
125
|
+
// re-open at once rather than waiting for a fresh run of failures.
|
|
126
|
+
if (state.trips > 0) {
|
|
127
|
+
open(id, state, `still failing after cool-off: ${message}`, now);
|
|
128
|
+
return;
|
|
129
|
+
}
|
|
130
|
+
if (state.consecutive >= BREAKER_FAILURE_THRESHOLD) {
|
|
131
|
+
open(
|
|
132
|
+
id,
|
|
133
|
+
state,
|
|
134
|
+
`${state.consecutive} consecutive failures: ${message}`,
|
|
135
|
+
now,
|
|
136
|
+
);
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** Record a successful run: the backend works, forget its failures. */
|
|
141
|
+
export function recordBackendRunSuccess(id: string): void {
|
|
142
|
+
breakers.delete(id);
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/** The breaker, when it is open at `now`; undefined when the backend may run. */
|
|
146
|
+
export function openBreaker(
|
|
147
|
+
id: string,
|
|
148
|
+
now = Date.now(),
|
|
149
|
+
): OpenBreaker | undefined {
|
|
150
|
+
const state = breakers.get(id);
|
|
151
|
+
if (!state?.openUntil || now >= state.openUntil) return undefined;
|
|
152
|
+
return { reason: state.reason ?? "failing", until: state.openUntil };
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
/** Test seam — close every breaker. */
|
|
156
|
+
export function resetBackendBreakersForTest(): void {
|
|
157
|
+
breakers.clear();
|
|
158
|
+
}
|
|
@@ -11,11 +11,21 @@
|
|
|
11
11
|
* the operator's soft budget (`config.backendBudgets`). The
|
|
12
12
|
* fallback for backends with no account API.
|
|
13
13
|
*
|
|
14
|
-
* A backend with neither reports `source: "none"` and headroom
|
|
15
|
-
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
14
|
+
* A backend with neither reports `source: "none"` and headroom 0: nothing
|
|
15
|
+
* is known about it, so it is never *preferred*. It still clears the
|
|
16
|
+
* ceiling (there is no window to be over), so it runs work when nothing
|
|
17
|
+
* measured is available. It used to read as 1. That made a Codex install
|
|
18
|
+
* with an expired login, and so no usage signal, the router's favourite
|
|
19
|
+
* for 36 hours while every run on it failed.
|
|
20
|
+
*
|
|
21
|
+
* On top of either source sit two "this backend is not working" signals.
|
|
22
|
+
* Either one zeroes the headroom and pins the limiting window at 100%, so
|
|
23
|
+
* the ceiling drops the backend whenever anything else is left:
|
|
24
|
+
*
|
|
25
|
+
* - the backend's own telemetry reports a rejected credential
|
|
26
|
+
* (`UsageTelemetry.getAuthFailure`, e.g. the Codex usage endpoint's 401);
|
|
27
|
+
* - the run breaker is open (`breaker.ts`): an auth failure or repeated
|
|
28
|
+
* failures on runs routed there.
|
|
19
29
|
*
|
|
20
30
|
* Reads are cached for 60s per backend: `/usage`, the router and the
|
|
21
31
|
* `plan_usage` tool all ask, and a plan lookup can be a subprocess spawn. A
|
|
@@ -31,6 +41,7 @@ import {
|
|
|
31
41
|
listAvailableBackends,
|
|
32
42
|
} from "../backend-controller/index.js";
|
|
33
43
|
import { ledgerUsage } from "./ledger.js";
|
|
44
|
+
import { openBreaker } from "./breaker.js";
|
|
34
45
|
|
|
35
46
|
/** How long a headroom reading is reused before the source is asked again. */
|
|
36
47
|
export const HEADROOM_CACHE_MS = 60_000;
|
|
@@ -63,6 +74,11 @@ export interface BackendHeadroom {
|
|
|
63
74
|
* same (cached) fetch the router ranked on, instead of asking twice.
|
|
64
75
|
*/
|
|
65
76
|
readonly plan?: PlanUsage;
|
|
77
|
+
/**
|
|
78
|
+
* Why the backend is treated as unusable right now (a rejected login, an
|
|
79
|
+
* open breaker). Set means headroom 0 and a 100% limiting window.
|
|
80
|
+
*/
|
|
81
|
+
readonly unavailable?: string;
|
|
66
82
|
}
|
|
67
83
|
|
|
68
84
|
interface CacheEntry {
|
|
@@ -183,13 +199,29 @@ export function headroomFromLedger(
|
|
|
183
199
|
};
|
|
184
200
|
}
|
|
185
201
|
|
|
186
|
-
/** The "nothing to measure" reading
|
|
202
|
+
/** The "nothing to measure" reading: headroom 0, but under any ceiling. */
|
|
187
203
|
function unknownHeadroom(
|
|
188
204
|
id: string,
|
|
189
205
|
label: string,
|
|
190
206
|
now: number,
|
|
191
207
|
): BackendHeadroom {
|
|
192
|
-
return { id, label, headroom:
|
|
208
|
+
return { id, label, headroom: 0, source: "none", fetchedAt: now };
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
/** What a backend's telemetry said: its plan windows and any auth failure. */
|
|
212
|
+
interface PlanRead {
|
|
213
|
+
usage?: PlanUsage;
|
|
214
|
+
authFailure?: string;
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
function authFailureOf(
|
|
218
|
+
usage: { getAuthFailure?(): string | undefined } | undefined,
|
|
219
|
+
): string | undefined {
|
|
220
|
+
try {
|
|
221
|
+
return usage?.getAuthFailure?.call(usage) || undefined;
|
|
222
|
+
} catch {
|
|
223
|
+
return undefined;
|
|
224
|
+
}
|
|
193
225
|
}
|
|
194
226
|
|
|
195
227
|
/**
|
|
@@ -202,22 +234,35 @@ function unknownHeadroom(
|
|
|
202
234
|
* is *not* currently in use. The read is the backend's own cached one, so at
|
|
203
235
|
* most one boot per cache window.
|
|
204
236
|
*/
|
|
205
|
-
async function readPlanUsage(id: string): Promise<
|
|
237
|
+
async function readPlanUsage(id: string): Promise<PlanRead> {
|
|
206
238
|
const pooled = getPooledBackend(id);
|
|
207
|
-
if (pooled
|
|
208
|
-
|
|
239
|
+
if (pooled) {
|
|
240
|
+
const usage = pooled.usage?.getPlanUsage
|
|
241
|
+
? await pooled.usage.getPlanUsage.call(pooled.usage)
|
|
242
|
+
: undefined; // pooled, but reports no plan windows
|
|
243
|
+
// Read after the plan fetch: that fetch is what discovers a 401.
|
|
244
|
+
const authFailure = authFailureOf(pooled.usage);
|
|
245
|
+
return {
|
|
246
|
+
...(usage ? { usage } : {}),
|
|
247
|
+
...(authFailure ? { authFailure } : {}),
|
|
248
|
+
};
|
|
209
249
|
}
|
|
210
|
-
if (pooled) return undefined; // pooled, but reports no plan windows
|
|
211
250
|
let acquired;
|
|
212
251
|
try {
|
|
213
252
|
acquired = await acquireBackendInstance(id);
|
|
214
253
|
} catch {
|
|
215
|
-
return
|
|
254
|
+
return {}; // can't boot it (not configured, no auth) — stay quiet
|
|
216
255
|
}
|
|
217
256
|
try {
|
|
218
|
-
const
|
|
219
|
-
|
|
220
|
-
|
|
257
|
+
const telemetry = acquired.backend.usage;
|
|
258
|
+
const usage = telemetry?.getPlanUsage
|
|
259
|
+
? await telemetry.getPlanUsage.call(telemetry)
|
|
260
|
+
: undefined;
|
|
261
|
+
const authFailure = authFailureOf(telemetry);
|
|
262
|
+
return {
|
|
263
|
+
...(usage ? { usage } : {}),
|
|
264
|
+
...(authFailure ? { authFailure } : {}),
|
|
265
|
+
};
|
|
221
266
|
} finally {
|
|
222
267
|
await acquired.release().catch(() => {});
|
|
223
268
|
}
|
|
@@ -238,16 +283,18 @@ export async function getBackendHeadroom(
|
|
|
238
283
|
const now = options?.now ?? Date.now();
|
|
239
284
|
const cached = cache.get(id);
|
|
240
285
|
if (!options?.force && cached && now - cached.cachedAt < HEADROOM_CACHE_MS) {
|
|
241
|
-
return cached.value;
|
|
286
|
+
return withBreaker(cached.value, now);
|
|
242
287
|
}
|
|
243
288
|
|
|
244
289
|
let value: BackendHeadroom;
|
|
245
290
|
try {
|
|
246
|
-
const
|
|
291
|
+
const read = await readPlanUsage(id);
|
|
292
|
+
const plan = headroomFromPlan(id, label, read.usage);
|
|
247
293
|
value =
|
|
248
294
|
plan ??
|
|
249
295
|
headroomFromLedger(id, label, config, now) ??
|
|
250
296
|
unknownHeadroom(id, label, now);
|
|
297
|
+
if (read.authFailure) value = unavailable(value, read.authFailure);
|
|
251
298
|
} catch {
|
|
252
299
|
// The source is unreachable this minute. Keeping the last good reading
|
|
253
300
|
// is the conservative answer: forgetting it would read as "empty" and
|
|
@@ -257,7 +304,29 @@ export async function getBackendHeadroom(
|
|
|
257
304
|
: unknownHeadroom(id, label, now);
|
|
258
305
|
}
|
|
259
306
|
cache.set(id, { value, cachedAt: now });
|
|
260
|
-
return value;
|
|
307
|
+
return withBreaker(value, now);
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/** Mark a reading unusable: zero headroom, limiting window pinned at 100%. */
|
|
311
|
+
function unavailable(value: BackendHeadroom, why: string): BackendHeadroom {
|
|
312
|
+
return {
|
|
313
|
+
...value,
|
|
314
|
+
headroom: 0,
|
|
315
|
+
limiting: { label: why, percent: 100 },
|
|
316
|
+
unavailable: why,
|
|
317
|
+
};
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
/**
|
|
321
|
+
* Overlay the run breaker. Applied on every read rather than cached: the
|
|
322
|
+
* breaker opens and closes on run outcomes, not on the headroom clock.
|
|
323
|
+
*/
|
|
324
|
+
function withBreaker(value: BackendHeadroom, now: number): BackendHeadroom {
|
|
325
|
+
if (value.unavailable) return value;
|
|
326
|
+
const breaker = openBreaker(value.id, now);
|
|
327
|
+
if (!breaker) return value;
|
|
328
|
+
const mins = Math.max(1, Math.ceil((breaker.until - now) / 60_000));
|
|
329
|
+
return unavailable(value, `breaker open ${mins}m — ${breaker.reason}`);
|
|
261
330
|
}
|
|
262
331
|
|
|
263
332
|
/** Headroom for every backend the config exposes, in config order. */
|
|
@@ -275,11 +344,10 @@ export async function collectBackendHeadroom(
|
|
|
275
344
|
|
|
276
345
|
/** One-line rendering shared by `/usage`, `plan_usage` and the router log. */
|
|
277
346
|
export function formatHeadroom(entry: BackendHeadroom): string {
|
|
347
|
+
if (entry.unavailable) return `0% — unavailable: ${entry.unavailable}`;
|
|
348
|
+
if (entry.source === "none") return "unmeasured — no usage signal";
|
|
278
349
|
const pct = `${Math.round(entry.headroom * 100)}%`;
|
|
279
|
-
const detail =
|
|
280
|
-
entry.source === "none"
|
|
281
|
-
? "no usage signal"
|
|
282
|
-
: `${entry.limiting?.label ?? "window"} ${Math.round(entry.limiting?.percent ?? 0)}% used`;
|
|
350
|
+
const detail = `${entry.limiting?.label ?? "window"} ${Math.round(entry.limiting?.percent ?? 0)}% used`;
|
|
283
351
|
const tag = entry.source === "ledger" ? " (local budget)" : "";
|
|
284
352
|
const stale = entry.stale ? " (stale)" : "";
|
|
285
353
|
return `${pct} — ${detail}${tag}${stale}`;
|
|
@@ -20,6 +20,17 @@ export {
|
|
|
20
20
|
LEDGER_RETENTION_MS,
|
|
21
21
|
LEDGER_SHORT_WINDOW_MS,
|
|
22
22
|
} from "./ledger.js";
|
|
23
|
+
export {
|
|
24
|
+
isAuthFailureMessage,
|
|
25
|
+
openBreaker,
|
|
26
|
+
recordBackendRunFailure,
|
|
27
|
+
recordBackendRunSuccess,
|
|
28
|
+
resetBackendBreakersForTest,
|
|
29
|
+
BREAKER_BASE_COOLOFF_MS,
|
|
30
|
+
BREAKER_FAILURE_THRESHOLD,
|
|
31
|
+
BREAKER_MAX_COOLOFF_MS,
|
|
32
|
+
type OpenBreaker,
|
|
33
|
+
} from "./breaker.js";
|
|
23
34
|
export {
|
|
24
35
|
collectBackendHeadroom,
|
|
25
36
|
formatHeadroom,
|
|
@@ -133,6 +133,9 @@ export const cronHandlers: SharedActionHandlers = {
|
|
|
133
133
|
: null,
|
|
134
134
|
endAt !== undefined ? `ends: ${new Date(endAt).toISOString()}` : null,
|
|
135
135
|
spec.catchup ? `catch-up: ${spec.catchup}` : null,
|
|
136
|
+
spec.timeoutMs !== undefined
|
|
137
|
+
? `timeout: ${Math.round(spec.timeoutMs / 1000)}s`
|
|
138
|
+
: null,
|
|
136
139
|
]
|
|
137
140
|
.filter(Boolean)
|
|
138
141
|
.join(", ");
|
|
@@ -173,6 +176,8 @@ export const cronHandlers: SharedActionHandlers = {
|
|
|
173
176
|
if (j.catchup && j.catchup !== "skip")
|
|
174
177
|
bounds.push(`catch-up: ${j.catchup}`);
|
|
175
178
|
if (j.model) bounds.push(`model: ${j.model}`);
|
|
179
|
+
if (j.timeoutMs !== undefined)
|
|
180
|
+
bounds.push(`timeout: ${Math.round(j.timeoutMs / 1000)}s`);
|
|
176
181
|
return [
|
|
177
182
|
`- ${j.name} (${status})`,
|
|
178
183
|
` ID: ${j.id}`,
|