@ours.network/fleet 0.17.1 → 0.17.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +38 -2
- package/dist/application/role-removal-service.js +1 -1
- package/dist/application/session-control.d.ts +14 -10
- package/dist/application/session-control.js +14 -3
- package/dist/atomic-file.d.ts +7 -1
- package/dist/atomic-file.js +33 -5
- package/dist/build-info.json +10 -0
- package/dist/capabilities.d.ts +20 -0
- package/dist/capabilities.js +21 -0
- package/dist/cli.js +98 -10
- package/dist/config.d.ts +9 -2
- package/dist/config.js +16 -2
- package/dist/creation.d.ts +16 -0
- package/dist/creation.js +28 -0
- package/dist/docs.d.ts +1 -1
- package/dist/docs.js +70 -4
- package/dist/doctor.d.ts +5 -0
- package/dist/doctor.js +87 -2
- package/dist/harness/acp-agent.d.ts +3 -0
- package/dist/harness/acp-agent.js +4 -1
- package/dist/harness/codex-app-server-proxy.d.ts +4 -0
- package/dist/harness/codex-app-server-proxy.js +133 -0
- package/dist/harness/codex.js +79 -11
- package/dist/index.d.ts +2 -1
- package/dist/index.js +1 -0
- package/dist/loops/manager.d.ts +42 -1
- package/dist/loops/manager.js +115 -16
- package/dist/loops/state.d.ts +46 -2
- package/dist/loops/state.js +81 -3
- package/dist/monitor.d.ts +21 -0
- package/dist/monitor.js +42 -0
- package/dist/ops.d.ts +6 -0
- package/dist/ops.js +46 -1
- package/dist/owner-channel/channel.d.ts +18 -2
- package/dist/owner-channel/channel.js +146 -2
- package/dist/owner-channel/commands.d.ts +2 -2
- package/dist/owner-channel/commands.js +7 -2
- package/dist/owner-channel/notices.d.ts +2 -0
- package/dist/owner-channel/notices.js +3 -0
- package/dist/provenance.d.ts +77 -0
- package/dist/provenance.js +283 -0
- package/dist/runner.d.ts +7 -1
- package/dist/runner.js +109 -16
- package/dist/session/acp.d.ts +40 -4
- package/dist/session/acp.js +157 -30
- package/dist/session/arbiter.d.ts +28 -2
- package/dist/session/arbiter.js +75 -4
- package/dist/session/control.js +12 -6
- package/dist/session/event-log.d.ts +109 -0
- package/dist/session/event-log.js +247 -0
- package/dist/session/events.d.ts +21 -0
- package/dist/session/events.js +105 -26
- package/dist/session/tmux.d.ts +3 -2
- package/dist/session/tmux.js +2 -0
- package/dist/session/types.d.ts +39 -2
- package/dist/session/types.js +11 -1
- package/dist/spawn.d.ts +3 -3
- package/dist/spawn.js +40 -14
- package/dist/temp-lifecycle.d.ts +62 -0
- package/dist/temp-lifecycle.js +437 -0
- package/dist/web-app/assets/{TerminalView-B3rnVWbo.js → TerminalView-BAVk1Bot.js} +1 -1
- package/dist/web-app/assets/{index-CliHATFt.js → index-C3S-xFRU.js} +5 -5
- package/dist/web-app/index.html +1 -1
- package/package.json +5 -3
package/dist/harness/codex.js
CHANGED
|
@@ -1,10 +1,11 @@
|
|
|
1
|
-
import { mkdirSync } from 'node:fs';
|
|
2
|
-
import { join } from 'node:path';
|
|
1
|
+
import { chmodSync, existsSync, mkdirSync, readFileSync, writeFileSync } from 'node:fs';
|
|
2
|
+
import { dirname, join, resolve } from 'node:path';
|
|
3
|
+
import { fileURLToPath } from 'node:url';
|
|
3
4
|
import { agentDir, home } from '../paths.js';
|
|
4
5
|
import { realExec } from '../exec.js';
|
|
5
6
|
import { registerAdapter } from './registry.js';
|
|
6
7
|
import { harnessRuntimeDir } from '../isolation/policy.js';
|
|
7
|
-
import { bundledAcpAgent } from './acp-agent.js';
|
|
8
|
+
import { bundledAcpAgent, resolveBundledAcpAgent } from './acp-agent.js';
|
|
8
9
|
const OPTION_KEYS = [
|
|
9
10
|
'launcher', 'sandbox', 'approval', 'permission_mode', 'search', 'profile', 'config', 'add_dirs',
|
|
10
11
|
'monitor',
|
|
@@ -15,6 +16,11 @@ const SANDBOX_MODES = ['read-only', 'workspace-write', 'danger-full-access'];
|
|
|
15
16
|
/** Codex CLI's accepted `--ask-for-approval` values. */
|
|
16
17
|
const APPROVAL_POLICIES = ['untrusted', 'on-request', 'never'];
|
|
17
18
|
const BUNDLED_CODEX_ACP_VERSION = '1.1.7';
|
|
19
|
+
const CODEX_ACP_PACKAGE = '@agentclientprotocol/codex-acp';
|
|
20
|
+
const CODEX_PROXY_APPROVAL_ENV = 'OURS_FLEET_CODEX_APPROVAL';
|
|
21
|
+
const CODEX_PROXY_SANDBOX_ENV = 'OURS_FLEET_CODEX_SANDBOX';
|
|
22
|
+
const CODEX_PROXY_REAL_PATH_ENV = 'OURS_FLEET_REAL_CODEX_PATH';
|
|
23
|
+
const CODEX_PROXY_MANIFEST_ENV = 'OURS_FLEET_CODEX_ACP_MANIFEST';
|
|
18
24
|
/**
|
|
19
25
|
* What an unattended role can actually do under Codex's native settings.
|
|
20
26
|
* `on-request` and `untrusted` stop to ask, and with no console attached that
|
|
@@ -93,6 +99,59 @@ function approvalPolicy(role) {
|
|
|
93
99
|
function launcherMode(role) {
|
|
94
100
|
return role.harness_options?.launcher ?? 'auto';
|
|
95
101
|
}
|
|
102
|
+
function bundledCodexAcp() {
|
|
103
|
+
return resolveBundledAcpAgent(CODEX_ACP_PACKAGE, 'codex-acp', 'codex-acp');
|
|
104
|
+
}
|
|
105
|
+
function canOverrideBundledAcpApproval() {
|
|
106
|
+
const resolution = bundledCodexAcp();
|
|
107
|
+
return resolution.bundled && resolution.version === BUNDLED_CODEX_ACP_VERSION
|
|
108
|
+
&& resolution.manifestPath !== undefined && compiledProxyModule() !== undefined;
|
|
109
|
+
}
|
|
110
|
+
function compiledProxyModule() {
|
|
111
|
+
const adjacent = fileURLToPath(new URL('./codex-app-server-proxy.js', import.meta.url));
|
|
112
|
+
if (existsSync(adjacent))
|
|
113
|
+
return adjacent;
|
|
114
|
+
// Vitest imports src/ directly; globalSetup builds the executable module in dist/.
|
|
115
|
+
const fromSource = resolve(dirname(fileURLToPath(import.meta.url)), '../../dist/harness', 'codex-app-server-proxy.js');
|
|
116
|
+
if (existsSync(fromSource))
|
|
117
|
+
return fromSource;
|
|
118
|
+
return undefined;
|
|
119
|
+
}
|
|
120
|
+
const shellQuote = (value) => `'${value.replaceAll("'", "'\\''")}'`;
|
|
121
|
+
/**
|
|
122
|
+
* Materialize the tiny CODEX_PATH executable inside the role's own state dir.
|
|
123
|
+
* Keeping it there makes the proxy available under both ordinary and isolated
|
|
124
|
+
* launches without adding another host path to the filesystem boundary.
|
|
125
|
+
*/
|
|
126
|
+
function codexAcpEnvironment(role, dirs) {
|
|
127
|
+
if (role.session !== 'acp' || role.session_options?.acp?.command != null)
|
|
128
|
+
return {};
|
|
129
|
+
const resolution = bundledCodexAcp();
|
|
130
|
+
if (!resolution.bundled || resolution.version !== BUNDLED_CODEX_ACP_VERSION
|
|
131
|
+
|| !resolution.manifestPath)
|
|
132
|
+
return {};
|
|
133
|
+
const runtimeDir = harnessRuntimeDir(dirs.stateDir, 'codex');
|
|
134
|
+
mkdirSync(runtimeDir, { recursive: true });
|
|
135
|
+
const proxyModule = join(runtimeDir, 'app-server-proxy.mjs');
|
|
136
|
+
const source = compiledProxyModule();
|
|
137
|
+
if (!source)
|
|
138
|
+
throw new Error('Codex app-server proxy is missing; rebuild ours-fleet');
|
|
139
|
+
writeFileSync(proxyModule, readFileSync(source, 'utf8'), { mode: 0o600 });
|
|
140
|
+
const windows = process.platform === 'win32';
|
|
141
|
+
const command = join(runtimeDir, windows ? 'codex-app-server-proxy.cmd' : 'codex-app-server-proxy');
|
|
142
|
+
const script = windows
|
|
143
|
+
? `@echo off\r\n"${process.execPath.replaceAll('"', '""')}" "${proxyModule.replaceAll('"', '""')}" %*\r\n`
|
|
144
|
+
: `#!/bin/sh\nexec ${shellQuote(process.execPath)} ${shellQuote(proxyModule)} "$@"\n`;
|
|
145
|
+
writeFileSync(command, script, { mode: 0o700 });
|
|
146
|
+
chmodSync(command, 0o700);
|
|
147
|
+
return {
|
|
148
|
+
CODEX_PATH: command,
|
|
149
|
+
[CODEX_PROXY_APPROVAL_ENV]: approvalPolicy(role) ?? 'on-request',
|
|
150
|
+
[CODEX_PROXY_SANDBOX_ENV]: sandboxMode(role) ?? 'workspace-write',
|
|
151
|
+
[CODEX_PROXY_MANIFEST_ENV]: resolution.manifestPath,
|
|
152
|
+
...(process.env.CODEX_PATH ? { [CODEX_PROXY_REAL_PATH_ENV]: process.env.CODEX_PATH } : {}),
|
|
153
|
+
};
|
|
154
|
+
}
|
|
96
155
|
function encodeTomlValue(value) {
|
|
97
156
|
if (typeof value === 'string')
|
|
98
157
|
return JSON.stringify(value);
|
|
@@ -226,7 +285,7 @@ export function makeCodexAdapter(exec = realExec) {
|
|
|
226
285
|
if (requested === 'ours-codex' && !hasOursCodex)
|
|
227
286
|
throw new Error('harness_options.launcher is ours-codex, but ours-codex is not on PATH; install @ours.network/codex or use launcher: auto');
|
|
228
287
|
const command = requested === 'codex' ? 'codex' : hasOursCodex ? 'ours-codex' : 'codex';
|
|
229
|
-
return { argv: [], env:
|
|
288
|
+
return { argv: [], env: codexAcpEnvironment(role, dirs), command };
|
|
230
289
|
},
|
|
231
290
|
buildLaunch(role, mode, _s, prep) {
|
|
232
291
|
const stateDir = roleStateDir(role);
|
|
@@ -244,7 +303,7 @@ export function makeCodexAdapter(exec = realExec) {
|
|
|
244
303
|
? [...configured]
|
|
245
304
|
: typeof configured === 'string'
|
|
246
305
|
? ['sh', '-c', configured]
|
|
247
|
-
: bundledAcpAgent(
|
|
306
|
+
: bundledAcpAgent(CODEX_ACP_PACKAGE, 'codex-acp', 'codex-acp');
|
|
248
307
|
const initialMode = acpAgentMode(role);
|
|
249
308
|
return {
|
|
250
309
|
argv,
|
|
@@ -301,17 +360,22 @@ export function makeCodexAdapter(exec = realExec) {
|
|
|
301
360
|
const sandbox = sandboxMode(role) ?? 'workspace-write';
|
|
302
361
|
if (role.session === 'acp') {
|
|
303
362
|
const mode = acpAgentMode(role) ?? 'agent';
|
|
304
|
-
const
|
|
363
|
+
const configured = role.session_options?.acp?.command;
|
|
364
|
+
const overrideAvailable = configured == null && canOverrideBundledAcpApproval();
|
|
365
|
+
const actual = overrideAvailable ? { approval, sandbox } : acpModePermissions(mode);
|
|
305
366
|
const exact = actual.approval === approval && actual.sandbox === sandbox;
|
|
306
367
|
return {
|
|
307
368
|
...translated,
|
|
308
369
|
native: { mode, ...actual },
|
|
309
370
|
exact,
|
|
310
|
-
warnings: exact ? [] : [
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
371
|
+
warnings: exact ? [] : [configured != null
|
|
372
|
+
? `custom ACP command cannot be verified against approval=${approval} sandbox=${sandbox}; `
|
|
373
|
+
+ `its '${mode}' mode is conservatively treated as approval=${actual.approval} `
|
|
374
|
+
+ `sandbox=${actual.sandbox}`
|
|
375
|
+
: `codex-acp mode '${mode}' actually uses approval=${actual.approval} `
|
|
376
|
+
+ `sandbox=${actual.sandbox}, and the bundled ${BUNDLED_CODEX_ACP_VERSION} `
|
|
377
|
+
+ `app-server override is unavailable; this does not exactly represent `
|
|
378
|
+
+ `approval=${approval} sandbox=${sandbox}`],
|
|
315
379
|
capabilities: codexCapabilities(actual.approval, actual.sandbox),
|
|
316
380
|
};
|
|
317
381
|
}
|
|
@@ -324,6 +388,10 @@ export function makeCodexAdapter(exec = realExec) {
|
|
|
324
388
|
effectivePermissionMode(role) {
|
|
325
389
|
if (role.session === 'acp') {
|
|
326
390
|
const nativeMode = acpAgentMode(role) ?? 'agent';
|
|
391
|
+
if (role.session_options?.acp?.command == null && canOverrideBundledAcpApproval()) {
|
|
392
|
+
const approval = approvalPolicy(role) ?? 'on-request';
|
|
393
|
+
return { fleetMode: fleetModeForApproval(approval), nativeMode };
|
|
394
|
+
}
|
|
327
395
|
return {
|
|
328
396
|
fleetMode: fleetModeForApproval(acpModePermissions(nativeMode).approval), nativeMode,
|
|
329
397
|
};
|
package/dist/index.d.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
export { loadConfig, findRole, resolvePermissions, ConfigError } from './config.js';
|
|
2
2
|
export type { FleetConfig, ResolvedRole, RoleConfig, OverseeEntry, SessionBackendId, CommonPermissions, FleetPermissionMode, ApprovalMode, SessionOptions, OwnerChannelConfig, } from './config.js';
|
|
3
3
|
export type { HarnessAdapter, BriefingVocab, ExitPolicy, PrereqReport, PrereqCheck, SessionPrep, SessionState, Launch, AcpLaunch, PermissionTranslation, RoleDirs, ValidationError, } from './harness/types.js';
|
|
4
|
-
export type { SessionHandle, SessionSnapshot, SessionEvent, TurnResult, } from './session/types.js';
|
|
4
|
+
export type { SessionHandle, SessionSnapshot, SessionEvent, TurnResult, InterruptOutcome, InterruptResult, TurnCancellationSource, QueuedPrompt, ExitRecord, } from './session/types.js';
|
|
5
|
+
export { interruptOutcome } from './session/types.js';
|
|
5
6
|
export { AcpSession } from './session/acp.js';
|
|
6
7
|
export { OwnerChannel } from './owner-channel/channel.js';
|
|
7
8
|
export { TmuxSession } from './session/tmux.js';
|
package/dist/index.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
export { loadConfig, findRole, resolvePermissions, ConfigError } from './config.js';
|
|
2
|
+
export { interruptOutcome } from './session/types.js';
|
|
2
3
|
export { AcpSession } from './session/acp.js';
|
|
3
4
|
export { OwnerChannel } from './owner-channel/channel.js';
|
|
4
5
|
export { TmuxSession } from './session/tmux.js';
|
package/dist/loops/manager.d.ts
CHANGED
|
@@ -1,12 +1,22 @@
|
|
|
1
1
|
import type { ResolvedRoleLoop } from './config.js';
|
|
2
|
-
import { type ScheduledLoopsFile } from './state.js';
|
|
2
|
+
import { type ScheduledLoopsFile, type StateWriter } from './state.js';
|
|
3
3
|
import { RoleTurnArbiter } from '../session/arbiter.js';
|
|
4
4
|
export interface LoopManagerDeps {
|
|
5
5
|
now(): number;
|
|
6
6
|
setTimer(callback: () => void, ms: number): unknown;
|
|
7
7
|
clearTimer(timer: unknown): void;
|
|
8
8
|
log(line: string): void;
|
|
9
|
+
/** Persistence seam; defaults to the atomic replace. Tests inject write faults here. */
|
|
10
|
+
writeState?: StateWriter;
|
|
11
|
+
/** Test seam for the post-cancellation abandon bound; production uses the default. */
|
|
12
|
+
cancelAbandonMs?: number;
|
|
9
13
|
}
|
|
14
|
+
/**
|
|
15
|
+
* Recovery delay for the nth consecutive failure: 1s doubling to the poll
|
|
16
|
+
* ceiling. Bounded at both ends — never a busy retry, never longer than the
|
|
17
|
+
* normal cadence, so recovery is noticed within a minute of the fault clearing.
|
|
18
|
+
*/
|
|
19
|
+
export declare function backoffMs(consecutiveFailures: number): number;
|
|
10
20
|
export interface LoopActionResult {
|
|
11
21
|
state: 'started' | 'skipped_busy' | 'disabled' | 'unavailable';
|
|
12
22
|
runId?: string;
|
|
@@ -29,6 +39,8 @@ export declare class ScheduledLoopManager implements ScheduledLoopManagerHandle
|
|
|
29
39
|
private timer?;
|
|
30
40
|
private readonly runTimeouts;
|
|
31
41
|
private stopping;
|
|
42
|
+
/** Consecutive unforeseen poll failures; drives the recovery backoff. */
|
|
43
|
+
private pollFailures;
|
|
32
44
|
constructor(role: string, definitions: ResolvedRoleLoop[], stateDir: string, arbiter: RoleTurnArbiter, deps: LoopManagerDeps);
|
|
33
45
|
start(): void;
|
|
34
46
|
stop(): Promise<void>;
|
|
@@ -40,10 +52,39 @@ export declare class ScheduledLoopManager implements ScheduledLoopManagerHandle
|
|
|
40
52
|
/** Public fake-clock seam; timer callbacks call the same transition. */
|
|
41
53
|
poll(): Promise<void>;
|
|
42
54
|
private attempt;
|
|
55
|
+
/**
|
|
56
|
+
* Last resort for a run whose cancellation never settles. Without it a single
|
|
57
|
+
* unsettled turn keeps `activeRunId` set for the life of the process, and
|
|
58
|
+
* every later tick reports `skipped_busy` forever — the loop looks scheduled
|
|
59
|
+
* while nothing has run since. Releasing the slot is honest, and it is safe
|
|
60
|
+
* because `finish` still refuses to double-report a run it no longer owns.
|
|
61
|
+
*
|
|
62
|
+
* Releasing the loop's own slot is only half of it. The cancellation that
|
|
63
|
+
* never settled is still holding the arbiter's admission boundary, so the
|
|
64
|
+
* generation it belongs to has to be retired in the same step — otherwise the
|
|
65
|
+
* loop believes it is free while every later producer, scheduled or owner,
|
|
66
|
+
* queues behind a promise that will never resolve.
|
|
67
|
+
*/
|
|
68
|
+
private armAbandon;
|
|
43
69
|
private finish;
|
|
44
70
|
private advance;
|
|
45
71
|
private skipMissed;
|
|
46
72
|
private skipRestartMisses;
|
|
73
|
+
/**
|
|
74
|
+
* A run the store could not record is dropped, not retried: the cursor has
|
|
75
|
+
* already moved, so this can never become a busy loop, and the outage is
|
|
76
|
+
* visible as a skip with a failed health rather than as silence.
|
|
77
|
+
*/
|
|
78
|
+
private skipUnpersisted;
|
|
47
79
|
private schedule;
|
|
80
|
+
private arm;
|
|
81
|
+
/**
|
|
82
|
+
* The residual safety net. Persistence failures are absorbed by the store, so
|
|
83
|
+
* reaching here means something unforeseen threw — and whatever it was, the
|
|
84
|
+
* manager must stay armed. The previous behaviour left no timer at all, which
|
|
85
|
+
* turned one transient ENOSPC into every loop on the role being silently gone
|
|
86
|
+
* until the process was restarted.
|
|
87
|
+
*/
|
|
88
|
+
private recover;
|
|
48
89
|
private envelope;
|
|
49
90
|
}
|
package/dist/loops/manager.js
CHANGED
|
@@ -1,5 +1,23 @@
|
|
|
1
1
|
import { randomUUID } from 'node:crypto';
|
|
2
2
|
import { ScheduledLoopStateStore, deterministicJitter, increment, } from './state.js';
|
|
3
|
+
/** The manager never sleeps longer than this, so one poll per minute is the floor rate. */
|
|
4
|
+
const POLL_CEILING_MS = 60_000;
|
|
5
|
+
/**
|
|
6
|
+
* Recovery delay for the nth consecutive failure: 1s doubling to the poll
|
|
7
|
+
* ceiling. Bounded at both ends — never a busy retry, never longer than the
|
|
8
|
+
* normal cadence, so recovery is noticed within a minute of the fault clearing.
|
|
9
|
+
*/
|
|
10
|
+
export function backoffMs(consecutiveFailures) {
|
|
11
|
+
const exponent = Math.min(Math.max(consecutiveFailures, 1) - 1, 6);
|
|
12
|
+
return Math.min(POLL_CEILING_MS, 1_000 * 2 ** exponent);
|
|
13
|
+
}
|
|
14
|
+
/**
|
|
15
|
+
* How long a cancelled run may stay `active` before the loop stops believing it
|
|
16
|
+
* will ever settle. Deliberately longer than the session layer's own escalation
|
|
17
|
+
* (cancel grace + SIGTERM->SIGKILL grace), so this fires only when that recovery
|
|
18
|
+
* did not, and never races a run that is about to report honestly.
|
|
19
|
+
*/
|
|
20
|
+
const LOOP_CANCEL_ABANDON_MS = 60_000;
|
|
3
21
|
export class ScheduledLoopManager {
|
|
4
22
|
role;
|
|
5
23
|
arbiter;
|
|
@@ -9,13 +27,15 @@ export class ScheduledLoopManager {
|
|
|
9
27
|
timer;
|
|
10
28
|
runTimeouts = new Set();
|
|
11
29
|
stopping = false;
|
|
30
|
+
/** Consecutive unforeseen poll failures; drives the recovery backoff. */
|
|
31
|
+
pollFailures = 0;
|
|
12
32
|
constructor(role, definitions, stateDir, arbiter, deps) {
|
|
13
33
|
this.role = role;
|
|
14
34
|
this.arbiter = arbiter;
|
|
15
35
|
this.deps = deps;
|
|
16
36
|
for (const definition of definitions)
|
|
17
37
|
this.definitions.set(definition.name, definition);
|
|
18
|
-
this.store = new ScheduledLoopStateStore(stateDir, role, definitions, deps.now(), deps.log);
|
|
38
|
+
this.store = new ScheduledLoopStateStore(stateDir, role, definitions, deps.now(), deps.log, deps.writeState);
|
|
19
39
|
}
|
|
20
40
|
start() {
|
|
21
41
|
if (!this.store.fresh)
|
|
@@ -102,10 +122,21 @@ export class ScheduledLoopManager {
|
|
|
102
122
|
}
|
|
103
123
|
const scheduledAt = Date.parse(state.nextScheduledAt);
|
|
104
124
|
this.advance(definition, state);
|
|
105
|
-
|
|
125
|
+
// The claim has to be durable before the turn is submitted: a run that is
|
|
126
|
+
// not on disk is a run a restart cannot see, and the occurrence would be
|
|
127
|
+
// submitted a second time. At-most-once outranks running this occurrence.
|
|
128
|
+
if (!this.store.persist()) {
|
|
129
|
+
this.skipUnpersisted(definition, state, now);
|
|
130
|
+
continue;
|
|
131
|
+
}
|
|
106
132
|
await this.attempt(definition, state, scheduledAt);
|
|
107
133
|
}
|
|
108
134
|
this.store.state.clock.lastWallMs = now;
|
|
135
|
+
if (this.pollFailures) {
|
|
136
|
+
this.store.clearAnomaly('manager_task_failed');
|
|
137
|
+
this.deps.log(`[${this.role}] scheduled loop manager recovered after ${this.pollFailures} failed poll(s)`);
|
|
138
|
+
this.pollFailures = 0;
|
|
139
|
+
}
|
|
109
140
|
this.store.persist();
|
|
110
141
|
this.schedule();
|
|
111
142
|
}
|
|
@@ -153,9 +184,16 @@ export class ScheduledLoopManager {
|
|
|
153
184
|
if (state.activeRunId !== runId || this.stopping)
|
|
154
185
|
return;
|
|
155
186
|
this.deps.log(`[${this.role}] loop ${definition.name} timed out run=${runId.slice(0, 11)} after ${timeoutMs}ms; cancelling`);
|
|
156
|
-
void this.arbiter.interrupt('scheduled-loop').
|
|
187
|
+
void this.arbiter.interrupt('scheduled-loop').then(outcome => {
|
|
188
|
+
// Forced recovery IS a successful cancellation; record how it ended
|
|
189
|
+
// rather than reporting the interrupt itself as a failure.
|
|
190
|
+
if (outcome?.state === 'forced')
|
|
191
|
+
this.deps.log(`[${this.role}] loop ${definition.name} cancellation enforced `
|
|
192
|
+
+ `run=${runId.slice(0, 11)} reason=${outcome.reasonCode ?? 'forced'}`);
|
|
193
|
+
}).catch(error => {
|
|
157
194
|
this.deps.log(`[${this.role}] loop ${definition.name} timeout cancellation failed: ${error?.name ?? 'Error'}`);
|
|
158
195
|
});
|
|
196
|
+
this.armAbandon(definition, state, runId);
|
|
159
197
|
}, timeoutMs);
|
|
160
198
|
this.runTimeouts.add(timeout);
|
|
161
199
|
void result.queued.completion.then(turn => {
|
|
@@ -165,6 +203,38 @@ export class ScheduledLoopManager {
|
|
|
165
203
|
});
|
|
166
204
|
return { state: 'started', runId };
|
|
167
205
|
}
|
|
206
|
+
/**
|
|
207
|
+
* Last resort for a run whose cancellation never settles. Without it a single
|
|
208
|
+
* unsettled turn keeps `activeRunId` set for the life of the process, and
|
|
209
|
+
* every later tick reports `skipped_busy` forever — the loop looks scheduled
|
|
210
|
+
* while nothing has run since. Releasing the slot is honest, and it is safe
|
|
211
|
+
* because `finish` still refuses to double-report a run it no longer owns.
|
|
212
|
+
*
|
|
213
|
+
* Releasing the loop's own slot is only half of it. The cancellation that
|
|
214
|
+
* never settled is still holding the arbiter's admission boundary, so the
|
|
215
|
+
* generation it belongs to has to be retired in the same step — otherwise the
|
|
216
|
+
* loop believes it is free while every later producer, scheduled or owner,
|
|
217
|
+
* queues behind a promise that will never resolve.
|
|
218
|
+
*/
|
|
219
|
+
armAbandon(definition, state, runId) {
|
|
220
|
+
const abandon = this.deps.setTimer(() => {
|
|
221
|
+
this.runTimeouts.delete(abandon);
|
|
222
|
+
if (state.activeRunId !== runId || this.stopping)
|
|
223
|
+
return;
|
|
224
|
+
state.activeRunId = null;
|
|
225
|
+
state.lastFinishedAt = new Date(this.deps.now()).toISOString();
|
|
226
|
+
state.counts.failed = increment(state.counts.failed);
|
|
227
|
+
state.lastOutcome = 'abandoned_unsettled';
|
|
228
|
+
state.lastError = { kind: 'abandoned_unsettled', at: state.lastFinishedAt };
|
|
229
|
+
this.store.state.health = 'degraded';
|
|
230
|
+
this.store.state.anomaly = 'abandoned_unsettled';
|
|
231
|
+
this.store.persist();
|
|
232
|
+
this.arbiter.retireStalledAdmission();
|
|
233
|
+
this.deps.log(`[${this.role}] loop ${definition.name} abandoned run=${runId.slice(0, 11)}: `
|
|
234
|
+
+ 'cancellation never settled; admission released');
|
|
235
|
+
}, this.deps.cancelAbandonMs ?? LOOP_CANCEL_ABANDON_MS);
|
|
236
|
+
this.runTimeouts.add(abandon);
|
|
237
|
+
}
|
|
168
238
|
finish(definition, state, runId, result) {
|
|
169
239
|
if (state.activeRunId !== runId)
|
|
170
240
|
return;
|
|
@@ -216,6 +286,19 @@ export class ScheduledLoopManager {
|
|
|
216
286
|
this.skipMissed(definition, state, now);
|
|
217
287
|
}
|
|
218
288
|
}
|
|
289
|
+
/**
|
|
290
|
+
* A run the store could not record is dropped, not retried: the cursor has
|
|
291
|
+
* already moved, so this can never become a busy loop, and the outage is
|
|
292
|
+
* visible as a skip with a failed health rather than as silence.
|
|
293
|
+
*/
|
|
294
|
+
skipUnpersisted(definition, state, now) {
|
|
295
|
+
state.counts.skipped = increment(state.counts.skipped);
|
|
296
|
+
state.lastOutcome = 'skipped_unpersisted';
|
|
297
|
+
state.lastFinishedAt = new Date(now).toISOString();
|
|
298
|
+
state.lastError = { kind: 'persist_failed', at: state.lastFinishedAt };
|
|
299
|
+
this.deps.log(`[${this.role}] loop ${definition.name} skipped_unpersisted `
|
|
300
|
+
+ `(${this.store.lastPersistError ?? 'state write failed'})`);
|
|
301
|
+
}
|
|
219
302
|
schedule() {
|
|
220
303
|
if (this.timer !== undefined)
|
|
221
304
|
this.deps.clearTimer(this.timer);
|
|
@@ -226,20 +309,36 @@ export class ScheduledLoopManager {
|
|
|
226
309
|
const state = this.store.state.loops[definition.name];
|
|
227
310
|
return definition.enabled && !state.operatorDisabled;
|
|
228
311
|
}).map(definition => Date.parse(this.store.state.loops[definition.name].nextDueAt));
|
|
229
|
-
|
|
312
|
+
// While writes are failing the manager keeps a bounded probe armed even with
|
|
313
|
+
// nothing due, so health stops lying the moment the disk comes back.
|
|
314
|
+
const probe = this.store.persistFailures ? backoffMs(this.store.persistFailures) : undefined;
|
|
315
|
+
if (!due.length && probe === undefined)
|
|
230
316
|
return;
|
|
231
|
-
const delay = Math.max(0, Math.min(...due) - this.deps.now());
|
|
232
|
-
this.
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
317
|
+
const delay = due.length ? Math.max(0, Math.min(...due) - this.deps.now()) : POLL_CEILING_MS;
|
|
318
|
+
this.arm(Math.min(delay, POLL_CEILING_MS, probe ?? POLL_CEILING_MS));
|
|
319
|
+
}
|
|
320
|
+
arm(delayMs) {
|
|
321
|
+
this.timer = this.deps.setTimer(() => { void this.poll().catch(error => this.recover(error)); }, delayMs);
|
|
322
|
+
}
|
|
323
|
+
/**
|
|
324
|
+
* The residual safety net. Persistence failures are absorbed by the store, so
|
|
325
|
+
* reaching here means something unforeseen threw — and whatever it was, the
|
|
326
|
+
* manager must stay armed. The previous behaviour left no timer at all, which
|
|
327
|
+
* turned one transient ENOSPC into every loop on the role being silently gone
|
|
328
|
+
* until the process was restarted.
|
|
329
|
+
*/
|
|
330
|
+
recover(error) {
|
|
331
|
+
this.pollFailures = increment(this.pollFailures);
|
|
332
|
+
this.store.state.health = 'failed';
|
|
333
|
+
this.store.state.anomaly = 'manager_task_failed';
|
|
334
|
+
this.store.persist();
|
|
335
|
+
this.deps.log(`[${this.role}] scheduled loop manager failed: ${error?.name ?? 'Error'}`
|
|
336
|
+
+ ` (attempt ${this.pollFailures}, retrying in ${backoffMs(this.pollFailures)}ms)`);
|
|
337
|
+
if (this.stopping)
|
|
338
|
+
return;
|
|
339
|
+
if (this.timer !== undefined)
|
|
340
|
+
this.deps.clearTimer(this.timer);
|
|
341
|
+
this.arm(backoffMs(this.pollFailures));
|
|
243
342
|
}
|
|
244
343
|
envelope(definition, runId, scheduledAt) {
|
|
245
344
|
return [
|
package/dist/loops/state.d.ts
CHANGED
|
@@ -42,13 +42,57 @@ export interface ScheduledLoopsFile {
|
|
|
42
42
|
export declare const increment: (value: number, amount?: number) => number;
|
|
43
43
|
export declare function deterministicJitter(role: string, loop: string, nominalMs: number, maximumMs: number): number;
|
|
44
44
|
export declare function scheduledLoopsPath(stateDir: string): string;
|
|
45
|
+
/**
|
|
46
|
+
* A live manager rewrites the checkpoint at least once per poll ceiling (60s).
|
|
47
|
+
* Past this bound the recorded `health` describes a manager that is no longer
|
|
48
|
+
* updating it — a dead scheduler, or one whose writes are failing, which is
|
|
49
|
+
* exactly the case where the field still reads `healthy` because the flip to
|
|
50
|
+
* `failed` could not be written. Readers must not repeat a stale field as fact.
|
|
51
|
+
*/
|
|
52
|
+
export declare const STORED_STATE_STALE_MS: number;
|
|
53
|
+
export interface StoredLoopVerdict {
|
|
54
|
+
health: ScheduledLoopsFile['health'] | 'stale';
|
|
55
|
+
recorded: ScheduledLoopsFile['health'];
|
|
56
|
+
stale: boolean;
|
|
57
|
+
ageMs: number;
|
|
58
|
+
/** Whether any loop still owes a run — nothing is expected of the file if not. */
|
|
59
|
+
scheduled: boolean;
|
|
60
|
+
}
|
|
61
|
+
/** Truthful health for a reader that only has the stored file to go on. */
|
|
62
|
+
export declare function storedLoopHealth(file: ScheduledLoopsFile, now: number): StoredLoopVerdict;
|
|
45
63
|
export declare function readScheduledLoops(stateDir: string): ScheduledLoopsFile | undefined;
|
|
64
|
+
/** Persistence seam. Production uses the atomic replace; faults are injected here. */
|
|
65
|
+
export type StateWriter = (path: string, contents: string, mode: number) => void;
|
|
46
66
|
export declare class ScheduledLoopStateStore {
|
|
47
67
|
private readonly log;
|
|
68
|
+
private readonly write;
|
|
48
69
|
readonly path: string;
|
|
49
70
|
readonly fresh: boolean;
|
|
50
71
|
state: ScheduledLoopsFile;
|
|
51
|
-
|
|
72
|
+
/** Consecutive failed checkpoints; 0 whenever the stored file is current. */
|
|
73
|
+
persistFailures: number;
|
|
74
|
+
lastPersistError: string | null;
|
|
75
|
+
/** Health displaced by the persist_failed marker, restored when a write lands. */
|
|
76
|
+
private suppressed;
|
|
77
|
+
constructor(stateDir: string, role: string, definitions: ResolvedRoleLoop[], now: number, log: (line: string) => void, write?: StateWriter);
|
|
52
78
|
reconcile(definitions: ResolvedRoleLoop[], now: number, recoverActive?: boolean): void;
|
|
53
|
-
|
|
79
|
+
/**
|
|
80
|
+
* Drop a transient anomaly once its cause is over — including one currently
|
|
81
|
+
* displaced by `persist_failed`, which would otherwise come back the moment a
|
|
82
|
+
* write finally lands.
|
|
83
|
+
*/
|
|
84
|
+
clearAnomaly(kind: string): void;
|
|
85
|
+
/**
|
|
86
|
+
* Checkpoint the state. A write failure — ENOSPC is the one seen in the field —
|
|
87
|
+
* must never propagate: every caller sits under a timer callback, and an
|
|
88
|
+
* exception there kills the scheduling chain for the life of the process while
|
|
89
|
+
* the last-written file goes on claiming the loops are healthy. So the failure
|
|
90
|
+
* is recorded in memory instead, where `status()` and the live control socket
|
|
91
|
+
* report it immediately, and the caller decides what to do with `false`.
|
|
92
|
+
*
|
|
93
|
+
* The `failed` marker is applied optimistically-in-reverse: it is rolled back
|
|
94
|
+
* just before each attempt, so that the write which finally lands records the
|
|
95
|
+
* real health rather than the outage that is now over.
|
|
96
|
+
*/
|
|
97
|
+
persist(): boolean;
|
|
54
98
|
}
|
package/dist/loops/state.js
CHANGED
|
@@ -17,6 +17,25 @@ export function deterministicJitter(role, loop, nominalMs, maximumMs) {
|
|
|
17
17
|
export function scheduledLoopsPath(stateDir) {
|
|
18
18
|
return join(stateDir, '.scheduled-loops.json');
|
|
19
19
|
}
|
|
20
|
+
/**
|
|
21
|
+
* A live manager rewrites the checkpoint at least once per poll ceiling (60s).
|
|
22
|
+
* Past this bound the recorded `health` describes a manager that is no longer
|
|
23
|
+
* updating it — a dead scheduler, or one whose writes are failing, which is
|
|
24
|
+
* exactly the case where the field still reads `healthy` because the flip to
|
|
25
|
+
* `failed` could not be written. Readers must not repeat a stale field as fact.
|
|
26
|
+
*/
|
|
27
|
+
export const STORED_STATE_STALE_MS = 5 * 60_000;
|
|
28
|
+
/** Truthful health for a reader that only has the stored file to go on. */
|
|
29
|
+
export function storedLoopHealth(file, now) {
|
|
30
|
+
const ageMs = Math.max(0, now - file.clock.lastWallMs);
|
|
31
|
+
// A role with nothing left to schedule stops polling on purpose, so its
|
|
32
|
+
// checkpoint stops advancing on purpose too. Only a role that still owes
|
|
33
|
+
// someone a run can be stale; calling a deliberately idle one stale would
|
|
34
|
+
// teach operators to ignore the word.
|
|
35
|
+
const scheduled = Object.values(file.loops).some(loop => loop.enabled && !loop.operatorDisabled);
|
|
36
|
+
const stale = scheduled && ageMs > STORED_STATE_STALE_MS;
|
|
37
|
+
return { health: stale ? 'stale' : file.health, recorded: file.health, stale, ageMs, scheduled };
|
|
38
|
+
}
|
|
20
39
|
export function readScheduledLoops(stateDir) {
|
|
21
40
|
try {
|
|
22
41
|
const path = scheduledLoopsPath(stateDir);
|
|
@@ -31,11 +50,18 @@ export function readScheduledLoops(stateDir) {
|
|
|
31
50
|
}
|
|
32
51
|
export class ScheduledLoopStateStore {
|
|
33
52
|
log;
|
|
53
|
+
write;
|
|
34
54
|
path;
|
|
35
55
|
fresh;
|
|
36
56
|
state;
|
|
37
|
-
|
|
57
|
+
/** Consecutive failed checkpoints; 0 whenever the stored file is current. */
|
|
58
|
+
persistFailures = 0;
|
|
59
|
+
lastPersistError = null;
|
|
60
|
+
/** Health displaced by the persist_failed marker, restored when a write lands. */
|
|
61
|
+
suppressed = null;
|
|
62
|
+
constructor(stateDir, role, definitions, now, log, write = replaceFileAtomically) {
|
|
38
63
|
this.log = log;
|
|
64
|
+
this.write = write;
|
|
39
65
|
this.path = scheduledLoopsPath(stateDir);
|
|
40
66
|
let restored;
|
|
41
67
|
let corrupt = false;
|
|
@@ -113,9 +139,61 @@ export class ScheduledLoopStateStore {
|
|
|
113
139
|
this.state.clock.lastWallMs = now;
|
|
114
140
|
this.persist();
|
|
115
141
|
}
|
|
142
|
+
/**
|
|
143
|
+
* Drop a transient anomaly once its cause is over — including one currently
|
|
144
|
+
* displaced by `persist_failed`, which would otherwise come back the moment a
|
|
145
|
+
* write finally lands.
|
|
146
|
+
*/
|
|
147
|
+
clearAnomaly(kind) {
|
|
148
|
+
if (this.suppressed?.anomaly === kind)
|
|
149
|
+
this.suppressed = { health: 'healthy', anomaly: null };
|
|
150
|
+
if (this.state.anomaly === kind) {
|
|
151
|
+
this.state.health = 'healthy';
|
|
152
|
+
this.state.anomaly = null;
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
/**
|
|
156
|
+
* Checkpoint the state. A write failure — ENOSPC is the one seen in the field —
|
|
157
|
+
* must never propagate: every caller sits under a timer callback, and an
|
|
158
|
+
* exception there kills the scheduling chain for the life of the process while
|
|
159
|
+
* the last-written file goes on claiming the loops are healthy. So the failure
|
|
160
|
+
* is recorded in memory instead, where `status()` and the live control socket
|
|
161
|
+
* report it immediately, and the caller decides what to do with `false`.
|
|
162
|
+
*
|
|
163
|
+
* The `failed` marker is applied optimistically-in-reverse: it is rolled back
|
|
164
|
+
* just before each attempt, so that the write which finally lands records the
|
|
165
|
+
* real health rather than the outage that is now over.
|
|
166
|
+
*/
|
|
116
167
|
persist() {
|
|
117
|
-
|
|
118
|
-
|
|
168
|
+
if (this.suppressed && this.state.anomaly === 'persist_failed') {
|
|
169
|
+
this.state.health = this.suppressed.health;
|
|
170
|
+
this.state.anomaly = this.suppressed.anomaly;
|
|
171
|
+
}
|
|
172
|
+
try {
|
|
173
|
+
this.write(this.path, JSON.stringify(this.state, null, 2) + '\n', 0o600);
|
|
174
|
+
chmodSync(this.path, 0o600);
|
|
175
|
+
}
|
|
176
|
+
catch (error) {
|
|
177
|
+
const code = error?.code ?? error?.name ?? 'Error';
|
|
178
|
+
if (this.state.anomaly !== 'persist_failed')
|
|
179
|
+
this.suppressed = { health: this.state.health, anomaly: this.state.anomaly };
|
|
180
|
+
this.state.health = 'failed';
|
|
181
|
+
this.state.anomaly = 'persist_failed';
|
|
182
|
+
this.lastPersistError = code;
|
|
183
|
+
this.persistFailures = increment(this.persistFailures);
|
|
184
|
+
if (this.persistFailures === 1)
|
|
185
|
+
this.log(`[${this.state.role}] scheduled loop state write failing (${code}); `
|
|
186
|
+
+ 'cadence held and health reported failed until it lands');
|
|
187
|
+
return false;
|
|
188
|
+
}
|
|
189
|
+
if (this.persistFailures) {
|
|
190
|
+
this.log(`[${this.state.role}] scheduled loop persistence recovered after `
|
|
191
|
+
+ `${this.persistFailures} failed write(s) (${this.lastPersistError})`);
|
|
192
|
+
this.persistFailures = 0;
|
|
193
|
+
this.lastPersistError = null;
|
|
194
|
+
}
|
|
195
|
+
this.suppressed = null;
|
|
196
|
+
return true;
|
|
119
197
|
}
|
|
120
198
|
}
|
|
121
199
|
function validFile(value) {
|
package/dist/monitor.d.ts
CHANGED
|
@@ -15,6 +15,11 @@ export interface FetchResponse {
|
|
|
15
15
|
json(): Promise<{
|
|
16
16
|
cursor?: number;
|
|
17
17
|
events?: NotifyEvent[];
|
|
18
|
+
identities?: Array<string | {
|
|
19
|
+
name?: unknown;
|
|
20
|
+
temporary?: unknown;
|
|
21
|
+
stale?: unknown;
|
|
22
|
+
}>;
|
|
18
23
|
}>;
|
|
19
24
|
}
|
|
20
25
|
export type FetchLike = (url: string, init?: {
|
|
@@ -92,8 +97,24 @@ export interface DaemonEndpoint {
|
|
|
92
97
|
url(name: string): string;
|
|
93
98
|
headers: Record<string, string>;
|
|
94
99
|
}
|
|
100
|
+
export type IdentityPresence = {
|
|
101
|
+
state: 'present';
|
|
102
|
+
temporary: boolean;
|
|
103
|
+
stale: boolean;
|
|
104
|
+
} | {
|
|
105
|
+
state: 'absent';
|
|
106
|
+
} | {
|
|
107
|
+
state: 'unknown';
|
|
108
|
+
detail: string;
|
|
109
|
+
};
|
|
95
110
|
/** Resolve the daemon endpoint + auth header from env → config → defaults. */
|
|
96
111
|
export declare function resolveEndpoint(env: NodeJS.ProcessEnv): DaemonEndpoint;
|
|
112
|
+
/**
|
|
113
|
+
* Ask the daemon's authoritative identity index. The notifications endpoint is
|
|
114
|
+
* intentionally unsuitable for lifecycle: it serves an empty 200 page for a
|
|
115
|
+
* valid but missing identity, which made a closed temp identity look healthy.
|
|
116
|
+
*/
|
|
117
|
+
export declare function probeIdentityPresence(name: string, fetch: FetchLike, env: NodeJS.ProcessEnv): Promise<IdentityPresence>;
|
|
97
118
|
/** Actionable, secret-free description of every token source for this profile. */
|
|
98
119
|
export declare function authResolutionHint(ep: DaemonEndpoint): string;
|
|
99
120
|
/** Keep only the events whose type the role asked to wake on. */
|