@enderfga/claw-orchestrator 5.0.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -28
- package/dist/bin/cli.js +107 -1
- package/dist/bin/cli.js.map +1 -1
- package/dist/src/acp-server.d.ts +5 -5
- package/dist/src/acp-server.js +3 -3
- package/dist/src/acp-server.js.map +1 -1
- package/dist/src/autoloop/dispatcher.d.ts +22 -0
- package/dist/src/autoloop/dispatcher.js +71 -13
- package/dist/src/autoloop/dispatcher.js.map +1 -1
- package/dist/src/autoloop/messages.d.ts +10 -0
- package/dist/src/autoloop/messages.js.map +1 -1
- package/dist/src/autoloop/runner.js +6 -0
- package/dist/src/autoloop/runner.js.map +1 -1
- package/dist/src/constants.d.ts +0 -6
- package/dist/src/constants.js +0 -6
- package/dist/src/constants.js.map +1 -1
- package/dist/src/council.d.ts +15 -0
- package/dist/src/council.js +48 -35
- package/dist/src/council.js.map +1 -1
- package/dist/src/dashboard/index.html +191 -6
- package/dist/src/embedded-server.js +132 -9
- package/dist/src/embedded-server.js.map +1 -1
- package/dist/src/fanout.d.ts +30 -1
- package/dist/src/fanout.js +32 -3
- package/dist/src/fanout.js.map +1 -1
- package/dist/src/index.d.ts +1 -0
- package/dist/src/index.js +360 -4
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/agent-step.d.ts +59 -0
- package/dist/src/kernel/agent-step.js +100 -0
- package/dist/src/kernel/agent-step.js.map +1 -0
- package/dist/src/kernel/conditions.d.ts +11 -0
- package/dist/src/kernel/conditions.js +24 -0
- package/dist/src/kernel/conditions.js.map +1 -0
- package/dist/src/kernel/engine.d.ts +319 -0
- package/dist/src/kernel/engine.js +1047 -0
- package/dist/src/kernel/engine.js.map +1 -0
- package/dist/src/kernel/exec.d.ts +43 -0
- package/dist/src/kernel/exec.js +112 -0
- package/dist/src/kernel/exec.js.map +1 -0
- package/dist/src/kernel/file-lock.d.ts +50 -0
- package/dist/src/kernel/file-lock.js +135 -0
- package/dist/src/kernel/file-lock.js.map +1 -0
- package/dist/src/kernel/nodes/agent.d.ts +4 -0
- package/dist/src/kernel/nodes/agent.js +35 -0
- package/dist/src/kernel/nodes/agent.js.map +1 -0
- package/dist/src/kernel/nodes/autoloop.d.ts +78 -0
- package/dist/src/kernel/nodes/autoloop.js +75 -0
- package/dist/src/kernel/nodes/autoloop.js.map +1 -0
- package/dist/src/kernel/nodes/council.d.ts +12 -0
- package/dist/src/kernel/nodes/council.js +88 -0
- package/dist/src/kernel/nodes/council.js.map +1 -0
- package/dist/src/kernel/nodes/fanout.d.ts +11 -0
- package/dist/src/kernel/nodes/fanout.js +63 -0
- package/dist/src/kernel/nodes/fanout.js.map +1 -0
- package/dist/src/kernel/nodes/human-gate.d.ts +4 -0
- package/dist/src/kernel/nodes/human-gate.js +7 -0
- package/dist/src/kernel/nodes/human-gate.js.map +1 -0
- package/dist/src/kernel/nodes/index.d.ts +12 -0
- package/dist/src/kernel/nodes/index.js +21 -0
- package/dist/src/kernel/nodes/index.js.map +1 -0
- package/dist/src/kernel/nodes/router.d.ts +4 -0
- package/dist/src/kernel/nodes/router.js +12 -0
- package/dist/src/kernel/nodes/router.js.map +1 -0
- package/dist/src/kernel/nodes/subflow.d.ts +13 -0
- package/dist/src/kernel/nodes/subflow.js +38 -0
- package/dist/src/kernel/nodes/subflow.js.map +1 -0
- package/dist/src/kernel/nodes/ultraapp.d.ts +60 -0
- package/dist/src/kernel/nodes/ultraapp.js +62 -0
- package/dist/src/kernel/nodes/ultraapp.js.map +1 -0
- package/dist/src/kernel/nodes/verifier.d.ts +14 -0
- package/dist/src/kernel/nodes/verifier.js +84 -0
- package/dist/src/kernel/nodes/verifier.js.map +1 -0
- package/dist/src/kernel/projections.d.ts +42 -0
- package/dist/src/kernel/projections.js +133 -0
- package/dist/src/kernel/projections.js.map +1 -0
- package/dist/src/kernel/repo.d.ts +13 -0
- package/dist/src/kernel/repo.js +64 -0
- package/dist/src/kernel/repo.js.map +1 -0
- package/dist/src/kernel/secrets.d.ts +25 -0
- package/dist/src/kernel/secrets.js +48 -0
- package/dist/src/kernel/secrets.js.map +1 -0
- package/dist/src/kernel/store.d.ts +225 -0
- package/dist/src/kernel/store.js +838 -0
- package/dist/src/kernel/store.js.map +1 -0
- package/dist/src/kernel/templates/index.d.ts +140 -0
- package/dist/src/kernel/templates/index.js +266 -0
- package/dist/src/kernel/templates/index.js.map +1 -0
- package/dist/src/kernel/types.d.ts +326 -0
- package/dist/src/kernel/types.js +19 -0
- package/dist/src/kernel/types.js.map +1 -0
- package/dist/src/models.d.ts +1 -1
- package/dist/src/models.js +31 -3
- package/dist/src/models.js.map +1 -1
- package/dist/src/persistent-cursor-session.js +6 -1
- package/dist/src/persistent-cursor-session.js.map +1 -1
- package/dist/src/persistent-grok-session.d.ts +40 -0
- package/dist/src/persistent-grok-session.js +197 -0
- package/dist/src/persistent-grok-session.js.map +1 -0
- package/dist/src/run-ledger.d.ts +57 -3
- package/dist/src/run-ledger.js +45 -2
- package/dist/src/run-ledger.js.map +1 -1
- package/dist/src/session-manager.d.ts +176 -129
- package/dist/src/session-manager.js +657 -603
- package/dist/src/session-manager.js.map +1 -1
- package/dist/src/types.d.ts +37 -4
- package/dist/src/types.js +15 -1
- package/dist/src/types.js.map +1 -1
- package/dist/src/ultraapp/build.d.ts +117 -3
- package/dist/src/ultraapp/build.js +319 -3
- package/dist/src/ultraapp/build.js.map +1 -1
- package/dist/src/ultraapp/contract.d.ts +52 -0
- package/dist/src/ultraapp/contract.js +83 -0
- package/dist/src/ultraapp/contract.js.map +1 -0
- package/dist/src/ultraapp/conventions.js +9 -2
- package/dist/src/ultraapp/conventions.js.map +1 -1
- package/dist/src/ultraapp/fix-on-failure.d.ts +21 -2
- package/dist/src/ultraapp/fix-on-failure.js +46 -62
- package/dist/src/ultraapp/fix-on-failure.js.map +1 -1
- package/dist/src/ultraapp/manager.d.ts +107 -2
- package/dist/src/ultraapp/manager.js +305 -86
- package/dist/src/ultraapp/manager.js.map +1 -1
- package/dist/src/verify/baseline.d.ts +73 -0
- package/dist/src/verify/baseline.js +186 -0
- package/dist/src/verify/baseline.js.map +1 -0
- package/dist/src/verify/contract.d.ts +116 -0
- package/dist/src/verify/contract.js +142 -0
- package/dist/src/verify/contract.js.map +1 -0
- package/dist/src/verify/evidence.d.ts +61 -0
- package/dist/src/verify/evidence.js +133 -0
- package/dist/src/verify/evidence.js.map +1 -0
- package/dist/src/verify/runner.d.ts +63 -0
- package/dist/src/verify/runner.js +317 -0
- package/dist/src/verify/runner.js.map +1 -0
- package/openclaw.plugin.json +8 -0
- package/package.json +2 -2
- package/skills/SKILL.md +121 -80
- package/skills/references/acp.md +18 -18
- package/skills/references/autoloop.md +148 -72
- package/skills/references/claude-cli-tracking.md +4 -4
- package/skills/references/cli.md +103 -60
- package/skills/references/council.md +109 -37
- package/skills/references/dashboard.md +34 -6
- package/skills/references/getting-started.md +14 -14
- package/skills/references/inbox.md +4 -4
- package/skills/references/mcp.md +39 -34
- package/skills/references/multi-engine.md +109 -51
- package/skills/references/observability.md +88 -27
- package/skills/references/openai-compat.md +40 -40
- package/skills/references/sessions.md +44 -26
- package/skills/references/tools.md +402 -309
- package/skills/references/ultra.md +45 -45
- package/skills/references/ultraapp.md +126 -50
- package/skills/references/verification.md +187 -0
- package/skills/references/workflow.md +362 -0
- package/dist/src/ultraapp/fix-on-failure-session.d.ts +0 -23
- package/dist/src/ultraapp/fix-on-failure-session.js +0 -51
- package/dist/src/ultraapp/fix-on-failure-session.js.map +0 -1
|
@@ -0,0 +1,838 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Durable run store.
|
|
3
|
+
*
|
|
4
|
+
* Layout, one directory per run under `~/.claw-orchestrator/wf/<runId>/`:
|
|
5
|
+
*
|
|
6
|
+
* spec.json the WorkflowSpec, written once at creation, never mutated
|
|
7
|
+
* run.json the mutable checkpoint, rewritten atomically on every transition
|
|
8
|
+
* events.jsonl append-only audit + stream source
|
|
9
|
+
* nodes/<id>/ per-node artifacts
|
|
10
|
+
* evidence/<id>/ evidence bundles (see verify/evidence.ts)
|
|
11
|
+
*
|
|
12
|
+
* Splitting the immutable spec from the mutable checkpoint is what carries crash
|
|
13
|
+
* recovery: if `run.json` is missing or half-written, the state is rebuilt by
|
|
14
|
+
* replaying `events.jsonl` against `spec.json`. The atomic rewrite makes that
|
|
15
|
+
* path rare; the replay makes it usually survivable.
|
|
16
|
+
*
|
|
17
|
+
* The checkpoint and the events it describes are written by the same
|
|
18
|
+
* transaction, so they cannot disagree: a commit that reports `committed` has
|
|
19
|
+
* both, and one that does not has neither. That was not always true — the event
|
|
20
|
+
* append used to swallow its own errors while the replay treated the same log as
|
|
21
|
+
* authoritative, which is a contradiction sitting in the recovery path. The
|
|
22
|
+
* replay is now a plain fallback for a lost `run.json`; a run that lost both it
|
|
23
|
+
* and the log is unrecoverable, and `loadRun` returns undefined rather than
|
|
24
|
+
* inventing a state.
|
|
25
|
+
*
|
|
26
|
+
* There is exactly one way to write to a run — `commit()` — and it is not
|
|
27
|
+
* optional: the checkpoint writer and the event appender are module-private, and
|
|
28
|
+
* a batch is published by a single atomic directory rename rather than by a
|
|
29
|
+
* sequence of writes that can each fail on their own. `atomicWriteJson` replaces
|
|
30
|
+
* four near-identical implementations that had grown across the codebase
|
|
31
|
+
* (`session-manager.ts` sync + async variants, `ultraapp/store.ts`,
|
|
32
|
+
* `ultraapp/patcher.ts`); the tmp-name shape is kept from the ultraapp one,
|
|
33
|
+
* whose comment records the CI bug it fixed: a reader catching a plain
|
|
34
|
+
* `writeFile` mid-flight and parsing a truncated object.
|
|
35
|
+
*/
|
|
36
|
+
import crypto from 'node:crypto';
|
|
37
|
+
import fs from 'node:fs';
|
|
38
|
+
import os from 'node:os';
|
|
39
|
+
import path from 'node:path';
|
|
40
|
+
import { withFileLock } from './file-lock.js';
|
|
41
|
+
export function wfDir() {
|
|
42
|
+
return process.env.CLAWO_WF_DIR || path.join(os.homedir(), '.claw-orchestrator', 'wf');
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* A run id is a single path segment: letters, digits, dot, dash, underscore.
|
|
46
|
+
*
|
|
47
|
+
* It has to be enforced, not merely expected. A run id can be supplied by the
|
|
48
|
+
* caller — including through a tool call, which means through an agent — and
|
|
49
|
+
* every path in this module is derived from it by `path.join`. `../escaped`
|
|
50
|
+
* resolves outside the store, and `deleteRunDir` is a recursive `rmSync`, so an
|
|
51
|
+
* unvalidated id turns a delete into arbitrary directory removal. A leading dot
|
|
52
|
+
* is refused too, so no id can produce `.` or `..` by itself.
|
|
53
|
+
*/
|
|
54
|
+
const VALID_RUN_ID = /^[A-Za-z0-9_][A-Za-z0-9._-]{0,127}$/;
|
|
55
|
+
export function isValidRunId(runId) {
|
|
56
|
+
return typeof runId === 'string' && VALID_RUN_ID.test(runId);
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Resolve a run's directory. Throws on an invalid id rather than returning a
|
|
60
|
+
* path, because this is the chokepoint every other path derives from — checking
|
|
61
|
+
* here means no caller can forget to.
|
|
62
|
+
*/
|
|
63
|
+
export function runDir(runId) {
|
|
64
|
+
if (!isValidRunId(runId)) {
|
|
65
|
+
throw new Error(`Invalid run id ${JSON.stringify(runId)}: must match ${VALID_RUN_ID.source} (a single path segment)`);
|
|
66
|
+
}
|
|
67
|
+
const dir = path.join(wfDir(), runId);
|
|
68
|
+
// Belt to the regex's braces: if any future change to the pattern lets a
|
|
69
|
+
// separator through, this still refuses to hand back a path outside the root.
|
|
70
|
+
const root = wfDir();
|
|
71
|
+
if (path.dirname(dir) !== root) {
|
|
72
|
+
throw new Error(`Invalid run id ${JSON.stringify(runId)}: resolves outside the run store`);
|
|
73
|
+
}
|
|
74
|
+
return dir;
|
|
75
|
+
}
|
|
76
|
+
export function nodeDir(runId, nodeId) {
|
|
77
|
+
return path.join(runDir(runId), 'nodes', nodeId.replace(/[^\w.-]/g, '_'));
|
|
78
|
+
}
|
|
79
|
+
// ─── Shared write primitives ────────────────────────────────────────────────
|
|
80
|
+
/** Write JSON so a concurrent reader sees either the old file or the new one, never a partial. */
|
|
81
|
+
export function atomicWriteJson(file, value) {
|
|
82
|
+
const tmp = `${file}.tmp.${process.pid}.${crypto.randomBytes(4).toString('hex')}`;
|
|
83
|
+
fs.mkdirSync(path.dirname(file), { recursive: true });
|
|
84
|
+
try {
|
|
85
|
+
fs.writeFileSync(tmp, JSON.stringify(value, null, 2));
|
|
86
|
+
fs.renameSync(tmp, file);
|
|
87
|
+
}
|
|
88
|
+
catch (err) {
|
|
89
|
+
try {
|
|
90
|
+
fs.unlinkSync(tmp);
|
|
91
|
+
}
|
|
92
|
+
catch {
|
|
93
|
+
// Nothing to clean up.
|
|
94
|
+
}
|
|
95
|
+
throw err;
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
function readJson(file) {
|
|
99
|
+
try {
|
|
100
|
+
return JSON.parse(fs.readFileSync(file, 'utf8'));
|
|
101
|
+
}
|
|
102
|
+
catch {
|
|
103
|
+
return undefined;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
// ─── Run lifecycle ──────────────────────────────────────────────────────────
|
|
107
|
+
export function runExists(runId) {
|
|
108
|
+
try {
|
|
109
|
+
return fs.existsSync(path.join(runDir(runId), 'spec.json'));
|
|
110
|
+
}
|
|
111
|
+
catch {
|
|
112
|
+
return false;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Keys whose values never reach disk.
|
|
117
|
+
*
|
|
118
|
+
* `customEngine` carries a `CustomEngineConfig`, whose `env` is explicitly for
|
|
119
|
+
* environment variables — API tokens included. An autoloop started with a custom
|
|
120
|
+
* engine wrote its whole options object into the node spec, and `spec.json`
|
|
121
|
+
* ended up holding the token in plain text.
|
|
122
|
+
*
|
|
123
|
+
* The real fix is to route them through `StartOptions.secrets`, which stays in
|
|
124
|
+
* memory. This is the second line: even a spec that should not contain one is
|
|
125
|
+
* scrubbed on the way out, so a future field cannot leak by omission.
|
|
126
|
+
*/
|
|
127
|
+
const NEVER_PERSIST = new Set(['customengine', 'plannercustomengine', 'codercustomengine', 'reviewercustomengine']);
|
|
128
|
+
export function sanitizeForDisk(value) {
|
|
129
|
+
if (Array.isArray(value))
|
|
130
|
+
return value.map((v) => sanitizeForDisk(v));
|
|
131
|
+
if (!value || typeof value !== 'object')
|
|
132
|
+
return value;
|
|
133
|
+
const out = {};
|
|
134
|
+
for (const [key, v] of Object.entries(value)) {
|
|
135
|
+
if (NEVER_PERSIST.has(key.toLowerCase()))
|
|
136
|
+
continue;
|
|
137
|
+
out[key] = sanitizeForDisk(v);
|
|
138
|
+
}
|
|
139
|
+
return out;
|
|
140
|
+
}
|
|
141
|
+
export function loadSpec(runId) {
|
|
142
|
+
// Lookups treat an invalid id as "no such run" rather than throwing: a query
|
|
143
|
+
// for a nonsense id has an answer, and it is "nothing".
|
|
144
|
+
if (!isValidRunId(runId))
|
|
145
|
+
return undefined;
|
|
146
|
+
return readJson(path.join(runDir(runId), 'spec.json'));
|
|
147
|
+
}
|
|
148
|
+
export function readEvents(runId, limit) {
|
|
149
|
+
if (!isValidRunId(runId))
|
|
150
|
+
return [];
|
|
151
|
+
// A published transaction is authoritative even before its files have been
|
|
152
|
+
// applied. Returning the old log while `.tx` is still present would expose a
|
|
153
|
+
// state from before a commit that already succeeded.
|
|
154
|
+
if (!recoverPending(runId)) {
|
|
155
|
+
throw new Error(`Run '${runId}' has a committed transaction that could not be applied`);
|
|
156
|
+
}
|
|
157
|
+
let raw;
|
|
158
|
+
try {
|
|
159
|
+
raw = fs.readFileSync(path.join(runDir(runId), 'events.jsonl'), 'utf8');
|
|
160
|
+
}
|
|
161
|
+
catch {
|
|
162
|
+
return [];
|
|
163
|
+
}
|
|
164
|
+
const out = [];
|
|
165
|
+
for (const line of raw.split('\n')) {
|
|
166
|
+
const t = line.trim();
|
|
167
|
+
if (!t)
|
|
168
|
+
continue;
|
|
169
|
+
try {
|
|
170
|
+
out.push(JSON.parse(t));
|
|
171
|
+
}
|
|
172
|
+
catch {
|
|
173
|
+
// One corrupt line must not make the stream unreadable.
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
return limit && limit > 0 ? out.slice(-limit) : out;
|
|
177
|
+
}
|
|
178
|
+
/**
|
|
179
|
+
* Rebuild a run's state from its spec plus its event log. Used when `run.json`
|
|
180
|
+
* is absent or unparseable — the belt to the atomic rewrite's braces.
|
|
181
|
+
*/
|
|
182
|
+
export function replayRun(runId, spec) {
|
|
183
|
+
const events = readEvents(runId);
|
|
184
|
+
if (events.length === 0)
|
|
185
|
+
return undefined;
|
|
186
|
+
const nodes = {};
|
|
187
|
+
for (const n of spec.nodes) {
|
|
188
|
+
nodes[n.id] = { id: n.id, kind: n.kind, state: 'pending', attempts: 0, visits: 0 };
|
|
189
|
+
}
|
|
190
|
+
const created = events.find((e) => e.type === 'run_created');
|
|
191
|
+
const record = {
|
|
192
|
+
runId,
|
|
193
|
+
workflow: spec.name,
|
|
194
|
+
spec,
|
|
195
|
+
state: 'pending',
|
|
196
|
+
outcome: 'unverified',
|
|
197
|
+
cwd: spec.cwd || process.cwd(),
|
|
198
|
+
createdAt: created?.ts ?? events[0].ts,
|
|
199
|
+
updatedAt: events[events.length - 1].ts,
|
|
200
|
+
nodes,
|
|
201
|
+
};
|
|
202
|
+
for (const e of events) {
|
|
203
|
+
switch (e.type) {
|
|
204
|
+
case 'run_state':
|
|
205
|
+
record.state = e.state;
|
|
206
|
+
if (e.outcome)
|
|
207
|
+
record.outcome = e.outcome;
|
|
208
|
+
if (e.error)
|
|
209
|
+
record.error = e.error;
|
|
210
|
+
break;
|
|
211
|
+
case 'node_state': {
|
|
212
|
+
const n = (nodes[e.node] ??= { id: e.node, kind: 'agent', state: 'pending', attempts: 0, visits: 0 });
|
|
213
|
+
n.state = e.state;
|
|
214
|
+
if (e.attempt !== undefined)
|
|
215
|
+
n.attempts = e.attempt;
|
|
216
|
+
if (e.state === 'running') {
|
|
217
|
+
n.visits++;
|
|
218
|
+
n.startedAt = e.ts;
|
|
219
|
+
}
|
|
220
|
+
if (e.state === 'succeeded' || e.state === 'failed' || e.state === 'skipped')
|
|
221
|
+
n.endedAt = e.ts;
|
|
222
|
+
if (e.error)
|
|
223
|
+
n.error = e.error;
|
|
224
|
+
record.currentNode = e.node;
|
|
225
|
+
break;
|
|
226
|
+
}
|
|
227
|
+
case 'node_output': {
|
|
228
|
+
const n = nodes[e.node];
|
|
229
|
+
if (n)
|
|
230
|
+
n.output = e.text;
|
|
231
|
+
break;
|
|
232
|
+
}
|
|
233
|
+
case 'evidence': {
|
|
234
|
+
const n = nodes[e.node];
|
|
235
|
+
if (n)
|
|
236
|
+
n.evidenceId = e.evidenceId;
|
|
237
|
+
record.evidenceId = e.evidenceId;
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
240
|
+
default:
|
|
241
|
+
break;
|
|
242
|
+
}
|
|
243
|
+
record.updatedAt = e.ts;
|
|
244
|
+
}
|
|
245
|
+
// A run whose process died mid-flight left no terminal event. Say so rather
|
|
246
|
+
// than reporting the stale `running` it was last seen in.
|
|
247
|
+
if (record.state === 'running' || record.state === 'verifying') {
|
|
248
|
+
record.error = record.error ?? 'process ended before the run reached a terminal state';
|
|
249
|
+
}
|
|
250
|
+
return record;
|
|
251
|
+
}
|
|
252
|
+
/** Load a run, preferring the checkpoint and falling back to an event replay. */
|
|
253
|
+
export function loadRun(runId) {
|
|
254
|
+
const spec = loadSpec(runId);
|
|
255
|
+
if (!spec)
|
|
256
|
+
return undefined;
|
|
257
|
+
// A transaction the last owner committed but died before applying is finished
|
|
258
|
+
// here, so a reader never sees the state from before a committed change.
|
|
259
|
+
if (!recoverPending(runId)) {
|
|
260
|
+
throw new Error(`Run '${runId}' has a committed transaction that could not be applied`);
|
|
261
|
+
}
|
|
262
|
+
const checkpoint = readJson(path.join(runDir(runId), 'run.json'));
|
|
263
|
+
if (checkpoint && checkpoint.runId === runId && checkpoint.nodes) {
|
|
264
|
+
checkpoint.spec = checkpoint.spec ?? spec;
|
|
265
|
+
return checkpoint;
|
|
266
|
+
}
|
|
267
|
+
return replayRun(runId, spec);
|
|
268
|
+
}
|
|
269
|
+
export function listRunIds() {
|
|
270
|
+
try {
|
|
271
|
+
return fs
|
|
272
|
+
.readdirSync(wfDir(), { withFileTypes: true })
|
|
273
|
+
.filter((d) => d.isDirectory())
|
|
274
|
+
.map((d) => d.name)
|
|
275
|
+
.filter(isValidRunId);
|
|
276
|
+
}
|
|
277
|
+
catch {
|
|
278
|
+
return [];
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
/**
|
|
282
|
+
* Remove a run's directory. Only ever called from an explicit user delete —
|
|
283
|
+
* nothing in the kernel prunes runs on its own, for the same reason the ledger
|
|
284
|
+
* does not: deleting someone's history is not ours to decide.
|
|
285
|
+
*/
|
|
286
|
+
export function deleteRunDir(runId, logger) {
|
|
287
|
+
// Validate before the try, so a bad id surfaces as an error instead of being
|
|
288
|
+
// swallowed alongside genuine I/O failures. This is a recursive rmSync — the
|
|
289
|
+
// one call in this module where a bad id is not merely wrong but destructive.
|
|
290
|
+
const dir = runDir(runId);
|
|
291
|
+
try {
|
|
292
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
293
|
+
}
|
|
294
|
+
catch (err) {
|
|
295
|
+
logger?.warn?.(`[kernel-store] delete failed for ${runId}: ${err.message}`);
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
/**
|
|
299
|
+
* Full text a node produced, preferring the artifact file over the record's
|
|
300
|
+
* preview. `NodeRecord.output` is capped so a checkpoint stays small, but the
|
|
301
|
+
* whole point of some nodes is their text — an ultraplan whose plan is silently
|
|
302
|
+
* cut at 4 kB has lost the deliverable.
|
|
303
|
+
*/
|
|
304
|
+
export function readNodeOutput(runId, nodeId) {
|
|
305
|
+
try {
|
|
306
|
+
return fs.readFileSync(path.join(nodeDir(runId, nodeId), 'output.txt'), 'utf8');
|
|
307
|
+
}
|
|
308
|
+
catch {
|
|
309
|
+
return loadRun(runId)?.nodes[nodeId]?.output;
|
|
310
|
+
}
|
|
311
|
+
}
|
|
312
|
+
/**
|
|
313
|
+
* Where a node artifact will live, without writing it.
|
|
314
|
+
*
|
|
315
|
+
* Pure, because the checkpoint that references an artifact and the artifact
|
|
316
|
+
* itself are written in the same {@link commit}: the caller needs the path
|
|
317
|
+
* before the write, to put it in the record it is committing.
|
|
318
|
+
*/
|
|
319
|
+
export function nodeArtifactPath(runId, nodeId, name) {
|
|
320
|
+
const file = path.join(nodeDir(runId, nodeId), name.replace(/[^\w.-]/g, '_'));
|
|
321
|
+
return path.relative(runDir(runId), file);
|
|
322
|
+
}
|
|
323
|
+
// ─── Ownership and the storage transaction ──────────────────────────────────
|
|
324
|
+
//
|
|
325
|
+
// A run has one owner at a time, every durable change it makes is checked
|
|
326
|
+
// against a capability only that owner holds, and each change lands in full or
|
|
327
|
+
// not at all. Each of those three was added after the previous two turned out to
|
|
328
|
+
// be decoration without it.
|
|
329
|
+
//
|
|
330
|
+
// The capability is a `RunGuard`, naming four things, all checked on every
|
|
331
|
+
// write:
|
|
332
|
+
//
|
|
333
|
+
// incarnationId which *creation* of this run id this is
|
|
334
|
+
// ownerId which RunKernel instance (never the pid — see `RunLease`)
|
|
335
|
+
// acquisitionId which claim by that owner
|
|
336
|
+
// fence monotonic within the incarnation, for ordering and logs
|
|
337
|
+
//
|
|
338
|
+
// `incarnationId` exists because a run id is reusable. Delete a run and start
|
|
339
|
+
// another with the same id and the directory — fence counter included — is
|
|
340
|
+
// gone, so the new run's first fence is 1 again and a stale attempt still
|
|
341
|
+
// holding `{ownerId, fence: 1}` from the old run became valid a second time.
|
|
342
|
+
// That is a textbook ABA, and not hypothetical: an abandoned timed-out attempt
|
|
343
|
+
// outlives its run by construction.
|
|
344
|
+
//
|
|
345
|
+
// Two distinctions this module is careful about, because collapsing either one
|
|
346
|
+
// caused a real failure:
|
|
347
|
+
//
|
|
348
|
+
// *Not the owner* is permanent; *could not take the lock* is transient. They
|
|
349
|
+
// are separate outcomes (`superseded` vs `blocked`), and a caller that treats
|
|
350
|
+
// a millisecond of contention as a loss stops forever while still holding its
|
|
351
|
+
// lease — so nothing can take the run over either.
|
|
352
|
+
//
|
|
353
|
+
// *Committed* means the whole batch is durable, not that most of it was
|
|
354
|
+
// attempted. A batch is staged in a scratch directory and published by a
|
|
355
|
+
// single atomic directory rename; the rename is the commit point, and what
|
|
356
|
+
// follows is replayable application of an already-committed transaction.
|
|
357
|
+
export const LEASE_TTL_MS = 60_000;
|
|
358
|
+
/** How often a live owner refreshes its claim, independently of any work it is doing. */
|
|
359
|
+
export const LEASE_HEARTBEAT_MS = 15_000;
|
|
360
|
+
/** Staged, not yet committed. Removed on the next lock; never read. */
|
|
361
|
+
const TX_STAGING = '.tx.staging';
|
|
362
|
+
/** Committed and awaiting application. Its presence IS the commit. */
|
|
363
|
+
const TX_COMMITTED = '.tx';
|
|
364
|
+
function leaseFile(runId) {
|
|
365
|
+
return path.join(runDir(runId), 'lease.json');
|
|
366
|
+
}
|
|
367
|
+
function lockFile(runId) {
|
|
368
|
+
return path.join(runDir(runId), 'lease.lock');
|
|
369
|
+
}
|
|
370
|
+
function incarnationFile(runId) {
|
|
371
|
+
return path.join(runDir(runId), 'incarnation.json');
|
|
372
|
+
}
|
|
373
|
+
function eventsFile(runId) {
|
|
374
|
+
return path.join(runDir(runId), 'events.jsonl');
|
|
375
|
+
}
|
|
376
|
+
function fileSize(file) {
|
|
377
|
+
try {
|
|
378
|
+
return fs.statSync(file).size;
|
|
379
|
+
}
|
|
380
|
+
catch {
|
|
381
|
+
return 0;
|
|
382
|
+
}
|
|
383
|
+
}
|
|
384
|
+
function readIncarnation(runId) {
|
|
385
|
+
const raw = readJson(incarnationFile(runId));
|
|
386
|
+
return raw && typeof raw.incarnationId === 'string' && raw.incarnationId ? raw : undefined;
|
|
387
|
+
}
|
|
388
|
+
/**
|
|
389
|
+
* The run's incarnation, minting one if the directory predates this field.
|
|
390
|
+
*
|
|
391
|
+
* Only ever called inside the lock. A directory with no incarnation is a run
|
|
392
|
+
* from an older layout; giving it one now is correct, because whatever guards
|
|
393
|
+
* existed before this file did are already unusable.
|
|
394
|
+
*/
|
|
395
|
+
function ensureIncarnationLocked(runId) {
|
|
396
|
+
const existing = readIncarnation(runId);
|
|
397
|
+
if (existing)
|
|
398
|
+
return existing;
|
|
399
|
+
const fresh = { incarnationId: crypto.randomUUID(), nextFence: 0 };
|
|
400
|
+
atomicWriteJson(incarnationFile(runId), fresh);
|
|
401
|
+
return fresh;
|
|
402
|
+
}
|
|
403
|
+
function processAlive(pid) {
|
|
404
|
+
try {
|
|
405
|
+
process.kill(pid, 0);
|
|
406
|
+
return true;
|
|
407
|
+
}
|
|
408
|
+
catch (err) {
|
|
409
|
+
// EPERM means it exists and belongs to someone else.
|
|
410
|
+
return err.code === 'EPERM';
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
export function readLease(runId) {
|
|
414
|
+
if (!isValidRunId(runId))
|
|
415
|
+
return undefined;
|
|
416
|
+
return readJson(leaseFile(runId));
|
|
417
|
+
}
|
|
418
|
+
/**
|
|
419
|
+
* True when a lease no longer represents a live owner.
|
|
420
|
+
*
|
|
421
|
+
* On the same host the pid is the authority and the heartbeat is not consulted:
|
|
422
|
+
* a process running one long node makes no checkpoints, and judging it dead for
|
|
423
|
+
* that reason is how two owners end up running the same work. The heartbeat is
|
|
424
|
+
* the fallback for a holder we cannot ask about — another machine, or a pid
|
|
425
|
+
* whose meaning we cannot trust across hosts.
|
|
426
|
+
*/
|
|
427
|
+
export function leaseIsStale(lease, now = Date.now()) {
|
|
428
|
+
if (!lease)
|
|
429
|
+
return true;
|
|
430
|
+
if (lease.host === os.hostname())
|
|
431
|
+
return !processAlive(lease.pid);
|
|
432
|
+
const age = now - Date.parse(lease.renewedAt);
|
|
433
|
+
return Number.isNaN(age) || age > LEASE_TTL_MS;
|
|
434
|
+
}
|
|
435
|
+
/**
|
|
436
|
+
* Apply a committed transaction. Idempotent, and safe to re-run after a crash.
|
|
437
|
+
*
|
|
438
|
+
* Caller holds the lock.
|
|
439
|
+
*/
|
|
440
|
+
function applyTxLocked(runId, logger) {
|
|
441
|
+
const dir = runDir(runId);
|
|
442
|
+
const tx = path.join(dir, TX_COMMITTED);
|
|
443
|
+
const manifest = readJson(path.join(tx, 'manifest.json'));
|
|
444
|
+
if (!manifest) {
|
|
445
|
+
// No manifest means the rename landed a half-written staging directory,
|
|
446
|
+
// which cannot happen — but if it somehow did, it is not a commitment.
|
|
447
|
+
fs.rmSync(tx, { recursive: true, force: true });
|
|
448
|
+
return;
|
|
449
|
+
}
|
|
450
|
+
// Written after the last data step and before cleanup, so a failure while
|
|
451
|
+
// removing the directory leaves something that says "the data is in, only the
|
|
452
|
+
// tidying is left". Without it, a half-removed transaction directory looks
|
|
453
|
+
// exactly like a corrupt one, and re-applying would either throw forever or
|
|
454
|
+
// have to guess.
|
|
455
|
+
const appliedMarker = path.join(tx, 'applied');
|
|
456
|
+
if (fs.existsSync(appliedMarker)) {
|
|
457
|
+
fs.rmSync(tx, { recursive: true, force: true });
|
|
458
|
+
return;
|
|
459
|
+
}
|
|
460
|
+
for (const rel of manifest.artifacts) {
|
|
461
|
+
const from = path.join(tx, 'files', rel);
|
|
462
|
+
const to = path.join(dir, rel);
|
|
463
|
+
try {
|
|
464
|
+
fs.mkdirSync(path.dirname(to), { recursive: true });
|
|
465
|
+
fs.renameSync(from, to);
|
|
466
|
+
}
|
|
467
|
+
catch (err) {
|
|
468
|
+
// ENOENT means a previous application already moved it.
|
|
469
|
+
if (err.code !== 'ENOENT')
|
|
470
|
+
throw err;
|
|
471
|
+
}
|
|
472
|
+
}
|
|
473
|
+
if (manifest.hasEvents) {
|
|
474
|
+
const events = eventsFile(runId);
|
|
475
|
+
try {
|
|
476
|
+
fs.truncateSync(events, manifest.eventsOffset);
|
|
477
|
+
}
|
|
478
|
+
catch (err) {
|
|
479
|
+
if (err.code !== 'ENOENT')
|
|
480
|
+
throw err;
|
|
481
|
+
}
|
|
482
|
+
fs.appendFileSync(events, fs.readFileSync(path.join(tx, 'events.jsonl')));
|
|
483
|
+
}
|
|
484
|
+
if (manifest.hasRecord) {
|
|
485
|
+
try {
|
|
486
|
+
fs.renameSync(path.join(tx, 'run.json'), path.join(dir, 'run.json'));
|
|
487
|
+
}
|
|
488
|
+
catch (err) {
|
|
489
|
+
if (err.code !== 'ENOENT')
|
|
490
|
+
throw err;
|
|
491
|
+
}
|
|
492
|
+
}
|
|
493
|
+
fs.writeFileSync(appliedMarker, '');
|
|
494
|
+
fs.rmSync(tx, { recursive: true, force: true });
|
|
495
|
+
logger?.debug?.(`[kernel-store] ${runId}: applied a pending transaction`);
|
|
496
|
+
}
|
|
497
|
+
/** Stage a batch and publish it with one atomic rename. Caller holds the lock. */
|
|
498
|
+
function stageLocked(runId, batch) {
|
|
499
|
+
const dir = runDir(runId);
|
|
500
|
+
const staging = path.join(dir, TX_STAGING);
|
|
501
|
+
fs.rmSync(staging, { recursive: true, force: true });
|
|
502
|
+
fs.mkdirSync(staging, { recursive: true });
|
|
503
|
+
const manifest = {
|
|
504
|
+
eventsOffset: fileSize(eventsFile(runId)),
|
|
505
|
+
hasRecord: Boolean(batch.record),
|
|
506
|
+
hasEvents: Boolean(batch.events?.length),
|
|
507
|
+
artifacts: [],
|
|
508
|
+
};
|
|
509
|
+
for (const artifact of batch.artifacts ?? []) {
|
|
510
|
+
const rel = nodeArtifactPath(runId, artifact.nodeId, artifact.name);
|
|
511
|
+
const dest = path.join(staging, 'files', rel);
|
|
512
|
+
fs.mkdirSync(path.dirname(dest), { recursive: true });
|
|
513
|
+
fs.writeFileSync(dest, artifact.body);
|
|
514
|
+
manifest.artifacts.push(rel);
|
|
515
|
+
}
|
|
516
|
+
if (batch.record) {
|
|
517
|
+
// The checkpoint embeds the spec, so it gets the same scrub.
|
|
518
|
+
fs.writeFileSync(path.join(staging, 'run.json'), JSON.stringify(sanitizeForDisk(batch.record), null, 2));
|
|
519
|
+
}
|
|
520
|
+
if (batch.events?.length) {
|
|
521
|
+
fs.writeFileSync(path.join(staging, 'events.jsonl'), batch.events.map((e) => JSON.stringify(e)).join('\n') + '\n');
|
|
522
|
+
}
|
|
523
|
+
fs.writeFileSync(path.join(staging, 'manifest.json'), JSON.stringify(manifest));
|
|
524
|
+
// THE COMMIT POINT. Renaming a directory is atomic, so the transaction is
|
|
525
|
+
// either wholly present or wholly absent — never a checkpoint whose events
|
|
526
|
+
// were silently dropped, or an artifact left behind by a write that failed
|
|
527
|
+
// afterwards. Both of those were real: the event append swallowed its own
|
|
528
|
+
// errors, and artifacts were written before the checkpoint they belonged to.
|
|
529
|
+
fs.renameSync(staging, path.join(dir, TX_COMMITTED));
|
|
530
|
+
}
|
|
531
|
+
/** Finish any transaction that was committed but not yet applied. Caller holds the lock. */
|
|
532
|
+
function recoverPendingLocked(runId, logger) {
|
|
533
|
+
const dir = runDir(runId);
|
|
534
|
+
try {
|
|
535
|
+
// Staging that outlived its process was never committed.
|
|
536
|
+
fs.rmSync(path.join(dir, TX_STAGING), { recursive: true, force: true });
|
|
537
|
+
}
|
|
538
|
+
catch {
|
|
539
|
+
// Debris; the next attempt overwrites it.
|
|
540
|
+
}
|
|
541
|
+
if (!fs.existsSync(path.join(dir, TX_COMMITTED)))
|
|
542
|
+
return;
|
|
543
|
+
try {
|
|
544
|
+
applyTxLocked(runId, logger);
|
|
545
|
+
}
|
|
546
|
+
catch (err) {
|
|
547
|
+
logger?.warn?.(`[kernel-store] ${runId}: could not apply a pending transaction: ${err.message}`);
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
/**
|
|
551
|
+
* Run `fn` inside the run's lock, with any pending transaction applied first.
|
|
552
|
+
*
|
|
553
|
+
* Everything that inspects or changes ownership goes through here, so a check
|
|
554
|
+
* and the write that depends on it cannot be separated by another process — and
|
|
555
|
+
* so no caller can observe a run mid-transaction.
|
|
556
|
+
*/
|
|
557
|
+
function withRunLock(runId, fn, logger) {
|
|
558
|
+
return withFileLock(lockFile(runId), () => {
|
|
559
|
+
recoverPendingLocked(runId, logger);
|
|
560
|
+
return fn();
|
|
561
|
+
},
|
|
562
|
+
// Never `createParent`: a run directory that has been deleted must stay
|
|
563
|
+
// deleted. Recreating it to hold a lock left an empty directory behind, and
|
|
564
|
+
// the id then looked taken by a run that no longer existed.
|
|
565
|
+
{ staleMs: LEASE_TTL_MS });
|
|
566
|
+
}
|
|
567
|
+
/** Whether the run directory is still there at all. A gone run is not contention. */
|
|
568
|
+
function runDirExists(runId) {
|
|
569
|
+
try {
|
|
570
|
+
return fs.existsSync(runDir(runId));
|
|
571
|
+
}
|
|
572
|
+
catch {
|
|
573
|
+
return false;
|
|
574
|
+
}
|
|
575
|
+
}
|
|
576
|
+
/** Apply anything a crashed owner committed but did not finish writing. */
|
|
577
|
+
export function recoverPending(runId, logger) {
|
|
578
|
+
// Nothing to apply is not the same as something stuck: an id that cannot name
|
|
579
|
+
// a run has no pending transaction, and readers must not be told otherwise.
|
|
580
|
+
if (!isValidRunId(runId))
|
|
581
|
+
return true;
|
|
582
|
+
try {
|
|
583
|
+
if (!fs.existsSync(path.join(runDir(runId), TX_COMMITTED)))
|
|
584
|
+
return true;
|
|
585
|
+
}
|
|
586
|
+
catch {
|
|
587
|
+
return true;
|
|
588
|
+
}
|
|
589
|
+
// A transaction is applied by whoever holds the lock, so finding the lock busy
|
|
590
|
+
// means someone is very likely applying it right now. Retry before concluding
|
|
591
|
+
// anything: reporting "cannot be applied" for a moment of contention would
|
|
592
|
+
// turn a healthy commit into a read error, which is the same conflation this
|
|
593
|
+
// module exists to avoid.
|
|
594
|
+
for (let attempt = 0; attempt < 3; attempt++) {
|
|
595
|
+
const locked = withRunLock(runId, () => undefined, logger);
|
|
596
|
+
if (locked.ok)
|
|
597
|
+
break;
|
|
598
|
+
if (!fs.existsSync(path.join(runDir(runId), TX_COMMITTED)))
|
|
599
|
+
return true;
|
|
600
|
+
}
|
|
601
|
+
// `recoverPendingLocked` deliberately leaves a committed transaction in
|
|
602
|
+
// place when application fails. Its continued presence means returning the
|
|
603
|
+
// old checkpoint or event log would be a lie.
|
|
604
|
+
return !fs.existsSync(path.join(runDir(runId), TX_COMMITTED));
|
|
605
|
+
}
|
|
606
|
+
// ─── Claiming ───────────────────────────────────────────────────────────────
|
|
607
|
+
function acquireLocked(runId, ownerId) {
|
|
608
|
+
const incarnation = ensureIncarnationLocked(runId);
|
|
609
|
+
const existing = readLease(runId);
|
|
610
|
+
if (existing && existing.ownerId !== ownerId && !leaseIsStale(existing)) {
|
|
611
|
+
throw new Error(`Run '${runId}' is owned by ${existing.ownerId} (pid ${existing.pid} on ${existing.host}, ` +
|
|
612
|
+
`last seen ${existing.renewedAt}) — only one owner may run it at a time`);
|
|
613
|
+
}
|
|
614
|
+
const fence = incarnation.nextFence + 1;
|
|
615
|
+
atomicWriteJson(incarnationFile(runId), { ...incarnation, nextFence: fence });
|
|
616
|
+
const now = new Date().toISOString();
|
|
617
|
+
const lease = {
|
|
618
|
+
runId,
|
|
619
|
+
incarnationId: incarnation.incarnationId,
|
|
620
|
+
ownerId,
|
|
621
|
+
acquisitionId: crypto.randomUUID(),
|
|
622
|
+
fence,
|
|
623
|
+
pid: process.pid,
|
|
624
|
+
host: os.hostname(),
|
|
625
|
+
acquiredAt: existing?.ownerId === ownerId ? existing.acquiredAt : now,
|
|
626
|
+
renewedAt: now,
|
|
627
|
+
};
|
|
628
|
+
atomicWriteJson(leaseFile(runId), lease);
|
|
629
|
+
return { runId, incarnationId: lease.incarnationId, ownerId, acquisitionId: lease.acquisitionId, fence };
|
|
630
|
+
}
|
|
631
|
+
/**
|
|
632
|
+
* Create a run and claim it, atomically.
|
|
633
|
+
*
|
|
634
|
+
* The non-recursive `mkdir` IS the claim: it fails with EEXIST for everyone but
|
|
635
|
+
* the first caller, so exactly one process can create a given run id. The
|
|
636
|
+
* previous version asked `runExists()` and then created with
|
|
637
|
+
* `{ recursive: true }`, which is a check-then-write race and lost it routinely
|
|
638
|
+
* — two processes creating the same id 80 times had both "succeed" 76 times,
|
|
639
|
+
* leaving one workflow executing while the other's `spec.json` sat on disk, and
|
|
640
|
+
* a lease belonging to an incarnation that had already been overwritten.
|
|
641
|
+
*/
|
|
642
|
+
export function createAndAcquire(runId, spec, ownerId) {
|
|
643
|
+
const dir = runDir(runId);
|
|
644
|
+
fs.mkdirSync(wfDir(), { recursive: true });
|
|
645
|
+
try {
|
|
646
|
+
fs.mkdirSync(dir);
|
|
647
|
+
}
|
|
648
|
+
catch (err) {
|
|
649
|
+
if (err.code === 'EEXIST') {
|
|
650
|
+
throw new Error(`Run '${runId}' already exists — pick another id, or resume it instead of starting over`);
|
|
651
|
+
}
|
|
652
|
+
throw err;
|
|
653
|
+
}
|
|
654
|
+
// Nobody else can be inside a directory that did not exist a moment ago, so
|
|
655
|
+
// this lock is uncontended by construction; taking it anyway keeps every
|
|
656
|
+
// write to the run under the same discipline.
|
|
657
|
+
const locked = withRunLock(runId, () => {
|
|
658
|
+
atomicWriteJson(incarnationFile(runId), { incarnationId: crypto.randomUUID(), nextFence: 0 });
|
|
659
|
+
atomicWriteJson(path.join(dir, 'spec.json'), sanitizeForDisk(spec));
|
|
660
|
+
return acquireLocked(runId, ownerId);
|
|
661
|
+
});
|
|
662
|
+
if (!locked.ok) {
|
|
663
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
664
|
+
throw new Error(`Run '${runId}' could not be created: ${locked.error}`);
|
|
665
|
+
}
|
|
666
|
+
return locked.value;
|
|
667
|
+
}
|
|
668
|
+
/**
|
|
669
|
+
* Claim an existing run and receive the capability to write to it.
|
|
670
|
+
*
|
|
671
|
+
* Refused when someone else's claim is still live. Allowed for the same owner —
|
|
672
|
+
* but the guard it returns is a NEW one, and the previous guard is dead from
|
|
673
|
+
* that moment: re-acquiring is a fresh claim, not a renewal of the old one, so
|
|
674
|
+
* anything still holding the old guard stops being able to write.
|
|
675
|
+
*/
|
|
676
|
+
export function acquireLease(runId, ownerId) {
|
|
677
|
+
if (!runDirExists(runId))
|
|
678
|
+
throw new Error(`Run '${runId}' not found`);
|
|
679
|
+
const locked = withRunLock(runId, () => acquireLocked(runId, ownerId));
|
|
680
|
+
if (!locked.ok) {
|
|
681
|
+
throw new Error(`Run '${runId}' could not be claimed right now: ${locked.error}`);
|
|
682
|
+
}
|
|
683
|
+
return locked.value;
|
|
684
|
+
}
|
|
685
|
+
/** Whether the on-disk lease still answers to this guard. Caller must hold the lock. */
|
|
686
|
+
function guardIsCurrentLocked(guard) {
|
|
687
|
+
const incarnation = readIncarnation(guard.runId);
|
|
688
|
+
if (!incarnation || incarnation.incarnationId !== guard.incarnationId)
|
|
689
|
+
return false;
|
|
690
|
+
const lease = readLease(guard.runId);
|
|
691
|
+
return Boolean(lease &&
|
|
692
|
+
lease.incarnationId === guard.incarnationId &&
|
|
693
|
+
lease.ownerId === guard.ownerId &&
|
|
694
|
+
lease.acquisitionId === guard.acquisitionId &&
|
|
695
|
+
lease.fence === guard.fence);
|
|
696
|
+
}
|
|
697
|
+
/** Heartbeat. Best-effort: losing a renewal must not break the run. */
|
|
698
|
+
export function renewLease(guard) {
|
|
699
|
+
if (!runDirExists(guard.runId))
|
|
700
|
+
return;
|
|
701
|
+
withRunLock(guard.runId, () => {
|
|
702
|
+
if (!guardIsCurrentLocked(guard))
|
|
703
|
+
return;
|
|
704
|
+
const existing = readLease(guard.runId);
|
|
705
|
+
try {
|
|
706
|
+
atomicWriteJson(leaseFile(guard.runId), { ...existing, renewedAt: new Date().toISOString() });
|
|
707
|
+
}
|
|
708
|
+
catch {
|
|
709
|
+
// The next renewal will try again.
|
|
710
|
+
}
|
|
711
|
+
});
|
|
712
|
+
}
|
|
713
|
+
/**
|
|
714
|
+
* The one way to change anything durable about a run.
|
|
715
|
+
*
|
|
716
|
+
* The guard is verified and the transaction is published inside a single
|
|
717
|
+
* critical section, so a holder that has been superseded cannot land a write
|
|
718
|
+
* between the check and the change.
|
|
719
|
+
*
|
|
720
|
+
* Callers pass a record they have already produced by copying and mutating,
|
|
721
|
+
* never the record they are still using: `commit` persists what it is given, and
|
|
722
|
+
* the caller adopts it only on `committed`. That is what keeps a refused write
|
|
723
|
+
* from leaving a run *in memory* claiming a state the disk rejected.
|
|
724
|
+
*/
|
|
725
|
+
export function commit(guard, batch, logger) {
|
|
726
|
+
// A deleted run is not contention, and must not be reported as something to
|
|
727
|
+
// retry: there is nothing left to write to, ever.
|
|
728
|
+
if (!runDirExists(guard.runId)) {
|
|
729
|
+
return { outcome: 'superseded', reason: `run '${guard.runId}' no longer exists` };
|
|
730
|
+
}
|
|
731
|
+
const locked = withRunLock(guard.runId, () => {
|
|
732
|
+
if (!guardIsCurrentLocked(guard)) {
|
|
733
|
+
const current = readLease(guard.runId);
|
|
734
|
+
return {
|
|
735
|
+
outcome: 'superseded',
|
|
736
|
+
reason: current
|
|
737
|
+
? `run is now held by ${current.ownerId} at fence ${current.fence}`
|
|
738
|
+
: `run '${guard.runId}' no longer exists, or its claim was released`,
|
|
739
|
+
};
|
|
740
|
+
}
|
|
741
|
+
try {
|
|
742
|
+
stageLocked(guard.runId, batch);
|
|
743
|
+
}
|
|
744
|
+
catch (err) {
|
|
745
|
+
try {
|
|
746
|
+
fs.rmSync(path.join(runDir(guard.runId), TX_STAGING), { recursive: true, force: true });
|
|
747
|
+
}
|
|
748
|
+
catch {
|
|
749
|
+
// Debris only; nothing was published.
|
|
750
|
+
}
|
|
751
|
+
return { outcome: 'blocked', reason: `could not stage the change: ${err.message}` };
|
|
752
|
+
}
|
|
753
|
+
// Past the commit point. Application is replayable, so a failure here is a
|
|
754
|
+
// delay, not a loss — and reporting it as "not committed" would be wrong.
|
|
755
|
+
try {
|
|
756
|
+
applyTxLocked(guard.runId, logger);
|
|
757
|
+
}
|
|
758
|
+
catch (err) {
|
|
759
|
+
logger?.warn?.(`[kernel-store] ${guard.runId}: change is committed but not yet applied ` +
|
|
760
|
+
`(${err.message}); it will be applied on the next lock`);
|
|
761
|
+
}
|
|
762
|
+
try {
|
|
763
|
+
const lease = readLease(guard.runId);
|
|
764
|
+
atomicWriteJson(leaseFile(guard.runId), { ...lease, renewedAt: new Date().toISOString() });
|
|
765
|
+
}
|
|
766
|
+
catch {
|
|
767
|
+
// The heartbeat is not part of the commitment.
|
|
768
|
+
}
|
|
769
|
+
return { outcome: 'committed' };
|
|
770
|
+
}, logger);
|
|
771
|
+
return locked.ok ? locked.value : { outcome: 'blocked', reason: locked.error };
|
|
772
|
+
}
|
|
773
|
+
/**
|
|
774
|
+
* Release the claim. The incarnation file is deliberately NOT removed — the
|
|
775
|
+
* fence must keep increasing for as long as this creation of the run exists.
|
|
776
|
+
*
|
|
777
|
+
* `blocked` is not `not-ours`: an owner that is standing down and cannot take
|
|
778
|
+
* the lock still has to hand the run back, so the caller retries rather than
|
|
779
|
+
* leaving a lease behind that no one can take over.
|
|
780
|
+
*/
|
|
781
|
+
export function releaseLease(guard) {
|
|
782
|
+
if (!runDirExists(guard.runId))
|
|
783
|
+
return 'not-ours';
|
|
784
|
+
const locked = withRunLock(guard.runId, () => {
|
|
785
|
+
if (!guardIsCurrentLocked(guard))
|
|
786
|
+
return 'not-ours';
|
|
787
|
+
fs.rmSync(leaseFile(guard.runId), { force: true });
|
|
788
|
+
return 'released';
|
|
789
|
+
});
|
|
790
|
+
return locked.ok ? locked.value : 'blocked';
|
|
791
|
+
}
|
|
792
|
+
export function summarize(record) {
|
|
793
|
+
return {
|
|
794
|
+
runId: record.runId,
|
|
795
|
+
workflow: record.workflow,
|
|
796
|
+
state: record.state,
|
|
797
|
+
outcome: record.outcome,
|
|
798
|
+
cwd: record.cwd,
|
|
799
|
+
createdAt: record.createdAt,
|
|
800
|
+
updatedAt: record.updatedAt,
|
|
801
|
+
endedAt: record.endedAt,
|
|
802
|
+
currentNode: record.currentNode,
|
|
803
|
+
evidenceId: record.evidenceId,
|
|
804
|
+
error: record.error,
|
|
805
|
+
costUsd: record.costUsd,
|
|
806
|
+
};
|
|
807
|
+
}
|
|
808
|
+
/**
|
|
809
|
+
* Every run on this machine, newest first — the cross-process listing the three
|
|
810
|
+
* mode-specific enumerations each did differently (council regex-scraped its own
|
|
811
|
+
* markdown transcripts, autoloop read a JSONL registry, ultraapp walked a store
|
|
812
|
+
* directory). No separate index file is needed here because every run lives
|
|
813
|
+
* under one root, so the directory *is* the index.
|
|
814
|
+
*/
|
|
815
|
+
export function listRuns(query = {}) {
|
|
816
|
+
const out = [];
|
|
817
|
+
for (const runId of listRunIds()) {
|
|
818
|
+
let record;
|
|
819
|
+
try {
|
|
820
|
+
record = loadRun(runId);
|
|
821
|
+
}
|
|
822
|
+
catch {
|
|
823
|
+
// One run whose committed transaction cannot currently be applied must
|
|
824
|
+
// not make every other run disappear from the listing.
|
|
825
|
+
continue;
|
|
826
|
+
}
|
|
827
|
+
if (!record)
|
|
828
|
+
continue;
|
|
829
|
+
if (query.workflow && record.workflow !== query.workflow)
|
|
830
|
+
continue;
|
|
831
|
+
if (query.state && record.state !== query.state)
|
|
832
|
+
continue;
|
|
833
|
+
out.push(summarize(record));
|
|
834
|
+
}
|
|
835
|
+
out.sort((a, b) => (a.createdAt < b.createdAt ? 1 : a.createdAt > b.createdAt ? -1 : 0));
|
|
836
|
+
return query.limit && query.limit > 0 ? out.slice(0, query.limit) : out;
|
|
837
|
+
}
|
|
838
|
+
//# sourceMappingURL=store.js.map
|