@enderfga/claw-orchestrator 5.1.0 → 6.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -26
- package/dist/bin/cli.js +107 -1
- package/dist/bin/cli.js.map +1 -1
- package/dist/src/acp-server.d.ts +5 -5
- package/dist/src/acp-server.js +3 -3
- package/dist/src/acp-server.js.map +1 -1
- package/dist/src/autoloop/dispatcher.d.ts +22 -0
- package/dist/src/autoloop/dispatcher.js +71 -13
- package/dist/src/autoloop/dispatcher.js.map +1 -1
- package/dist/src/autoloop/messages.d.ts +10 -0
- package/dist/src/autoloop/messages.js.map +1 -1
- package/dist/src/autoloop/runner.js +6 -0
- package/dist/src/autoloop/runner.js.map +1 -1
- package/dist/src/constants.d.ts +0 -6
- package/dist/src/constants.js +0 -6
- package/dist/src/constants.js.map +1 -1
- package/dist/src/council.d.ts +15 -0
- package/dist/src/council.js +48 -35
- package/dist/src/council.js.map +1 -1
- package/dist/src/dashboard/index.html +191 -6
- package/dist/src/embedded-server.js +132 -9
- package/dist/src/embedded-server.js.map +1 -1
- package/dist/src/fanout.d.ts +30 -1
- package/dist/src/fanout.js +32 -3
- package/dist/src/fanout.js.map +1 -1
- package/dist/src/index.js +359 -4
- package/dist/src/index.js.map +1 -1
- package/dist/src/kernel/agent-step.d.ts +59 -0
- package/dist/src/kernel/agent-step.js +100 -0
- package/dist/src/kernel/agent-step.js.map +1 -0
- package/dist/src/kernel/conditions.d.ts +11 -0
- package/dist/src/kernel/conditions.js +24 -0
- package/dist/src/kernel/conditions.js.map +1 -0
- package/dist/src/kernel/engine.d.ts +319 -0
- package/dist/src/kernel/engine.js +1047 -0
- package/dist/src/kernel/engine.js.map +1 -0
- package/dist/src/kernel/exec.d.ts +43 -0
- package/dist/src/kernel/exec.js +112 -0
- package/dist/src/kernel/exec.js.map +1 -0
- package/dist/src/kernel/file-lock.d.ts +50 -0
- package/dist/src/kernel/file-lock.js +135 -0
- package/dist/src/kernel/file-lock.js.map +1 -0
- package/dist/src/kernel/nodes/agent.d.ts +4 -0
- package/dist/src/kernel/nodes/agent.js +35 -0
- package/dist/src/kernel/nodes/agent.js.map +1 -0
- package/dist/src/kernel/nodes/autoloop.d.ts +78 -0
- package/dist/src/kernel/nodes/autoloop.js +75 -0
- package/dist/src/kernel/nodes/autoloop.js.map +1 -0
- package/dist/src/kernel/nodes/council.d.ts +12 -0
- package/dist/src/kernel/nodes/council.js +88 -0
- package/dist/src/kernel/nodes/council.js.map +1 -0
- package/dist/src/kernel/nodes/fanout.d.ts +11 -0
- package/dist/src/kernel/nodes/fanout.js +63 -0
- package/dist/src/kernel/nodes/fanout.js.map +1 -0
- package/dist/src/kernel/nodes/human-gate.d.ts +4 -0
- package/dist/src/kernel/nodes/human-gate.js +7 -0
- package/dist/src/kernel/nodes/human-gate.js.map +1 -0
- package/dist/src/kernel/nodes/index.d.ts +12 -0
- package/dist/src/kernel/nodes/index.js +21 -0
- package/dist/src/kernel/nodes/index.js.map +1 -0
- package/dist/src/kernel/nodes/router.d.ts +4 -0
- package/dist/src/kernel/nodes/router.js +12 -0
- package/dist/src/kernel/nodes/router.js.map +1 -0
- package/dist/src/kernel/nodes/subflow.d.ts +13 -0
- package/dist/src/kernel/nodes/subflow.js +38 -0
- package/dist/src/kernel/nodes/subflow.js.map +1 -0
- package/dist/src/kernel/nodes/ultraapp.d.ts +60 -0
- package/dist/src/kernel/nodes/ultraapp.js +62 -0
- package/dist/src/kernel/nodes/ultraapp.js.map +1 -0
- package/dist/src/kernel/nodes/verifier.d.ts +14 -0
- package/dist/src/kernel/nodes/verifier.js +84 -0
- package/dist/src/kernel/nodes/verifier.js.map +1 -0
- package/dist/src/kernel/projections.d.ts +42 -0
- package/dist/src/kernel/projections.js +133 -0
- package/dist/src/kernel/projections.js.map +1 -0
- package/dist/src/kernel/repo.d.ts +13 -0
- package/dist/src/kernel/repo.js +64 -0
- package/dist/src/kernel/repo.js.map +1 -0
- package/dist/src/kernel/secrets.d.ts +25 -0
- package/dist/src/kernel/secrets.js +48 -0
- package/dist/src/kernel/secrets.js.map +1 -0
- package/dist/src/kernel/store.d.ts +225 -0
- package/dist/src/kernel/store.js +838 -0
- package/dist/src/kernel/store.js.map +1 -0
- package/dist/src/kernel/templates/index.d.ts +140 -0
- package/dist/src/kernel/templates/index.js +266 -0
- package/dist/src/kernel/templates/index.js.map +1 -0
- package/dist/src/kernel/types.d.ts +326 -0
- package/dist/src/kernel/types.js +19 -0
- package/dist/src/kernel/types.js.map +1 -0
- package/dist/src/models.d.ts +7 -0
- package/dist/src/models.js +43 -14
- package/dist/src/models.js.map +1 -1
- package/dist/src/persistent-custom-session.js +8 -3
- package/dist/src/persistent-custom-session.js.map +1 -1
- package/dist/src/run-ledger.d.ts +57 -3
- package/dist/src/run-ledger.js +45 -2
- package/dist/src/run-ledger.js.map +1 -1
- package/dist/src/session-manager.d.ts +176 -129
- package/dist/src/session-manager.js +652 -603
- package/dist/src/session-manager.js.map +1 -1
- package/dist/src/types.d.ts +33 -3
- package/dist/src/ultraapp/build.d.ts +117 -3
- package/dist/src/ultraapp/build.js +319 -3
- package/dist/src/ultraapp/build.js.map +1 -1
- package/dist/src/ultraapp/contract.d.ts +52 -0
- package/dist/src/ultraapp/contract.js +83 -0
- package/dist/src/ultraapp/contract.js.map +1 -0
- package/dist/src/ultraapp/conventions.js +9 -2
- package/dist/src/ultraapp/conventions.js.map +1 -1
- package/dist/src/ultraapp/fix-on-failure.d.ts +21 -2
- package/dist/src/ultraapp/fix-on-failure.js +46 -62
- package/dist/src/ultraapp/fix-on-failure.js.map +1 -1
- package/dist/src/ultraapp/manager.d.ts +107 -2
- package/dist/src/ultraapp/manager.js +305 -86
- package/dist/src/ultraapp/manager.js.map +1 -1
- package/dist/src/verify/baseline.d.ts +73 -0
- package/dist/src/verify/baseline.js +186 -0
- package/dist/src/verify/baseline.js.map +1 -0
- package/dist/src/verify/contract.d.ts +116 -0
- package/dist/src/verify/contract.js +142 -0
- package/dist/src/verify/contract.js.map +1 -0
- package/dist/src/verify/evidence.d.ts +61 -0
- package/dist/src/verify/evidence.js +133 -0
- package/dist/src/verify/evidence.js.map +1 -0
- package/dist/src/verify/runner.d.ts +63 -0
- package/dist/src/verify/runner.js +317 -0
- package/dist/src/verify/runner.js.map +1 -0
- package/openclaw.plugin.json +38 -1
- package/package.json +2 -2
- package/skills/SKILL.md +120 -79
- package/skills/references/acp.md +17 -17
- package/skills/references/autoloop.md +139 -65
- package/skills/references/claude-cli-tracking.md +4 -4
- package/skills/references/cli.md +101 -59
- package/skills/references/council.md +109 -37
- package/skills/references/dashboard.md +34 -6
- package/skills/references/getting-started.md +13 -13
- package/skills/references/inbox.md +4 -4
- package/skills/references/mcp.md +39 -34
- package/skills/references/multi-engine.md +51 -47
- package/skills/references/observability.md +115 -28
- package/skills/references/openai-compat.md +39 -39
- package/skills/references/sessions.md +43 -25
- package/skills/references/tools.md +402 -309
- package/skills/references/ultra.md +45 -45
- package/skills/references/ultraapp.md +126 -50
- package/skills/references/verification.md +187 -0
- package/skills/references/workflow.md +362 -0
- package/dist/src/ultraapp/fix-on-failure-session.d.ts +0 -23
- package/dist/src/ultraapp/fix-on-failure-session.js +0 -51
- package/dist/src/ultraapp/fix-on-failure-session.js.map +0 -1
|
@@ -0,0 +1,1047 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The run kernel.
|
|
3
|
+
*
|
|
4
|
+
* A durable executor for workflow runs. It owns what the existing state machines
|
|
5
|
+
* each owned separately and differently: what is running, what to do when a step
|
|
6
|
+
* fails, when to give up, how to stop, and — the part none of them had — how to
|
|
7
|
+
* come back after the process dies.
|
|
8
|
+
*
|
|
9
|
+
* Every mode goes through it. `council_start`, `fanout_start`, `ultraplan_start`,
|
|
10
|
+
* `ultrareview_start` and `autoloop_start` each create a run here; the engines
|
|
11
|
+
* they wrap (`Council`, `Fanout`, the autoloop dispatcher) still do the work,
|
|
12
|
+
* but they no longer own a lifecycle. What that replaced: five result maps, four
|
|
13
|
+
* 30-minute eviction timers, a 5-second poller, two `Set`s fencing a start
|
|
14
|
+
* against a delete, and two incompatible ways of listing past runs across
|
|
15
|
+
* processes.
|
|
16
|
+
*
|
|
17
|
+
* Durability contract, stated precisely: every state transition is checkpointed
|
|
18
|
+
* (atomic `run.json`) and appended to `events.jsonl` before the next step
|
|
19
|
+
* begins, and `resume` restarts at the last node boundary. That makes node
|
|
20
|
+
* execution **at-least-once**, not exactly-once — there is no idempotency key,
|
|
21
|
+
* no attempt lease, and no side-effect commit marker, so a node that died after
|
|
22
|
+
* writing files but before its checkpoint runs again from the top. Workflows
|
|
23
|
+
* whose nodes are not safe to repeat need to make them safe.
|
|
24
|
+
*
|
|
25
|
+
* Ownership contract, equally precisely: this class never writes to a run
|
|
26
|
+
* directly. It holds a `RunGuard` from the store and every change — checkpoint,
|
|
27
|
+
* event, node artifact, terminal verdict — goes through `RunTxn`, which applies
|
|
28
|
+
* the change to a *copy*, commits it under the guard, and adopts the copy only
|
|
29
|
+
* if the disk accepted it. There is no code path from here to `run.json` that
|
|
30
|
+
* skips that, because the raw writers are no longer exported. An owner that has
|
|
31
|
+
* been superseded therefore cannot change the run in memory either: its record
|
|
32
|
+
* stops advancing at the last write the disk agreed to.
|
|
33
|
+
*
|
|
34
|
+
* Control flow is linear with an explicit `router` node for branches and loops.
|
|
35
|
+
* Parallelism is the `fanout` node rather than a general parallel/join construct:
|
|
36
|
+
* fan-out is the shape every existing mode actually needed, and a join barrier
|
|
37
|
+
* would add failure modes (partial joins, orphaned branches) that nothing here
|
|
38
|
+
* would exercise.
|
|
39
|
+
*
|
|
40
|
+
* What a node timeout can and cannot do: it stops the kernel waiting and marks
|
|
41
|
+
* the node failed, but an in-flight agent turn is owned by the session layer and
|
|
42
|
+
* finishes on its own schedule. The kernel does not pretend otherwise.
|
|
43
|
+
*/
|
|
44
|
+
import { EventEmitter } from 'node:events';
|
|
45
|
+
import crypto from 'node:crypto';
|
|
46
|
+
import { createConsoleLogger } from '../logger.js';
|
|
47
|
+
import { normalizeContract } from '../verify/contract.js';
|
|
48
|
+
import { captureBaseline, treeFingerprint } from '../verify/baseline.js';
|
|
49
|
+
import { acquireLease, commit, createAndAcquire, LEASE_HEARTBEAT_MS, renewLease, isValidRunId, releaseLease, deleteRunDir, listRuns, loadRun, nodeArtifactPath, runDir, summarize, } from './store.js';
|
|
50
|
+
import { isTerminalRunState, } from './types.js';
|
|
51
|
+
export const DEFAULT_NODE_TIMEOUT_MS = 30 * 60_000;
|
|
52
|
+
export const DEFAULT_MAX_NODE_VISITS = 50;
|
|
53
|
+
/**
|
|
54
|
+
* How long a run about to claim `verified` waits for abandoned attempts to stop.
|
|
55
|
+
*
|
|
56
|
+
* Short on purpose. The wait exists to catch an attempt that is about to finish,
|
|
57
|
+
* not to hold a run hostage to one that never will — a node stuck forever must
|
|
58
|
+
* not stop the run from ending, only from being called verified.
|
|
59
|
+
*/
|
|
60
|
+
export const QUIESCE_GRACE_MS = 2_000;
|
|
61
|
+
/** Terminal verifier appended when the workflow declares a run-level contract. */
|
|
62
|
+
export const IMPLICIT_VERIFIER_ID = '__verify';
|
|
63
|
+
/** How much node text the checkpoint carries inline. The rest goes to disk. */
|
|
64
|
+
export const OUTPUT_PREVIEW_CHARS = 4000;
|
|
65
|
+
/**
|
|
66
|
+
* One owner's transactional view of a run.
|
|
67
|
+
*
|
|
68
|
+
* It holds the guard and the last record the disk accepted, and it is the only
|
|
69
|
+
* thing in this module that can change either. Three properties matter:
|
|
70
|
+
*
|
|
71
|
+
* - **Copy-on-write.** A change is applied to a clone, committed, and adopted
|
|
72
|
+
* only if the commit succeeded. A refused write therefore leaves the record
|
|
73
|
+
* this process hands to its callers exactly as the disk has it. The previous
|
|
74
|
+
* version mutated first and committed second in most places, so a superseded
|
|
75
|
+
* owner returned a record saying `completed` while `run.json` correctly said
|
|
76
|
+
* `running` — the same lie, one layer up.
|
|
77
|
+
* - **Everything, not most things.** Checkpoints, events and node artifacts all
|
|
78
|
+
* go through here. There is no unfenced variant to reach for, because the
|
|
79
|
+
* store no longer exports one.
|
|
80
|
+
* - **One-way.** The first refusal sets `lost`, and nothing is attempted after
|
|
81
|
+
* that. An owner that has been replaced does not keep trying.
|
|
82
|
+
*/
|
|
83
|
+
class RunTxn {
|
|
84
|
+
guard;
|
|
85
|
+
logger;
|
|
86
|
+
onEvents;
|
|
87
|
+
onStop;
|
|
88
|
+
_record;
|
|
89
|
+
/** The run was taken over. Permanent: this owner may never write again. */
|
|
90
|
+
lost = false;
|
|
91
|
+
/**
|
|
92
|
+
* A write could not get through, but nothing says we lost the run.
|
|
93
|
+
*
|
|
94
|
+
* Kept apart from `lost` because the responses are opposite. Treating a
|
|
95
|
+
* millisecond of lock contention as a takeover made a run stop forever while
|
|
96
|
+
* still holding its lease — so nobody could take it over either, and a live
|
|
97
|
+
* local pid is never judged stale. Stalling stops the run and hands the claim
|
|
98
|
+
* back, which leaves it resumable.
|
|
99
|
+
*/
|
|
100
|
+
stalled = false;
|
|
101
|
+
constructor(guard, record, logger,
|
|
102
|
+
/** Notified with the events of each accepted commit, for the in-process stream. */
|
|
103
|
+
onEvents,
|
|
104
|
+
/** Called once, when this owner stops writing, with which of the two it was. */
|
|
105
|
+
onStop) {
|
|
106
|
+
this.guard = guard;
|
|
107
|
+
this.logger = logger;
|
|
108
|
+
this.onEvents = onEvents;
|
|
109
|
+
this.onStop = onStop;
|
|
110
|
+
this._record = record;
|
|
111
|
+
}
|
|
112
|
+
get record() {
|
|
113
|
+
return this._record;
|
|
114
|
+
}
|
|
115
|
+
/** True once this owner has stopped writing, for either reason. */
|
|
116
|
+
get finished() {
|
|
117
|
+
return this.lost || this.stalled;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Apply a change, persist it, and adopt it — or refuse all three.
|
|
121
|
+
*
|
|
122
|
+
* `mutate` receives a clone. Mutating `txn.record` inside it would defeat the
|
|
123
|
+
* whole mechanism, which is why the clone is what is handed over.
|
|
124
|
+
*/
|
|
125
|
+
apply(mutate, events = [], artifacts = []) {
|
|
126
|
+
if (this.finished)
|
|
127
|
+
return false;
|
|
128
|
+
const draft = JSON.parse(JSON.stringify(this._record));
|
|
129
|
+
mutate(draft);
|
|
130
|
+
const result = commit(this.guard, { record: draft, events, artifacts }, this.logger);
|
|
131
|
+
if (result.outcome !== 'committed')
|
|
132
|
+
return this._stop(result.outcome, result.reason);
|
|
133
|
+
this._record = draft;
|
|
134
|
+
this.onEvents(events);
|
|
135
|
+
return true;
|
|
136
|
+
}
|
|
137
|
+
/** Append events without changing state. Fenced like every other write. */
|
|
138
|
+
emit(...events) {
|
|
139
|
+
if (this.finished)
|
|
140
|
+
return false;
|
|
141
|
+
const result = commit(this.guard, { events }, this.logger);
|
|
142
|
+
if (result.outcome !== 'committed')
|
|
143
|
+
return this._stop(result.outcome, result.reason);
|
|
144
|
+
this.onEvents(events);
|
|
145
|
+
return true;
|
|
146
|
+
}
|
|
147
|
+
_stop(outcome, reason) {
|
|
148
|
+
if (this.finished)
|
|
149
|
+
return false;
|
|
150
|
+
const why = reason ?? 'no reason given';
|
|
151
|
+
if (outcome === 'superseded') {
|
|
152
|
+
this.lost = true;
|
|
153
|
+
this.logger.warn?.(`[kernel] ${this.guard.runId}: write refused — this owner (fence ${this.guard.fence}) no longer holds ` +
|
|
154
|
+
`the run (${why}), so it stops here rather than carrying on in memory`);
|
|
155
|
+
}
|
|
156
|
+
else {
|
|
157
|
+
this.stalled = true;
|
|
158
|
+
this.logger.warn?.(`[kernel] ${this.guard.runId}: write could not be committed (${why}) — the run stops and its claim is ` +
|
|
159
|
+
`handed back, so it can be resumed rather than wedged`);
|
|
160
|
+
}
|
|
161
|
+
this.onStop(outcome, why);
|
|
162
|
+
return false;
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
|
166
|
+
/**
|
|
167
|
+
* Append the implicit terminal verifier when a run-level contract exists and the
|
|
168
|
+
* author did not already place one. This is what makes `completed` mean "checked".
|
|
169
|
+
*/
|
|
170
|
+
/**
|
|
171
|
+
* Reject a spec that cannot execute correctly, before anything runs.
|
|
172
|
+
*
|
|
173
|
+
* These are all mistakes that used to surface much later and much worse: a
|
|
174
|
+
* duplicate node id silently overwrote a record so one of the two nodes had no
|
|
175
|
+
* state; a `next` or router target naming a node that does not exist failed the
|
|
176
|
+
* run halfway through with `unknown node`; a negative retry or visit bound was
|
|
177
|
+
* accepted and then behaved arbitrarily.
|
|
178
|
+
*/
|
|
179
|
+
export function validateSpec(spec) {
|
|
180
|
+
if (!spec || typeof spec.name !== 'string' || !spec.name.trim()) {
|
|
181
|
+
throw new Error('WorkflowSpec needs a name');
|
|
182
|
+
}
|
|
183
|
+
if (!Array.isArray(spec.nodes) || spec.nodes.length === 0) {
|
|
184
|
+
throw new Error(`Workflow '${spec.name}' has no nodes`);
|
|
185
|
+
}
|
|
186
|
+
const ids = new Set();
|
|
187
|
+
for (const node of spec.nodes) {
|
|
188
|
+
if (!node?.id || typeof node.id !== 'string')
|
|
189
|
+
throw new Error(`Workflow '${spec.name}' has a node with no id`);
|
|
190
|
+
if (ids.has(node.id))
|
|
191
|
+
throw new Error(`Workflow '${spec.name}' has duplicate node id '${node.id}'`);
|
|
192
|
+
ids.add(node.id);
|
|
193
|
+
if (node.retry && (!Number.isInteger(node.retry.max) || node.retry.max < 0)) {
|
|
194
|
+
throw new Error(`Node '${node.id}': retry.max must be a non-negative integer`);
|
|
195
|
+
}
|
|
196
|
+
if (node.timeoutMs !== undefined && (!Number.isFinite(node.timeoutMs) || node.timeoutMs <= 0)) {
|
|
197
|
+
throw new Error(`Node '${node.id}': timeoutMs must be a positive number`);
|
|
198
|
+
}
|
|
199
|
+
}
|
|
200
|
+
if (spec.maxNodeVisits !== undefined && (!Number.isInteger(spec.maxNodeVisits) || spec.maxNodeVisits < 1)) {
|
|
201
|
+
throw new Error(`Workflow '${spec.name}': maxNodeVisits must be a positive integer`);
|
|
202
|
+
}
|
|
203
|
+
const check = (target, where) => {
|
|
204
|
+
if (target !== undefined && !ids.has(target)) {
|
|
205
|
+
throw new Error(`${where} points at unknown node '${target}'`);
|
|
206
|
+
}
|
|
207
|
+
};
|
|
208
|
+
for (const node of spec.nodes) {
|
|
209
|
+
check(node.next, `Node '${node.id}' next`);
|
|
210
|
+
if (node.kind === 'router') {
|
|
211
|
+
check(node.default, `Router '${node.id}' default`);
|
|
212
|
+
for (const route of node.routes ?? []) {
|
|
213
|
+
check(route.to, `Router '${node.id}' route`);
|
|
214
|
+
if (!route.when?.type)
|
|
215
|
+
throw new Error(`Router '${node.id}' has a route with no condition`);
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
export function prepareSpec(spec) {
|
|
221
|
+
if (!spec.contract)
|
|
222
|
+
return spec;
|
|
223
|
+
const hasRunVerifier = spec.nodes.some((n) => n.kind === 'verifier' && n.contract === 'run');
|
|
224
|
+
if (hasRunVerifier)
|
|
225
|
+
return spec;
|
|
226
|
+
return {
|
|
227
|
+
...spec,
|
|
228
|
+
nodes: [...spec.nodes, { id: IMPLICIT_VERIFIER_ID, kind: 'verifier', contract: 'run' }],
|
|
229
|
+
};
|
|
230
|
+
}
|
|
231
|
+
export class RunKernel extends EventEmitter {
|
|
232
|
+
manager;
|
|
233
|
+
logger;
|
|
234
|
+
executors;
|
|
235
|
+
nodeTimeoutMs;
|
|
236
|
+
live = new Map();
|
|
237
|
+
/**
|
|
238
|
+
* This kernel's identity as a run owner.
|
|
239
|
+
*
|
|
240
|
+
* Per instance, not per process: two `RunKernel`s in one process — two
|
|
241
|
+
* SessionManagers is not exotic — would otherwise share a pid and each treat
|
|
242
|
+
* the other's lease as its own, so both would execute the same run.
|
|
243
|
+
*/
|
|
244
|
+
ownerId = `owner-${crypto.randomUUID()}`;
|
|
245
|
+
/**
|
|
246
|
+
* Per-run secrets, in memory only. Keyed by run id so a resume in this
|
|
247
|
+
* process can re-use what the start supplied; a resume elsewhere must be
|
|
248
|
+
* given them again.
|
|
249
|
+
*/
|
|
250
|
+
_secrets = new Map();
|
|
251
|
+
/** Caller label for the most recent start of each run. */
|
|
252
|
+
_tags = new Map();
|
|
253
|
+
constructor(opts = {}) {
|
|
254
|
+
super();
|
|
255
|
+
this.manager = opts.manager;
|
|
256
|
+
this.logger = opts.logger ?? createConsoleLogger('kernel');
|
|
257
|
+
this.executors = opts.executors ?? {};
|
|
258
|
+
this.nodeTimeoutMs = opts.nodeTimeoutMs ?? DEFAULT_NODE_TIMEOUT_MS;
|
|
259
|
+
}
|
|
260
|
+
/** Stop and await any live run holding this id, so a reuse cannot overlap it. */
|
|
261
|
+
async _retire(runId) {
|
|
262
|
+
const existing = this.live.get(runId);
|
|
263
|
+
if (!existing)
|
|
264
|
+
return;
|
|
265
|
+
this.cancel(runId);
|
|
266
|
+
await existing.done.catch(() => undefined);
|
|
267
|
+
if (this.live.get(runId) === existing)
|
|
268
|
+
this.live.delete(runId);
|
|
269
|
+
}
|
|
270
|
+
/** Register (or replace) the executor for one node kind. */
|
|
271
|
+
setExecutor(kind, executor) {
|
|
272
|
+
this.executors[kind] = executor;
|
|
273
|
+
}
|
|
274
|
+
/**
|
|
275
|
+
* Open a transaction over a run this kernel has just claimed.
|
|
276
|
+
*
|
|
277
|
+
* The abort signal is created here rather than in `_launch` because the
|
|
278
|
+
* transaction has to be able to stop the run the moment a write is refused,
|
|
279
|
+
* and that can happen before the handle exists.
|
|
280
|
+
*/
|
|
281
|
+
_open(guard, record) {
|
|
282
|
+
const signal = { aborted: false };
|
|
283
|
+
const txn = new RunTxn(guard, record, this.logger, (events) => {
|
|
284
|
+
for (const event of events) {
|
|
285
|
+
this.emit('kernel-event', { runId: guard.runId, event });
|
|
286
|
+
this.emit(guard.runId, event);
|
|
287
|
+
}
|
|
288
|
+
}, (outcome, reason) => {
|
|
289
|
+
signal.aborted = true;
|
|
290
|
+
// Being superseded means someone else already owns the run; handing it
|
|
291
|
+
// back would take it from them. Being blocked means we still hold a
|
|
292
|
+
// claim we can no longer use, and leaving that behind is what wedges a
|
|
293
|
+
// run permanently — a live local pid is never judged stale, so nobody
|
|
294
|
+
// else could ever take it.
|
|
295
|
+
if (outcome === 'blocked')
|
|
296
|
+
this._scheduleRelease(guard, reason);
|
|
297
|
+
});
|
|
298
|
+
return { txn, signal };
|
|
299
|
+
}
|
|
300
|
+
/**
|
|
301
|
+
* Hand a claim back, retrying while the lock is merely busy.
|
|
302
|
+
*
|
|
303
|
+
* Best-effort with a bound: contention here is measured in microseconds, so a
|
|
304
|
+
* few backed-off attempts cover everything short of a wedged filesystem — and
|
|
305
|
+
* if it is wedged, saying so beats retrying forever.
|
|
306
|
+
*/
|
|
307
|
+
_scheduleRelease(guard, reason, attempt = 0) {
|
|
308
|
+
const outcome = releaseLease(guard);
|
|
309
|
+
if (outcome !== 'blocked')
|
|
310
|
+
return;
|
|
311
|
+
if (attempt >= 6) {
|
|
312
|
+
this.logger.error?.(`[kernel] ${guard.runId}: could not hand the claim back after ${attempt} attempts` +
|
|
313
|
+
`${reason ? ` (${reason})` : ''} — resuming it elsewhere will have to wait for the lease to expire`);
|
|
314
|
+
return;
|
|
315
|
+
}
|
|
316
|
+
const timer = setTimeout(() => this._scheduleRelease(guard, reason, attempt + 1), 250 * 2 ** attempt);
|
|
317
|
+
if (typeof timer.unref === 'function')
|
|
318
|
+
timer.unref();
|
|
319
|
+
}
|
|
320
|
+
// ─── Lifecycle ────────────────────────────────────────────────────────────
|
|
321
|
+
async start(rawSpec, opts = {}) {
|
|
322
|
+
let contract = rawSpec.contract;
|
|
323
|
+
if (opts.contract !== undefined) {
|
|
324
|
+
contract = normalizeContract(opts.contract);
|
|
325
|
+
// A contract that survives normalisation with nothing in it used to leave
|
|
326
|
+
// the run with no contract at all, which then completed `unverified` — a
|
|
327
|
+
// caller who asked to be checked was told nothing had checked, and the
|
|
328
|
+
// reason was a typo they never saw.
|
|
329
|
+
if (!contract) {
|
|
330
|
+
throw new Error('The supplied acceptance contract has no recognised checks. Every check needs a `type` of ' +
|
|
331
|
+
'command | http | screenshot | diff_policy | file, and a `command` check needs `cmd`.');
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
const spec = prepareSpec({ ...rawSpec, contract });
|
|
335
|
+
validateSpec(spec);
|
|
336
|
+
const runId = opts.runId || `wf-${Date.now().toString(36)}-${crypto.randomBytes(3).toString('hex')}`;
|
|
337
|
+
// Validated here as well as in the store: failing before any directory is
|
|
338
|
+
// created keeps a rejected id from leaving a half-made run behind.
|
|
339
|
+
if (!isValidRunId(runId)) {
|
|
340
|
+
throw new Error(`Invalid run id ${JSON.stringify(runId)}: must be a single path segment ([A-Za-z0-9._-])`);
|
|
341
|
+
}
|
|
342
|
+
const cwd = opts.cwd || spec.cwd || process.cwd();
|
|
343
|
+
const now = new Date().toISOString();
|
|
344
|
+
const nodes = {};
|
|
345
|
+
for (const n of spec.nodes) {
|
|
346
|
+
nodes[n.id] = { id: n.id, kind: n.kind, state: 'pending', attempts: 0, visits: 0 };
|
|
347
|
+
}
|
|
348
|
+
const record = {
|
|
349
|
+
runId,
|
|
350
|
+
workflow: spec.name,
|
|
351
|
+
spec,
|
|
352
|
+
state: 'pending',
|
|
353
|
+
outcome: 'unverified',
|
|
354
|
+
cwd,
|
|
355
|
+
createdAt: now,
|
|
356
|
+
updatedAt: now,
|
|
357
|
+
nodes,
|
|
358
|
+
costUsd: 0,
|
|
359
|
+
};
|
|
360
|
+
if (opts.secrets)
|
|
361
|
+
this._secrets.set(runId, opts.secrets);
|
|
362
|
+
if (opts.tag)
|
|
363
|
+
this._tags.set(runId, opts.tag);
|
|
364
|
+
// A run id may be reused once its previous run is gone — a failed start
|
|
365
|
+
// frees it. But two live runs with the same id in one process would write
|
|
366
|
+
// over each other's checkpoints, so the old one is stopped and waited for
|
|
367
|
+
// first. This is the in-process half of what the lease does across
|
|
368
|
+
// processes.
|
|
369
|
+
await this._retire(runId);
|
|
370
|
+
// Creating and claiming are one step, and the directory creation is itself
|
|
371
|
+
// the claim: two processes cannot both come away believing they made this
|
|
372
|
+
// run. A fresh directory also means a fresh incarnation id, which is what
|
|
373
|
+
// stops a guard from the previous life of this run id from ever being valid
|
|
374
|
+
// again.
|
|
375
|
+
const guard = createAndAcquire(runId, spec, this.ownerId);
|
|
376
|
+
record.baseSha = await captureBaseline(cwd);
|
|
377
|
+
const { txn, signal } = this._open(guard, record);
|
|
378
|
+
if (!txn.apply(() => undefined, [{ ts: now, type: 'run_created', runId, workflow: spec.name }])) {
|
|
379
|
+
throw new Error(`Run '${runId}' could not be checkpointed: the claim was lost before it started`);
|
|
380
|
+
}
|
|
381
|
+
this._launch(txn, signal);
|
|
382
|
+
return txn.record;
|
|
383
|
+
}
|
|
384
|
+
/**
|
|
385
|
+
* Re-attach to a run whose process died. Nodes already marked succeeded are not
|
|
386
|
+
* re-run; the node that was in flight when the process ended is retried from
|
|
387
|
+
* the start, because a half-finished node left no result to trust.
|
|
388
|
+
*/
|
|
389
|
+
async resume(runId, opts = {}) {
|
|
390
|
+
if (opts.secrets)
|
|
391
|
+
this._secrets.set(runId, opts.secrets);
|
|
392
|
+
if (opts.tag)
|
|
393
|
+
this._tags.set(runId, opts.tag);
|
|
394
|
+
if (this.live.has(runId)) {
|
|
395
|
+
const existing = loadRun(runId);
|
|
396
|
+
if (!existing)
|
|
397
|
+
throw new Error(`Run '${runId}' not found`);
|
|
398
|
+
return existing;
|
|
399
|
+
}
|
|
400
|
+
const record = loadRun(runId);
|
|
401
|
+
if (!record)
|
|
402
|
+
throw new Error(`Run '${runId}' not found`);
|
|
403
|
+
// A finished run stays finished by default: silently re-running a completed
|
|
404
|
+
// workflow because someone polled `resume` would be a nasty surprise.
|
|
405
|
+
// `restart` is for callers whose "resume" genuinely means "bring it back up"
|
|
406
|
+
// — autoloop, whose runs terminate and are expected to be restartable.
|
|
407
|
+
//
|
|
408
|
+
// Checked BEFORE the lease is taken. Taking it first meant that merely
|
|
409
|
+
// polling `resume` on a completed run minted a lease nobody would ever
|
|
410
|
+
// release, blocking the next process from restarting it.
|
|
411
|
+
if (isTerminalRunState(record.state) && !opts.restart)
|
|
412
|
+
return record;
|
|
413
|
+
// Claim it: two processes resuming the same run would each execute its
|
|
414
|
+
// nodes, with every side effect happening twice.
|
|
415
|
+
const guard = acquireLease(runId, this.ownerId);
|
|
416
|
+
const { txn, signal } = this._open(guard, record);
|
|
417
|
+
const wasTerminal = isTerminalRunState(record.state);
|
|
418
|
+
if (!txn.apply((draft) => {
|
|
419
|
+
for (const n of Object.values(draft.nodes)) {
|
|
420
|
+
if (n.state === 'running' || (opts.restart && wasTerminal)) {
|
|
421
|
+
n.state = 'pending';
|
|
422
|
+
n.attempts = 0;
|
|
423
|
+
n.error = undefined;
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
if (opts.restart) {
|
|
427
|
+
draft.endedAt = undefined;
|
|
428
|
+
draft.outcome = 'unverified';
|
|
429
|
+
draft.outcomeReason = undefined;
|
|
430
|
+
draft.verdict = undefined;
|
|
431
|
+
}
|
|
432
|
+
draft.error = undefined;
|
|
433
|
+
draft.updatedAt = new Date().toISOString();
|
|
434
|
+
})) {
|
|
435
|
+
throw new Error(`Run '${runId}' could not be checkpointed: the claim was lost before it resumed`);
|
|
436
|
+
}
|
|
437
|
+
this._launch(txn, signal);
|
|
438
|
+
return txn.record;
|
|
439
|
+
}
|
|
440
|
+
cancel(runId) {
|
|
441
|
+
const handle = this.live.get(runId);
|
|
442
|
+
if (!handle)
|
|
443
|
+
return false;
|
|
444
|
+
handle.signal.aborted = true;
|
|
445
|
+
handle.gate?.resolve(false);
|
|
446
|
+
// Reach the engines too. Setting a flag only works for runners that check
|
|
447
|
+
// it; `Council` and `Fanout` have their own `abort()`, and without calling
|
|
448
|
+
// it a shutdown waits out the node timeout instead of stopping.
|
|
449
|
+
for (const [, live] of handle.handles) {
|
|
450
|
+
const abortable = live;
|
|
451
|
+
try {
|
|
452
|
+
abortable.abort?.();
|
|
453
|
+
}
|
|
454
|
+
catch {
|
|
455
|
+
// Best-effort: an engine that fails to abort must not block the others.
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
// Cancellation propagates to subflows. Without this a cancelled parent
|
|
459
|
+
// leaves its children running and spending, with nothing pointing at them.
|
|
460
|
+
for (const node of Object.values(handle.txn.record.nodes)) {
|
|
461
|
+
if (node.childRunId)
|
|
462
|
+
this.cancel(node.childRunId);
|
|
463
|
+
}
|
|
464
|
+
return true;
|
|
465
|
+
}
|
|
466
|
+
/** Queue text for the current (or next) agent node. */
|
|
467
|
+
steer(runId, text) {
|
|
468
|
+
const handle = this.live.get(runId);
|
|
469
|
+
if (!handle)
|
|
470
|
+
return false;
|
|
471
|
+
handle.steer.push(text);
|
|
472
|
+
handle.txn.emit({
|
|
473
|
+
ts: new Date().toISOString(),
|
|
474
|
+
type: 'steer',
|
|
475
|
+
node: handle.txn.record.currentNode ?? '',
|
|
476
|
+
text,
|
|
477
|
+
});
|
|
478
|
+
return true;
|
|
479
|
+
}
|
|
480
|
+
/** Answer a parked `human_gate`. */
|
|
481
|
+
approve(runId, approved) {
|
|
482
|
+
const handle = this.live.get(runId);
|
|
483
|
+
if (!handle?.gate)
|
|
484
|
+
return false;
|
|
485
|
+
handle.gate.resolve(approved);
|
|
486
|
+
return true;
|
|
487
|
+
}
|
|
488
|
+
get(runId) {
|
|
489
|
+
return loadRun(runId);
|
|
490
|
+
}
|
|
491
|
+
/**
|
|
492
|
+
* The live engine object a running node published, if the run is still going
|
|
493
|
+
* in this process. Undefined once the node finishes or the process restarts —
|
|
494
|
+
* which is the honest answer, not a gap.
|
|
495
|
+
*/
|
|
496
|
+
handle(runId, nodeId) {
|
|
497
|
+
return this.live.get(runId)?.handles.get(nodeId);
|
|
498
|
+
}
|
|
499
|
+
list(query = {}) {
|
|
500
|
+
return listRuns(query);
|
|
501
|
+
}
|
|
502
|
+
/**
|
|
503
|
+
* Remove a run.
|
|
504
|
+
*
|
|
505
|
+
* `expectTag` guards against deleting the wrong incarnation: a run id is
|
|
506
|
+
* reused when a failed start frees it, so a dying start's cleanup could
|
|
507
|
+
* otherwise delete the retry that had already taken the id. When the tag does
|
|
508
|
+
* not match, nothing is touched.
|
|
509
|
+
*
|
|
510
|
+
* Deleting takes the incarnation with it, which is what makes the id safe to
|
|
511
|
+
* reuse: any guard still held by an abandoned attempt of the deleted run names
|
|
512
|
+
* an incarnation that no longer exists, so it cannot write to whatever takes
|
|
513
|
+
* the id next.
|
|
514
|
+
*/
|
|
515
|
+
delete(runId, opts = {}) {
|
|
516
|
+
if (opts.expectTag !== undefined && this._tags.get(runId) !== opts.expectTag)
|
|
517
|
+
return false;
|
|
518
|
+
// Claim before deleting and never release first. Releasing here used to open
|
|
519
|
+
// a window in which another process could legally resume the run, only for
|
|
520
|
+
// this process to delete the directory out from under its new owner.
|
|
521
|
+
let claimed = false;
|
|
522
|
+
try {
|
|
523
|
+
acquireLease(runId, this.ownerId);
|
|
524
|
+
claimed = true;
|
|
525
|
+
}
|
|
526
|
+
catch {
|
|
527
|
+
claimed = false;
|
|
528
|
+
}
|
|
529
|
+
this.cancel(runId);
|
|
530
|
+
// Whatever happened to the directory, this process is done with the run, so
|
|
531
|
+
// its in-memory traces go — a refusal must not leave the caller's secrets
|
|
532
|
+
// sitting in a map for a run we are no longer tracking. (`acquireLease`
|
|
533
|
+
// throws for a run that no longer exists as well as for one someone else
|
|
534
|
+
// owns, and forgetting the credentials is right in both cases.)
|
|
535
|
+
this._secrets.delete(runId);
|
|
536
|
+
this._tags.delete(runId);
|
|
537
|
+
if (!claimed)
|
|
538
|
+
return false;
|
|
539
|
+
deleteRunDir(runId, this.logger);
|
|
540
|
+
return true;
|
|
541
|
+
}
|
|
542
|
+
/**
|
|
543
|
+
* Resolves when the run reaches a terminal state. A run that already finished
|
|
544
|
+
* (or belongs to another process) resolves from disk, so the caller does not
|
|
545
|
+
* have to race the completion.
|
|
546
|
+
*/
|
|
547
|
+
wait(runId) {
|
|
548
|
+
const handle = this.live.get(runId);
|
|
549
|
+
return handle ? handle.done : Promise.resolve(loadRun(runId));
|
|
550
|
+
}
|
|
551
|
+
async shutdown() {
|
|
552
|
+
for (const [runId] of this.live)
|
|
553
|
+
this.cancel(runId);
|
|
554
|
+
await Promise.allSettled([...this.live.values()].map((h) => h.done));
|
|
555
|
+
this.live.clear();
|
|
556
|
+
}
|
|
557
|
+
// ─── Execution ────────────────────────────────────────────────────────────
|
|
558
|
+
_launch(txn, signal) {
|
|
559
|
+
const runId = txn.guard.runId;
|
|
560
|
+
const handle = {
|
|
561
|
+
txn,
|
|
562
|
+
signal,
|
|
563
|
+
steer: [],
|
|
564
|
+
tag: this._tags.get(runId),
|
|
565
|
+
inflight: new Set(),
|
|
566
|
+
secrets: this._secrets.get(runId) ?? {},
|
|
567
|
+
handles: new Map(),
|
|
568
|
+
done: Promise.resolve(txn.record),
|
|
569
|
+
};
|
|
570
|
+
handle.heartbeat = setInterval(() => renewLease(txn.guard), LEASE_HEARTBEAT_MS);
|
|
571
|
+
if (typeof handle.heartbeat.unref === 'function')
|
|
572
|
+
handle.heartbeat.unref();
|
|
573
|
+
// Registered before the run starts: a workflow with no nodes finishes
|
|
574
|
+
// synchronously up to its first await, and the `finally` below would
|
|
575
|
+
// otherwise delete an entry that had not been added yet.
|
|
576
|
+
this.live.set(runId, handle);
|
|
577
|
+
handle.done = this._run(handle).finally(() => {
|
|
578
|
+
if (handle.heartbeat)
|
|
579
|
+
clearInterval(handle.heartbeat);
|
|
580
|
+
// A run that stopped because it could not write still holds its claim.
|
|
581
|
+
// `_scheduleRelease` was already started by the stop callback; this covers
|
|
582
|
+
// the case where the run ended for another reason while stalled.
|
|
583
|
+
if (txn.stalled)
|
|
584
|
+
this._scheduleRelease(txn.guard);
|
|
585
|
+
// Identity-checked: a run id can be reused, and deleting by key alone
|
|
586
|
+
// meant a finishing run evicted the handle of the run that had just
|
|
587
|
+
// replaced it.
|
|
588
|
+
if (this.live.get(runId) === handle)
|
|
589
|
+
this.live.delete(runId);
|
|
590
|
+
});
|
|
591
|
+
// The caller gets the record immediately; failures surface through the record.
|
|
592
|
+
handle.done.catch(() => undefined);
|
|
593
|
+
}
|
|
594
|
+
_setRunState(handle, state, error) {
|
|
595
|
+
const txn = handle.txn;
|
|
596
|
+
const ts = new Date().toISOString();
|
|
597
|
+
const terminal = isTerminalRunState(state);
|
|
598
|
+
const outcome = txn.record.outcome;
|
|
599
|
+
const ok = txn.apply((draft) => {
|
|
600
|
+
draft.state = state;
|
|
601
|
+
draft.updatedAt = ts;
|
|
602
|
+
if (error)
|
|
603
|
+
draft.error = error;
|
|
604
|
+
// Terminal states carry `endedAt`; this is the one place that stamps it.
|
|
605
|
+
if (terminal)
|
|
606
|
+
draft.endedAt = ts;
|
|
607
|
+
}, [{ ts, type: 'run_state', state, outcome, error }]);
|
|
608
|
+
if (!ok)
|
|
609
|
+
return false;
|
|
610
|
+
// Hand the run back once it is over, so another owner can pick it up
|
|
611
|
+
// without waiting out the lease. After the writes, never before.
|
|
612
|
+
if (terminal)
|
|
613
|
+
this._scheduleRelease(txn.guard);
|
|
614
|
+
return true;
|
|
615
|
+
}
|
|
616
|
+
_setNodeState(handle, nodeId, state, extra = {}) {
|
|
617
|
+
const ts = new Date().toISOString();
|
|
618
|
+
return handle.txn.apply((draft) => {
|
|
619
|
+
const node = draft.nodes[nodeId];
|
|
620
|
+
if (!node)
|
|
621
|
+
return;
|
|
622
|
+
node.state = state;
|
|
623
|
+
if (extra.attempt !== undefined)
|
|
624
|
+
node.attempts = extra.attempt;
|
|
625
|
+
if (extra.error)
|
|
626
|
+
node.error = extra.error;
|
|
627
|
+
if (state === 'running')
|
|
628
|
+
node.startedAt = ts;
|
|
629
|
+
if (state === 'succeeded' || state === 'failed' || state === 'skipped' || state === 'cancelled') {
|
|
630
|
+
node.endedAt = ts;
|
|
631
|
+
}
|
|
632
|
+
draft.currentNode = nodeId;
|
|
633
|
+
draft.updatedAt = ts;
|
|
634
|
+
}, [{ ts, type: 'node_state', node: nodeId, state, ...extra }]);
|
|
635
|
+
}
|
|
636
|
+
_nodeSpec(record, id) {
|
|
637
|
+
return record.spec.nodes.find((n) => n.id === id);
|
|
638
|
+
}
|
|
639
|
+
_nextInOrder(record, id) {
|
|
640
|
+
const idx = record.spec.nodes.findIndex((n) => n.id === id);
|
|
641
|
+
return idx >= 0 && idx + 1 < record.spec.nodes.length ? record.spec.nodes[idx + 1].id : undefined;
|
|
642
|
+
}
|
|
643
|
+
_firstPending(record) {
|
|
644
|
+
for (const n of record.spec.nodes) {
|
|
645
|
+
const rec = record.nodes[n.id];
|
|
646
|
+
if (!rec || rec.state === 'pending' || rec.state === 'awaiting_human')
|
|
647
|
+
return n.id;
|
|
648
|
+
}
|
|
649
|
+
return undefined;
|
|
650
|
+
}
|
|
651
|
+
async _executeNode(node, ctx, timeoutMs, attemptSignal, inflight) {
|
|
652
|
+
const executor = this.executors[node.kind];
|
|
653
|
+
if (!executor) {
|
|
654
|
+
return { ok: false, error: `no executor registered for node kind '${node.kind}'` };
|
|
655
|
+
}
|
|
656
|
+
let timer;
|
|
657
|
+
const timeout = new Promise((resolve) => {
|
|
658
|
+
timer = setTimeout(() => {
|
|
659
|
+
// Aborts this attempt only. A timeout is a node failure — it still gets
|
|
660
|
+
// its retries and still honours `onFailure` — whereas cancelling the run
|
|
661
|
+
// is a separate, user-initiated thing. Conflating the two made a hung
|
|
662
|
+
// node report the whole run as `cancelled`.
|
|
663
|
+
attemptSignal.aborted = true;
|
|
664
|
+
resolve({ ok: false, error: `node timed out after ${timeoutMs}ms` });
|
|
665
|
+
}, timeoutMs);
|
|
666
|
+
if (typeof timer.unref === 'function')
|
|
667
|
+
timer.unref();
|
|
668
|
+
});
|
|
669
|
+
// The executor promise is tracked, not just raced. A timeout abandons the
|
|
670
|
+
// wait, not the work — the executor keeps running and can still write — so
|
|
671
|
+
// the run has to know it is out there before it calls anything verified.
|
|
672
|
+
const running = Promise.resolve()
|
|
673
|
+
.then(() => executor(node, ctx))
|
|
674
|
+
.catch((err) => ({ ok: false, error: err.message }))
|
|
675
|
+
.finally(() => inflight.delete(running));
|
|
676
|
+
inflight.add(running);
|
|
677
|
+
try {
|
|
678
|
+
return await Promise.race([running, timeout]);
|
|
679
|
+
}
|
|
680
|
+
catch (err) {
|
|
681
|
+
return { ok: false, error: err.message };
|
|
682
|
+
}
|
|
683
|
+
finally {
|
|
684
|
+
if (timer)
|
|
685
|
+
clearTimeout(timer);
|
|
686
|
+
}
|
|
687
|
+
}
|
|
688
|
+
/**
|
|
689
|
+
* Wait for abandoned attempts to stop, so a terminal verdict describes a tree
|
|
690
|
+
* nobody is still writing to.
|
|
691
|
+
*
|
|
692
|
+
* A timed-out node is not killed — JS gives us no way to — so a run used to
|
|
693
|
+
* stamp `completed / verified`, then have the abandoned attempt write to the
|
|
694
|
+
* workspace afterwards. The evidence was accurate at the moment it was taken
|
|
695
|
+
* and wrong seconds later, with nothing recording that.
|
|
696
|
+
*/
|
|
697
|
+
async _awaitQuiescence(handle, graceMs) {
|
|
698
|
+
if (handle.inflight.size === 0)
|
|
699
|
+
return true;
|
|
700
|
+
let timer;
|
|
701
|
+
const grace = new Promise((resolve) => {
|
|
702
|
+
timer = setTimeout(() => resolve('timeout'), graceMs);
|
|
703
|
+
if (typeof timer.unref === 'function')
|
|
704
|
+
timer.unref();
|
|
705
|
+
});
|
|
706
|
+
const settled = Promise.allSettled([...handle.inflight]).then(() => 'settled');
|
|
707
|
+
try {
|
|
708
|
+
return (await Promise.race([settled, grace])) === 'settled';
|
|
709
|
+
}
|
|
710
|
+
finally {
|
|
711
|
+
if (timer)
|
|
712
|
+
clearTimeout(timer);
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
async _run(handle) {
|
|
716
|
+
const txn = handle.txn;
|
|
717
|
+
const maxVisits = txn.record.spec.maxNodeVisits ?? DEFAULT_MAX_NODE_VISITS;
|
|
718
|
+
this._setRunState(handle, 'running');
|
|
719
|
+
let cursor = this._firstPending(txn.record);
|
|
720
|
+
while (cursor) {
|
|
721
|
+
// Losing the run outranks everything else: this owner may not write, so
|
|
722
|
+
// there is nothing left for it to do but stop.
|
|
723
|
+
if (txn.finished)
|
|
724
|
+
return txn.record;
|
|
725
|
+
if (handle.signal.aborted) {
|
|
726
|
+
this._setRunState(handle, 'cancelled');
|
|
727
|
+
return txn.record;
|
|
728
|
+
}
|
|
729
|
+
const nodeId = cursor;
|
|
730
|
+
const spec = this._nodeSpec(txn.record, nodeId);
|
|
731
|
+
if (!spec || !txn.record.nodes[nodeId]) {
|
|
732
|
+
this._setRunState(handle, 'failed', `unknown node '${nodeId}'`);
|
|
733
|
+
return txn.record;
|
|
734
|
+
}
|
|
735
|
+
// The visit counter is part of the loop bound, so it is committed like any
|
|
736
|
+
// other state rather than incremented on a record nobody agreed to.
|
|
737
|
+
if (!txn.apply((draft) => {
|
|
738
|
+
const node = draft.nodes[nodeId];
|
|
739
|
+
if (node)
|
|
740
|
+
node.visits = (node.visits ?? 0) + 1;
|
|
741
|
+
draft.updatedAt = new Date().toISOString();
|
|
742
|
+
})) {
|
|
743
|
+
return txn.record;
|
|
744
|
+
}
|
|
745
|
+
if ((txn.record.nodes[nodeId]?.visits ?? 0) > maxVisits) {
|
|
746
|
+
this._setNodeState(handle, nodeId, 'failed', { error: `visit limit ${maxVisits} exceeded` });
|
|
747
|
+
this._setRunState(handle, 'failed', `node '${nodeId}' exceeded the ${maxVisits}-visit loop bound`);
|
|
748
|
+
return txn.record;
|
|
749
|
+
}
|
|
750
|
+
if (spec.kind === 'verifier')
|
|
751
|
+
this._setRunState(handle, 'verifying');
|
|
752
|
+
const result = await this._runWithRetry(handle, spec);
|
|
753
|
+
if (txn.finished)
|
|
754
|
+
return txn.record;
|
|
755
|
+
// Cancel wins regardless of what the node returned. A runner that never
|
|
756
|
+
// looked at the signal and reported success anyway used to carry the run
|
|
757
|
+
// all the way to `completed` — so "I cancelled it" and "it completed"
|
|
758
|
+
// could both be true, which makes cancellation meaningless.
|
|
759
|
+
if (handle.signal.aborted) {
|
|
760
|
+
this._absorb(handle, nodeId, result);
|
|
761
|
+
this._setNodeState(handle, nodeId, 'cancelled');
|
|
762
|
+
this._setRunState(handle, 'cancelled');
|
|
763
|
+
return txn.record;
|
|
764
|
+
}
|
|
765
|
+
if (result.awaitHuman) {
|
|
766
|
+
const approved = await this._park(handle, nodeId);
|
|
767
|
+
if (txn.finished)
|
|
768
|
+
return txn.record;
|
|
769
|
+
if (!approved) {
|
|
770
|
+
this._setNodeState(handle, nodeId, 'failed', { error: 'rejected at human gate' });
|
|
771
|
+
this._setRunState(handle, handle.signal.aborted ? 'cancelled' : 'failed', 'rejected at human gate');
|
|
772
|
+
return txn.record;
|
|
773
|
+
}
|
|
774
|
+
this._setNodeState(handle, nodeId, 'succeeded');
|
|
775
|
+
cursor = spec.next ?? this._nextInOrder(txn.record, nodeId);
|
|
776
|
+
continue;
|
|
777
|
+
}
|
|
778
|
+
this._absorb(handle, nodeId, result);
|
|
779
|
+
if (txn.finished)
|
|
780
|
+
return txn.record;
|
|
781
|
+
if (!result.ok) {
|
|
782
|
+
this._setNodeState(handle, nodeId, 'failed', { error: result.error });
|
|
783
|
+
if ((spec.onFailure ?? 'fail') === 'fail') {
|
|
784
|
+
await this._finish(handle, `node '${nodeId}' failed: ${result.error ?? 'unknown'}`);
|
|
785
|
+
return txn.record;
|
|
786
|
+
}
|
|
787
|
+
// `continue` — record it and move on.
|
|
788
|
+
}
|
|
789
|
+
else {
|
|
790
|
+
this._setNodeState(handle, nodeId, 'succeeded');
|
|
791
|
+
}
|
|
792
|
+
if (spec.kind === 'router') {
|
|
793
|
+
cursor = result.goto ?? spec.default ?? this._nextInOrder(txn.record, nodeId);
|
|
794
|
+
continue;
|
|
795
|
+
}
|
|
796
|
+
cursor = spec.next ?? this._nextInOrder(txn.record, nodeId);
|
|
797
|
+
}
|
|
798
|
+
await this._finish(handle);
|
|
799
|
+
return txn.record;
|
|
800
|
+
}
|
|
801
|
+
async _runWithRetry(handle, spec) {
|
|
802
|
+
const txn = handle.txn;
|
|
803
|
+
const maxAttempts = (spec.retry?.max ?? 0) + 1;
|
|
804
|
+
const timeoutMs = spec.timeoutMs ?? this.nodeTimeoutMs;
|
|
805
|
+
let last = { ok: false, error: 'not attempted' };
|
|
806
|
+
for (let attempt = 1; attempt <= maxAttempts; attempt++) {
|
|
807
|
+
if (handle.signal.aborted)
|
|
808
|
+
return { ok: false, error: 'cancelled' };
|
|
809
|
+
// The attempt counter goes in with the state change, so a refused write
|
|
810
|
+
// means the attempt never officially started.
|
|
811
|
+
if (!this._setNodeState(handle, spec.id, 'running', { attempt })) {
|
|
812
|
+
return { ok: false, error: 'the run was taken over by another owner' };
|
|
813
|
+
}
|
|
814
|
+
// A node sees one signal that is the union of "this attempt gave up" and
|
|
815
|
+
// "the whole run was cancelled"; the kernel keeps them apart.
|
|
816
|
+
const attemptSignal = { aborted: false };
|
|
817
|
+
const ctx = {
|
|
818
|
+
runId: txn.guard.runId,
|
|
819
|
+
// A live view of the committed record, so a node holding `ctx` across an
|
|
820
|
+
// await never reads a state the disk refused.
|
|
821
|
+
get record() {
|
|
822
|
+
return txn.record;
|
|
823
|
+
},
|
|
824
|
+
cwd: txn.record.cwd,
|
|
825
|
+
attempt,
|
|
826
|
+
manager: this.manager,
|
|
827
|
+
logger: this.logger,
|
|
828
|
+
signal: {
|
|
829
|
+
get aborted() {
|
|
830
|
+
return handle.signal.aborted || attemptSignal.aborted;
|
|
831
|
+
},
|
|
832
|
+
set aborted(v) {
|
|
833
|
+
attemptSignal.aborted = v;
|
|
834
|
+
},
|
|
835
|
+
},
|
|
836
|
+
takeSteer: () => handle.steer.splice(0, handle.steer.length),
|
|
837
|
+
// Fenced like everything else. It used to be the one context method that
|
|
838
|
+
// wrote unguarded, so a superseded owner's node could still append to the
|
|
839
|
+
// log its replacement was reading.
|
|
840
|
+
emit: (event) => {
|
|
841
|
+
txn.emit(event);
|
|
842
|
+
},
|
|
843
|
+
runContract: txn.record.spec.contract,
|
|
844
|
+
setHandle: (h) => handle.handles.set(spec.id, h),
|
|
845
|
+
publish: (data) => {
|
|
846
|
+
txn.apply((draft) => {
|
|
847
|
+
const node = draft.nodes[spec.id];
|
|
848
|
+
if (node)
|
|
849
|
+
node.data = data;
|
|
850
|
+
draft.updatedAt = new Date().toISOString();
|
|
851
|
+
});
|
|
852
|
+
},
|
|
853
|
+
secrets: handle.secrets,
|
|
854
|
+
tag: handle.tag,
|
|
855
|
+
setChild: (childRunId) => {
|
|
856
|
+
txn.apply((draft) => {
|
|
857
|
+
const node = draft.nodes[spec.id];
|
|
858
|
+
if (node)
|
|
859
|
+
node.childRunId = childRunId;
|
|
860
|
+
draft.updatedAt = new Date().toISOString();
|
|
861
|
+
});
|
|
862
|
+
},
|
|
863
|
+
};
|
|
864
|
+
last = await this._executeNode(spec, ctx, timeoutMs, attemptSignal, handle.inflight);
|
|
865
|
+
if (last.ok || handle.signal.aborted)
|
|
866
|
+
return last;
|
|
867
|
+
if (attempt < maxAttempts) {
|
|
868
|
+
const backoff = (spec.retry?.backoffMs ?? 1000) * attempt;
|
|
869
|
+
this.logger.warn?.(`[kernel] ${txn.guard.runId}/${spec.id} attempt ${attempt}/${maxAttempts} failed: ${last.error} — retrying in ${backoff}ms`);
|
|
870
|
+
await sleep(backoff);
|
|
871
|
+
}
|
|
872
|
+
}
|
|
873
|
+
return last;
|
|
874
|
+
}
|
|
875
|
+
async _park(handle, nodeId) {
|
|
876
|
+
// If either write is refused the run is no longer ours, and parking on a
|
|
877
|
+
// gate we can never record the answer to would hang the executor forever.
|
|
878
|
+
if (!this._setNodeState(handle, nodeId, 'awaiting_human'))
|
|
879
|
+
return false;
|
|
880
|
+
if (!this._setRunState(handle, 'awaiting_human'))
|
|
881
|
+
return false;
|
|
882
|
+
const approved = await new Promise((resolve) => {
|
|
883
|
+
handle.gate = { resolve };
|
|
884
|
+
});
|
|
885
|
+
handle.gate = undefined;
|
|
886
|
+
if (!handle.signal.aborted)
|
|
887
|
+
this._setRunState(handle, 'running');
|
|
888
|
+
return approved;
|
|
889
|
+
}
|
|
890
|
+
/** Node kinds that can change the workspace. Routers and gates cannot. */
|
|
891
|
+
static SIDE_EFFECT_KINDS = ['agent', 'fanout', 'council', 'subflow'];
|
|
892
|
+
/**
|
|
893
|
+
* Fold a node's result into the run — output, artifacts, cost, votes, verdict
|
|
894
|
+
* — as one committed change.
|
|
895
|
+
*
|
|
896
|
+
* Every one of these used to be assigned straight onto the live record, with
|
|
897
|
+
* only the events fenced. A superseded owner therefore returned a record
|
|
898
|
+
* carrying an output, a cost and a passing verdict that the disk had refused.
|
|
899
|
+
*/
|
|
900
|
+
_absorb(handle, nodeId, result) {
|
|
901
|
+
const txn = handle.txn;
|
|
902
|
+
const ts = new Date().toISOString();
|
|
903
|
+
const kind = txn.record.nodes[nodeId]?.kind;
|
|
904
|
+
const events = [];
|
|
905
|
+
const artifacts = [];
|
|
906
|
+
// The record keeps a preview so checkpoints stay small; the full text goes
|
|
907
|
+
// to the node's artifact directory, because for some nodes the text *is*
|
|
908
|
+
// the deliverable and a silent 4 kB cut would lose it. The artifact is
|
|
909
|
+
// written inside the same commit as the record that references it, so a
|
|
910
|
+
// refused write leaves neither behind.
|
|
911
|
+
let preview;
|
|
912
|
+
let outputArtifact;
|
|
913
|
+
if (result.output !== undefined) {
|
|
914
|
+
if (result.output.length > OUTPUT_PREVIEW_CHARS) {
|
|
915
|
+
outputArtifact = nodeArtifactPath(txn.guard.runId, nodeId, 'output.txt');
|
|
916
|
+
artifacts.push({ nodeId, name: 'output.txt', body: result.output });
|
|
917
|
+
preview =
|
|
918
|
+
result.output.slice(0, OUTPUT_PREVIEW_CHARS) + `\n…[truncated — full text in nodes/${nodeId}/output.txt]`;
|
|
919
|
+
}
|
|
920
|
+
else {
|
|
921
|
+
preview = result.output;
|
|
922
|
+
}
|
|
923
|
+
events.push({ ts, type: 'node_output', node: nodeId, text: preview });
|
|
924
|
+
}
|
|
925
|
+
if (result.evidenceId) {
|
|
926
|
+
events.push({
|
|
927
|
+
ts,
|
|
928
|
+
type: 'evidence',
|
|
929
|
+
node: nodeId,
|
|
930
|
+
evidenceId: result.evidenceId,
|
|
931
|
+
passed: Boolean(result.passed),
|
|
932
|
+
});
|
|
933
|
+
}
|
|
934
|
+
return txn.apply((draft) => {
|
|
935
|
+
const node = draft.nodes[nodeId];
|
|
936
|
+
if (!node)
|
|
937
|
+
return;
|
|
938
|
+
if (kind && RunKernel.SIDE_EFFECT_KINDS.includes(kind)) {
|
|
939
|
+
draft.sideEffectSeq = (draft.sideEffectSeq ?? 0) + 1;
|
|
940
|
+
}
|
|
941
|
+
if (preview !== undefined)
|
|
942
|
+
node.output = preview;
|
|
943
|
+
if (outputArtifact)
|
|
944
|
+
node.artifacts = [...new Set([...(node.artifacts ?? []), outputArtifact])];
|
|
945
|
+
if (result.artifacts?.length)
|
|
946
|
+
node.artifacts = result.artifacts;
|
|
947
|
+
if (result.data !== undefined)
|
|
948
|
+
node.data = result.data;
|
|
949
|
+
if (result.childRunId)
|
|
950
|
+
node.childRunId = result.childRunId;
|
|
951
|
+
if (typeof result.costUsd === 'number')
|
|
952
|
+
draft.costUsd = (draft.costUsd ?? 0) + result.costUsd;
|
|
953
|
+
if (result.consensusVotes?.length) {
|
|
954
|
+
draft.consensusVotes = [...(draft.consensusVotes ?? []), ...result.consensusVotes];
|
|
955
|
+
}
|
|
956
|
+
if (result.evidenceId) {
|
|
957
|
+
node.evidenceId = result.evidenceId;
|
|
958
|
+
draft.evidenceId = result.evidenceId;
|
|
959
|
+
draft.outcome = result.passed ? 'verified' : 'refuted';
|
|
960
|
+
draft.outcomeReason = undefined;
|
|
961
|
+
draft.verdict = {
|
|
962
|
+
node: nodeId,
|
|
963
|
+
evidenceId: result.evidenceId,
|
|
964
|
+
treeFingerprint: result.treeFingerprint,
|
|
965
|
+
sideEffectSeq: draft.sideEffectSeq ?? 0,
|
|
966
|
+
};
|
|
967
|
+
}
|
|
968
|
+
draft.updatedAt = ts;
|
|
969
|
+
}, events, artifacts);
|
|
970
|
+
}
|
|
971
|
+
/**
|
|
972
|
+
* Decide the terminal state. `completed` requires that nothing refuted the run;
|
|
973
|
+
* a run with no contract completes as `unverified`, which says we did not check
|
|
974
|
+
* rather than claiming success.
|
|
975
|
+
*/
|
|
976
|
+
async _finish(handle, error) {
|
|
977
|
+
const txn = handle.txn;
|
|
978
|
+
let outcome = txn.record.outcome;
|
|
979
|
+
let downgrade;
|
|
980
|
+
// Nothing may still be writing when a verdict is stamped. An abandoned
|
|
981
|
+
// attempt keeps running after its timeout, so wait briefly for it — and if
|
|
982
|
+
// it will not stop, say so instead of vouching for a tree it may yet change.
|
|
983
|
+
//
|
|
984
|
+
// Only when there is a verdict to protect: a run that was never verified has
|
|
985
|
+
// nothing to lose by ending promptly, and blocking it would make one hung
|
|
986
|
+
// node delay every run that contained it.
|
|
987
|
+
if (outcome === 'verified' && !(await this._awaitQuiescence(handle, QUIESCE_GRACE_MS))) {
|
|
988
|
+
outcome = 'unverified';
|
|
989
|
+
downgrade =
|
|
990
|
+
`evidence ${txn.record.verdict?.evidenceId ?? '(none)'} passed, but ${handle.inflight.size} abandoned ` +
|
|
991
|
+
`attempt(s) were still running when the run ended — they can still change the tree, so the verdict ` +
|
|
992
|
+
`cannot stand`;
|
|
993
|
+
}
|
|
994
|
+
if (outcome === 'verified') {
|
|
995
|
+
const stale = await this._verdictWentStale(txn.record);
|
|
996
|
+
if (stale) {
|
|
997
|
+
outcome = 'unverified';
|
|
998
|
+
downgrade = stale;
|
|
999
|
+
}
|
|
1000
|
+
}
|
|
1001
|
+
if (downgrade) {
|
|
1002
|
+
const reason = downgrade;
|
|
1003
|
+
txn.apply((draft) => {
|
|
1004
|
+
draft.outcome = outcome;
|
|
1005
|
+
draft.outcomeReason = reason;
|
|
1006
|
+
}, [{ ts: new Date().toISOString(), type: 'log', level: 'warn', message: `[verify] ${reason}` }]);
|
|
1007
|
+
if (txn.finished)
|
|
1008
|
+
return;
|
|
1009
|
+
}
|
|
1010
|
+
if (error || outcome === 'refuted') {
|
|
1011
|
+
this._setRunState(handle, 'failed', error ?? 'acceptance contract was not satisfied');
|
|
1012
|
+
return;
|
|
1013
|
+
}
|
|
1014
|
+
this._setRunState(handle, 'completed');
|
|
1015
|
+
}
|
|
1016
|
+
/**
|
|
1017
|
+
* Why a passing verdict no longer stands, or undefined if it still does.
|
|
1018
|
+
*
|
|
1019
|
+
* `prepareSpec` cannot enforce this structurally: a router can send control
|
|
1020
|
+
* anywhere, so which node runs last is not a property of the spec. And a
|
|
1021
|
+
* "nothing after the verifier" rule would be the wrong rule anyway — what
|
|
1022
|
+
* matters is not that a node ran, but that the tree moved. So this measures.
|
|
1023
|
+
*
|
|
1024
|
+
* The caller drops the outcome to `unverified`, not `refuted`: no check
|
|
1025
|
+
* failed. We simply no longer know, and saying so is the whole point of having
|
|
1026
|
+
* three outcomes.
|
|
1027
|
+
*/
|
|
1028
|
+
async _verdictWentStale(record) {
|
|
1029
|
+
if (record.outcome !== 'verified' || !record.verdict)
|
|
1030
|
+
return undefined;
|
|
1031
|
+
// Nothing that could touch the workspace ran after the checks, so the
|
|
1032
|
+
// verdict still describes the tree. This is the ordinary case, and it must
|
|
1033
|
+
// not depend on git: a contract that passed in a plain directory passed.
|
|
1034
|
+
if ((record.sideEffectSeq ?? 0) === record.verdict.sideEffectSeq)
|
|
1035
|
+
return undefined;
|
|
1036
|
+
const before = record.verdict.treeFingerprint;
|
|
1037
|
+
const after = await treeFingerprint(record.cwd);
|
|
1038
|
+
if (before !== undefined && after !== undefined && before === after)
|
|
1039
|
+
return undefined;
|
|
1040
|
+
return before === undefined || after === undefined
|
|
1041
|
+
? `evidence ${record.verdict.evidenceId} passed, but nodes ran afterwards and ${record.cwd} is not a git repository, so we cannot tell whether it still describes the tree`
|
|
1042
|
+
: `evidence ${record.verdict.evidenceId} passed, but the working tree changed afterwards — the verdict describes an earlier state`;
|
|
1043
|
+
}
|
|
1044
|
+
}
|
|
1045
|
+
/** Run directory for a given run — re-exported so callers need not import the store. */
|
|
1046
|
+
export { runDir, summarize };
|
|
1047
|
+
//# sourceMappingURL=engine.js.map
|