@enderfga/claw-orchestrator 5.1.0 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/README.md +26 -26
  2. package/dist/bin/cli.js +107 -1
  3. package/dist/bin/cli.js.map +1 -1
  4. package/dist/src/acp-server.d.ts +5 -5
  5. package/dist/src/acp-server.js +3 -3
  6. package/dist/src/acp-server.js.map +1 -1
  7. package/dist/src/autoloop/dispatcher.d.ts +22 -0
  8. package/dist/src/autoloop/dispatcher.js +71 -13
  9. package/dist/src/autoloop/dispatcher.js.map +1 -1
  10. package/dist/src/autoloop/messages.d.ts +10 -0
  11. package/dist/src/autoloop/messages.js.map +1 -1
  12. package/dist/src/autoloop/runner.js +6 -0
  13. package/dist/src/autoloop/runner.js.map +1 -1
  14. package/dist/src/constants.d.ts +0 -6
  15. package/dist/src/constants.js +0 -6
  16. package/dist/src/constants.js.map +1 -1
  17. package/dist/src/council.d.ts +15 -0
  18. package/dist/src/council.js +48 -35
  19. package/dist/src/council.js.map +1 -1
  20. package/dist/src/dashboard/index.html +191 -6
  21. package/dist/src/embedded-server.js +132 -9
  22. package/dist/src/embedded-server.js.map +1 -1
  23. package/dist/src/fanout.d.ts +30 -1
  24. package/dist/src/fanout.js +32 -3
  25. package/dist/src/fanout.js.map +1 -1
  26. package/dist/src/index.js +359 -4
  27. package/dist/src/index.js.map +1 -1
  28. package/dist/src/kernel/agent-step.d.ts +59 -0
  29. package/dist/src/kernel/agent-step.js +100 -0
  30. package/dist/src/kernel/agent-step.js.map +1 -0
  31. package/dist/src/kernel/conditions.d.ts +11 -0
  32. package/dist/src/kernel/conditions.js +24 -0
  33. package/dist/src/kernel/conditions.js.map +1 -0
  34. package/dist/src/kernel/engine.d.ts +319 -0
  35. package/dist/src/kernel/engine.js +1047 -0
  36. package/dist/src/kernel/engine.js.map +1 -0
  37. package/dist/src/kernel/exec.d.ts +43 -0
  38. package/dist/src/kernel/exec.js +112 -0
  39. package/dist/src/kernel/exec.js.map +1 -0
  40. package/dist/src/kernel/file-lock.d.ts +50 -0
  41. package/dist/src/kernel/file-lock.js +135 -0
  42. package/dist/src/kernel/file-lock.js.map +1 -0
  43. package/dist/src/kernel/nodes/agent.d.ts +4 -0
  44. package/dist/src/kernel/nodes/agent.js +35 -0
  45. package/dist/src/kernel/nodes/agent.js.map +1 -0
  46. package/dist/src/kernel/nodes/autoloop.d.ts +78 -0
  47. package/dist/src/kernel/nodes/autoloop.js +75 -0
  48. package/dist/src/kernel/nodes/autoloop.js.map +1 -0
  49. package/dist/src/kernel/nodes/council.d.ts +12 -0
  50. package/dist/src/kernel/nodes/council.js +88 -0
  51. package/dist/src/kernel/nodes/council.js.map +1 -0
  52. package/dist/src/kernel/nodes/fanout.d.ts +11 -0
  53. package/dist/src/kernel/nodes/fanout.js +63 -0
  54. package/dist/src/kernel/nodes/fanout.js.map +1 -0
  55. package/dist/src/kernel/nodes/human-gate.d.ts +4 -0
  56. package/dist/src/kernel/nodes/human-gate.js +7 -0
  57. package/dist/src/kernel/nodes/human-gate.js.map +1 -0
  58. package/dist/src/kernel/nodes/index.d.ts +12 -0
  59. package/dist/src/kernel/nodes/index.js +21 -0
  60. package/dist/src/kernel/nodes/index.js.map +1 -0
  61. package/dist/src/kernel/nodes/router.d.ts +4 -0
  62. package/dist/src/kernel/nodes/router.js +12 -0
  63. package/dist/src/kernel/nodes/router.js.map +1 -0
  64. package/dist/src/kernel/nodes/subflow.d.ts +13 -0
  65. package/dist/src/kernel/nodes/subflow.js +38 -0
  66. package/dist/src/kernel/nodes/subflow.js.map +1 -0
  67. package/dist/src/kernel/nodes/ultraapp.d.ts +60 -0
  68. package/dist/src/kernel/nodes/ultraapp.js +62 -0
  69. package/dist/src/kernel/nodes/ultraapp.js.map +1 -0
  70. package/dist/src/kernel/nodes/verifier.d.ts +14 -0
  71. package/dist/src/kernel/nodes/verifier.js +84 -0
  72. package/dist/src/kernel/nodes/verifier.js.map +1 -0
  73. package/dist/src/kernel/projections.d.ts +42 -0
  74. package/dist/src/kernel/projections.js +133 -0
  75. package/dist/src/kernel/projections.js.map +1 -0
  76. package/dist/src/kernel/repo.d.ts +13 -0
  77. package/dist/src/kernel/repo.js +64 -0
  78. package/dist/src/kernel/repo.js.map +1 -0
  79. package/dist/src/kernel/secrets.d.ts +25 -0
  80. package/dist/src/kernel/secrets.js +48 -0
  81. package/dist/src/kernel/secrets.js.map +1 -0
  82. package/dist/src/kernel/store.d.ts +225 -0
  83. package/dist/src/kernel/store.js +838 -0
  84. package/dist/src/kernel/store.js.map +1 -0
  85. package/dist/src/kernel/templates/index.d.ts +140 -0
  86. package/dist/src/kernel/templates/index.js +266 -0
  87. package/dist/src/kernel/templates/index.js.map +1 -0
  88. package/dist/src/kernel/types.d.ts +326 -0
  89. package/dist/src/kernel/types.js +19 -0
  90. package/dist/src/kernel/types.js.map +1 -0
  91. package/dist/src/models.d.ts +7 -0
  92. package/dist/src/models.js +43 -14
  93. package/dist/src/models.js.map +1 -1
  94. package/dist/src/persistent-custom-session.js +8 -3
  95. package/dist/src/persistent-custom-session.js.map +1 -1
  96. package/dist/src/run-ledger.d.ts +57 -3
  97. package/dist/src/run-ledger.js +45 -2
  98. package/dist/src/run-ledger.js.map +1 -1
  99. package/dist/src/session-manager.d.ts +176 -129
  100. package/dist/src/session-manager.js +652 -603
  101. package/dist/src/session-manager.js.map +1 -1
  102. package/dist/src/types.d.ts +33 -3
  103. package/dist/src/ultraapp/build.d.ts +117 -3
  104. package/dist/src/ultraapp/build.js +319 -3
  105. package/dist/src/ultraapp/build.js.map +1 -1
  106. package/dist/src/ultraapp/contract.d.ts +52 -0
  107. package/dist/src/ultraapp/contract.js +83 -0
  108. package/dist/src/ultraapp/contract.js.map +1 -0
  109. package/dist/src/ultraapp/conventions.js +9 -2
  110. package/dist/src/ultraapp/conventions.js.map +1 -1
  111. package/dist/src/ultraapp/fix-on-failure.d.ts +21 -2
  112. package/dist/src/ultraapp/fix-on-failure.js +46 -62
  113. package/dist/src/ultraapp/fix-on-failure.js.map +1 -1
  114. package/dist/src/ultraapp/manager.d.ts +107 -2
  115. package/dist/src/ultraapp/manager.js +305 -86
  116. package/dist/src/ultraapp/manager.js.map +1 -1
  117. package/dist/src/verify/baseline.d.ts +73 -0
  118. package/dist/src/verify/baseline.js +186 -0
  119. package/dist/src/verify/baseline.js.map +1 -0
  120. package/dist/src/verify/contract.d.ts +116 -0
  121. package/dist/src/verify/contract.js +142 -0
  122. package/dist/src/verify/contract.js.map +1 -0
  123. package/dist/src/verify/evidence.d.ts +61 -0
  124. package/dist/src/verify/evidence.js +133 -0
  125. package/dist/src/verify/evidence.js.map +1 -0
  126. package/dist/src/verify/runner.d.ts +63 -0
  127. package/dist/src/verify/runner.js +317 -0
  128. package/dist/src/verify/runner.js.map +1 -0
  129. package/openclaw.plugin.json +38 -1
  130. package/package.json +2 -2
  131. package/skills/SKILL.md +120 -79
  132. package/skills/references/acp.md +17 -17
  133. package/skills/references/autoloop.md +139 -65
  134. package/skills/references/claude-cli-tracking.md +4 -4
  135. package/skills/references/cli.md +101 -59
  136. package/skills/references/council.md +109 -37
  137. package/skills/references/dashboard.md +34 -6
  138. package/skills/references/getting-started.md +13 -13
  139. package/skills/references/inbox.md +4 -4
  140. package/skills/references/mcp.md +39 -34
  141. package/skills/references/multi-engine.md +51 -47
  142. package/skills/references/observability.md +115 -28
  143. package/skills/references/openai-compat.md +39 -39
  144. package/skills/references/sessions.md +43 -25
  145. package/skills/references/tools.md +402 -309
  146. package/skills/references/ultra.md +45 -45
  147. package/skills/references/ultraapp.md +126 -50
  148. package/skills/references/verification.md +187 -0
  149. package/skills/references/workflow.md +362 -0
  150. package/dist/src/ultraapp/fix-on-failure-session.d.ts +0 -23
  151. package/dist/src/ultraapp/fix-on-failure-session.js +0 -51
  152. package/dist/src/ultraapp/fix-on-failure-session.js.map +0 -1
@@ -0,0 +1,1047 @@
1
+ /**
2
+ * The run kernel.
3
+ *
4
+ * A durable executor for workflow runs. It owns what the existing state machines
5
+ * each owned separately and differently: what is running, what to do when a step
6
+ * fails, when to give up, how to stop, and — the part none of them had — how to
7
+ * come back after the process dies.
8
+ *
9
+ * Every mode goes through it. `council_start`, `fanout_start`, `ultraplan_start`,
10
+ * `ultrareview_start` and `autoloop_start` each create a run here; the engines
11
+ * they wrap (`Council`, `Fanout`, the autoloop dispatcher) still do the work,
12
+ * but they no longer own a lifecycle. What that replaced: five result maps, four
13
+ * 30-minute eviction timers, a 5-second poller, two `Set`s fencing a start
14
+ * against a delete, and two incompatible ways of listing past runs across
15
+ * processes.
16
+ *
17
+ * Durability contract, stated precisely: every state transition is checkpointed
18
+ * (atomic `run.json`) and appended to `events.jsonl` before the next step
19
+ * begins, and `resume` restarts at the last node boundary. That makes node
20
+ * execution **at-least-once**, not exactly-once — there is no idempotency key,
21
+ * no attempt lease, and no side-effect commit marker, so a node that died after
22
+ * writing files but before its checkpoint runs again from the top. Workflows
23
+ * whose nodes are not safe to repeat need to make them safe.
24
+ *
25
+ * Ownership contract, equally precisely: this class never writes to a run
26
+ * directly. It holds a `RunGuard` from the store and every change — checkpoint,
27
+ * event, node artifact, terminal verdict — goes through `RunTxn`, which applies
28
+ * the change to a *copy*, commits it under the guard, and adopts the copy only
29
+ * if the disk accepted it. There is no code path from here to `run.json` that
30
+ * skips that, because the raw writers are no longer exported. An owner that has
31
+ * been superseded therefore cannot change the run in memory either: its record
32
+ * stops advancing at the last write the disk agreed to.
33
+ *
34
+ * Control flow is linear with an explicit `router` node for branches and loops.
35
+ * Parallelism is the `fanout` node rather than a general parallel/join construct:
36
+ * fan-out is the shape every existing mode actually needed, and a join barrier
37
+ * would add failure modes (partial joins, orphaned branches) that nothing here
38
+ * would exercise.
39
+ *
40
+ * What a node timeout can and cannot do: it stops the kernel waiting and marks
41
+ * the node failed, but an in-flight agent turn is owned by the session layer and
42
+ * finishes on its own schedule. The kernel does not pretend otherwise.
43
+ */
44
+ import { EventEmitter } from 'node:events';
45
+ import crypto from 'node:crypto';
46
+ import { createConsoleLogger } from '../logger.js';
47
+ import { normalizeContract } from '../verify/contract.js';
48
+ import { captureBaseline, treeFingerprint } from '../verify/baseline.js';
49
+ import { acquireLease, commit, createAndAcquire, LEASE_HEARTBEAT_MS, renewLease, isValidRunId, releaseLease, deleteRunDir, listRuns, loadRun, nodeArtifactPath, runDir, summarize, } from './store.js';
50
+ import { isTerminalRunState, } from './types.js';
51
+ export const DEFAULT_NODE_TIMEOUT_MS = 30 * 60_000;
52
+ export const DEFAULT_MAX_NODE_VISITS = 50;
53
+ /**
54
+ * How long a run about to claim `verified` waits for abandoned attempts to stop.
55
+ *
56
+ * Short on purpose. The wait exists to catch an attempt that is about to finish,
57
+ * not to hold a run hostage to one that never will — a node stuck forever must
58
+ * not stop the run from ending, only from being called verified.
59
+ */
60
+ export const QUIESCE_GRACE_MS = 2_000;
61
+ /** Terminal verifier appended when the workflow declares a run-level contract. */
62
+ export const IMPLICIT_VERIFIER_ID = '__verify';
63
+ /** How much node text the checkpoint carries inline. The rest goes to disk. */
64
+ export const OUTPUT_PREVIEW_CHARS = 4000;
65
+ /**
66
+ * One owner's transactional view of a run.
67
+ *
68
+ * It holds the guard and the last record the disk accepted, and it is the only
69
+ * thing in this module that can change either. Three properties matter:
70
+ *
71
+ * - **Copy-on-write.** A change is applied to a clone, committed, and adopted
72
+ * only if the commit succeeded. A refused write therefore leaves the record
73
+ * this process hands to its callers exactly as the disk has it. The previous
74
+ * version mutated first and committed second in most places, so a superseded
75
+ * owner returned a record saying `completed` while `run.json` correctly said
76
+ * `running` — the same lie, one layer up.
77
+ * - **Everything, not most things.** Checkpoints, events and node artifacts all
78
+ * go through here. There is no unfenced variant to reach for, because the
79
+ * store no longer exports one.
80
+ * - **One-way.** The first refusal sets `lost`, and nothing is attempted after
81
+ * that. An owner that has been replaced does not keep trying.
82
+ */
83
+ class RunTxn {
84
+ guard;
85
+ logger;
86
+ onEvents;
87
+ onStop;
88
+ _record;
89
+ /** The run was taken over. Permanent: this owner may never write again. */
90
+ lost = false;
91
+ /**
92
+ * A write could not get through, but nothing says we lost the run.
93
+ *
94
+ * Kept apart from `lost` because the responses are opposite. Treating a
95
+ * millisecond of lock contention as a takeover made a run stop forever while
96
+ * still holding its lease — so nobody could take it over either, and a live
97
+ * local pid is never judged stale. Stalling stops the run and hands the claim
98
+ * back, which leaves it resumable.
99
+ */
100
+ stalled = false;
101
+ constructor(guard, record, logger,
102
+ /** Notified with the events of each accepted commit, for the in-process stream. */
103
+ onEvents,
104
+ /** Called once, when this owner stops writing, with which of the two it was. */
105
+ onStop) {
106
+ this.guard = guard;
107
+ this.logger = logger;
108
+ this.onEvents = onEvents;
109
+ this.onStop = onStop;
110
+ this._record = record;
111
+ }
112
+ get record() {
113
+ return this._record;
114
+ }
115
+ /** True once this owner has stopped writing, for either reason. */
116
+ get finished() {
117
+ return this.lost || this.stalled;
118
+ }
119
+ /**
120
+ * Apply a change, persist it, and adopt it — or refuse all three.
121
+ *
122
+ * `mutate` receives a clone. Mutating `txn.record` inside it would defeat the
123
+ * whole mechanism, which is why the clone is what is handed over.
124
+ */
125
+ apply(mutate, events = [], artifacts = []) {
126
+ if (this.finished)
127
+ return false;
128
+ const draft = JSON.parse(JSON.stringify(this._record));
129
+ mutate(draft);
130
+ const result = commit(this.guard, { record: draft, events, artifacts }, this.logger);
131
+ if (result.outcome !== 'committed')
132
+ return this._stop(result.outcome, result.reason);
133
+ this._record = draft;
134
+ this.onEvents(events);
135
+ return true;
136
+ }
137
+ /** Append events without changing state. Fenced like every other write. */
138
+ emit(...events) {
139
+ if (this.finished)
140
+ return false;
141
+ const result = commit(this.guard, { events }, this.logger);
142
+ if (result.outcome !== 'committed')
143
+ return this._stop(result.outcome, result.reason);
144
+ this.onEvents(events);
145
+ return true;
146
+ }
147
+ _stop(outcome, reason) {
148
+ if (this.finished)
149
+ return false;
150
+ const why = reason ?? 'no reason given';
151
+ if (outcome === 'superseded') {
152
+ this.lost = true;
153
+ this.logger.warn?.(`[kernel] ${this.guard.runId}: write refused — this owner (fence ${this.guard.fence}) no longer holds ` +
154
+ `the run (${why}), so it stops here rather than carrying on in memory`);
155
+ }
156
+ else {
157
+ this.stalled = true;
158
+ this.logger.warn?.(`[kernel] ${this.guard.runId}: write could not be committed (${why}) — the run stops and its claim is ` +
159
+ `handed back, so it can be resumed rather than wedged`);
160
+ }
161
+ this.onStop(outcome, why);
162
+ return false;
163
+ }
164
+ }
165
+ const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
166
+ /**
167
+ * Append the implicit terminal verifier when a run-level contract exists and the
168
+ * author did not already place one. This is what makes `completed` mean "checked".
169
+ */
170
+ /**
171
+ * Reject a spec that cannot execute correctly, before anything runs.
172
+ *
173
+ * These are all mistakes that used to surface much later and much worse: a
174
+ * duplicate node id silently overwrote a record so one of the two nodes had no
175
+ * state; a `next` or router target naming a node that does not exist failed the
176
+ * run halfway through with `unknown node`; a negative retry or visit bound was
177
+ * accepted and then behaved arbitrarily.
178
+ */
179
+ export function validateSpec(spec) {
180
+ if (!spec || typeof spec.name !== 'string' || !spec.name.trim()) {
181
+ throw new Error('WorkflowSpec needs a name');
182
+ }
183
+ if (!Array.isArray(spec.nodes) || spec.nodes.length === 0) {
184
+ throw new Error(`Workflow '${spec.name}' has no nodes`);
185
+ }
186
+ const ids = new Set();
187
+ for (const node of spec.nodes) {
188
+ if (!node?.id || typeof node.id !== 'string')
189
+ throw new Error(`Workflow '${spec.name}' has a node with no id`);
190
+ if (ids.has(node.id))
191
+ throw new Error(`Workflow '${spec.name}' has duplicate node id '${node.id}'`);
192
+ ids.add(node.id);
193
+ if (node.retry && (!Number.isInteger(node.retry.max) || node.retry.max < 0)) {
194
+ throw new Error(`Node '${node.id}': retry.max must be a non-negative integer`);
195
+ }
196
+ if (node.timeoutMs !== undefined && (!Number.isFinite(node.timeoutMs) || node.timeoutMs <= 0)) {
197
+ throw new Error(`Node '${node.id}': timeoutMs must be a positive number`);
198
+ }
199
+ }
200
+ if (spec.maxNodeVisits !== undefined && (!Number.isInteger(spec.maxNodeVisits) || spec.maxNodeVisits < 1)) {
201
+ throw new Error(`Workflow '${spec.name}': maxNodeVisits must be a positive integer`);
202
+ }
203
+ const check = (target, where) => {
204
+ if (target !== undefined && !ids.has(target)) {
205
+ throw new Error(`${where} points at unknown node '${target}'`);
206
+ }
207
+ };
208
+ for (const node of spec.nodes) {
209
+ check(node.next, `Node '${node.id}' next`);
210
+ if (node.kind === 'router') {
211
+ check(node.default, `Router '${node.id}' default`);
212
+ for (const route of node.routes ?? []) {
213
+ check(route.to, `Router '${node.id}' route`);
214
+ if (!route.when?.type)
215
+ throw new Error(`Router '${node.id}' has a route with no condition`);
216
+ }
217
+ }
218
+ }
219
+ }
220
+ export function prepareSpec(spec) {
221
+ if (!spec.contract)
222
+ return spec;
223
+ const hasRunVerifier = spec.nodes.some((n) => n.kind === 'verifier' && n.contract === 'run');
224
+ if (hasRunVerifier)
225
+ return spec;
226
+ return {
227
+ ...spec,
228
+ nodes: [...spec.nodes, { id: IMPLICIT_VERIFIER_ID, kind: 'verifier', contract: 'run' }],
229
+ };
230
+ }
231
+ export class RunKernel extends EventEmitter {
232
+ manager;
233
+ logger;
234
+ executors;
235
+ nodeTimeoutMs;
236
+ live = new Map();
237
+ /**
238
+ * This kernel's identity as a run owner.
239
+ *
240
+ * Per instance, not per process: two `RunKernel`s in one process — two
241
+ * SessionManagers is not exotic — would otherwise share a pid and each treat
242
+ * the other's lease as its own, so both would execute the same run.
243
+ */
244
+ ownerId = `owner-${crypto.randomUUID()}`;
245
+ /**
246
+ * Per-run secrets, in memory only. Keyed by run id so a resume in this
247
+ * process can re-use what the start supplied; a resume elsewhere must be
248
+ * given them again.
249
+ */
250
+ _secrets = new Map();
251
+ /** Caller label for the most recent start of each run. */
252
+ _tags = new Map();
253
+ constructor(opts = {}) {
254
+ super();
255
+ this.manager = opts.manager;
256
+ this.logger = opts.logger ?? createConsoleLogger('kernel');
257
+ this.executors = opts.executors ?? {};
258
+ this.nodeTimeoutMs = opts.nodeTimeoutMs ?? DEFAULT_NODE_TIMEOUT_MS;
259
+ }
260
+ /** Stop and await any live run holding this id, so a reuse cannot overlap it. */
261
+ async _retire(runId) {
262
+ const existing = this.live.get(runId);
263
+ if (!existing)
264
+ return;
265
+ this.cancel(runId);
266
+ await existing.done.catch(() => undefined);
267
+ if (this.live.get(runId) === existing)
268
+ this.live.delete(runId);
269
+ }
270
+ /** Register (or replace) the executor for one node kind. */
271
+ setExecutor(kind, executor) {
272
+ this.executors[kind] = executor;
273
+ }
274
+ /**
275
+ * Open a transaction over a run this kernel has just claimed.
276
+ *
277
+ * The abort signal is created here rather than in `_launch` because the
278
+ * transaction has to be able to stop the run the moment a write is refused,
279
+ * and that can happen before the handle exists.
280
+ */
281
+ _open(guard, record) {
282
+ const signal = { aborted: false };
283
+ const txn = new RunTxn(guard, record, this.logger, (events) => {
284
+ for (const event of events) {
285
+ this.emit('kernel-event', { runId: guard.runId, event });
286
+ this.emit(guard.runId, event);
287
+ }
288
+ }, (outcome, reason) => {
289
+ signal.aborted = true;
290
+ // Being superseded means someone else already owns the run; handing it
291
+ // back would take it from them. Being blocked means we still hold a
292
+ // claim we can no longer use, and leaving that behind is what wedges a
293
+ // run permanently — a live local pid is never judged stale, so nobody
294
+ // else could ever take it.
295
+ if (outcome === 'blocked')
296
+ this._scheduleRelease(guard, reason);
297
+ });
298
+ return { txn, signal };
299
+ }
300
+ /**
301
+ * Hand a claim back, retrying while the lock is merely busy.
302
+ *
303
+ * Best-effort with a bound: contention here is measured in microseconds, so a
304
+ * few backed-off attempts cover everything short of a wedged filesystem — and
305
+ * if it is wedged, saying so beats retrying forever.
306
+ */
307
+ _scheduleRelease(guard, reason, attempt = 0) {
308
+ const outcome = releaseLease(guard);
309
+ if (outcome !== 'blocked')
310
+ return;
311
+ if (attempt >= 6) {
312
+ this.logger.error?.(`[kernel] ${guard.runId}: could not hand the claim back after ${attempt} attempts` +
313
+ `${reason ? ` (${reason})` : ''} — resuming it elsewhere will have to wait for the lease to expire`);
314
+ return;
315
+ }
316
+ const timer = setTimeout(() => this._scheduleRelease(guard, reason, attempt + 1), 250 * 2 ** attempt);
317
+ if (typeof timer.unref === 'function')
318
+ timer.unref();
319
+ }
320
+ // ─── Lifecycle ────────────────────────────────────────────────────────────
321
+ async start(rawSpec, opts = {}) {
322
+ let contract = rawSpec.contract;
323
+ if (opts.contract !== undefined) {
324
+ contract = normalizeContract(opts.contract);
325
+ // A contract that survives normalisation with nothing in it used to leave
326
+ // the run with no contract at all, which then completed `unverified` — a
327
+ // caller who asked to be checked was told nothing had checked, and the
328
+ // reason was a typo they never saw.
329
+ if (!contract) {
330
+ throw new Error('The supplied acceptance contract has no recognised checks. Every check needs a `type` of ' +
331
+ 'command | http | screenshot | diff_policy | file, and a `command` check needs `cmd`.');
332
+ }
333
+ }
334
+ const spec = prepareSpec({ ...rawSpec, contract });
335
+ validateSpec(spec);
336
+ const runId = opts.runId || `wf-${Date.now().toString(36)}-${crypto.randomBytes(3).toString('hex')}`;
337
+ // Validated here as well as in the store: failing before any directory is
338
+ // created keeps a rejected id from leaving a half-made run behind.
339
+ if (!isValidRunId(runId)) {
340
+ throw new Error(`Invalid run id ${JSON.stringify(runId)}: must be a single path segment ([A-Za-z0-9._-])`);
341
+ }
342
+ const cwd = opts.cwd || spec.cwd || process.cwd();
343
+ const now = new Date().toISOString();
344
+ const nodes = {};
345
+ for (const n of spec.nodes) {
346
+ nodes[n.id] = { id: n.id, kind: n.kind, state: 'pending', attempts: 0, visits: 0 };
347
+ }
348
+ const record = {
349
+ runId,
350
+ workflow: spec.name,
351
+ spec,
352
+ state: 'pending',
353
+ outcome: 'unverified',
354
+ cwd,
355
+ createdAt: now,
356
+ updatedAt: now,
357
+ nodes,
358
+ costUsd: 0,
359
+ };
360
+ if (opts.secrets)
361
+ this._secrets.set(runId, opts.secrets);
362
+ if (opts.tag)
363
+ this._tags.set(runId, opts.tag);
364
+ // A run id may be reused once its previous run is gone — a failed start
365
+ // frees it. But two live runs with the same id in one process would write
366
+ // over each other's checkpoints, so the old one is stopped and waited for
367
+ // first. This is the in-process half of what the lease does across
368
+ // processes.
369
+ await this._retire(runId);
370
+ // Creating and claiming are one step, and the directory creation is itself
371
+ // the claim: two processes cannot both come away believing they made this
372
+ // run. A fresh directory also means a fresh incarnation id, which is what
373
+ // stops a guard from the previous life of this run id from ever being valid
374
+ // again.
375
+ const guard = createAndAcquire(runId, spec, this.ownerId);
376
+ record.baseSha = await captureBaseline(cwd);
377
+ const { txn, signal } = this._open(guard, record);
378
+ if (!txn.apply(() => undefined, [{ ts: now, type: 'run_created', runId, workflow: spec.name }])) {
379
+ throw new Error(`Run '${runId}' could not be checkpointed: the claim was lost before it started`);
380
+ }
381
+ this._launch(txn, signal);
382
+ return txn.record;
383
+ }
384
+ /**
385
+ * Re-attach to a run whose process died. Nodes already marked succeeded are not
386
+ * re-run; the node that was in flight when the process ended is retried from
387
+ * the start, because a half-finished node left no result to trust.
388
+ */
389
+ async resume(runId, opts = {}) {
390
+ if (opts.secrets)
391
+ this._secrets.set(runId, opts.secrets);
392
+ if (opts.tag)
393
+ this._tags.set(runId, opts.tag);
394
+ if (this.live.has(runId)) {
395
+ const existing = loadRun(runId);
396
+ if (!existing)
397
+ throw new Error(`Run '${runId}' not found`);
398
+ return existing;
399
+ }
400
+ const record = loadRun(runId);
401
+ if (!record)
402
+ throw new Error(`Run '${runId}' not found`);
403
+ // A finished run stays finished by default: silently re-running a completed
404
+ // workflow because someone polled `resume` would be a nasty surprise.
405
+ // `restart` is for callers whose "resume" genuinely means "bring it back up"
406
+ // — autoloop, whose runs terminate and are expected to be restartable.
407
+ //
408
+ // Checked BEFORE the lease is taken. Taking it first meant that merely
409
+ // polling `resume` on a completed run minted a lease nobody would ever
410
+ // release, blocking the next process from restarting it.
411
+ if (isTerminalRunState(record.state) && !opts.restart)
412
+ return record;
413
+ // Claim it: two processes resuming the same run would each execute its
414
+ // nodes, with every side effect happening twice.
415
+ const guard = acquireLease(runId, this.ownerId);
416
+ const { txn, signal } = this._open(guard, record);
417
+ const wasTerminal = isTerminalRunState(record.state);
418
+ if (!txn.apply((draft) => {
419
+ for (const n of Object.values(draft.nodes)) {
420
+ if (n.state === 'running' || (opts.restart && wasTerminal)) {
421
+ n.state = 'pending';
422
+ n.attempts = 0;
423
+ n.error = undefined;
424
+ }
425
+ }
426
+ if (opts.restart) {
427
+ draft.endedAt = undefined;
428
+ draft.outcome = 'unverified';
429
+ draft.outcomeReason = undefined;
430
+ draft.verdict = undefined;
431
+ }
432
+ draft.error = undefined;
433
+ draft.updatedAt = new Date().toISOString();
434
+ })) {
435
+ throw new Error(`Run '${runId}' could not be checkpointed: the claim was lost before it resumed`);
436
+ }
437
+ this._launch(txn, signal);
438
+ return txn.record;
439
+ }
440
+ cancel(runId) {
441
+ const handle = this.live.get(runId);
442
+ if (!handle)
443
+ return false;
444
+ handle.signal.aborted = true;
445
+ handle.gate?.resolve(false);
446
+ // Reach the engines too. Setting a flag only works for runners that check
447
+ // it; `Council` and `Fanout` have their own `abort()`, and without calling
448
+ // it a shutdown waits out the node timeout instead of stopping.
449
+ for (const [, live] of handle.handles) {
450
+ const abortable = live;
451
+ try {
452
+ abortable.abort?.();
453
+ }
454
+ catch {
455
+ // Best-effort: an engine that fails to abort must not block the others.
456
+ }
457
+ }
458
+ // Cancellation propagates to subflows. Without this a cancelled parent
459
+ // leaves its children running and spending, with nothing pointing at them.
460
+ for (const node of Object.values(handle.txn.record.nodes)) {
461
+ if (node.childRunId)
462
+ this.cancel(node.childRunId);
463
+ }
464
+ return true;
465
+ }
466
+ /** Queue text for the current (or next) agent node. */
467
+ steer(runId, text) {
468
+ const handle = this.live.get(runId);
469
+ if (!handle)
470
+ return false;
471
+ handle.steer.push(text);
472
+ handle.txn.emit({
473
+ ts: new Date().toISOString(),
474
+ type: 'steer',
475
+ node: handle.txn.record.currentNode ?? '',
476
+ text,
477
+ });
478
+ return true;
479
+ }
480
+ /** Answer a parked `human_gate`. */
481
+ approve(runId, approved) {
482
+ const handle = this.live.get(runId);
483
+ if (!handle?.gate)
484
+ return false;
485
+ handle.gate.resolve(approved);
486
+ return true;
487
+ }
488
+ get(runId) {
489
+ return loadRun(runId);
490
+ }
491
+ /**
492
+ * The live engine object a running node published, if the run is still going
493
+ * in this process. Undefined once the node finishes or the process restarts —
494
+ * which is the honest answer, not a gap.
495
+ */
496
+ handle(runId, nodeId) {
497
+ return this.live.get(runId)?.handles.get(nodeId);
498
+ }
499
+ list(query = {}) {
500
+ return listRuns(query);
501
+ }
502
+ /**
503
+ * Remove a run.
504
+ *
505
+ * `expectTag` guards against deleting the wrong incarnation: a run id is
506
+ * reused when a failed start frees it, so a dying start's cleanup could
507
+ * otherwise delete the retry that had already taken the id. When the tag does
508
+ * not match, nothing is touched.
509
+ *
510
+ * Deleting takes the incarnation with it, which is what makes the id safe to
511
+ * reuse: any guard still held by an abandoned attempt of the deleted run names
512
+ * an incarnation that no longer exists, so it cannot write to whatever takes
513
+ * the id next.
514
+ */
515
+ delete(runId, opts = {}) {
516
+ if (opts.expectTag !== undefined && this._tags.get(runId) !== opts.expectTag)
517
+ return false;
518
+ // Claim before deleting and never release first. Releasing here used to open
519
+ // a window in which another process could legally resume the run, only for
520
+ // this process to delete the directory out from under its new owner.
521
+ let claimed = false;
522
+ try {
523
+ acquireLease(runId, this.ownerId);
524
+ claimed = true;
525
+ }
526
+ catch {
527
+ claimed = false;
528
+ }
529
+ this.cancel(runId);
530
+ // Whatever happened to the directory, this process is done with the run, so
531
+ // its in-memory traces go — a refusal must not leave the caller's secrets
532
+ // sitting in a map for a run we are no longer tracking. (`acquireLease`
533
+ // throws for a run that no longer exists as well as for one someone else
534
+ // owns, and forgetting the credentials is right in both cases.)
535
+ this._secrets.delete(runId);
536
+ this._tags.delete(runId);
537
+ if (!claimed)
538
+ return false;
539
+ deleteRunDir(runId, this.logger);
540
+ return true;
541
+ }
542
+ /**
543
+ * Resolves when the run reaches a terminal state. A run that already finished
544
+ * (or belongs to another process) resolves from disk, so the caller does not
545
+ * have to race the completion.
546
+ */
547
+ wait(runId) {
548
+ const handle = this.live.get(runId);
549
+ return handle ? handle.done : Promise.resolve(loadRun(runId));
550
+ }
551
+ async shutdown() {
552
+ for (const [runId] of this.live)
553
+ this.cancel(runId);
554
+ await Promise.allSettled([...this.live.values()].map((h) => h.done));
555
+ this.live.clear();
556
+ }
557
+ // ─── Execution ────────────────────────────────────────────────────────────
558
+ _launch(txn, signal) {
559
+ const runId = txn.guard.runId;
560
+ const handle = {
561
+ txn,
562
+ signal,
563
+ steer: [],
564
+ tag: this._tags.get(runId),
565
+ inflight: new Set(),
566
+ secrets: this._secrets.get(runId) ?? {},
567
+ handles: new Map(),
568
+ done: Promise.resolve(txn.record),
569
+ };
570
+ handle.heartbeat = setInterval(() => renewLease(txn.guard), LEASE_HEARTBEAT_MS);
571
+ if (typeof handle.heartbeat.unref === 'function')
572
+ handle.heartbeat.unref();
573
+ // Registered before the run starts: a workflow with no nodes finishes
574
+ // synchronously up to its first await, and the `finally` below would
575
+ // otherwise delete an entry that had not been added yet.
576
+ this.live.set(runId, handle);
577
+ handle.done = this._run(handle).finally(() => {
578
+ if (handle.heartbeat)
579
+ clearInterval(handle.heartbeat);
580
+ // A run that stopped because it could not write still holds its claim.
581
+ // `_scheduleRelease` was already started by the stop callback; this covers
582
+ // the case where the run ended for another reason while stalled.
583
+ if (txn.stalled)
584
+ this._scheduleRelease(txn.guard);
585
+ // Identity-checked: a run id can be reused, and deleting by key alone
586
+ // meant a finishing run evicted the handle of the run that had just
587
+ // replaced it.
588
+ if (this.live.get(runId) === handle)
589
+ this.live.delete(runId);
590
+ });
591
+ // The caller gets the record immediately; failures surface through the record.
592
+ handle.done.catch(() => undefined);
593
+ }
594
+ _setRunState(handle, state, error) {
595
+ const txn = handle.txn;
596
+ const ts = new Date().toISOString();
597
+ const terminal = isTerminalRunState(state);
598
+ const outcome = txn.record.outcome;
599
+ const ok = txn.apply((draft) => {
600
+ draft.state = state;
601
+ draft.updatedAt = ts;
602
+ if (error)
603
+ draft.error = error;
604
+ // Terminal states carry `endedAt`; this is the one place that stamps it.
605
+ if (terminal)
606
+ draft.endedAt = ts;
607
+ }, [{ ts, type: 'run_state', state, outcome, error }]);
608
+ if (!ok)
609
+ return false;
610
+ // Hand the run back once it is over, so another owner can pick it up
611
+ // without waiting out the lease. After the writes, never before.
612
+ if (terminal)
613
+ this._scheduleRelease(txn.guard);
614
+ return true;
615
+ }
616
+ _setNodeState(handle, nodeId, state, extra = {}) {
617
+ const ts = new Date().toISOString();
618
+ return handle.txn.apply((draft) => {
619
+ const node = draft.nodes[nodeId];
620
+ if (!node)
621
+ return;
622
+ node.state = state;
623
+ if (extra.attempt !== undefined)
624
+ node.attempts = extra.attempt;
625
+ if (extra.error)
626
+ node.error = extra.error;
627
+ if (state === 'running')
628
+ node.startedAt = ts;
629
+ if (state === 'succeeded' || state === 'failed' || state === 'skipped' || state === 'cancelled') {
630
+ node.endedAt = ts;
631
+ }
632
+ draft.currentNode = nodeId;
633
+ draft.updatedAt = ts;
634
+ }, [{ ts, type: 'node_state', node: nodeId, state, ...extra }]);
635
+ }
636
+ _nodeSpec(record, id) {
637
+ return record.spec.nodes.find((n) => n.id === id);
638
+ }
639
+ _nextInOrder(record, id) {
640
+ const idx = record.spec.nodes.findIndex((n) => n.id === id);
641
+ return idx >= 0 && idx + 1 < record.spec.nodes.length ? record.spec.nodes[idx + 1].id : undefined;
642
+ }
643
+ _firstPending(record) {
644
+ for (const n of record.spec.nodes) {
645
+ const rec = record.nodes[n.id];
646
+ if (!rec || rec.state === 'pending' || rec.state === 'awaiting_human')
647
+ return n.id;
648
+ }
649
+ return undefined;
650
+ }
651
+ async _executeNode(node, ctx, timeoutMs, attemptSignal, inflight) {
652
+ const executor = this.executors[node.kind];
653
+ if (!executor) {
654
+ return { ok: false, error: `no executor registered for node kind '${node.kind}'` };
655
+ }
656
+ let timer;
657
+ const timeout = new Promise((resolve) => {
658
+ timer = setTimeout(() => {
659
+ // Aborts this attempt only. A timeout is a node failure — it still gets
660
+ // its retries and still honours `onFailure` — whereas cancelling the run
661
+ // is a separate, user-initiated thing. Conflating the two made a hung
662
+ // node report the whole run as `cancelled`.
663
+ attemptSignal.aborted = true;
664
+ resolve({ ok: false, error: `node timed out after ${timeoutMs}ms` });
665
+ }, timeoutMs);
666
+ if (typeof timer.unref === 'function')
667
+ timer.unref();
668
+ });
669
+ // The executor promise is tracked, not just raced. A timeout abandons the
670
+ // wait, not the work — the executor keeps running and can still write — so
671
+ // the run has to know it is out there before it calls anything verified.
672
+ const running = Promise.resolve()
673
+ .then(() => executor(node, ctx))
674
+ .catch((err) => ({ ok: false, error: err.message }))
675
+ .finally(() => inflight.delete(running));
676
+ inflight.add(running);
677
+ try {
678
+ return await Promise.race([running, timeout]);
679
+ }
680
+ catch (err) {
681
+ return { ok: false, error: err.message };
682
+ }
683
+ finally {
684
+ if (timer)
685
+ clearTimeout(timer);
686
+ }
687
+ }
688
+ /**
689
+ * Wait for abandoned attempts to stop, so a terminal verdict describes a tree
690
+ * nobody is still writing to.
691
+ *
692
+ * A timed-out node is not killed — JS gives us no way to — so a run used to
693
+ * stamp `completed / verified`, then have the abandoned attempt write to the
694
+ * workspace afterwards. The evidence was accurate at the moment it was taken
695
+ * and wrong seconds later, with nothing recording that.
696
+ */
697
+ async _awaitQuiescence(handle, graceMs) {
698
+ if (handle.inflight.size === 0)
699
+ return true;
700
+ let timer;
701
+ const grace = new Promise((resolve) => {
702
+ timer = setTimeout(() => resolve('timeout'), graceMs);
703
+ if (typeof timer.unref === 'function')
704
+ timer.unref();
705
+ });
706
+ const settled = Promise.allSettled([...handle.inflight]).then(() => 'settled');
707
+ try {
708
+ return (await Promise.race([settled, grace])) === 'settled';
709
+ }
710
+ finally {
711
+ if (timer)
712
+ clearTimeout(timer);
713
+ }
714
+ }
715
+ async _run(handle) {
716
+ const txn = handle.txn;
717
+ const maxVisits = txn.record.spec.maxNodeVisits ?? DEFAULT_MAX_NODE_VISITS;
718
+ this._setRunState(handle, 'running');
719
+ let cursor = this._firstPending(txn.record);
720
+ while (cursor) {
721
+ // Losing the run outranks everything else: this owner may not write, so
722
+ // there is nothing left for it to do but stop.
723
+ if (txn.finished)
724
+ return txn.record;
725
+ if (handle.signal.aborted) {
726
+ this._setRunState(handle, 'cancelled');
727
+ return txn.record;
728
+ }
729
+ const nodeId = cursor;
730
+ const spec = this._nodeSpec(txn.record, nodeId);
731
+ if (!spec || !txn.record.nodes[nodeId]) {
732
+ this._setRunState(handle, 'failed', `unknown node '${nodeId}'`);
733
+ return txn.record;
734
+ }
735
+ // The visit counter is part of the loop bound, so it is committed like any
736
+ // other state rather than incremented on a record nobody agreed to.
737
+ if (!txn.apply((draft) => {
738
+ const node = draft.nodes[nodeId];
739
+ if (node)
740
+ node.visits = (node.visits ?? 0) + 1;
741
+ draft.updatedAt = new Date().toISOString();
742
+ })) {
743
+ return txn.record;
744
+ }
745
+ if ((txn.record.nodes[nodeId]?.visits ?? 0) > maxVisits) {
746
+ this._setNodeState(handle, nodeId, 'failed', { error: `visit limit ${maxVisits} exceeded` });
747
+ this._setRunState(handle, 'failed', `node '${nodeId}' exceeded the ${maxVisits}-visit loop bound`);
748
+ return txn.record;
749
+ }
750
+ if (spec.kind === 'verifier')
751
+ this._setRunState(handle, 'verifying');
752
+ const result = await this._runWithRetry(handle, spec);
753
+ if (txn.finished)
754
+ return txn.record;
755
+ // Cancel wins regardless of what the node returned. A runner that never
756
+ // looked at the signal and reported success anyway used to carry the run
757
+ // all the way to `completed` — so "I cancelled it" and "it completed"
758
+ // could both be true, which makes cancellation meaningless.
759
+ if (handle.signal.aborted) {
760
+ this._absorb(handle, nodeId, result);
761
+ this._setNodeState(handle, nodeId, 'cancelled');
762
+ this._setRunState(handle, 'cancelled');
763
+ return txn.record;
764
+ }
765
+ if (result.awaitHuman) {
766
+ const approved = await this._park(handle, nodeId);
767
+ if (txn.finished)
768
+ return txn.record;
769
+ if (!approved) {
770
+ this._setNodeState(handle, nodeId, 'failed', { error: 'rejected at human gate' });
771
+ this._setRunState(handle, handle.signal.aborted ? 'cancelled' : 'failed', 'rejected at human gate');
772
+ return txn.record;
773
+ }
774
+ this._setNodeState(handle, nodeId, 'succeeded');
775
+ cursor = spec.next ?? this._nextInOrder(txn.record, nodeId);
776
+ continue;
777
+ }
778
+ this._absorb(handle, nodeId, result);
779
+ if (txn.finished)
780
+ return txn.record;
781
+ if (!result.ok) {
782
+ this._setNodeState(handle, nodeId, 'failed', { error: result.error });
783
+ if ((spec.onFailure ?? 'fail') === 'fail') {
784
+ await this._finish(handle, `node '${nodeId}' failed: ${result.error ?? 'unknown'}`);
785
+ return txn.record;
786
+ }
787
+ // `continue` — record it and move on.
788
+ }
789
+ else {
790
+ this._setNodeState(handle, nodeId, 'succeeded');
791
+ }
792
+ if (spec.kind === 'router') {
793
+ cursor = result.goto ?? spec.default ?? this._nextInOrder(txn.record, nodeId);
794
+ continue;
795
+ }
796
+ cursor = spec.next ?? this._nextInOrder(txn.record, nodeId);
797
+ }
798
+ await this._finish(handle);
799
+ return txn.record;
800
+ }
801
+ async _runWithRetry(handle, spec) {
802
+ const txn = handle.txn;
803
+ const maxAttempts = (spec.retry?.max ?? 0) + 1;
804
+ const timeoutMs = spec.timeoutMs ?? this.nodeTimeoutMs;
805
+ let last = { ok: false, error: 'not attempted' };
806
+ for (let attempt = 1; attempt <= maxAttempts; attempt++) {
807
+ if (handle.signal.aborted)
808
+ return { ok: false, error: 'cancelled' };
809
+ // The attempt counter goes in with the state change, so a refused write
810
+ // means the attempt never officially started.
811
+ if (!this._setNodeState(handle, spec.id, 'running', { attempt })) {
812
+ return { ok: false, error: 'the run was taken over by another owner' };
813
+ }
814
+ // A node sees one signal that is the union of "this attempt gave up" and
815
+ // "the whole run was cancelled"; the kernel keeps them apart.
816
+ const attemptSignal = { aborted: false };
817
+ const ctx = {
818
+ runId: txn.guard.runId,
819
+ // A live view of the committed record, so a node holding `ctx` across an
820
+ // await never reads a state the disk refused.
821
+ get record() {
822
+ return txn.record;
823
+ },
824
+ cwd: txn.record.cwd,
825
+ attempt,
826
+ manager: this.manager,
827
+ logger: this.logger,
828
+ signal: {
829
+ get aborted() {
830
+ return handle.signal.aborted || attemptSignal.aborted;
831
+ },
832
+ set aborted(v) {
833
+ attemptSignal.aborted = v;
834
+ },
835
+ },
836
+ takeSteer: () => handle.steer.splice(0, handle.steer.length),
837
+ // Fenced like everything else. It used to be the one context method that
838
+ // wrote unguarded, so a superseded owner's node could still append to the
839
+ // log its replacement was reading.
840
+ emit: (event) => {
841
+ txn.emit(event);
842
+ },
843
+ runContract: txn.record.spec.contract,
844
+ setHandle: (h) => handle.handles.set(spec.id, h),
845
+ publish: (data) => {
846
+ txn.apply((draft) => {
847
+ const node = draft.nodes[spec.id];
848
+ if (node)
849
+ node.data = data;
850
+ draft.updatedAt = new Date().toISOString();
851
+ });
852
+ },
853
+ secrets: handle.secrets,
854
+ tag: handle.tag,
855
+ setChild: (childRunId) => {
856
+ txn.apply((draft) => {
857
+ const node = draft.nodes[spec.id];
858
+ if (node)
859
+ node.childRunId = childRunId;
860
+ draft.updatedAt = new Date().toISOString();
861
+ });
862
+ },
863
+ };
864
+ last = await this._executeNode(spec, ctx, timeoutMs, attemptSignal, handle.inflight);
865
+ if (last.ok || handle.signal.aborted)
866
+ return last;
867
+ if (attempt < maxAttempts) {
868
+ const backoff = (spec.retry?.backoffMs ?? 1000) * attempt;
869
+ this.logger.warn?.(`[kernel] ${txn.guard.runId}/${spec.id} attempt ${attempt}/${maxAttempts} failed: ${last.error} — retrying in ${backoff}ms`);
870
+ await sleep(backoff);
871
+ }
872
+ }
873
+ return last;
874
+ }
875
+ async _park(handle, nodeId) {
876
+ // If either write is refused the run is no longer ours, and parking on a
877
+ // gate we can never record the answer to would hang the executor forever.
878
+ if (!this._setNodeState(handle, nodeId, 'awaiting_human'))
879
+ return false;
880
+ if (!this._setRunState(handle, 'awaiting_human'))
881
+ return false;
882
+ const approved = await new Promise((resolve) => {
883
+ handle.gate = { resolve };
884
+ });
885
+ handle.gate = undefined;
886
+ if (!handle.signal.aborted)
887
+ this._setRunState(handle, 'running');
888
+ return approved;
889
+ }
890
+ /** Node kinds that can change the workspace. Routers and gates cannot. */
891
+ static SIDE_EFFECT_KINDS = ['agent', 'fanout', 'council', 'subflow'];
892
+ /**
893
+ * Fold a node's result into the run — output, artifacts, cost, votes, verdict
894
+ * — as one committed change.
895
+ *
896
+ * Every one of these used to be assigned straight onto the live record, with
897
+ * only the events fenced. A superseded owner therefore returned a record
898
+ * carrying an output, a cost and a passing verdict that the disk had refused.
899
+ */
900
+ _absorb(handle, nodeId, result) {
901
+ const txn = handle.txn;
902
+ const ts = new Date().toISOString();
903
+ const kind = txn.record.nodes[nodeId]?.kind;
904
+ const events = [];
905
+ const artifacts = [];
906
+ // The record keeps a preview so checkpoints stay small; the full text goes
907
+ // to the node's artifact directory, because for some nodes the text *is*
908
+ // the deliverable and a silent 4 kB cut would lose it. The artifact is
909
+ // written inside the same commit as the record that references it, so a
910
+ // refused write leaves neither behind.
911
+ let preview;
912
+ let outputArtifact;
913
+ if (result.output !== undefined) {
914
+ if (result.output.length > OUTPUT_PREVIEW_CHARS) {
915
+ outputArtifact = nodeArtifactPath(txn.guard.runId, nodeId, 'output.txt');
916
+ artifacts.push({ nodeId, name: 'output.txt', body: result.output });
917
+ preview =
918
+ result.output.slice(0, OUTPUT_PREVIEW_CHARS) + `\n…[truncated — full text in nodes/${nodeId}/output.txt]`;
919
+ }
920
+ else {
921
+ preview = result.output;
922
+ }
923
+ events.push({ ts, type: 'node_output', node: nodeId, text: preview });
924
+ }
925
+ if (result.evidenceId) {
926
+ events.push({
927
+ ts,
928
+ type: 'evidence',
929
+ node: nodeId,
930
+ evidenceId: result.evidenceId,
931
+ passed: Boolean(result.passed),
932
+ });
933
+ }
934
+ return txn.apply((draft) => {
935
+ const node = draft.nodes[nodeId];
936
+ if (!node)
937
+ return;
938
+ if (kind && RunKernel.SIDE_EFFECT_KINDS.includes(kind)) {
939
+ draft.sideEffectSeq = (draft.sideEffectSeq ?? 0) + 1;
940
+ }
941
+ if (preview !== undefined)
942
+ node.output = preview;
943
+ if (outputArtifact)
944
+ node.artifacts = [...new Set([...(node.artifacts ?? []), outputArtifact])];
945
+ if (result.artifacts?.length)
946
+ node.artifacts = result.artifacts;
947
+ if (result.data !== undefined)
948
+ node.data = result.data;
949
+ if (result.childRunId)
950
+ node.childRunId = result.childRunId;
951
+ if (typeof result.costUsd === 'number')
952
+ draft.costUsd = (draft.costUsd ?? 0) + result.costUsd;
953
+ if (result.consensusVotes?.length) {
954
+ draft.consensusVotes = [...(draft.consensusVotes ?? []), ...result.consensusVotes];
955
+ }
956
+ if (result.evidenceId) {
957
+ node.evidenceId = result.evidenceId;
958
+ draft.evidenceId = result.evidenceId;
959
+ draft.outcome = result.passed ? 'verified' : 'refuted';
960
+ draft.outcomeReason = undefined;
961
+ draft.verdict = {
962
+ node: nodeId,
963
+ evidenceId: result.evidenceId,
964
+ treeFingerprint: result.treeFingerprint,
965
+ sideEffectSeq: draft.sideEffectSeq ?? 0,
966
+ };
967
+ }
968
+ draft.updatedAt = ts;
969
+ }, events, artifacts);
970
+ }
971
+ /**
972
+ * Decide the terminal state. `completed` requires that nothing refuted the run;
973
+ * a run with no contract completes as `unverified`, which says we did not check
974
+ * rather than claiming success.
975
+ */
976
+ async _finish(handle, error) {
977
+ const txn = handle.txn;
978
+ let outcome = txn.record.outcome;
979
+ let downgrade;
980
+ // Nothing may still be writing when a verdict is stamped. An abandoned
981
+ // attempt keeps running after its timeout, so wait briefly for it — and if
982
+ // it will not stop, say so instead of vouching for a tree it may yet change.
983
+ //
984
+ // Only when there is a verdict to protect: a run that was never verified has
985
+ // nothing to lose by ending promptly, and blocking it would make one hung
986
+ // node delay every run that contained it.
987
+ if (outcome === 'verified' && !(await this._awaitQuiescence(handle, QUIESCE_GRACE_MS))) {
988
+ outcome = 'unverified';
989
+ downgrade =
990
+ `evidence ${txn.record.verdict?.evidenceId ?? '(none)'} passed, but ${handle.inflight.size} abandoned ` +
991
+ `attempt(s) were still running when the run ended — they can still change the tree, so the verdict ` +
992
+ `cannot stand`;
993
+ }
994
+ if (outcome === 'verified') {
995
+ const stale = await this._verdictWentStale(txn.record);
996
+ if (stale) {
997
+ outcome = 'unverified';
998
+ downgrade = stale;
999
+ }
1000
+ }
1001
+ if (downgrade) {
1002
+ const reason = downgrade;
1003
+ txn.apply((draft) => {
1004
+ draft.outcome = outcome;
1005
+ draft.outcomeReason = reason;
1006
+ }, [{ ts: new Date().toISOString(), type: 'log', level: 'warn', message: `[verify] ${reason}` }]);
1007
+ if (txn.finished)
1008
+ return;
1009
+ }
1010
+ if (error || outcome === 'refuted') {
1011
+ this._setRunState(handle, 'failed', error ?? 'acceptance contract was not satisfied');
1012
+ return;
1013
+ }
1014
+ this._setRunState(handle, 'completed');
1015
+ }
1016
+ /**
1017
+ * Why a passing verdict no longer stands, or undefined if it still does.
1018
+ *
1019
+ * `prepareSpec` cannot enforce this structurally: a router can send control
1020
+ * anywhere, so which node runs last is not a property of the spec. And a
1021
+ * "nothing after the verifier" rule would be the wrong rule anyway — what
1022
+ * matters is not that a node ran, but that the tree moved. So this measures.
1023
+ *
1024
+ * The caller drops the outcome to `unverified`, not `refuted`: no check
1025
+ * failed. We simply no longer know, and saying so is the whole point of having
1026
+ * three outcomes.
1027
+ */
1028
+ async _verdictWentStale(record) {
1029
+ if (record.outcome !== 'verified' || !record.verdict)
1030
+ return undefined;
1031
+ // Nothing that could touch the workspace ran after the checks, so the
1032
+ // verdict still describes the tree. This is the ordinary case, and it must
1033
+ // not depend on git: a contract that passed in a plain directory passed.
1034
+ if ((record.sideEffectSeq ?? 0) === record.verdict.sideEffectSeq)
1035
+ return undefined;
1036
+ const before = record.verdict.treeFingerprint;
1037
+ const after = await treeFingerprint(record.cwd);
1038
+ if (before !== undefined && after !== undefined && before === after)
1039
+ return undefined;
1040
+ return before === undefined || after === undefined
1041
+ ? `evidence ${record.verdict.evidenceId} passed, but nodes ran afterwards and ${record.cwd} is not a git repository, so we cannot tell whether it still describes the tree`
1042
+ : `evidence ${record.verdict.evidenceId} passed, but the working tree changed afterwards — the verdict describes an earlier state`;
1043
+ }
1044
+ }
1045
+ /** Run directory for a given run — re-exported so callers need not import the store. */
1046
+ export { runDir, summarize };
1047
+ //# sourceMappingURL=engine.js.map