@enderfga/claw-orchestrator 5.1.0 → 6.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (152) hide show
  1. package/README.md +26 -26
  2. package/dist/bin/cli.js +107 -1
  3. package/dist/bin/cli.js.map +1 -1
  4. package/dist/src/acp-server.d.ts +5 -5
  5. package/dist/src/acp-server.js +3 -3
  6. package/dist/src/acp-server.js.map +1 -1
  7. package/dist/src/autoloop/dispatcher.d.ts +22 -0
  8. package/dist/src/autoloop/dispatcher.js +71 -13
  9. package/dist/src/autoloop/dispatcher.js.map +1 -1
  10. package/dist/src/autoloop/messages.d.ts +10 -0
  11. package/dist/src/autoloop/messages.js.map +1 -1
  12. package/dist/src/autoloop/runner.js +6 -0
  13. package/dist/src/autoloop/runner.js.map +1 -1
  14. package/dist/src/constants.d.ts +0 -6
  15. package/dist/src/constants.js +0 -6
  16. package/dist/src/constants.js.map +1 -1
  17. package/dist/src/council.d.ts +15 -0
  18. package/dist/src/council.js +48 -35
  19. package/dist/src/council.js.map +1 -1
  20. package/dist/src/dashboard/index.html +191 -6
  21. package/dist/src/embedded-server.js +132 -9
  22. package/dist/src/embedded-server.js.map +1 -1
  23. package/dist/src/fanout.d.ts +30 -1
  24. package/dist/src/fanout.js +32 -3
  25. package/dist/src/fanout.js.map +1 -1
  26. package/dist/src/index.js +359 -4
  27. package/dist/src/index.js.map +1 -1
  28. package/dist/src/kernel/agent-step.d.ts +59 -0
  29. package/dist/src/kernel/agent-step.js +100 -0
  30. package/dist/src/kernel/agent-step.js.map +1 -0
  31. package/dist/src/kernel/conditions.d.ts +11 -0
  32. package/dist/src/kernel/conditions.js +24 -0
  33. package/dist/src/kernel/conditions.js.map +1 -0
  34. package/dist/src/kernel/engine.d.ts +319 -0
  35. package/dist/src/kernel/engine.js +1047 -0
  36. package/dist/src/kernel/engine.js.map +1 -0
  37. package/dist/src/kernel/exec.d.ts +43 -0
  38. package/dist/src/kernel/exec.js +112 -0
  39. package/dist/src/kernel/exec.js.map +1 -0
  40. package/dist/src/kernel/file-lock.d.ts +50 -0
  41. package/dist/src/kernel/file-lock.js +135 -0
  42. package/dist/src/kernel/file-lock.js.map +1 -0
  43. package/dist/src/kernel/nodes/agent.d.ts +4 -0
  44. package/dist/src/kernel/nodes/agent.js +35 -0
  45. package/dist/src/kernel/nodes/agent.js.map +1 -0
  46. package/dist/src/kernel/nodes/autoloop.d.ts +78 -0
  47. package/dist/src/kernel/nodes/autoloop.js +75 -0
  48. package/dist/src/kernel/nodes/autoloop.js.map +1 -0
  49. package/dist/src/kernel/nodes/council.d.ts +12 -0
  50. package/dist/src/kernel/nodes/council.js +88 -0
  51. package/dist/src/kernel/nodes/council.js.map +1 -0
  52. package/dist/src/kernel/nodes/fanout.d.ts +11 -0
  53. package/dist/src/kernel/nodes/fanout.js +63 -0
  54. package/dist/src/kernel/nodes/fanout.js.map +1 -0
  55. package/dist/src/kernel/nodes/human-gate.d.ts +4 -0
  56. package/dist/src/kernel/nodes/human-gate.js +7 -0
  57. package/dist/src/kernel/nodes/human-gate.js.map +1 -0
  58. package/dist/src/kernel/nodes/index.d.ts +12 -0
  59. package/dist/src/kernel/nodes/index.js +21 -0
  60. package/dist/src/kernel/nodes/index.js.map +1 -0
  61. package/dist/src/kernel/nodes/router.d.ts +4 -0
  62. package/dist/src/kernel/nodes/router.js +12 -0
  63. package/dist/src/kernel/nodes/router.js.map +1 -0
  64. package/dist/src/kernel/nodes/subflow.d.ts +13 -0
  65. package/dist/src/kernel/nodes/subflow.js +38 -0
  66. package/dist/src/kernel/nodes/subflow.js.map +1 -0
  67. package/dist/src/kernel/nodes/ultraapp.d.ts +60 -0
  68. package/dist/src/kernel/nodes/ultraapp.js +62 -0
  69. package/dist/src/kernel/nodes/ultraapp.js.map +1 -0
  70. package/dist/src/kernel/nodes/verifier.d.ts +14 -0
  71. package/dist/src/kernel/nodes/verifier.js +84 -0
  72. package/dist/src/kernel/nodes/verifier.js.map +1 -0
  73. package/dist/src/kernel/projections.d.ts +42 -0
  74. package/dist/src/kernel/projections.js +133 -0
  75. package/dist/src/kernel/projections.js.map +1 -0
  76. package/dist/src/kernel/repo.d.ts +13 -0
  77. package/dist/src/kernel/repo.js +64 -0
  78. package/dist/src/kernel/repo.js.map +1 -0
  79. package/dist/src/kernel/secrets.d.ts +25 -0
  80. package/dist/src/kernel/secrets.js +48 -0
  81. package/dist/src/kernel/secrets.js.map +1 -0
  82. package/dist/src/kernel/store.d.ts +225 -0
  83. package/dist/src/kernel/store.js +838 -0
  84. package/dist/src/kernel/store.js.map +1 -0
  85. package/dist/src/kernel/templates/index.d.ts +140 -0
  86. package/dist/src/kernel/templates/index.js +266 -0
  87. package/dist/src/kernel/templates/index.js.map +1 -0
  88. package/dist/src/kernel/types.d.ts +326 -0
  89. package/dist/src/kernel/types.js +19 -0
  90. package/dist/src/kernel/types.js.map +1 -0
  91. package/dist/src/models.d.ts +7 -0
  92. package/dist/src/models.js +43 -14
  93. package/dist/src/models.js.map +1 -1
  94. package/dist/src/persistent-custom-session.js +8 -3
  95. package/dist/src/persistent-custom-session.js.map +1 -1
  96. package/dist/src/run-ledger.d.ts +57 -3
  97. package/dist/src/run-ledger.js +45 -2
  98. package/dist/src/run-ledger.js.map +1 -1
  99. package/dist/src/session-manager.d.ts +176 -129
  100. package/dist/src/session-manager.js +652 -603
  101. package/dist/src/session-manager.js.map +1 -1
  102. package/dist/src/types.d.ts +33 -3
  103. package/dist/src/ultraapp/build.d.ts +117 -3
  104. package/dist/src/ultraapp/build.js +319 -3
  105. package/dist/src/ultraapp/build.js.map +1 -1
  106. package/dist/src/ultraapp/contract.d.ts +52 -0
  107. package/dist/src/ultraapp/contract.js +83 -0
  108. package/dist/src/ultraapp/contract.js.map +1 -0
  109. package/dist/src/ultraapp/conventions.js +9 -2
  110. package/dist/src/ultraapp/conventions.js.map +1 -1
  111. package/dist/src/ultraapp/fix-on-failure.d.ts +21 -2
  112. package/dist/src/ultraapp/fix-on-failure.js +46 -62
  113. package/dist/src/ultraapp/fix-on-failure.js.map +1 -1
  114. package/dist/src/ultraapp/manager.d.ts +107 -2
  115. package/dist/src/ultraapp/manager.js +305 -86
  116. package/dist/src/ultraapp/manager.js.map +1 -1
  117. package/dist/src/verify/baseline.d.ts +73 -0
  118. package/dist/src/verify/baseline.js +186 -0
  119. package/dist/src/verify/baseline.js.map +1 -0
  120. package/dist/src/verify/contract.d.ts +116 -0
  121. package/dist/src/verify/contract.js +142 -0
  122. package/dist/src/verify/contract.js.map +1 -0
  123. package/dist/src/verify/evidence.d.ts +61 -0
  124. package/dist/src/verify/evidence.js +133 -0
  125. package/dist/src/verify/evidence.js.map +1 -0
  126. package/dist/src/verify/runner.d.ts +63 -0
  127. package/dist/src/verify/runner.js +317 -0
  128. package/dist/src/verify/runner.js.map +1 -0
  129. package/openclaw.plugin.json +38 -1
  130. package/package.json +2 -2
  131. package/skills/SKILL.md +120 -79
  132. package/skills/references/acp.md +17 -17
  133. package/skills/references/autoloop.md +139 -65
  134. package/skills/references/claude-cli-tracking.md +4 -4
  135. package/skills/references/cli.md +101 -59
  136. package/skills/references/council.md +109 -37
  137. package/skills/references/dashboard.md +34 -6
  138. package/skills/references/getting-started.md +13 -13
  139. package/skills/references/inbox.md +4 -4
  140. package/skills/references/mcp.md +39 -34
  141. package/skills/references/multi-engine.md +51 -47
  142. package/skills/references/observability.md +115 -28
  143. package/skills/references/openai-compat.md +39 -39
  144. package/skills/references/sessions.md +43 -25
  145. package/skills/references/tools.md +402 -309
  146. package/skills/references/ultra.md +45 -45
  147. package/skills/references/ultraapp.md +126 -50
  148. package/skills/references/verification.md +187 -0
  149. package/skills/references/workflow.md +362 -0
  150. package/dist/src/ultraapp/fix-on-failure-session.d.ts +0 -23
  151. package/dist/src/ultraapp/fix-on-failure-session.js +0 -51
  152. package/dist/src/ultraapp/fix-on-failure-session.js.map +0 -1
@@ -0,0 +1,838 @@
1
+ /**
2
+ * Durable run store.
3
+ *
4
+ * Layout, one directory per run under `~/.claw-orchestrator/wf/<runId>/`:
5
+ *
6
+ * spec.json the WorkflowSpec, written once at creation, never mutated
7
+ * run.json the mutable checkpoint, rewritten atomically on every transition
8
+ * events.jsonl append-only audit + stream source
9
+ * nodes/<id>/ per-node artifacts
10
+ * evidence/<id>/ evidence bundles (see verify/evidence.ts)
11
+ *
12
+ * Splitting the immutable spec from the mutable checkpoint is what carries crash
13
+ * recovery: if `run.json` is missing or half-written, the state is rebuilt by
14
+ * replaying `events.jsonl` against `spec.json`. The atomic rewrite makes that
15
+ * path rare; the replay makes it usually survivable.
16
+ *
17
+ * The checkpoint and the events it describes are written by the same
18
+ * transaction, so they cannot disagree: a commit that reports `committed` has
19
+ * both, and one that does not has neither. That was not always true — the event
20
+ * append used to swallow its own errors while the replay treated the same log as
21
+ * authoritative, which is a contradiction sitting in the recovery path. The
22
+ * replay is now a plain fallback for a lost `run.json`; a run that lost both it
23
+ * and the log is unrecoverable, and `loadRun` returns undefined rather than
24
+ * inventing a state.
25
+ *
26
+ * There is exactly one way to write to a run — `commit()` — and it is not
27
+ * optional: the checkpoint writer and the event appender are module-private, and
28
+ * a batch is published by a single atomic directory rename rather than by a
29
+ * sequence of writes that can each fail on their own. `atomicWriteJson` replaces
30
+ * four near-identical implementations that had grown across the codebase
31
+ * (`session-manager.ts` sync + async variants, `ultraapp/store.ts`,
32
+ * `ultraapp/patcher.ts`); the tmp-name shape is kept from the ultraapp one,
33
+ * whose comment records the CI bug it fixed: a reader catching a plain
34
+ * `writeFile` mid-flight and parsing a truncated object.
35
+ */
36
+ import crypto from 'node:crypto';
37
+ import fs from 'node:fs';
38
+ import os from 'node:os';
39
+ import path from 'node:path';
40
+ import { withFileLock } from './file-lock.js';
41
+ export function wfDir() {
42
+ return process.env.CLAWO_WF_DIR || path.join(os.homedir(), '.claw-orchestrator', 'wf');
43
+ }
44
+ /**
45
+ * A run id is a single path segment: letters, digits, dot, dash, underscore.
46
+ *
47
+ * It has to be enforced, not merely expected. A run id can be supplied by the
48
+ * caller — including through a tool call, which means through an agent — and
49
+ * every path in this module is derived from it by `path.join`. `../escaped`
50
+ * resolves outside the store, and `deleteRunDir` is a recursive `rmSync`, so an
51
+ * unvalidated id turns a delete into arbitrary directory removal. A leading dot
52
+ * is refused too, so no id can produce `.` or `..` by itself.
53
+ */
54
+ const VALID_RUN_ID = /^[A-Za-z0-9_][A-Za-z0-9._-]{0,127}$/;
55
+ export function isValidRunId(runId) {
56
+ return typeof runId === 'string' && VALID_RUN_ID.test(runId);
57
+ }
58
+ /**
59
+ * Resolve a run's directory. Throws on an invalid id rather than returning a
60
+ * path, because this is the chokepoint every other path derives from — checking
61
+ * here means no caller can forget to.
62
+ */
63
+ export function runDir(runId) {
64
+ if (!isValidRunId(runId)) {
65
+ throw new Error(`Invalid run id ${JSON.stringify(runId)}: must match ${VALID_RUN_ID.source} (a single path segment)`);
66
+ }
67
+ const dir = path.join(wfDir(), runId);
68
+ // Belt to the regex's braces: if any future change to the pattern lets a
69
+ // separator through, this still refuses to hand back a path outside the root.
70
+ const root = wfDir();
71
+ if (path.dirname(dir) !== root) {
72
+ throw new Error(`Invalid run id ${JSON.stringify(runId)}: resolves outside the run store`);
73
+ }
74
+ return dir;
75
+ }
76
+ export function nodeDir(runId, nodeId) {
77
+ return path.join(runDir(runId), 'nodes', nodeId.replace(/[^\w.-]/g, '_'));
78
+ }
79
+ // ─── Shared write primitives ────────────────────────────────────────────────
80
+ /** Write JSON so a concurrent reader sees either the old file or the new one, never a partial. */
81
+ export function atomicWriteJson(file, value) {
82
+ const tmp = `${file}.tmp.${process.pid}.${crypto.randomBytes(4).toString('hex')}`;
83
+ fs.mkdirSync(path.dirname(file), { recursive: true });
84
+ try {
85
+ fs.writeFileSync(tmp, JSON.stringify(value, null, 2));
86
+ fs.renameSync(tmp, file);
87
+ }
88
+ catch (err) {
89
+ try {
90
+ fs.unlinkSync(tmp);
91
+ }
92
+ catch {
93
+ // Nothing to clean up.
94
+ }
95
+ throw err;
96
+ }
97
+ }
98
+ function readJson(file) {
99
+ try {
100
+ return JSON.parse(fs.readFileSync(file, 'utf8'));
101
+ }
102
+ catch {
103
+ return undefined;
104
+ }
105
+ }
106
+ // ─── Run lifecycle ──────────────────────────────────────────────────────────
107
+ export function runExists(runId) {
108
+ try {
109
+ return fs.existsSync(path.join(runDir(runId), 'spec.json'));
110
+ }
111
+ catch {
112
+ return false;
113
+ }
114
+ }
115
+ /**
116
+ * Keys whose values never reach disk.
117
+ *
118
+ * `customEngine` carries a `CustomEngineConfig`, whose `env` is explicitly for
119
+ * environment variables — API tokens included. An autoloop started with a custom
120
+ * engine wrote its whole options object into the node spec, and `spec.json`
121
+ * ended up holding the token in plain text.
122
+ *
123
+ * The real fix is to route them through `StartOptions.secrets`, which stays in
124
+ * memory. This is the second line: even a spec that should not contain one is
125
+ * scrubbed on the way out, so a future field cannot leak by omission.
126
+ */
127
+ const NEVER_PERSIST = new Set(['customengine', 'plannercustomengine', 'codercustomengine', 'reviewercustomengine']);
128
+ export function sanitizeForDisk(value) {
129
+ if (Array.isArray(value))
130
+ return value.map((v) => sanitizeForDisk(v));
131
+ if (!value || typeof value !== 'object')
132
+ return value;
133
+ const out = {};
134
+ for (const [key, v] of Object.entries(value)) {
135
+ if (NEVER_PERSIST.has(key.toLowerCase()))
136
+ continue;
137
+ out[key] = sanitizeForDisk(v);
138
+ }
139
+ return out;
140
+ }
141
+ export function loadSpec(runId) {
142
+ // Lookups treat an invalid id as "no such run" rather than throwing: a query
143
+ // for a nonsense id has an answer, and it is "nothing".
144
+ if (!isValidRunId(runId))
145
+ return undefined;
146
+ return readJson(path.join(runDir(runId), 'spec.json'));
147
+ }
148
+ export function readEvents(runId, limit) {
149
+ if (!isValidRunId(runId))
150
+ return [];
151
+ // A published transaction is authoritative even before its files have been
152
+ // applied. Returning the old log while `.tx` is still present would expose a
153
+ // state from before a commit that already succeeded.
154
+ if (!recoverPending(runId)) {
155
+ throw new Error(`Run '${runId}' has a committed transaction that could not be applied`);
156
+ }
157
+ let raw;
158
+ try {
159
+ raw = fs.readFileSync(path.join(runDir(runId), 'events.jsonl'), 'utf8');
160
+ }
161
+ catch {
162
+ return [];
163
+ }
164
+ const out = [];
165
+ for (const line of raw.split('\n')) {
166
+ const t = line.trim();
167
+ if (!t)
168
+ continue;
169
+ try {
170
+ out.push(JSON.parse(t));
171
+ }
172
+ catch {
173
+ // One corrupt line must not make the stream unreadable.
174
+ }
175
+ }
176
+ return limit && limit > 0 ? out.slice(-limit) : out;
177
+ }
178
+ /**
179
+ * Rebuild a run's state from its spec plus its event log. Used when `run.json`
180
+ * is absent or unparseable — the belt to the atomic rewrite's braces.
181
+ */
182
+ export function replayRun(runId, spec) {
183
+ const events = readEvents(runId);
184
+ if (events.length === 0)
185
+ return undefined;
186
+ const nodes = {};
187
+ for (const n of spec.nodes) {
188
+ nodes[n.id] = { id: n.id, kind: n.kind, state: 'pending', attempts: 0, visits: 0 };
189
+ }
190
+ const created = events.find((e) => e.type === 'run_created');
191
+ const record = {
192
+ runId,
193
+ workflow: spec.name,
194
+ spec,
195
+ state: 'pending',
196
+ outcome: 'unverified',
197
+ cwd: spec.cwd || process.cwd(),
198
+ createdAt: created?.ts ?? events[0].ts,
199
+ updatedAt: events[events.length - 1].ts,
200
+ nodes,
201
+ };
202
+ for (const e of events) {
203
+ switch (e.type) {
204
+ case 'run_state':
205
+ record.state = e.state;
206
+ if (e.outcome)
207
+ record.outcome = e.outcome;
208
+ if (e.error)
209
+ record.error = e.error;
210
+ break;
211
+ case 'node_state': {
212
+ const n = (nodes[e.node] ??= { id: e.node, kind: 'agent', state: 'pending', attempts: 0, visits: 0 });
213
+ n.state = e.state;
214
+ if (e.attempt !== undefined)
215
+ n.attempts = e.attempt;
216
+ if (e.state === 'running') {
217
+ n.visits++;
218
+ n.startedAt = e.ts;
219
+ }
220
+ if (e.state === 'succeeded' || e.state === 'failed' || e.state === 'skipped')
221
+ n.endedAt = e.ts;
222
+ if (e.error)
223
+ n.error = e.error;
224
+ record.currentNode = e.node;
225
+ break;
226
+ }
227
+ case 'node_output': {
228
+ const n = nodes[e.node];
229
+ if (n)
230
+ n.output = e.text;
231
+ break;
232
+ }
233
+ case 'evidence': {
234
+ const n = nodes[e.node];
235
+ if (n)
236
+ n.evidenceId = e.evidenceId;
237
+ record.evidenceId = e.evidenceId;
238
+ break;
239
+ }
240
+ default:
241
+ break;
242
+ }
243
+ record.updatedAt = e.ts;
244
+ }
245
+ // A run whose process died mid-flight left no terminal event. Say so rather
246
+ // than reporting the stale `running` it was last seen in.
247
+ if (record.state === 'running' || record.state === 'verifying') {
248
+ record.error = record.error ?? 'process ended before the run reached a terminal state';
249
+ }
250
+ return record;
251
+ }
252
+ /** Load a run, preferring the checkpoint and falling back to an event replay. */
253
+ export function loadRun(runId) {
254
+ const spec = loadSpec(runId);
255
+ if (!spec)
256
+ return undefined;
257
+ // A transaction the last owner committed but died before applying is finished
258
+ // here, so a reader never sees the state from before a committed change.
259
+ if (!recoverPending(runId)) {
260
+ throw new Error(`Run '${runId}' has a committed transaction that could not be applied`);
261
+ }
262
+ const checkpoint = readJson(path.join(runDir(runId), 'run.json'));
263
+ if (checkpoint && checkpoint.runId === runId && checkpoint.nodes) {
264
+ checkpoint.spec = checkpoint.spec ?? spec;
265
+ return checkpoint;
266
+ }
267
+ return replayRun(runId, spec);
268
+ }
269
+ export function listRunIds() {
270
+ try {
271
+ return fs
272
+ .readdirSync(wfDir(), { withFileTypes: true })
273
+ .filter((d) => d.isDirectory())
274
+ .map((d) => d.name)
275
+ .filter(isValidRunId);
276
+ }
277
+ catch {
278
+ return [];
279
+ }
280
+ }
281
+ /**
282
+ * Remove a run's directory. Only ever called from an explicit user delete —
283
+ * nothing in the kernel prunes runs on its own, for the same reason the ledger
284
+ * does not: deleting someone's history is not ours to decide.
285
+ */
286
+ export function deleteRunDir(runId, logger) {
287
+ // Validate before the try, so a bad id surfaces as an error instead of being
288
+ // swallowed alongside genuine I/O failures. This is a recursive rmSync — the
289
+ // one call in this module where a bad id is not merely wrong but destructive.
290
+ const dir = runDir(runId);
291
+ try {
292
+ fs.rmSync(dir, { recursive: true, force: true });
293
+ }
294
+ catch (err) {
295
+ logger?.warn?.(`[kernel-store] delete failed for ${runId}: ${err.message}`);
296
+ }
297
+ }
298
+ /**
299
+ * Full text a node produced, preferring the artifact file over the record's
300
+ * preview. `NodeRecord.output` is capped so a checkpoint stays small, but the
301
+ * whole point of some nodes is their text — an ultraplan whose plan is silently
302
+ * cut at 4 kB has lost the deliverable.
303
+ */
304
+ export function readNodeOutput(runId, nodeId) {
305
+ try {
306
+ return fs.readFileSync(path.join(nodeDir(runId, nodeId), 'output.txt'), 'utf8');
307
+ }
308
+ catch {
309
+ return loadRun(runId)?.nodes[nodeId]?.output;
310
+ }
311
+ }
312
+ /**
313
+ * Where a node artifact will live, without writing it.
314
+ *
315
+ * Pure, because the checkpoint that references an artifact and the artifact
316
+ * itself are written in the same {@link commit}: the caller needs the path
317
+ * before the write, to put it in the record it is committing.
318
+ */
319
+ export function nodeArtifactPath(runId, nodeId, name) {
320
+ const file = path.join(nodeDir(runId, nodeId), name.replace(/[^\w.-]/g, '_'));
321
+ return path.relative(runDir(runId), file);
322
+ }
323
+ // ─── Ownership and the storage transaction ──────────────────────────────────
324
+ //
325
+ // A run has one owner at a time, every durable change it makes is checked
326
+ // against a capability only that owner holds, and each change lands in full or
327
+ // not at all. Each of those three was added after the previous two turned out to
328
+ // be decoration without it.
329
+ //
330
+ // The capability is a `RunGuard`, naming four things, all checked on every
331
+ // write:
332
+ //
333
+ // incarnationId which *creation* of this run id this is
334
+ // ownerId which RunKernel instance (never the pid — see `RunLease`)
335
+ // acquisitionId which claim by that owner
336
+ // fence monotonic within the incarnation, for ordering and logs
337
+ //
338
+ // `incarnationId` exists because a run id is reusable. Delete a run and start
339
+ // another with the same id and the directory — fence counter included — is
340
+ // gone, so the new run's first fence is 1 again and a stale attempt still
341
+ // holding `{ownerId, fence: 1}` from the old run became valid a second time.
342
+ // That is a textbook ABA, and not hypothetical: an abandoned timed-out attempt
343
+ // outlives its run by construction.
344
+ //
345
+ // Two distinctions this module is careful about, because collapsing either one
346
+ // caused a real failure:
347
+ //
348
+ // *Not the owner* is permanent; *could not take the lock* is transient. They
349
+ // are separate outcomes (`superseded` vs `blocked`), and a caller that treats
350
+ // a millisecond of contention as a loss stops forever while still holding its
351
+ // lease — so nothing can take the run over either.
352
+ //
353
+ // *Committed* means the whole batch is durable, not that most of it was
354
+ // attempted. A batch is staged in a scratch directory and published by a
355
+ // single atomic directory rename; the rename is the commit point, and what
356
+ // follows is replayable application of an already-committed transaction.
357
+ export const LEASE_TTL_MS = 60_000;
358
+ /** How often a live owner refreshes its claim, independently of any work it is doing. */
359
+ export const LEASE_HEARTBEAT_MS = 15_000;
360
+ /** Staged, not yet committed. Removed on the next lock; never read. */
361
+ const TX_STAGING = '.tx.staging';
362
+ /** Committed and awaiting application. Its presence IS the commit. */
363
+ const TX_COMMITTED = '.tx';
364
+ function leaseFile(runId) {
365
+ return path.join(runDir(runId), 'lease.json');
366
+ }
367
+ function lockFile(runId) {
368
+ return path.join(runDir(runId), 'lease.lock');
369
+ }
370
+ function incarnationFile(runId) {
371
+ return path.join(runDir(runId), 'incarnation.json');
372
+ }
373
+ function eventsFile(runId) {
374
+ return path.join(runDir(runId), 'events.jsonl');
375
+ }
376
+ function fileSize(file) {
377
+ try {
378
+ return fs.statSync(file).size;
379
+ }
380
+ catch {
381
+ return 0;
382
+ }
383
+ }
384
+ function readIncarnation(runId) {
385
+ const raw = readJson(incarnationFile(runId));
386
+ return raw && typeof raw.incarnationId === 'string' && raw.incarnationId ? raw : undefined;
387
+ }
388
+ /**
389
+ * The run's incarnation, minting one if the directory predates this field.
390
+ *
391
+ * Only ever called inside the lock. A directory with no incarnation is a run
392
+ * from an older layout; giving it one now is correct, because whatever guards
393
+ * existed before this file did are already unusable.
394
+ */
395
+ function ensureIncarnationLocked(runId) {
396
+ const existing = readIncarnation(runId);
397
+ if (existing)
398
+ return existing;
399
+ const fresh = { incarnationId: crypto.randomUUID(), nextFence: 0 };
400
+ atomicWriteJson(incarnationFile(runId), fresh);
401
+ return fresh;
402
+ }
403
+ function processAlive(pid) {
404
+ try {
405
+ process.kill(pid, 0);
406
+ return true;
407
+ }
408
+ catch (err) {
409
+ // EPERM means it exists and belongs to someone else.
410
+ return err.code === 'EPERM';
411
+ }
412
+ }
413
+ export function readLease(runId) {
414
+ if (!isValidRunId(runId))
415
+ return undefined;
416
+ return readJson(leaseFile(runId));
417
+ }
418
+ /**
419
+ * True when a lease no longer represents a live owner.
420
+ *
421
+ * On the same host the pid is the authority and the heartbeat is not consulted:
422
+ * a process running one long node makes no checkpoints, and judging it dead for
423
+ * that reason is how two owners end up running the same work. The heartbeat is
424
+ * the fallback for a holder we cannot ask about — another machine, or a pid
425
+ * whose meaning we cannot trust across hosts.
426
+ */
427
+ export function leaseIsStale(lease, now = Date.now()) {
428
+ if (!lease)
429
+ return true;
430
+ if (lease.host === os.hostname())
431
+ return !processAlive(lease.pid);
432
+ const age = now - Date.parse(lease.renewedAt);
433
+ return Number.isNaN(age) || age > LEASE_TTL_MS;
434
+ }
435
+ /**
436
+ * Apply a committed transaction. Idempotent, and safe to re-run after a crash.
437
+ *
438
+ * Caller holds the lock.
439
+ */
440
+ function applyTxLocked(runId, logger) {
441
+ const dir = runDir(runId);
442
+ const tx = path.join(dir, TX_COMMITTED);
443
+ const manifest = readJson(path.join(tx, 'manifest.json'));
444
+ if (!manifest) {
445
+ // No manifest means the rename landed a half-written staging directory,
446
+ // which cannot happen — but if it somehow did, it is not a commitment.
447
+ fs.rmSync(tx, { recursive: true, force: true });
448
+ return;
449
+ }
450
+ // Written after the last data step and before cleanup, so a failure while
451
+ // removing the directory leaves something that says "the data is in, only the
452
+ // tidying is left". Without it, a half-removed transaction directory looks
453
+ // exactly like a corrupt one, and re-applying would either throw forever or
454
+ // have to guess.
455
+ const appliedMarker = path.join(tx, 'applied');
456
+ if (fs.existsSync(appliedMarker)) {
457
+ fs.rmSync(tx, { recursive: true, force: true });
458
+ return;
459
+ }
460
+ for (const rel of manifest.artifacts) {
461
+ const from = path.join(tx, 'files', rel);
462
+ const to = path.join(dir, rel);
463
+ try {
464
+ fs.mkdirSync(path.dirname(to), { recursive: true });
465
+ fs.renameSync(from, to);
466
+ }
467
+ catch (err) {
468
+ // ENOENT means a previous application already moved it.
469
+ if (err.code !== 'ENOENT')
470
+ throw err;
471
+ }
472
+ }
473
+ if (manifest.hasEvents) {
474
+ const events = eventsFile(runId);
475
+ try {
476
+ fs.truncateSync(events, manifest.eventsOffset);
477
+ }
478
+ catch (err) {
479
+ if (err.code !== 'ENOENT')
480
+ throw err;
481
+ }
482
+ fs.appendFileSync(events, fs.readFileSync(path.join(tx, 'events.jsonl')));
483
+ }
484
+ if (manifest.hasRecord) {
485
+ try {
486
+ fs.renameSync(path.join(tx, 'run.json'), path.join(dir, 'run.json'));
487
+ }
488
+ catch (err) {
489
+ if (err.code !== 'ENOENT')
490
+ throw err;
491
+ }
492
+ }
493
+ fs.writeFileSync(appliedMarker, '');
494
+ fs.rmSync(tx, { recursive: true, force: true });
495
+ logger?.debug?.(`[kernel-store] ${runId}: applied a pending transaction`);
496
+ }
497
+ /** Stage a batch and publish it with one atomic rename. Caller holds the lock. */
498
+ function stageLocked(runId, batch) {
499
+ const dir = runDir(runId);
500
+ const staging = path.join(dir, TX_STAGING);
501
+ fs.rmSync(staging, { recursive: true, force: true });
502
+ fs.mkdirSync(staging, { recursive: true });
503
+ const manifest = {
504
+ eventsOffset: fileSize(eventsFile(runId)),
505
+ hasRecord: Boolean(batch.record),
506
+ hasEvents: Boolean(batch.events?.length),
507
+ artifacts: [],
508
+ };
509
+ for (const artifact of batch.artifacts ?? []) {
510
+ const rel = nodeArtifactPath(runId, artifact.nodeId, artifact.name);
511
+ const dest = path.join(staging, 'files', rel);
512
+ fs.mkdirSync(path.dirname(dest), { recursive: true });
513
+ fs.writeFileSync(dest, artifact.body);
514
+ manifest.artifacts.push(rel);
515
+ }
516
+ if (batch.record) {
517
+ // The checkpoint embeds the spec, so it gets the same scrub.
518
+ fs.writeFileSync(path.join(staging, 'run.json'), JSON.stringify(sanitizeForDisk(batch.record), null, 2));
519
+ }
520
+ if (batch.events?.length) {
521
+ fs.writeFileSync(path.join(staging, 'events.jsonl'), batch.events.map((e) => JSON.stringify(e)).join('\n') + '\n');
522
+ }
523
+ fs.writeFileSync(path.join(staging, 'manifest.json'), JSON.stringify(manifest));
524
+ // THE COMMIT POINT. Renaming a directory is atomic, so the transaction is
525
+ // either wholly present or wholly absent — never a checkpoint whose events
526
+ // were silently dropped, or an artifact left behind by a write that failed
527
+ // afterwards. Both of those were real: the event append swallowed its own
528
+ // errors, and artifacts were written before the checkpoint they belonged to.
529
+ fs.renameSync(staging, path.join(dir, TX_COMMITTED));
530
+ }
531
+ /** Finish any transaction that was committed but not yet applied. Caller holds the lock. */
532
+ function recoverPendingLocked(runId, logger) {
533
+ const dir = runDir(runId);
534
+ try {
535
+ // Staging that outlived its process was never committed.
536
+ fs.rmSync(path.join(dir, TX_STAGING), { recursive: true, force: true });
537
+ }
538
+ catch {
539
+ // Debris; the next attempt overwrites it.
540
+ }
541
+ if (!fs.existsSync(path.join(dir, TX_COMMITTED)))
542
+ return;
543
+ try {
544
+ applyTxLocked(runId, logger);
545
+ }
546
+ catch (err) {
547
+ logger?.warn?.(`[kernel-store] ${runId}: could not apply a pending transaction: ${err.message}`);
548
+ }
549
+ }
550
+ /**
551
+ * Run `fn` inside the run's lock, with any pending transaction applied first.
552
+ *
553
+ * Everything that inspects or changes ownership goes through here, so a check
554
+ * and the write that depends on it cannot be separated by another process — and
555
+ * so no caller can observe a run mid-transaction.
556
+ */
557
+ function withRunLock(runId, fn, logger) {
558
+ return withFileLock(lockFile(runId), () => {
559
+ recoverPendingLocked(runId, logger);
560
+ return fn();
561
+ },
562
+ // Never `createParent`: a run directory that has been deleted must stay
563
+ // deleted. Recreating it to hold a lock left an empty directory behind, and
564
+ // the id then looked taken by a run that no longer existed.
565
+ { staleMs: LEASE_TTL_MS });
566
+ }
567
+ /** Whether the run directory is still there at all. A gone run is not contention. */
568
+ function runDirExists(runId) {
569
+ try {
570
+ return fs.existsSync(runDir(runId));
571
+ }
572
+ catch {
573
+ return false;
574
+ }
575
+ }
576
+ /** Apply anything a crashed owner committed but did not finish writing. */
577
+ export function recoverPending(runId, logger) {
578
+ // Nothing to apply is not the same as something stuck: an id that cannot name
579
+ // a run has no pending transaction, and readers must not be told otherwise.
580
+ if (!isValidRunId(runId))
581
+ return true;
582
+ try {
583
+ if (!fs.existsSync(path.join(runDir(runId), TX_COMMITTED)))
584
+ return true;
585
+ }
586
+ catch {
587
+ return true;
588
+ }
589
+ // A transaction is applied by whoever holds the lock, so finding the lock busy
590
+ // means someone is very likely applying it right now. Retry before concluding
591
+ // anything: reporting "cannot be applied" for a moment of contention would
592
+ // turn a healthy commit into a read error, which is the same conflation this
593
+ // module exists to avoid.
594
+ for (let attempt = 0; attempt < 3; attempt++) {
595
+ const locked = withRunLock(runId, () => undefined, logger);
596
+ if (locked.ok)
597
+ break;
598
+ if (!fs.existsSync(path.join(runDir(runId), TX_COMMITTED)))
599
+ return true;
600
+ }
601
+ // `recoverPendingLocked` deliberately leaves a committed transaction in
602
+ // place when application fails. Its continued presence means returning the
603
+ // old checkpoint or event log would be a lie.
604
+ return !fs.existsSync(path.join(runDir(runId), TX_COMMITTED));
605
+ }
606
+ // ─── Claiming ───────────────────────────────────────────────────────────────
607
+ function acquireLocked(runId, ownerId) {
608
+ const incarnation = ensureIncarnationLocked(runId);
609
+ const existing = readLease(runId);
610
+ if (existing && existing.ownerId !== ownerId && !leaseIsStale(existing)) {
611
+ throw new Error(`Run '${runId}' is owned by ${existing.ownerId} (pid ${existing.pid} on ${existing.host}, ` +
612
+ `last seen ${existing.renewedAt}) — only one owner may run it at a time`);
613
+ }
614
+ const fence = incarnation.nextFence + 1;
615
+ atomicWriteJson(incarnationFile(runId), { ...incarnation, nextFence: fence });
616
+ const now = new Date().toISOString();
617
+ const lease = {
618
+ runId,
619
+ incarnationId: incarnation.incarnationId,
620
+ ownerId,
621
+ acquisitionId: crypto.randomUUID(),
622
+ fence,
623
+ pid: process.pid,
624
+ host: os.hostname(),
625
+ acquiredAt: existing?.ownerId === ownerId ? existing.acquiredAt : now,
626
+ renewedAt: now,
627
+ };
628
+ atomicWriteJson(leaseFile(runId), lease);
629
+ return { runId, incarnationId: lease.incarnationId, ownerId, acquisitionId: lease.acquisitionId, fence };
630
+ }
631
+ /**
632
+ * Create a run and claim it, atomically.
633
+ *
634
+ * The non-recursive `mkdir` IS the claim: it fails with EEXIST for everyone but
635
+ * the first caller, so exactly one process can create a given run id. The
636
+ * previous version asked `runExists()` and then created with
637
+ * `{ recursive: true }`, which is a check-then-write race and lost it routinely
638
+ * — two processes creating the same id 80 times had both "succeed" 76 times,
639
+ * leaving one workflow executing while the other's `spec.json` sat on disk, and
640
+ * a lease belonging to an incarnation that had already been overwritten.
641
+ */
642
+ export function createAndAcquire(runId, spec, ownerId) {
643
+ const dir = runDir(runId);
644
+ fs.mkdirSync(wfDir(), { recursive: true });
645
+ try {
646
+ fs.mkdirSync(dir);
647
+ }
648
+ catch (err) {
649
+ if (err.code === 'EEXIST') {
650
+ throw new Error(`Run '${runId}' already exists — pick another id, or resume it instead of starting over`);
651
+ }
652
+ throw err;
653
+ }
654
+ // Nobody else can be inside a directory that did not exist a moment ago, so
655
+ // this lock is uncontended by construction; taking it anyway keeps every
656
+ // write to the run under the same discipline.
657
+ const locked = withRunLock(runId, () => {
658
+ atomicWriteJson(incarnationFile(runId), { incarnationId: crypto.randomUUID(), nextFence: 0 });
659
+ atomicWriteJson(path.join(dir, 'spec.json'), sanitizeForDisk(spec));
660
+ return acquireLocked(runId, ownerId);
661
+ });
662
+ if (!locked.ok) {
663
+ fs.rmSync(dir, { recursive: true, force: true });
664
+ throw new Error(`Run '${runId}' could not be created: ${locked.error}`);
665
+ }
666
+ return locked.value;
667
+ }
668
+ /**
669
+ * Claim an existing run and receive the capability to write to it.
670
+ *
671
+ * Refused when someone else's claim is still live. Allowed for the same owner —
672
+ * but the guard it returns is a NEW one, and the previous guard is dead from
673
+ * that moment: re-acquiring is a fresh claim, not a renewal of the old one, so
674
+ * anything still holding the old guard stops being able to write.
675
+ */
676
+ export function acquireLease(runId, ownerId) {
677
+ if (!runDirExists(runId))
678
+ throw new Error(`Run '${runId}' not found`);
679
+ const locked = withRunLock(runId, () => acquireLocked(runId, ownerId));
680
+ if (!locked.ok) {
681
+ throw new Error(`Run '${runId}' could not be claimed right now: ${locked.error}`);
682
+ }
683
+ return locked.value;
684
+ }
685
+ /** Whether the on-disk lease still answers to this guard. Caller must hold the lock. */
686
+ function guardIsCurrentLocked(guard) {
687
+ const incarnation = readIncarnation(guard.runId);
688
+ if (!incarnation || incarnation.incarnationId !== guard.incarnationId)
689
+ return false;
690
+ const lease = readLease(guard.runId);
691
+ return Boolean(lease &&
692
+ lease.incarnationId === guard.incarnationId &&
693
+ lease.ownerId === guard.ownerId &&
694
+ lease.acquisitionId === guard.acquisitionId &&
695
+ lease.fence === guard.fence);
696
+ }
697
+ /** Heartbeat. Best-effort: losing a renewal must not break the run. */
698
+ export function renewLease(guard) {
699
+ if (!runDirExists(guard.runId))
700
+ return;
701
+ withRunLock(guard.runId, () => {
702
+ if (!guardIsCurrentLocked(guard))
703
+ return;
704
+ const existing = readLease(guard.runId);
705
+ try {
706
+ atomicWriteJson(leaseFile(guard.runId), { ...existing, renewedAt: new Date().toISOString() });
707
+ }
708
+ catch {
709
+ // The next renewal will try again.
710
+ }
711
+ });
712
+ }
713
+ /**
714
+ * The one way to change anything durable about a run.
715
+ *
716
+ * The guard is verified and the transaction is published inside a single
717
+ * critical section, so a holder that has been superseded cannot land a write
718
+ * between the check and the change.
719
+ *
720
+ * Callers pass a record they have already produced by copying and mutating,
721
+ * never the record they are still using: `commit` persists what it is given, and
722
+ * the caller adopts it only on `committed`. That is what keeps a refused write
723
+ * from leaving a run *in memory* claiming a state the disk rejected.
724
+ */
725
+ export function commit(guard, batch, logger) {
726
+ // A deleted run is not contention, and must not be reported as something to
727
+ // retry: there is nothing left to write to, ever.
728
+ if (!runDirExists(guard.runId)) {
729
+ return { outcome: 'superseded', reason: `run '${guard.runId}' no longer exists` };
730
+ }
731
+ const locked = withRunLock(guard.runId, () => {
732
+ if (!guardIsCurrentLocked(guard)) {
733
+ const current = readLease(guard.runId);
734
+ return {
735
+ outcome: 'superseded',
736
+ reason: current
737
+ ? `run is now held by ${current.ownerId} at fence ${current.fence}`
738
+ : `run '${guard.runId}' no longer exists, or its claim was released`,
739
+ };
740
+ }
741
+ try {
742
+ stageLocked(guard.runId, batch);
743
+ }
744
+ catch (err) {
745
+ try {
746
+ fs.rmSync(path.join(runDir(guard.runId), TX_STAGING), { recursive: true, force: true });
747
+ }
748
+ catch {
749
+ // Debris only; nothing was published.
750
+ }
751
+ return { outcome: 'blocked', reason: `could not stage the change: ${err.message}` };
752
+ }
753
+ // Past the commit point. Application is replayable, so a failure here is a
754
+ // delay, not a loss — and reporting it as "not committed" would be wrong.
755
+ try {
756
+ applyTxLocked(guard.runId, logger);
757
+ }
758
+ catch (err) {
759
+ logger?.warn?.(`[kernel-store] ${guard.runId}: change is committed but not yet applied ` +
760
+ `(${err.message}); it will be applied on the next lock`);
761
+ }
762
+ try {
763
+ const lease = readLease(guard.runId);
764
+ atomicWriteJson(leaseFile(guard.runId), { ...lease, renewedAt: new Date().toISOString() });
765
+ }
766
+ catch {
767
+ // The heartbeat is not part of the commitment.
768
+ }
769
+ return { outcome: 'committed' };
770
+ }, logger);
771
+ return locked.ok ? locked.value : { outcome: 'blocked', reason: locked.error };
772
+ }
773
+ /**
774
+ * Release the claim. The incarnation file is deliberately NOT removed — the
775
+ * fence must keep increasing for as long as this creation of the run exists.
776
+ *
777
+ * `blocked` is not `not-ours`: an owner that is standing down and cannot take
778
+ * the lock still has to hand the run back, so the caller retries rather than
779
+ * leaving a lease behind that no one can take over.
780
+ */
781
+ export function releaseLease(guard) {
782
+ if (!runDirExists(guard.runId))
783
+ return 'not-ours';
784
+ const locked = withRunLock(guard.runId, () => {
785
+ if (!guardIsCurrentLocked(guard))
786
+ return 'not-ours';
787
+ fs.rmSync(leaseFile(guard.runId), { force: true });
788
+ return 'released';
789
+ });
790
+ return locked.ok ? locked.value : 'blocked';
791
+ }
792
+ export function summarize(record) {
793
+ return {
794
+ runId: record.runId,
795
+ workflow: record.workflow,
796
+ state: record.state,
797
+ outcome: record.outcome,
798
+ cwd: record.cwd,
799
+ createdAt: record.createdAt,
800
+ updatedAt: record.updatedAt,
801
+ endedAt: record.endedAt,
802
+ currentNode: record.currentNode,
803
+ evidenceId: record.evidenceId,
804
+ error: record.error,
805
+ costUsd: record.costUsd,
806
+ };
807
+ }
808
+ /**
809
+ * Every run on this machine, newest first — the cross-process listing the three
810
+ * mode-specific enumerations each did differently (council regex-scraped its own
811
+ * markdown transcripts, autoloop read a JSONL registry, ultraapp walked a store
812
+ * directory). No separate index file is needed here because every run lives
813
+ * under one root, so the directory *is* the index.
814
+ */
815
+ export function listRuns(query = {}) {
816
+ const out = [];
817
+ for (const runId of listRunIds()) {
818
+ let record;
819
+ try {
820
+ record = loadRun(runId);
821
+ }
822
+ catch {
823
+ // One run whose committed transaction cannot currently be applied must
824
+ // not make every other run disappear from the listing.
825
+ continue;
826
+ }
827
+ if (!record)
828
+ continue;
829
+ if (query.workflow && record.workflow !== query.workflow)
830
+ continue;
831
+ if (query.state && record.state !== query.state)
832
+ continue;
833
+ out.push(summarize(record));
834
+ }
835
+ out.sort((a, b) => (a.createdAt < b.createdAt ? 1 : a.createdAt > b.createdAt ? -1 : 0));
836
+ return query.limit && query.limit > 0 ? out.slice(0, query.limit) : out;
837
+ }
838
+ //# sourceMappingURL=store.js.map