@enderfga/claw-orchestrator 5.0.0 → 6.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (158) hide show
  1. package/README.md +28 -28
  2. package/dist/bin/cli.js +107 -1
  3. package/dist/bin/cli.js.map +1 -1
  4. package/dist/src/acp-server.d.ts +5 -5
  5. package/dist/src/acp-server.js +3 -3
  6. package/dist/src/acp-server.js.map +1 -1
  7. package/dist/src/autoloop/dispatcher.d.ts +22 -0
  8. package/dist/src/autoloop/dispatcher.js +71 -13
  9. package/dist/src/autoloop/dispatcher.js.map +1 -1
  10. package/dist/src/autoloop/messages.d.ts +10 -0
  11. package/dist/src/autoloop/messages.js.map +1 -1
  12. package/dist/src/autoloop/runner.js +6 -0
  13. package/dist/src/autoloop/runner.js.map +1 -1
  14. package/dist/src/constants.d.ts +0 -6
  15. package/dist/src/constants.js +0 -6
  16. package/dist/src/constants.js.map +1 -1
  17. package/dist/src/council.d.ts +15 -0
  18. package/dist/src/council.js +48 -35
  19. package/dist/src/council.js.map +1 -1
  20. package/dist/src/dashboard/index.html +191 -6
  21. package/dist/src/embedded-server.js +132 -9
  22. package/dist/src/embedded-server.js.map +1 -1
  23. package/dist/src/fanout.d.ts +30 -1
  24. package/dist/src/fanout.js +32 -3
  25. package/dist/src/fanout.js.map +1 -1
  26. package/dist/src/index.d.ts +1 -0
  27. package/dist/src/index.js +360 -4
  28. package/dist/src/index.js.map +1 -1
  29. package/dist/src/kernel/agent-step.d.ts +59 -0
  30. package/dist/src/kernel/agent-step.js +100 -0
  31. package/dist/src/kernel/agent-step.js.map +1 -0
  32. package/dist/src/kernel/conditions.d.ts +11 -0
  33. package/dist/src/kernel/conditions.js +24 -0
  34. package/dist/src/kernel/conditions.js.map +1 -0
  35. package/dist/src/kernel/engine.d.ts +319 -0
  36. package/dist/src/kernel/engine.js +1047 -0
  37. package/dist/src/kernel/engine.js.map +1 -0
  38. package/dist/src/kernel/exec.d.ts +43 -0
  39. package/dist/src/kernel/exec.js +112 -0
  40. package/dist/src/kernel/exec.js.map +1 -0
  41. package/dist/src/kernel/file-lock.d.ts +50 -0
  42. package/dist/src/kernel/file-lock.js +135 -0
  43. package/dist/src/kernel/file-lock.js.map +1 -0
  44. package/dist/src/kernel/nodes/agent.d.ts +4 -0
  45. package/dist/src/kernel/nodes/agent.js +35 -0
  46. package/dist/src/kernel/nodes/agent.js.map +1 -0
  47. package/dist/src/kernel/nodes/autoloop.d.ts +78 -0
  48. package/dist/src/kernel/nodes/autoloop.js +75 -0
  49. package/dist/src/kernel/nodes/autoloop.js.map +1 -0
  50. package/dist/src/kernel/nodes/council.d.ts +12 -0
  51. package/dist/src/kernel/nodes/council.js +88 -0
  52. package/dist/src/kernel/nodes/council.js.map +1 -0
  53. package/dist/src/kernel/nodes/fanout.d.ts +11 -0
  54. package/dist/src/kernel/nodes/fanout.js +63 -0
  55. package/dist/src/kernel/nodes/fanout.js.map +1 -0
  56. package/dist/src/kernel/nodes/human-gate.d.ts +4 -0
  57. package/dist/src/kernel/nodes/human-gate.js +7 -0
  58. package/dist/src/kernel/nodes/human-gate.js.map +1 -0
  59. package/dist/src/kernel/nodes/index.d.ts +12 -0
  60. package/dist/src/kernel/nodes/index.js +21 -0
  61. package/dist/src/kernel/nodes/index.js.map +1 -0
  62. package/dist/src/kernel/nodes/router.d.ts +4 -0
  63. package/dist/src/kernel/nodes/router.js +12 -0
  64. package/dist/src/kernel/nodes/router.js.map +1 -0
  65. package/dist/src/kernel/nodes/subflow.d.ts +13 -0
  66. package/dist/src/kernel/nodes/subflow.js +38 -0
  67. package/dist/src/kernel/nodes/subflow.js.map +1 -0
  68. package/dist/src/kernel/nodes/ultraapp.d.ts +60 -0
  69. package/dist/src/kernel/nodes/ultraapp.js +62 -0
  70. package/dist/src/kernel/nodes/ultraapp.js.map +1 -0
  71. package/dist/src/kernel/nodes/verifier.d.ts +14 -0
  72. package/dist/src/kernel/nodes/verifier.js +84 -0
  73. package/dist/src/kernel/nodes/verifier.js.map +1 -0
  74. package/dist/src/kernel/projections.d.ts +42 -0
  75. package/dist/src/kernel/projections.js +133 -0
  76. package/dist/src/kernel/projections.js.map +1 -0
  77. package/dist/src/kernel/repo.d.ts +13 -0
  78. package/dist/src/kernel/repo.js +64 -0
  79. package/dist/src/kernel/repo.js.map +1 -0
  80. package/dist/src/kernel/secrets.d.ts +25 -0
  81. package/dist/src/kernel/secrets.js +48 -0
  82. package/dist/src/kernel/secrets.js.map +1 -0
  83. package/dist/src/kernel/store.d.ts +225 -0
  84. package/dist/src/kernel/store.js +838 -0
  85. package/dist/src/kernel/store.js.map +1 -0
  86. package/dist/src/kernel/templates/index.d.ts +140 -0
  87. package/dist/src/kernel/templates/index.js +266 -0
  88. package/dist/src/kernel/templates/index.js.map +1 -0
  89. package/dist/src/kernel/types.d.ts +326 -0
  90. package/dist/src/kernel/types.js +19 -0
  91. package/dist/src/kernel/types.js.map +1 -0
  92. package/dist/src/models.d.ts +1 -1
  93. package/dist/src/models.js +31 -3
  94. package/dist/src/models.js.map +1 -1
  95. package/dist/src/persistent-cursor-session.js +6 -1
  96. package/dist/src/persistent-cursor-session.js.map +1 -1
  97. package/dist/src/persistent-grok-session.d.ts +40 -0
  98. package/dist/src/persistent-grok-session.js +197 -0
  99. package/dist/src/persistent-grok-session.js.map +1 -0
  100. package/dist/src/run-ledger.d.ts +57 -3
  101. package/dist/src/run-ledger.js +45 -2
  102. package/dist/src/run-ledger.js.map +1 -1
  103. package/dist/src/session-manager.d.ts +176 -129
  104. package/dist/src/session-manager.js +657 -603
  105. package/dist/src/session-manager.js.map +1 -1
  106. package/dist/src/types.d.ts +37 -4
  107. package/dist/src/types.js +15 -1
  108. package/dist/src/types.js.map +1 -1
  109. package/dist/src/ultraapp/build.d.ts +117 -3
  110. package/dist/src/ultraapp/build.js +319 -3
  111. package/dist/src/ultraapp/build.js.map +1 -1
  112. package/dist/src/ultraapp/contract.d.ts +52 -0
  113. package/dist/src/ultraapp/contract.js +83 -0
  114. package/dist/src/ultraapp/contract.js.map +1 -0
  115. package/dist/src/ultraapp/conventions.js +9 -2
  116. package/dist/src/ultraapp/conventions.js.map +1 -1
  117. package/dist/src/ultraapp/fix-on-failure.d.ts +21 -2
  118. package/dist/src/ultraapp/fix-on-failure.js +46 -62
  119. package/dist/src/ultraapp/fix-on-failure.js.map +1 -1
  120. package/dist/src/ultraapp/manager.d.ts +107 -2
  121. package/dist/src/ultraapp/manager.js +305 -86
  122. package/dist/src/ultraapp/manager.js.map +1 -1
  123. package/dist/src/verify/baseline.d.ts +73 -0
  124. package/dist/src/verify/baseline.js +186 -0
  125. package/dist/src/verify/baseline.js.map +1 -0
  126. package/dist/src/verify/contract.d.ts +116 -0
  127. package/dist/src/verify/contract.js +142 -0
  128. package/dist/src/verify/contract.js.map +1 -0
  129. package/dist/src/verify/evidence.d.ts +61 -0
  130. package/dist/src/verify/evidence.js +133 -0
  131. package/dist/src/verify/evidence.js.map +1 -0
  132. package/dist/src/verify/runner.d.ts +63 -0
  133. package/dist/src/verify/runner.js +317 -0
  134. package/dist/src/verify/runner.js.map +1 -0
  135. package/openclaw.plugin.json +8 -0
  136. package/package.json +2 -2
  137. package/skills/SKILL.md +121 -80
  138. package/skills/references/acp.md +18 -18
  139. package/skills/references/autoloop.md +148 -72
  140. package/skills/references/claude-cli-tracking.md +4 -4
  141. package/skills/references/cli.md +103 -60
  142. package/skills/references/council.md +109 -37
  143. package/skills/references/dashboard.md +34 -6
  144. package/skills/references/getting-started.md +14 -14
  145. package/skills/references/inbox.md +4 -4
  146. package/skills/references/mcp.md +39 -34
  147. package/skills/references/multi-engine.md +109 -51
  148. package/skills/references/observability.md +88 -27
  149. package/skills/references/openai-compat.md +40 -40
  150. package/skills/references/sessions.md +44 -26
  151. package/skills/references/tools.md +402 -309
  152. package/skills/references/ultra.md +45 -45
  153. package/skills/references/ultraapp.md +126 -50
  154. package/skills/references/verification.md +187 -0
  155. package/skills/references/workflow.md +362 -0
  156. package/dist/src/ultraapp/fix-on-failure-session.d.ts +0 -23
  157. package/dist/src/ultraapp/fix-on-failure-session.js +0 -51
  158. package/dist/src/ultraapp/fix-on-failure-session.js.map +0 -1
@@ -8,6 +8,7 @@ import * as fs from 'node:fs';
8
8
  import * as path from 'node:path';
9
9
  import * as os from 'node:os';
10
10
  import { execFile, execFileSync } from 'node:child_process';
11
+ import { randomUUID } from 'node:crypto';
11
12
  import { promisify } from 'node:util';
12
13
  const execFileAsync = promisify(execFile);
13
14
  import * as http from 'node:http';
@@ -103,7 +104,17 @@ function makeDebounced(fn, ms) {
103
104
  }
104
105
  import { createConsoleLogger } from './logger.js';
105
106
  import { CircuitBreaker } from './circuit-breaker.js';
106
- import { appendRunRow, readRunLedger, summarizeRuns, } from './run-ledger.js';
107
+ import { detectRepoLang } from './kernel/repo.js';
108
+ import { RunKernel, runDir as kernelRunDir } from './kernel/engine.js';
109
+ import { registerDefaultExecutors } from './kernel/nodes/index.js';
110
+ import { autoloopStateFromRecord, makeAutoloopExecutor } from './kernel/nodes/autoloop.js';
111
+ import { loadRun, readNodeOutput } from './kernel/store.js';
112
+ import { LEGACY_NODE, joinFindings, toCouncilSession, toFanoutSession, toUltraplanResult, toUltrareviewResult, } from './kernel/projections.js';
113
+ import { legacyCouncilWorkflow, legacyFanoutWorkflow, legacyUltraplanWorkflow, splitAgentSecrets, } from './kernel/templates/index.js';
114
+ import { normalizeContract } from './verify/contract.js';
115
+ import { runContract } from './verify/runner.js';
116
+ import { evidenceDir, listEvidence, readEvidence, writeEvidence } from './verify/evidence.js';
117
+ import { annotateVerdicts, appendRunRow, readRunLedger, summarizeRuns, } from './run-ledger.js';
107
118
  import { checkBudget, isBudgetExceeded } from './budget.js';
108
119
  import { InboxManager } from './inbox-manager.js';
109
120
  import { sanitizeCwd, validateName } from './validation.js';
@@ -112,6 +123,7 @@ import { PersistentGeminiSession } from './persistent-gemini-session.js';
112
123
  import { PersistentCodexSession } from './persistent-codex-session.js';
113
124
  import { PersistentCodexAppServerSession } from './persistent-codex-app-session.js';
114
125
  import { PersistentCursorSession } from './persistent-cursor-session.js';
126
+ import { PersistentGrokSession } from './persistent-grok-session.js';
115
127
  import { PersistentOpencodeSession } from './persistent-opencode-session.js';
116
128
  import { PersistentAgySession } from './persistent-agy-session.js';
117
129
  import { PersistentCustomSession } from './persistent-custom-session.js';
@@ -119,7 +131,6 @@ import { ENGINE_TYPES, overrideModelPricing, } from './types.js';
119
131
  import { resolveAlias, isClaudeModel } from './models.js';
120
132
  import { isAgyConversationId } from './agy-conversation.js';
121
133
  import { Council } from './council.js';
122
- import { Fanout } from './fanout.js';
123
134
  import { AutoloopRunner } from './autoloop/runner.js';
124
135
  import { ClaudeAgentDispatcher } from './autoloop/dispatcher.js';
125
136
  import { DEFAULT_PUSH_POLICY } from './autoloop/types.js';
@@ -127,16 +138,7 @@ import { Msg as AutoloopMsg } from './autoloop/messages.js';
127
138
  import { appendPushLog, notifyUserFallbackChain } from './autoloop/notify.js';
128
139
  import { UltraappManager } from './ultraapp/manager.js';
129
140
  import { UltraappStore, defaultStoreRoot } from './ultraapp/store.js';
130
- import { PERSIST_DISK_TTL_MS, DEBOUNCED_SAVE_MS, CLEANUP_INTERVAL_MS, TURN_TIMEOUT_MS, GREP_HISTORY_FETCH, RESULT_TTL_MS, ULTRAPLAN_TIMEOUT_MS, ULTRAREVIEW_POLL_INTERVAL_MS, STOP_SIGKILL_DELAY_MS, SESSION_EVENT, DEFAULT_HISTORY_LIMIT, } from './constants.js';
131
- // ─── Disk enumeration (cross-process visibility) ────────────────────────────
132
- //
133
- // When the dashboard's standalone clawo-serve and the OpenClaw plugin run as
134
- // separate processes, each has its own in-memory map of active runs. To make
135
- // past runs visible across processes we read what's persisted on disk:
136
- // - Council: transcripts at ~/.openclaw/council-logs/council-*.md
137
- // - Autoloop: registry at ~/.claw-orchestrator/autoloop-registry.jsonl
138
- const DEFAULT_COUNCIL_LOG_DIR = path.join(os.homedir(), '.openclaw', 'council-logs');
139
- const DEFAULT_AUTOLOOP_REGISTRY = path.join(os.homedir(), '.claw-orchestrator', 'autoloop-registry.jsonl');
141
+ import { PERSIST_DISK_TTL_MS, DEBOUNCED_SAVE_MS, CLEANUP_INTERVAL_MS, TURN_TIMEOUT_MS, GREP_HISTORY_FETCH, ULTRAPLAN_TIMEOUT_MS, STOP_SIGKILL_DELAY_MS, SESSION_EVENT, DEFAULT_HISTORY_LIMIT, } from './constants.js';
140
142
  function isStringRecord(value) {
141
143
  return (typeof value === 'object' &&
142
144
  value !== null &&
@@ -197,128 +199,6 @@ function validateAutoloopRole(role, engine, customEngine) {
197
199
  }
198
200
  return resolved;
199
201
  }
200
- /** Append-only registry write. Safe under concurrent writers — append is atomic for short lines. */
201
- export function appendAutoloopRegistry(file, entry) {
202
- fs.mkdirSync(path.dirname(file), { recursive: true });
203
- fs.appendFileSync(file, JSON.stringify(entry) + '\n');
204
- }
205
- /**
206
- * Write the current row for a run, dropping any older rows for the same id.
207
- *
208
- * The registry is append-only and `listAutoloopsFromRegistry` dedups on read
209
- * (newest wins), so correctness never depended on cleanup — but a run now emits
210
- * a row at start, another on every successful `spawn_subagents`, and another on
211
- * every resume, none of which were ever removed. The file grew monotonically and
212
- * every list / resume parses all of it. Callers use this AFTER the operation
213
- * succeeds, so a failed start still leaves the previous row intact.
214
- */
215
- export function upsertAutoloopRegistry(file, entry) {
216
- fs.mkdirSync(path.dirname(file), { recursive: true });
217
- removeAutoloopFromRegistry(file, entry.run_id);
218
- fs.appendFileSync(file, JSON.stringify(entry) + '\n');
219
- }
220
- /**
221
- * Read the registry, dedup by run_id (newest entry wins), drop entries whose
222
- * ledger_dir no longer exists on disk (cleanup of moved/deleted workspaces).
223
- * Returns entries newest-first.
224
- */
225
- export function listAutoloopsFromRegistry(file = DEFAULT_AUTOLOOP_REGISTRY) {
226
- if (!fs.existsSync(file))
227
- return [];
228
- const lines = fs.readFileSync(file, 'utf-8').split('\n').filter(Boolean);
229
- const seen = new Set();
230
- const out = [];
231
- // Walk in reverse so the latest entry for a given run_id wins.
232
- for (const line of [...lines].reverse()) {
233
- try {
234
- const e = JSON.parse(line);
235
- if (seen.has(e.run_id))
236
- continue;
237
- seen.add(e.run_id);
238
- if (!fs.existsSync(e.ledger_dir))
239
- continue; // stale entry, ledger gone
240
- out.push(e);
241
- }
242
- catch {
243
- // malformed line; skip
244
- }
245
- }
246
- return out; // already newest-first because we reversed
247
- }
248
- /**
249
- * Rewrite the registry with every line for the given run_id filtered out.
250
- * Used by autoloopDelete to scrub a run from cross-process visibility.
251
- * No-op if the file does not exist. Returns the number of lines removed.
252
- */
253
- export function removeAutoloopFromRegistry(file, runId) {
254
- if (!fs.existsSync(file))
255
- return 0;
256
- const lines = fs.readFileSync(file, 'utf-8').split('\n');
257
- let removed = 0;
258
- const kept = [];
259
- for (const line of lines) {
260
- if (!line) {
261
- kept.push(line);
262
- continue;
263
- }
264
- try {
265
- const e = JSON.parse(line);
266
- if (e.run_id === runId) {
267
- removed += 1;
268
- continue;
269
- }
270
- }
271
- catch {
272
- // malformed line — keep it, we only filter recognizable entries
273
- }
274
- kept.push(line);
275
- }
276
- if (removed === 0)
277
- return 0;
278
- const tmp = `${file}.${process.pid}.tmp`;
279
- fs.writeFileSync(tmp, kept.join('\n'));
280
- fs.renameSync(tmp, file);
281
- return removed;
282
- }
283
- /**
284
- * Enumerate council sessions from on-disk transcripts. Called by
285
- * SessionManager.councilList() to surface runs that the current process didn't
286
- * spawn itself (e.g. runs started in another process whose transcripts have
287
- * already been flushed to ~/.openclaw/council-logs/).
288
- *
289
- * Format parsed (matches src/council.ts saveTranscript):
290
- * - **ID**: <session.id>
291
- * - **Time**: <iso>
292
- * - **Task**: <text>
293
- * - **Status**: <consensus|max_rounds|...>
294
- *
295
- * Legacy transcripts written before the ID field was added fall back to a
296
- * filename-derived id (basename without .md). That's stable across reruns
297
- * even if uncomfortable as a display id.
298
- */
299
- export function listCouncilsFromDisk(logDir = DEFAULT_COUNCIL_LOG_DIR) {
300
- if (!fs.existsSync(logDir))
301
- return [];
302
- const out = [];
303
- for (const entry of fs.readdirSync(logDir)) {
304
- if (!entry.startsWith('council-') || !entry.endsWith('.md'))
305
- continue;
306
- let head;
307
- try {
308
- head = fs.readFileSync(path.join(logDir, entry), 'utf-8').slice(0, 2000);
309
- }
310
- catch {
311
- continue;
312
- }
313
- const id = /^-\s+\*\*ID\*\*:\s*([^\n]+)/m.exec(head)?.[1]?.trim() || entry.replace(/\.md$/, '');
314
- const task = /^-\s+\*\*Task\*\*:\s*([^\n]+)/m.exec(head)?.[1]?.trim() || '(no task recorded)';
315
- const startTime = /^-\s+\*\*Time\*\*:\s*([^\n]+)/m.exec(head)?.[1]?.trim() || '';
316
- const status = /^-\s+\*\*Status\*\*:\s*([^\n]+)/m.exec(head)?.[1]?.trim() || 'unknown';
317
- out.push({ id, task, status, startTime });
318
- }
319
- return out;
320
- }
321
- // ─── SessionManager ──────────────────────────────────────────────────────────
322
202
  export class SessionManager {
323
203
  sessions = new Map();
324
204
  _pendingSessions = new Map();
@@ -331,6 +211,8 @@ export class SessionManager {
331
211
  _activePids = new Map();
332
212
  _circuitBreaker = new CircuitBreaker();
333
213
  _inbox = new InboxManager();
214
+ /** cwd → detected language, so the manifest probe runs once per directory. */
215
+ _repoLangCache = new Map();
334
216
  logger;
335
217
  _ultraappManager = null;
336
218
  _ultraappRouter = null;
@@ -370,6 +252,10 @@ export class SessionManager {
370
252
  sessionManager: this,
371
253
  router: this._ultraappRouter ?? undefined,
372
254
  runtimeMode: this._ultraappRuntimeMode,
255
+ // The same kernel every other mode runs on, so an ultraapp build is a
256
+ // run like any other: listed by `workflow_list`, visible in the Runs
257
+ // tab, owned by one process, and resumable at a node boundary.
258
+ kernel: this.kernel,
373
259
  });
374
260
  }
375
261
  return this._ultraappManager;
@@ -625,7 +511,10 @@ export class SessionManager {
625
511
  throw err;
626
512
  }
627
513
  finally {
628
- this._recordRunTurn(name, managed, ledgerBefore, startedAt, turnError, options.parentRunId);
514
+ this._recordRunTurn(name, managed, ledgerBefore, startedAt, turnError, options.parentRunId, {
515
+ nodeKind: options.nodeKind,
516
+ taskKind: options.taskKind,
517
+ });
629
518
  }
630
519
  }
631
520
  finally {
@@ -685,7 +574,7 @@ export class SessionManager {
685
574
  * Rows carry per-turn deltas rather than session totals so that summing a
686
575
  * query gives the spend for that window without double-counting.
687
576
  */
688
- _recordRunTurn(name, managed, before, startedAt, error, parent) {
577
+ _recordRunTurn(name, managed, before, startedAt, error, parent, dims = {}) {
689
578
  const after = this._statsSnapshot(managed);
690
579
  const delta = (a, b) => Math.max(0, a - b);
691
580
  const row = {
@@ -720,11 +609,34 @@ export class SessionManager {
720
609
  row.error = error.slice(0, 500);
721
610
  if (parent)
722
611
  row.parent = parent;
612
+ if (dims.nodeKind)
613
+ row.nodeKind = dims.nodeKind;
614
+ if (dims.taskKind)
615
+ row.taskKind = dims.taskKind;
616
+ // Detected from a manifest, never guessed. `verified` is deliberately absent
617
+ // here: the verdict does not exist yet at turn time, and is joined in at read
618
+ // time by annotateVerdicts().
619
+ const repoLang = this._repoLang(managed.cwd);
620
+ if (repoLang)
621
+ row.repoLang = repoLang;
723
622
  appendRunRow(row, this.logger);
724
623
  if (isBudgetExceeded(after.costUsd, managed.config.maxBudgetUsd)) {
725
624
  managed.budgetExhausted = true;
726
625
  }
727
626
  }
627
+ /**
628
+ * Repo language for the ledger row, memoised per cwd — the detector stats a
629
+ * handful of manifest paths and a turn-rate filesystem probe is wasteful when
630
+ * a session's cwd never changes.
631
+ */
632
+ _repoLang(cwd) {
633
+ if (!cwd)
634
+ return undefined;
635
+ if (!this._repoLangCache.has(cwd)) {
636
+ this._repoLangCache.set(cwd, detectRepoLang(cwd));
637
+ }
638
+ return this._repoLangCache.get(cwd);
639
+ }
728
640
  _reportedModel(managed) {
729
641
  try {
730
642
  return managed.session.getCost()?.model || undefined;
@@ -746,8 +658,159 @@ export class SessionManager {
746
658
  * process restart and covers sessions this manager never owned.
747
659
  */
748
660
  getRunLedger(query = {}) {
749
- const rows = readRunLedger(query, this.logger);
750
- return { rows, summary: summarizeRuns(rows) };
661
+ // Join each row to the verdict of the run it belonged to. The turns that did
662
+ // the work all finish before the verifier that judged it, so the verdict
663
+ // cannot be written at turn time — see `annotateVerdicts`.
664
+ //
665
+ // `verified` is deliberately withheld from the read: applying it there would
666
+ // filter on a field no row carries yet and return nothing. It is applied
667
+ // after the join instead.
668
+ const { verified, ...readQuery } = query;
669
+ const rows = annotateVerdicts(readRunLedger(readQuery, this.logger), (parent) => {
670
+ const record = loadRun(parent);
671
+ if (!record || record.outcome === 'unverified')
672
+ return undefined;
673
+ return {
674
+ verified: record.outcome === 'verified',
675
+ evidenceId: record.evidenceId,
676
+ contractId: record.spec?.contract?.id,
677
+ };
678
+ });
679
+ const filtered = verified === undefined ? rows : rows.filter((r) => r.verified === verified);
680
+ return { rows: filtered, summary: summarizeRuns(filtered) };
681
+ }
682
+ // ─── Workflow kernel ──────────────────────────────────────────────────────
683
+ /**
684
+ * Lazily built, like every other subsystem here — constructing it at plugin
685
+ * load would create run directories for a process that may never run anything.
686
+ */
687
+ get kernel() {
688
+ if (!this._kernel) {
689
+ const kernel = registerDefaultExecutors(new RunKernel({ manager: this, logger: this.logger }), (name) => this._resolveTemplate(name));
690
+ // The autoloop engine needs sessions, prompt files and push channels, so
691
+ // its executor is registered here with a builder closed over `this`
692
+ // rather than living in the kernel.
693
+ kernel.setExecutor('autoloop', makeAutoloopExecutor({
694
+ boot: (config, secrets) => this._bootAutoloop({
695
+ ...config,
696
+ // Custom-engine configs never reach the spec, so they come from
697
+ // the run's in-memory secret bag — supplied at start, and
698
+ // re-supplied by the caller on a resume.
699
+ ...secrets,
700
+ }),
701
+ ready: (key, value) => {
702
+ const deferred = this._autoloopReady.get(key);
703
+ if (!deferred)
704
+ return;
705
+ if (value instanceof Error)
706
+ deferred.reject(value);
707
+ else
708
+ deferred.resolve(value);
709
+ },
710
+ waitForExit: (handle, signal) => this._awaitAutoloopExit(handle, signal),
711
+ registerPublisher: (runId, publish) => this._autoloopPublishers.set(runId, publish),
712
+ unregisterPublisher: (runId) => this._autoloopPublishers.delete(runId),
713
+ extra: (runId) => {
714
+ const roleSelection = this._autoloopSelection.get(runId);
715
+ return roleSelection ? { roleSelection } : {};
716
+ },
717
+ }));
718
+ this._kernel = kernel;
719
+ }
720
+ return this._kernel;
721
+ }
722
+ /** Named built-ins available to `subflow` nodes and to `workflow_start`. */
723
+ _resolveTemplate(name) {
724
+ // Built-ins need caller arguments, so a bare name only resolves to a
725
+ // previously started run's spec — a subflow referencing a template by name
726
+ // without arguments has nothing to run.
727
+ const record = loadRun(name);
728
+ return record?.spec;
729
+ }
730
+ /**
731
+ * Subscribe to kernel events (for the SSE endpoint). Returns an unsubscribe
732
+ * function — SessionManager is not an EventEmitter, and making it one just for
733
+ * this would widen its surface for one consumer.
734
+ */
735
+ onWorkflowEvent(listener) {
736
+ const k = this.kernel;
737
+ k.on('kernel-event', listener);
738
+ return () => {
739
+ k.off('kernel-event', listener);
740
+ };
741
+ }
742
+ async workflowStart(spec, opts = {}) {
743
+ return this.kernel.start(spec, opts);
744
+ }
745
+ workflowStatus(runId) {
746
+ const record = this.kernel.get(runId);
747
+ if (!record)
748
+ throw new Error(`Workflow run '${runId}' not found`);
749
+ return record;
750
+ }
751
+ workflowList(query = {}) {
752
+ return this.kernel.list(query);
753
+ }
754
+ workflowCancel(runId) {
755
+ return { cancelled: this.kernel.cancel(runId) };
756
+ }
757
+ /**
758
+ * Re-attach to a run.
759
+ *
760
+ * `secrets` re-supplies the material the spec deliberately does not carry —
761
+ * per-agent custom-engine configs, keyed as `{ agentCustomEngines: { <name>: cfg } }`.
762
+ * A run that used one cannot be resumed in a fresh process without them,
763
+ * because they were never written down.
764
+ */
765
+ async workflowResume(runId, opts = {}) {
766
+ return this.kernel.resume(runId, { secrets: opts.secrets });
767
+ }
768
+ workflowSteer(runId, text) {
769
+ return { steered: this.kernel.steer(runId, text) };
770
+ }
771
+ workflowApprove(runId, approved) {
772
+ return { answered: this.kernel.approve(runId, approved) };
773
+ }
774
+ workflowDelete(runId) {
775
+ this.kernel.delete(runId);
776
+ }
777
+ workflowEvidence(runId, evidenceId) {
778
+ const dir = kernelRunDir(runId);
779
+ const id = evidenceId ?? this.kernel.get(runId)?.evidenceId ?? listEvidence(dir).at(-1);
780
+ return id ? readEvidence(dir, id) : undefined;
781
+ }
782
+ /**
783
+ * Run an acceptance contract against a directory, outside any workflow.
784
+ *
785
+ * This is the escape hatch for work that did not come through the kernel — a
786
+ * plain `session_send` that edited a repo, or a run from an older version. The
787
+ * contract comes from the caller and is normalized before anything executes.
788
+ */
789
+ async verifyRun(args) {
790
+ const contract = normalizeContract(args.contract);
791
+ if (!contract)
792
+ throw new Error('verifyRun requires a contract with at least one recognised check');
793
+ const runId = args.label || `verify-${Date.now().toString(36)}`;
794
+ const dir = kernelRunDir(runId);
795
+ const evidenceId = 'verify-01';
796
+ const { results, rounds } = await runContract(contract, {
797
+ cwd: args.cwd,
798
+ artifactDir: evidenceDir(dir, evidenceId),
799
+ baseSha: args.baseSha,
800
+ logger: this.logger,
801
+ });
802
+ return writeEvidence({
803
+ runDir: dir,
804
+ runId,
805
+ node: 'run',
806
+ evidenceId,
807
+ cwd: args.cwd,
808
+ baseSha: args.baseSha,
809
+ contractId: contract.id,
810
+ results,
811
+ rounds,
812
+ logger: this.logger,
813
+ });
751
814
  }
752
815
  async stopSession(name, opts = {}) {
753
816
  const managed = this._getSession(name);
@@ -1127,32 +1190,14 @@ export class SessionManager {
1127
1190
  clearInterval(this.cleanupTimer);
1128
1191
  this.cleanupTimer = null;
1129
1192
  }
1130
- // Stop ultrareview pollers
1131
- for (const [, timer] of this.ultrareviewPollers)
1132
- clearInterval(timer);
1133
- this.ultrareviewPollers.clear();
1134
- // Clear council/fanout cleanup timers — their 30-min closures capture `this`
1135
- // and would otherwise fire after shutdown (and council timers, before this
1136
- // fix, were not unref'd so they blocked a clean process exit).
1137
- for (const [, timer] of this.councilCleanupTimers)
1138
- clearTimeout(timer);
1139
- this.councilCleanupTimers.clear();
1140
- this.councils.clear();
1141
- for (const [, timer] of this.fanoutCleanupTimers)
1142
- clearTimeout(timer);
1143
- this.fanoutCleanupTimers.clear();
1144
- this.fanouts.clear();
1145
- // Stop autoloops (graceful: dispatch a terminate envelope so each run
1146
- // shuts down its three persistent agents and cleans up the ledger lock).
1147
- for (const [, ctx] of this.autoloops) {
1148
- try {
1149
- await ctx.runner.send(AutoloopMsg.terminate(ctx.runner.state.iter, { reason: 'manager-shutdown' }));
1150
- }
1151
- catch {
1152
- // Best-effort.
1153
- }
1154
- }
1155
- this.autoloops.clear();
1193
+ // Council, fan-out, ultraplan and ultrareview no longer have timers or maps
1194
+ // to tear down here: the kernel owns their lifecycle, and `shutdown` on it
1195
+ // cancels every live run. Four separate 30-minute TTL closures used to sit
1196
+ // in this method, each capturing `this`.
1197
+ // Autoloops included: cancelling their run stops the loop, which shuts down
1198
+ // its three persistent agents. One teardown path for every mode.
1199
+ if (this._kernel)
1200
+ await this._kernel.shutdown();
1156
1201
  // Stop all sessions
1157
1202
  for (const [name, managed] of this.sessions) {
1158
1203
  try {
@@ -1862,6 +1907,8 @@ export class SessionManager {
1862
1907
  return isAgyConversationId(id) ? id : undefined;
1863
1908
  if (engine === 'codex')
1864
1909
  return id && !/^codex-\d+-/.test(id) ? id : undefined;
1910
+ if (engine === 'grok')
1911
+ return id && !/^grok-\d+-/.test(id) ? id : undefined;
1865
1912
  return id;
1866
1913
  }
1867
1914
  _listMdFiles(dir) {
@@ -1888,6 +1935,8 @@ export class SessionManager {
1888
1935
  return new PersistentCodexAppServerSession(config, process.env.CODEX_BIN);
1889
1936
  case 'cursor':
1890
1937
  return new PersistentCursorSession(config, process.env.CURSOR_BIN);
1938
+ case 'grok':
1939
+ return new PersistentGrokSession(config, process.env.GROK_BIN);
1891
1940
  case 'opencode':
1892
1941
  return new PersistentOpencodeSession(config, process.env.OPENCODE_BIN);
1893
1942
  case 'custom':
@@ -1900,178 +1949,151 @@ export class SessionManager {
1900
1949
  }
1901
1950
  }
1902
1951
  // ─── Council ──────────────────────────────────────────────────────────
1903
- councils = new Map();
1904
- councilCleanupTimers = new Map();
1905
- councilStart(task, config) {
1906
- const council = new Council(config, this, this.logger);
1907
- const initialSession = council.init(task);
1908
- // Store BEFORE running so council_status/abort/inject work while it's active
1909
- this.councils.set(initialSession.id, council);
1910
- // Run in background — callers poll via councilStatus()
1911
- council
1912
- .run()
1913
- .then(() => {
1914
- // Keep completed council queryable; schedule cleanup after TTL
1915
- this._scheduleCouncilCleanup(initialSession.id);
1916
- })
1917
- .catch((err) => {
1918
- this.logger.error(`Council ${initialSession.id} failed:`, err);
1919
- this._scheduleCouncilCleanup(initialSession.id);
1920
- });
1921
- return initialSession;
1922
- }
1923
- _scheduleCouncilCleanup(id) {
1924
- // Clear any existing timer before scheduling a new one
1925
- const existing = this.councilCleanupTimers.get(id);
1926
- if (existing)
1927
- clearTimeout(existing);
1928
- const timer = setTimeout(() => {
1929
- // Abort if still running to prevent orphaned background tasks
1930
- const council = this.councils.get(id);
1931
- if (council) {
1932
- const session = council.getSession();
1933
- if (session?.status === 'running') {
1934
- this.logger.info(`Council ${id} still running at TTL expiry — aborting`);
1935
- council.abort();
1936
- }
1937
- }
1938
- this.councils.delete(id);
1939
- this.councilCleanupTimers.delete(id);
1940
- }, RESULT_TTL_MS);
1941
- // Don't let a pending 30-min cleanup timer keep the process alive or block shutdown.
1942
- timer.unref();
1943
- this.councilCleanupTimers.set(id, timer);
1944
- }
1945
- /** Clear and forget a cleanup timer (used on abort/shutdown so it can't fire late). */
1946
- _clearCleanupTimer(map, id) {
1947
- const t = map.get(id);
1948
- if (t) {
1949
- clearTimeout(t);
1950
- map.delete(id);
1951
- }
1952
+ //
1953
+ // The council's lifecycle belongs to the run kernel now. What used to live
1954
+ // here — a `Map` of live `Council` objects, a 30-minute TTL timer per entry,
1955
+ // and a `councilList` that regex-scraped markdown transcripts to see runs from
1956
+ // other processes — is gone. A council is a one-node workflow; its state is
1957
+ // the run record, which is durable, cross-process, and does not evaporate.
1958
+ //
1959
+ // What still needs a live object is in-flight control: `inject` and `abort`
1960
+ // have to reach the `Council` instance that is running right now. The kernel
1961
+ // publishes it for the duration of the node, and says so honestly — after a
1962
+ // restart the run is readable and resumable, but there is no turn to inject
1963
+ // into.
1964
+ async councilStart(task, config) {
1965
+ const runId = `council-${Date.now().toString(36)}-${randomUUID().slice(0, 8)}`;
1966
+ const { agents, secrets } = splitAgentSecrets(config.agents);
1967
+ const record = await this.kernel.start(legacyCouncilWorkflow({
1968
+ task,
1969
+ cwd: config.projectDir,
1970
+ agents,
1971
+ maxRounds: config.maxRounds,
1972
+ timeoutMs: config.agentTimeoutMs,
1973
+ maxTurnsPerAgent: config.maxTurnsPerAgent,
1974
+ maxBudgetUsd: config.maxBudgetUsd,
1975
+ defaultPermissionMode: config.defaultPermissionMode,
1976
+ }), { runId, cwd: config.projectDir, secrets: { agentCustomEngines: secrets } });
1977
+ return toCouncilSession(record);
1952
1978
  }
1953
1979
  councilStatus(id) {
1954
- const council = this.councils.get(id);
1955
- return council?.getSession();
1980
+ const record = loadRun(id);
1981
+ if (!record || record.workflow !== 'council')
1982
+ return undefined;
1983
+ return toCouncilSession(record);
1956
1984
  }
1957
1985
  /**
1958
- * List all council sessions visible to this process.
1986
+ * Every council this machine has run, newest first.
1959
1987
  *
1960
- * Includes (a) in-memory sessions managed by this SessionManager and (b)
1961
- * sessions reconstructed from on-disk transcripts at ~/.openclaw/council-logs/.
1962
- * The disk path lets the dashboard see runs started in OTHER processes
1963
- * (e.g. plugin-managed runs visible to a standalone clawo-serve dashboard).
1964
- * Dedup by id; in-memory wins. Sorted by startTime descending so the newest
1965
- * appears at the top of the sidebar.
1988
+ * Cross-process visibility used to come from scraping `~/.openclaw/council-logs/*.md`
1989
+ * with a regex and fabricating a stub session with no responses and an empty
1990
+ * config. Runs are stored records now, so the dashboard sees the real thing.
1966
1991
  */
1967
1992
  councilList() {
1968
- const inMemory = Array.from(this.councils.values())
1969
- .map((c) => c.getSession())
1970
- .filter((s) => s !== null && s !== undefined);
1971
- const inMemIds = new Set(inMemory.map((s) => s.id));
1972
- const fromDisk = listCouncilsFromDisk()
1973
- .filter((r) => !inMemIds.has(r.id))
1974
- .map((r) => ({
1975
- id: r.id,
1976
- task: r.task,
1977
- status: r.status,
1978
- startTime: r.startTime,
1979
- responses: [],
1980
- config: { agents: [], maxRounds: 0, projectDir: '' },
1981
- }));
1982
- return [...inMemory, ...fromDisk].sort((a, b) => (b.startTime || '').localeCompare(a.startTime || ''));
1993
+ return this.kernel
1994
+ .list({ workflow: 'council' })
1995
+ .map((r) => loadRun(r.runId))
1996
+ .filter((r) => Boolean(r))
1997
+ .map(toCouncilSession);
1983
1998
  }
1984
1999
  /** Used by embedded-server to subscribe to a council's event stream. */
1985
2000
  getCouncil(id) {
1986
- return this.councils.get(id);
2001
+ return this.kernel.handle(id, LEGACY_NODE);
2002
+ }
2003
+ /** The live council for a run, or a clear error about why there isn't one. */
2004
+ _liveCouncil(id) {
2005
+ const council = this.kernel.handle(id, LEGACY_NODE);
2006
+ if (council)
2007
+ return council;
2008
+ const record = loadRun(id);
2009
+ if (!record)
2010
+ throw new Error(`Council '${id}' not found`);
2011
+ throw new Error(`Council '${id}' is ${record.state} and not running in this process — its record is readable, but there is no live round to act on`);
1987
2012
  }
1988
2013
  councilAbort(id) {
1989
- const council = this.councils.get(id);
1990
- if (!council)
2014
+ // Cancel the run first so the kernel stops advancing, then abort the engine
2015
+ // so the current round tears down its worktrees.
2016
+ if (!this.kernel.cancel(id) && !loadRun(id))
1991
2017
  throw new Error(`Council '${id}' not found`);
1992
- council.abort();
1993
- this.councils.delete(id);
1994
- // Drop the orphaned cleanup timer so it doesn't fire later on a deleted council.
1995
- this._clearCleanupTimer(this.councilCleanupTimers, id);
2018
+ this.kernel.handle(id, LEGACY_NODE)?.abort();
1996
2019
  }
1997
2020
  councilInject(id, message) {
1998
- const council = this.councils.get(id);
1999
- if (!council)
2000
- throw new Error(`Council '${id}' not found`);
2001
- council.injectMessage(message);
2021
+ this._liveCouncil(id).injectMessage(message);
2002
2022
  }
2003
2023
  async councilReview(id) {
2004
- const council = this.councils.get(id);
2005
- if (!council)
2006
- throw new Error(`Council '${id}' not found`);
2007
- this._scheduleCouncilCleanup(id); // reset TTL — user is actively reviewing
2008
- return council.review();
2024
+ return this._councilForPostProcessing(id).review();
2009
2025
  }
2010
2026
  async councilAccept(id) {
2011
- const council = this.councils.get(id);
2012
- if (!council)
2013
- throw new Error(`Council '${id}' not found`);
2014
- const result = await council.accept();
2015
- // Accepted — no longer needed, clean up after short grace period
2016
- this._scheduleCouncilCleanup(id);
2017
- return result;
2027
+ return this._councilForPostProcessing(id).accept();
2018
2028
  }
2019
2029
  async councilReject(id, feedback) {
2020
- const council = this.councils.get(id);
2021
- if (!council)
2030
+ return this._councilForPostProcessing(id).reject(feedback);
2031
+ }
2032
+ /**
2033
+ * A `Council` for review / accept / reject.
2034
+ *
2035
+ * These three act on the git state a finished council left behind — branches,
2036
+ * worktrees, plan.md — so they do not need the instance that produced it, only
2037
+ * one pointed at the same project directory. Reconstructing from the run
2038
+ * record is what makes them work after a restart, which the in-memory map made
2039
+ * impossible.
2040
+ */
2041
+ _councilForPostProcessing(id) {
2042
+ const live = this.kernel.handle(id, LEGACY_NODE);
2043
+ if (live)
2044
+ return live;
2045
+ const record = loadRun(id);
2046
+ if (!record || record.workflow !== 'council')
2022
2047
  throw new Error(`Council '${id}' not found`);
2023
- const result = await council.reject(feedback);
2024
- this._scheduleCouncilCleanup(id); // reset TTL — council may be restarted
2025
- return result;
2048
+ const session = toCouncilSession(record);
2049
+ const council = new Council(session.config, this, this.logger);
2050
+ council.adoptSession(session);
2051
+ return council;
2026
2052
  }
2027
2053
  // ─── Fan-out (parallel multi-engine task, no consensus) ────────────────
2028
- fanouts = new Map();
2029
- fanoutCleanupTimers = new Map();
2054
+ //
2055
+ // Also a one-node workflow. This is the mode the old design failed hardest:
2056
+ // a fan-out wrote nothing to disk at all, so 30 minutes after it finished
2057
+ // `fanoutStatus` threw "not found" and the results were simply gone.
2030
2058
  /**
2031
2059
  * Start a fan-out: run the task across N engine/model agents in parallel and
2032
2060
  * collect their answers (optional synthesis). Runs in the background; poll
2033
2061
  * with fanoutStatus. Distinct from council — no rounds, votes, or worktrees.
2034
2062
  */
2035
- fanoutStart(config) {
2063
+ async fanoutStart(config) {
2036
2064
  if (!config.agents?.length)
2037
2065
  throw new Error('fanoutStart: at least one agent is required');
2038
2066
  const names = config.agents.map((a) => a.name);
2039
2067
  if (new Set(names).size !== names.length) {
2040
2068
  throw new Error('fanoutStart: agent names must be unique (they form session names)');
2041
2069
  }
2042
- const fanout = new Fanout(config, this, this.logger);
2043
- const session = fanout.init();
2044
- this.fanouts.set(session.id, fanout);
2045
- fanout
2046
- .run()
2047
- .catch((err) => this.logger.error(`Fanout ${session.id} failed:`, err))
2048
- .finally(() => this._scheduleFanoutCleanup(session.id));
2049
- return session;
2070
+ const runId = `fanout-${Date.now().toString(36)}-${randomUUID().slice(0, 8)}`;
2071
+ const { agents, secrets } = splitAgentSecrets(config.agents);
2072
+ const record = await this.kernel.start(legacyFanoutWorkflow({
2073
+ task: config.task,
2074
+ cwd: config.projectDir,
2075
+ agents,
2076
+ synthesize: config.synthesize,
2077
+ synthesisEngine: config.synthesisEngine,
2078
+ synthesisModel: config.synthesisModel,
2079
+ synthesisPermissionMode: config.synthesisPermissionMode,
2080
+ maxTurnsPerAgent: config.maxTurnsPerAgent,
2081
+ maxBudgetUsd: config.maxBudgetUsd,
2082
+ timeoutMs: config.agentTimeoutMs,
2083
+ }), { runId, cwd: config.projectDir, secrets: { agentCustomEngines: secrets } });
2084
+ return toFanoutSession(record);
2050
2085
  }
2051
2086
  fanoutStatus(id) {
2052
- const fanout = this.fanouts.get(id);
2053
- if (!fanout)
2087
+ const record = loadRun(id);
2088
+ if (!record)
2054
2089
  throw new Error(`Fanout '${id}' not found`);
2055
- return fanout.getSession();
2090
+ return toFanoutSession(record);
2056
2091
  }
2057
2092
  fanoutAbort(id) {
2058
- const fanout = this.fanouts.get(id);
2059
- if (!fanout)
2093
+ if (!loadRun(id))
2060
2094
  throw new Error(`Fanout '${id}' not found`);
2061
- fanout.abort();
2062
- this._scheduleFanoutCleanup(id);
2063
- }
2064
- _scheduleFanoutCleanup(id) {
2065
- const existing = this.fanoutCleanupTimers.get(id);
2066
- if (existing)
2067
- clearTimeout(existing);
2068
- const timer = setTimeout(() => {
2069
- this.fanouts.delete(id);
2070
- this.fanoutCleanupTimers.delete(id);
2071
- }, RESULT_TTL_MS);
2072
- if (typeof timer.unref === 'function')
2073
- timer.unref();
2074
- this.fanoutCleanupTimers.set(id, timer);
2095
+ this.kernel.cancel(id);
2096
+ this.kernel.handle(id, LEGACY_NODE)?.abort();
2075
2097
  }
2076
2098
  // ─── Inbox (cross-session messaging) — delegated to InboxManager ────
2077
2099
  get _sessionLookup() {
@@ -2093,79 +2115,66 @@ export class SessionManager {
2093
2115
  return this._inbox.deliverInbox(name, this._sessionLookup);
2094
2116
  }
2095
2117
  // ─── Ultraplan ────────────────────────────────────────────────────────
2096
- ultraplans = new Map();
2097
- ultraplanStart(task, opts) {
2098
- const id = `ultraplan-${Date.now()}-${Math.random().toString(36).slice(2, 6)}`;
2099
- const sessionName = `ultraplan-${id}`;
2100
- const timeout = opts?.timeout || ULTRAPLAN_TIMEOUT_MS;
2101
- const result = {
2102
- id,
2103
- status: 'running',
2104
- sessionName,
2105
- startTime: new Date().toISOString(),
2106
- };
2107
- this.ultraplans.set(id, result);
2108
- // Run in background
2109
- this._runUltraplan(id, sessionName, task, opts?.model || 'opus', opts?.cwd || process.cwd(), timeout)
2110
- .catch((err) => {
2111
- result.status = 'error';
2112
- result.error = err.message;
2113
- result.endTime = new Date().toISOString();
2114
- })
2115
- .finally(() => {
2116
- // Cleanup session
2117
- this.stopSession(sessionName).catch((err) => {
2118
- this.logger.error(`Failed to stop ultraplan session '${sessionName}':`, err);
2119
- });
2120
- const ttlTimer = setTimeout(() => {
2121
- // Mark as error if still running at TTL expiry
2122
- const plan = this.ultraplans.get(id);
2123
- if (plan?.status === 'running') {
2124
- this.logger.info(`Ultraplan ${id} still running at TTL expiry — marking as error`);
2125
- plan.status = 'error';
2126
- plan.error = 'Timed out (TTL expired)';
2127
- plan.endTime = new Date().toISOString();
2128
- }
2129
- this.ultraplans.delete(id);
2130
- }, RESULT_TTL_MS);
2131
- ttlTimer.unref(); // don't block process exit on a 30-min TTL timer
2132
- });
2133
- return result;
2134
- }
2135
- async _runUltraplan(id, sessionName, task, model, cwd, timeout) {
2136
- const result = this.ultraplans.get(id);
2137
- await this.startSession({
2138
- name: sessionName,
2118
+ //
2119
+ // A one-node workflow. What is gone: a `Map` of results, and an inline
2120
+ // 30-minute timer that doubled as the timeout — a plan still running when the
2121
+ // TTL fired was rewritten as `error: 'Timed out (TTL expired)'` and then
2122
+ // deleted, so a long plan could be destroyed by its own eviction timer. The
2123
+ // node's `timeoutMs` is the timeout now, and the record does not expire.
2124
+ _kernel = null;
2125
+ /**
2126
+ * Deferreds resolved by the `autoloop` node once its engine is up, so
2127
+ * `autoloopStart` can return the Planner session name the caller expects
2128
+ * without polling.
2129
+ */
2130
+ /**
2131
+ * Deferreds resolved by the `autoloop` node once its engine is up.
2132
+ *
2133
+ * Keyed by the start's tag rather than its run id. A run id gets reused — a
2134
+ * start that failed frees it for a retry — so keying on the id let a dying
2135
+ * start settle, or clear, the retry's deferred instead of its own, and the
2136
+ * retry then waited forever for a signal with nowhere to land.
2137
+ */
2138
+ _autoloopReady = new Map();
2139
+ /**
2140
+ * Run ids with a start in flight — the window between "run created" and
2141
+ * "engine up". Deleting inside it would drop the run while its Planner
2142
+ * session is still being created, orphaning a session that finishes a moment
2143
+ * later with nothing pointing at it.
2144
+ */
2145
+ _autoloopStarting = new Map();
2146
+ /** Latest role selection per run, published into the node payload. */
2147
+ _autoloopSelection = new Map();
2148
+ /** Per-run checkpoint refreshers, registered by the autoloop node executor. */
2149
+ _autoloopPublishers = new Map();
2150
+ async ultraplanStart(task, opts) {
2151
+ const runId = `ultraplan-${Date.now().toString(36)}-${randomUUID().slice(0, 8)}`;
2152
+ const cwd = opts?.cwd || process.cwd();
2153
+ const record = await this.kernel.start(legacyUltraplanWorkflow({
2154
+ task,
2139
2155
  cwd,
2140
- model,
2141
- permissionMode: 'plan',
2142
- effort: 'max',
2143
- appendSystemPrompt: 'You are in ultraplan mode. Explore the project thoroughly, analyze feasibility, and produce a detailed, actionable plan. Do NOT write code — plan only. Output your final plan in a clear markdown format.',
2144
- });
2145
- const planPrompt = `# Ultraplan Task\n\n${task}\n\nExplore the project, understand the codebase, analyze feasibility, and produce a comprehensive implementation plan. Take your time (up to 30 minutes). Be thorough.`;
2146
- const sendResult = await this.sendMessage(sessionName, planPrompt, { timeout });
2147
- // Detect error responses: empty output or output that looks like an error message
2148
- const output = sendResult.output?.trim() || '';
2149
- const looksLikeError = !output ||
2150
- /^(Error|not logged in|authentication|auth failed|permission denied)/i.test(output) ||
2151
- (sendResult.error && sendResult.error.length > 0);
2152
- if (looksLikeError) {
2153
- result.status = 'error';
2154
- result.error = sendResult.error || output || 'Empty response from engine';
2155
- }
2156
- else {
2157
- result.plan = output;
2158
- result.status = 'completed';
2159
- }
2160
- result.endTime = new Date().toISOString();
2156
+ model: opts?.model || 'opus',
2157
+ timeoutMs: opts?.timeout || ULTRAPLAN_TIMEOUT_MS,
2158
+ }), { runId, cwd });
2159
+ return toUltraplanResult(record, undefined);
2161
2160
  }
2162
2161
  ultraplanStatus(id) {
2163
- return this.ultraplans.get(id);
2162
+ const record = loadRun(id);
2163
+ if (!record || record.workflow !== 'ultraplan')
2164
+ return undefined;
2165
+ // Read the plan from the node artifact, not the record's preview: a plan is
2166
+ // routinely longer than the inline cap, and returning a truncated one would
2167
+ // quietly hand back a broken deliverable.
2168
+ return toUltraplanResult(record, readNodeOutput(id, LEGACY_NODE));
2164
2169
  }
2165
2170
  // ─── Ultrareview ──────────────────────────────────────────────────────
2166
- ultrareviews = new Map();
2167
- ultrareviewPollers = new Map();
2168
- ultrareviewStart(cwd, opts) {
2171
+ // No map and no poller. Ultrareview used to hold its results in a `Map`, then
2172
+ // `setInterval` every 5 seconds asking the fan-out whether it had finished —
2173
+ // which meant its correctness depended on the fan-out's 30-minute eviction
2174
+ // timer: evict first and the poll threw, the interval was cleared, and the
2175
+ // review stayed `running` forever. It is one run now, and there is nothing to
2176
+ // poll.
2177
+ async ultrareviewStart(cwd, opts) {
2169
2178
  const id = `ultrareview-${Date.now()}-${Math.random().toString(36).slice(2, 6)}`;
2170
2179
  const agentCount = Math.min(20, Math.max(1, opts?.agentCount || 5));
2171
2180
  const result = {
@@ -2175,7 +2184,6 @@ export class SessionManager {
2175
2184
  agentCount,
2176
2185
  startTime: new Date().toISOString(),
2177
2186
  };
2178
- this.ultrareviews.set(id, result);
2179
2187
  // Build reviewer agents
2180
2188
  const reviewAngles = [
2181
2189
  {
@@ -2299,92 +2307,93 @@ export class SessionManager {
2299
2307
  // opt-in via `engines`, run under their engine's default sandbox.)
2300
2308
  permissionMode: 'plan',
2301
2309
  }));
2302
- let fanoutSession;
2303
- try {
2304
- fanoutSession = this.fanoutStart({
2305
- task: reviewInstruction,
2306
- projectDir: cwd,
2307
- agents,
2308
- synthesize: true,
2309
- agentTimeoutMs: maxMinutes * 60 * 1000,
2310
- maxTurnsPerAgent: 20,
2311
- });
2312
- }
2313
- catch (err) {
2314
- // Fan-out failed to even start (e.g. validation) — surface it on the
2315
- // stored result instead of leaving it frozen at 'running'.
2316
- result.status = 'error';
2317
- result.error = err.message;
2318
- result.endTime = new Date().toISOString();
2319
- setTimeout(() => this.ultrareviews.delete(id), RESULT_TTL_MS);
2320
- return result;
2321
- }
2322
- // `councilId` is kept for the UltrareviewResult contract; it now holds the
2323
- // fan-out id (an opaque run id used only by ultrareview_status).
2324
- result.councilId = fanoutSession.id;
2325
- // Poll the fan-out for completion (store ref for shutdown cleanup).
2326
- const pollInterval = setInterval(() => {
2327
- try {
2328
- const status = this.fanoutStatus(fanoutSession.id);
2329
- if (!status || status.status === 'running')
2330
- return;
2331
- clearInterval(pollInterval);
2332
- this.ultrareviewPollers.delete(id);
2333
- result.status = status.status === 'error' ? 'error' : 'completed';
2334
- result.endTime = new Date().toISOString();
2335
- // Prefer the synthesis pass; fall back to joining successful results.
2336
- if (status.synthesis) {
2337
- result.findings = status.synthesis;
2338
- }
2339
- else if (status.results.length > 0) {
2340
- result.findings = status.results
2341
- .filter((r) => r.ok)
2342
- .map((r) => `## ${r.agent}\n\n${r.output}`)
2343
- .join('\n\n---\n\n');
2344
- }
2345
- {
2346
- const ttlDelete = setTimeout(() => this.ultrareviews.delete(id), RESULT_TTL_MS);
2347
- ttlDelete.unref();
2348
- }
2349
- }
2350
- catch {
2351
- // Fan-out may have been cleaned up; stop polling.
2352
- clearInterval(pollInterval);
2353
- this.ultrareviewPollers.delete(id);
2354
- }
2355
- }, ULTRAREVIEW_POLL_INTERVAL_MS);
2356
- this.ultrareviewPollers.set(id, pollInterval);
2310
+ const runId = id;
2311
+ await this.kernel.start(legacyFanoutWorkflow({
2312
+ name: 'ultrareview',
2313
+ task: reviewInstruction,
2314
+ cwd,
2315
+ // Each reviewer's own prompt and `permissionMode: 'plan'` travel with it.
2316
+ // They were being dropped, so every reviewer got the shared task under
2317
+ // `bypassPermissions` — a read-only review that could edit the code.
2318
+ agents,
2319
+ synthesize: true,
2320
+ // The synthesiser reads the reviewers' text, not the code, and it shares
2321
+ // the project directory — so it is held to the same read-only rule. It
2322
+ // was not, which meant an ultrareview could still write through its
2323
+ // final pass.
2324
+ synthesisPermissionMode: 'plan',
2325
+ maxTurnsPerAgent: 20,
2326
+ timeoutMs: maxMinutes * 60 * 1000,
2327
+ }), { runId, cwd });
2328
+ // `councilId` is kept for the UltrareviewResult contract; it holds the run
2329
+ // id, which is also the fan-out id — they are the same run now.
2330
+ result.councilId = runId;
2357
2331
  return result;
2358
2332
  }
2359
2333
  ultrareviewStatus(id) {
2360
- return this.ultrareviews.get(id);
2334
+ const record = loadRun(id);
2335
+ if (!record || record.workflow !== 'ultrareview')
2336
+ return undefined;
2337
+ const data = record.nodes[LEGACY_NODE]?.data;
2338
+ return toUltrareviewResult(record, joinFindings(data));
2361
2339
  }
2362
2340
  // ─── Autoloop (three-agent architecture) ───────────────────────────
2363
- autoloops = new Map();
2364
- // runIds currently being torn down by autoloopDelete. Guards against a
2365
- // concurrent autoloopStart recreating the same id (or autoloopChat using a
2366
- // dispatcher mid-shutdown) during the async delete window.
2367
- _deletingAutoloops = new Set();
2341
+ // No map, no registry file, and no start/delete fences.
2342
+ //
2343
+ // What used to live here: `autoloops`, holding the live runner and dispatcher;
2344
+ // `_deletingAutoloops` and `_startingAutoloops`, two `Set`s that existed only
2345
+ // because a start and a delete could race each other over that map; and four
2346
+ // bespoke helpers over `~/.claw-orchestrator/autoloop-registry.jsonl` for
2347
+ // cross-process listing. A run has exactly one owner now, run ids collide in
2348
+ // the run store rather than in a map that only saw this process, and the
2349
+ // record is the registry.
2368
2350
  /**
2369
- * Runs whose Planner is mid-startup. The delete fence was one-directional:
2370
- * a start could not race a delete, but a delete COULD race a start — it would
2371
- * resolve `true`, drop the registry row, and leave the still-starting Planner
2372
- * session orphaned (no run to stop it, no entry to find it by). Deleting a run
2373
- * that is still coming up is rejected instead.
2351
+ * Build and start the Planner/Coder/Reviewer engine for a run.
2352
+ *
2353
+ * This is everything `autoloopStart` used to be except the bookkeeping: the
2354
+ * `autoloops` map and the private JSONL registry are gone, and the kernel owns
2355
+ * the lifecycle. Called from the `autoloop` node executor, which holds the
2356
+ * returned objects for as long as the loop runs.
2374
2357
  */
2375
- _startingAutoloops = new Set();
2376
2358
  /**
2377
- * Start a v2 autoloop in chat mode. Creates the Planner persistent session,
2378
- * returns the run handle. Coder/Reviewer are NOT started until S3's
2379
- * spawn_subagents tool is called.
2359
+ * Resolve until the loop stops.
2360
+ *
2361
+ * The runner is an event emitter, not a promise: it settles when a
2362
+ * `terminate` envelope is drained or the phase-error circuit trips. Cancelling
2363
+ * the run stops it too, which is what makes `workflow_cancel` work on an
2364
+ * autoloop.
2380
2365
  */
2381
- async autoloopStart(opts) {
2382
- if (this.autoloops.has(opts.runId)) {
2383
- throw new Error(`Autoloop with id '${opts.runId}' already exists`);
2384
- }
2385
- if (this._deletingAutoloops.has(opts.runId)) {
2386
- throw new Error(`Autoloop with id '${opts.runId}' is being deleted`);
2387
- }
2366
+ _awaitAutoloopExit(handle, signal) {
2367
+ return new Promise((resolve) => {
2368
+ const runner = handle.runner;
2369
+ const done = () => runner.state.status === 'terminated' || runner.state.status === 'crashed';
2370
+ if (done())
2371
+ return resolve();
2372
+ const check = () => {
2373
+ if (done() || signal.aborted) {
2374
+ runner.off('state', check);
2375
+ clearInterval(poll);
2376
+ if (signal.aborted) {
2377
+ // Cancelling a run has to tear the loop down the way a stop does.
2378
+ // Without this the three persistent agents keep running and their
2379
+ // session names stay claimed, so the run cannot be restarted — the
2380
+ // failure looks like "session name already in use" a long way from
2381
+ // its cause.
2382
+ runner.stop();
2383
+ void handle.dispatcher.shutdown('cancelled').catch(() => undefined);
2384
+ }
2385
+ resolve();
2386
+ }
2387
+ };
2388
+ runner.on('state', check);
2389
+ // The runner emits on state changes, but a cancel arrives out of band and
2390
+ // a crashed loop may emit nothing at all, so poll as the backstop.
2391
+ const poll = setInterval(check, 1000);
2392
+ if (typeof poll.unref === 'function')
2393
+ poll.unref();
2394
+ });
2395
+ }
2396
+ async _bootAutoloop(opts) {
2388
2397
  const plannerEngine = validateAutoloopRole('planner', opts.plannerEngine, opts.plannerCustomEngine);
2389
2398
  const coderEngine = validateAutoloopRole('coder', opts.coderEngine, opts.coderCustomEngine);
2390
2399
  const reviewerEngine = validateAutoloopRole('reviewer', opts.reviewerEngine, opts.reviewerCustomEngine);
@@ -2427,24 +2436,11 @@ export class SessionManager {
2427
2436
  runnerRef?.markSubagentsSpawned();
2428
2437
  },
2429
2438
  onRoleSelectionChanged: async (selection) => {
2430
- try {
2431
- upsertAutoloopRegistry(DEFAULT_AUTOLOOP_REGISTRY, {
2432
- run_id: runId,
2433
- workspace: opts.workspace,
2434
- ledger_dir: ledgerDir,
2435
- started_at: runnerRef?.state.started_at ?? new Date().toISOString(),
2436
- planner_session: dispatcherRef?.sessionNames.planner ?? `autoloop-${runId}-planner`,
2437
- planner_engine: plannerEngine,
2438
- planner_model: opts.plannerModel,
2439
- coder_engine: selection.coder.engine,
2440
- coder_model: selection.coder.model,
2441
- reviewer_engine: selection.reviewer.engine,
2442
- reviewer_model: selection.reviewer.model,
2443
- });
2444
- }
2445
- catch (err) {
2446
- this.logger.warn?.(`[autoloop/${runId}] registry update after spawn failed: ${err.message}`);
2447
- }
2439
+ // Used to write a row into a private append-only registry file. The run
2440
+ // record is the registry now, so this just refreshes the published
2441
+ // payload the `autoloop_status` projection reads.
2442
+ this._autoloopSelection.set(runId, selection);
2443
+ this._autoloopPublishers.get(runId)?.();
2448
2444
  },
2449
2445
  };
2450
2446
  const dispatcher = new ClaudeAgentDispatcher(dispatcherConfig);
@@ -2475,19 +2471,10 @@ export class SessionManager {
2475
2471
  dispatcher,
2476
2472
  });
2477
2473
  runnerRef = runner;
2478
- this.autoloops.set(opts.runId, {
2479
- runner,
2480
- dispatcher,
2481
- workspace: opts.workspace,
2482
- ledgerDir,
2483
- pushPolicy,
2484
- });
2485
- this._startingAutoloops.add(opts.runId);
2486
2474
  try {
2487
2475
  await runner.start();
2488
2476
  }
2489
2477
  catch (err) {
2490
- this.autoloops.delete(opts.runId);
2491
2478
  try {
2492
2479
  await dispatcher.shutdown('start-failed', { purge: true });
2493
2480
  }
@@ -2497,44 +2484,82 @@ export class SessionManager {
2497
2484
  runner.stop();
2498
2485
  throw err;
2499
2486
  }
2500
- finally {
2501
- this._startingAutoloops.delete(opts.runId);
2487
+ return { runner, dispatcher, ledgerDir, pushPolicy };
2488
+ }
2489
+ /**
2490
+ * Start a v2 autoloop in chat mode. Creates the Planner persistent session,
2491
+ * returns the run handle. Coder/Reviewer are NOT started until S3's
2492
+ * spawn_subagents tool is called.
2493
+ *
2494
+ * The run is a kernel run whose single `autoloop` node holds the loop for as
2495
+ * long as it lives. That is what replaced the `autoloops` map, the
2496
+ * `autoloop-registry.jsonl` file with its four bespoke read/write helpers, and
2497
+ * the two `Set`s that fenced start against delete: a run has one owner now,
2498
+ * and `runId` collisions are refused by the run store rather than by a map
2499
+ * lookup that only saw this process.
2500
+ */
2501
+ async autoloopStart(opts) {
2502
+ // Fail before the run directory exists, so a rejected start leaves nothing.
2503
+ validateAutoloopRole('planner', opts.plannerEngine, opts.plannerCustomEngine);
2504
+ validateAutoloopRole('coder', opts.coderEngine, opts.coderCustomEngine);
2505
+ validateAutoloopRole('reviewer', opts.reviewerEngine, opts.reviewerCustomEngine);
2506
+ for (const role of ['planner', 'coder', 'reviewer']) {
2507
+ const sessionName = `autoloop-${opts.runId}-${role}`;
2508
+ if (this.sessions.has(sessionName) || this._pendingSessions.has(sessionName)) {
2509
+ throw new Error(`Autoloop session name '${sessionName}' is already in use`);
2510
+ }
2502
2511
  }
2503
- // Record into the cross-process registry so the dashboard / another
2504
- // SessionManager instance can list this run even after it ends. Best
2505
- // effort — registry failure should not block the run.
2512
+ const tag = `${opts.runId}:${randomUUID()}`;
2513
+ const ready = new Promise((resolve, reject) => {
2514
+ this._autoloopReady.set(tag, { resolve, reject });
2515
+ });
2516
+ this._autoloopStarting.set(tag, opts.runId);
2517
+ // Custom-engine configs hold credentials and the spec is written to disk, so
2518
+ // they travel in memory. Without this split, `spec.json` contained the token
2519
+ // from `CustomEngineConfig.env` in plain text.
2520
+ const { plannerCustomEngine, coderCustomEngine, reviewerCustomEngine, ...persistable } = opts;
2521
+ await this.kernel.start({
2522
+ name: 'autoloop',
2523
+ cwd: opts.workspace,
2524
+ nodes: [
2525
+ {
2526
+ id: LEGACY_NODE,
2527
+ kind: 'autoloop',
2528
+ workspace: opts.workspace,
2529
+ config: persistable,
2530
+ },
2531
+ ],
2532
+ }, {
2533
+ runId: opts.runId,
2534
+ cwd: opts.workspace,
2535
+ tag,
2536
+ secrets: { plannerCustomEngine, coderCustomEngine, reviewerCustomEngine },
2537
+ });
2506
2538
  try {
2507
- upsertAutoloopRegistry(DEFAULT_AUTOLOOP_REGISTRY, {
2508
- run_id: opts.runId,
2509
- workspace: opts.workspace,
2510
- ledger_dir: ledgerDir,
2511
- started_at: runner.state.started_at,
2512
- planner_session: dispatcher.sessionNames.planner,
2513
- planner_engine: plannerEngine,
2514
- planner_model: opts.plannerModel,
2515
- coder_engine: coderEngine,
2516
- coder_model: opts.coderModel,
2517
- reviewer_engine: reviewerEngine,
2518
- reviewer_model: opts.reviewerModel,
2519
- });
2539
+ const { plannerSession, state } = await ready;
2540
+ return { runId: opts.runId, plannerSession, state };
2520
2541
  }
2521
2542
  catch (err) {
2522
- this.logger.warn?.(`[autoloop/${runId}] registry append failed: ${err.message}`);
2543
+ // A start that never came up must not leave the id claimed. The store
2544
+ // refuses to reuse a run id, so without this a failed Planner startup
2545
+ // would make that id permanently unusable.
2546
+ //
2547
+ // Tag-guarded: by the time this runs, a retry may already hold the id, and
2548
+ // deleting it would take out the run that replaced us.
2549
+ this.kernel.delete(opts.runId, { expectTag: tag });
2550
+ throw err;
2551
+ }
2552
+ finally {
2553
+ this._autoloopReady.delete(tag);
2554
+ this._autoloopStarting.delete(tag);
2523
2555
  }
2524
- return {
2525
- runId: opts.runId,
2526
- plannerSession: dispatcher.sessionNames.planner,
2527
- state: runner.state,
2528
- };
2529
2556
  }
2530
2557
  /**
2531
2558
  * Inject a user chat message into a v2 run's Planner. Returns the Planner's
2532
2559
  * natural-language reply.
2533
2560
  */
2534
2561
  async autoloopChat(runId, text) {
2535
- const ctx = this.autoloops.get(runId);
2536
- if (!ctx || this._deletingAutoloops.has(runId))
2537
- throw new Error(`Autoloop run '${runId}' not found`);
2562
+ const ctx = this._liveAutoloop(runId);
2538
2563
  let reply = '';
2539
2564
  const onReply = (...args) => {
2540
2565
  const t = args[0];
@@ -2550,74 +2575,54 @@ export class SessionManager {
2550
2575
  }
2551
2576
  return { reply };
2552
2577
  }
2578
+ /**
2579
+ * The running loop for a run, or a clear reason why there is not one.
2580
+ *
2581
+ * Chatting with a Planner needs the live dispatcher; a run that finished or
2582
+ * belongs to another process has a readable record and no one to talk to.
2583
+ */
2584
+ _liveAutoloop(runId) {
2585
+ const handle = this.kernel.handle(runId, LEGACY_NODE);
2586
+ if (handle)
2587
+ return handle;
2588
+ const record = loadRun(runId);
2589
+ if (!record || record.workflow !== 'autoloop')
2590
+ throw new Error(`Autoloop run '${runId}' not found`);
2591
+ throw new Error(`Autoloop run '${runId}' is ${record.state} and not running in this process — resume it before chatting`);
2592
+ }
2553
2593
  autoloopStatus(runId) {
2554
- const live = this.autoloops.get(runId)?.runner.state;
2594
+ const live = this.kernel.handle(runId, LEGACY_NODE)?.runner.state;
2555
2595
  if (live)
2556
2596
  return live;
2557
- // Fallback: rebuild a terminated-state shape from the cross-process
2558
- // registry so the dashboard can open historical runs (read chat
2559
- // history, view plan.md, push_log) instead of hanging on a 404 forever.
2560
- const entry = listAutoloopsFromRegistry().find((e) => e.run_id === runId);
2561
- if (!entry)
2597
+ // Not running here. The record still holds the last state the loop
2598
+ // published, so a historical run opens with its real iteration count and
2599
+ // workspace instead of the all-zero stub the registry fallback produced.
2600
+ const record = loadRun(runId);
2601
+ if (!record || record.workflow !== 'autoloop')
2562
2602
  return undefined;
2563
- return {
2564
- run_id: entry.run_id,
2565
- status: 'terminated',
2566
- iter: 0,
2567
- subagents_spawned: false,
2568
- started_at: entry.started_at,
2569
- workspace: entry.workspace,
2570
- ledger_dir: entry.ledger_dir,
2571
- push_log_count: 0,
2572
- status_reason: 'reconstructed from registry — not in current process memory',
2573
- consecutive_phase_errors: 0,
2574
- recent_phase_errors: [],
2575
- metric_history: [],
2576
- last_activity_at: 0,
2577
- };
2603
+ return autoloopStateFromRecord(record);
2578
2604
  }
2579
2605
  autoloopList() {
2580
- const inMemory = Array.from(this.autoloops.values()).map((c) => c.runner.state);
2581
- const inMemIds = new Set(inMemory.map((s) => s.run_id));
2582
- const fromDisk = listAutoloopsFromRegistry()
2583
- .filter((e) => !inMemIds.has(e.run_id))
2584
- .map((e) => ({
2585
- run_id: e.run_id,
2586
- status: 'terminated',
2587
- iter: 0,
2588
- subagents_spawned: false,
2589
- started_at: e.started_at,
2590
- workspace: e.workspace,
2591
- ledger_dir: e.ledger_dir,
2592
- push_log_count: 0,
2593
- status_reason: 'reconstructed from registry — not in current process memory',
2594
- consecutive_phase_errors: 0,
2595
- recent_phase_errors: [],
2596
- metric_history: [],
2597
- last_activity_at: 0,
2598
- }));
2599
- return [...inMemory, ...fromDisk].sort((a, b) => (b.started_at || '').localeCompare(a.started_at || ''));
2606
+ return this.kernel
2607
+ .list({ workflow: 'autoloop' })
2608
+ .map((r) => this.autoloopStatus(r.runId))
2609
+ .filter((s) => Boolean(s));
2600
2610
  }
2601
- /**
2602
- * Reset a single subagent on a v2 run. Useful when an agent has drifted
2603
- * (chat memory implies hallucination, repeated rejects, or context bloat).
2604
- * Coder/Reviewer: safe to reset; the next directive/review_request will
2605
- * re-prime from system prompt + ledger artifacts.
2606
- * Planner: requires force=true and discards user-conversation context.
2607
- */
2608
2611
  async autoloopResetAgent(runId, agent, opts = {}) {
2609
- const ctx = this.autoloops.get(runId);
2612
+ const ctx = this.kernel.handle(runId, LEGACY_NODE);
2610
2613
  if (!ctx)
2611
2614
  return false;
2612
2615
  await ctx.dispatcher.resetAgent(agent, opts);
2613
2616
  return true;
2614
2617
  }
2615
2618
  async autoloopStop(runId, reason = 'user-stop') {
2616
- const ctx = this.autoloops.get(runId);
2619
+ const ctx = this.kernel.handle(runId, LEGACY_NODE);
2617
2620
  if (!ctx)
2618
2621
  return false;
2622
+ // Soft stop: a terminate envelope, so the three persistent agents shut down
2623
+ // and the persisted sessions survive for a later resume. The node's exit
2624
+ // watcher sees the status change and lets the run finish on its own.
2619
2625
  await ctx.runner.send(AutoloopMsg.terminate(ctx.runner.state.iter, { reason }));
2620
- this.autoloops.delete(runId);
2621
2626
  return true;
2622
2627
  }
2623
2628
  /**
@@ -2637,62 +2642,113 @@ export class SessionManager {
2637
2642
  * is still served via /autoloop/<id>/chat_history so the dashboard can
2638
2643
  * replay the conversation visually.
2639
2644
  */
2645
+ /**
2646
+ * Which roles of a stored autoloop run need a custom-engine config before it
2647
+ * can be resumed.
2648
+ *
2649
+ * Custom-engine configs are never persisted, so a resume has to be given them
2650
+ * again — and a caller that cannot find out which roles need one can only
2651
+ * guess. The dashboard's Resume button used to send an empty body
2652
+ * unconditionally, which meant a custom-engine run could be resumed from the
2653
+ * library and from the HTTP API but not from the UI that offers the button.
2654
+ *
2655
+ * Returns role names only. Nothing here is sensitive: the engine kind is
2656
+ * already in `spec.json`, and the answer is a list of roles, not credentials.
2657
+ */
2658
+ autoloopResumeRequirements(runId) {
2659
+ const record = loadRun(runId);
2660
+ if (!record || record.workflow !== 'autoloop')
2661
+ throw new Error(`Autoloop run '${runId}' not found`);
2662
+ const config = record.spec.nodes.find((n) => n.id === LEGACY_NODE)
2663
+ ?.config;
2664
+ const roles = [];
2665
+ for (const role of ['planner', 'coder', 'reviewer']) {
2666
+ if (config?.[`${role}Engine`] === 'custom')
2667
+ roles.push(role);
2668
+ }
2669
+ return { runId, rolesNeedingCustomEngine: roles };
2670
+ }
2640
2671
  async autoloopResume(runId, opts = {}) {
2641
- const existing = this.autoloops.get(runId);
2642
- if (existing)
2643
- return existing.runner.state;
2644
- const entry = listAutoloopsFromRegistry().find((e) => e.run_id === runId);
2645
- if (!entry)
2646
- throw new Error(`Autoloop run '${runId}' not found in registry`);
2647
- // Validate the full restart configuration before touching the registry.
2648
- // Old rows omit these fields and intentionally recover the legacy Claude defaults.
2649
- const plannerEngine = validateAutoloopRole('planner', entry.planner_engine, opts.plannerCustomEngine);
2650
- const coderEngine = validateAutoloopRole('coder', entry.coder_engine, opts.coderCustomEngine);
2651
- const reviewerEngine = validateAutoloopRole('reviewer', entry.reviewer_engine, opts.reviewerCustomEngine);
2652
- // The registry is append-only and newest entry wins. Leave the prior row
2653
- // untouched while starting so a transient failure cannot erase or restore
2654
- // stale cross-process state. A successful start appends the replacement.
2655
- return (await this.autoloopStart({
2656
- runId: entry.run_id,
2657
- workspace: entry.workspace,
2658
- plannerEngine,
2659
- plannerModel: entry.planner_model,
2672
+ const live = this.kernel.handle(runId, LEGACY_NODE);
2673
+ if (live)
2674
+ return live.runner.state;
2675
+ const record = loadRun(runId);
2676
+ if (!record || record.workflow !== 'autoloop')
2677
+ throw new Error(`Autoloop run '${runId}' not found`);
2678
+ const config = record.spec.nodes.find((n) => n.id === LEGACY_NODE)
2679
+ ?.config;
2680
+ if (!config)
2681
+ throw new Error(`Autoloop run '${runId}' has no stored configuration to restart from`);
2682
+ // Validate the full restart configuration before touching anything. The
2683
+ // spec is the immutable record of how the run was started, so a resume
2684
+ // reproduces it exactly instead of reconstructing it from a registry row
2685
+ // whose older versions omitted the engine fields entirely.
2686
+ validateAutoloopRole('planner', config.plannerEngine, opts.plannerCustomEngine);
2687
+ validateAutoloopRole('coder', config.coderEngine, opts.coderCustomEngine);
2688
+ validateAutoloopRole('reviewer', config.reviewerEngine, opts.reviewerCustomEngine);
2689
+ // Custom-engine configs are never persisted (they can carry secrets), so a
2690
+ // resume must be given them again by the caller.
2691
+ const resumed = await this._resumeAutoloopRun(runId, {
2692
+ ...config,
2660
2693
  plannerCustomEngine: opts.plannerCustomEngine,
2661
- coderEngine,
2662
- coderModel: entry.coder_model,
2663
2694
  coderCustomEngine: opts.coderCustomEngine,
2664
- reviewerEngine,
2665
- reviewerModel: entry.reviewer_model,
2666
2695
  reviewerCustomEngine: opts.reviewerCustomEngine,
2667
- })).state;
2696
+ });
2697
+ return resumed;
2698
+ }
2699
+ /** Re-attach a stored autoloop run: same run id, same spec, fresh engine. */
2700
+ async _resumeAutoloopRun(runId, config) {
2701
+ const tag = `${runId}:${randomUUID()}`;
2702
+ const ready = new Promise((resolve, reject) => {
2703
+ this._autoloopReady.set(tag, { resolve, reject });
2704
+ });
2705
+ this._autoloopStarting.set(tag, runId);
2706
+ // The custom-engine configs the caller re-supplied go into the run's secret
2707
+ // bag, which is where the node reads them from. They used to be stashed in a
2708
+ // separate map the executor no longer consulted, so a resume in a fresh
2709
+ // process — the case that matters — silently got none of them.
2710
+ const secrets = {
2711
+ plannerCustomEngine: config.plannerCustomEngine,
2712
+ coderCustomEngine: config.coderCustomEngine,
2713
+ reviewerCustomEngine: config.reviewerCustomEngine,
2714
+ };
2715
+ try {
2716
+ // `restart: true` because an autoloop resume means "bring the loop back
2717
+ // up", not "carry on from where the kernel left off" — the run is
2718
+ // normally terminated when someone resumes it.
2719
+ const record = await this.kernel.resume(runId, { restart: true, secrets, tag });
2720
+ // Race readiness against the run ending: a node that fails before it
2721
+ // publishes would otherwise leave this awaiting a signal that is never
2722
+ // coming.
2723
+ const finished = this.kernel
2724
+ .wait(record.runId)
2725
+ .then((r) => Promise.reject(new Error(r?.error ?? `autoloop run '${runId}' ended before it came up`)));
2726
+ const { state } = await Promise.race([ready, finished]);
2727
+ return state;
2728
+ }
2729
+ finally {
2730
+ this._autoloopReady.delete(tag);
2731
+ this._autoloopStarting.delete(tag);
2732
+ }
2668
2733
  }
2669
2734
  /**
2670
- * Delete a run from the system: stop the runner if it's still alive in this
2671
- * process, then scrub the row from the cross-process registry so it stops
2672
- * appearing in `autoloop_list` / the dashboard. The ledger directory on disk
2673
- * is NOT removed — postmortem artifacts (chat history, push log, plan.md)
2674
- * are kept for the user to inspect or `rm` manually.
2735
+ * Delete a run: really gone, not paused.
2675
2736
  *
2676
- * Returns true if anything was removed (in-memory entry OR registry row).
2737
+ * The two `Set` fences this used to open with — one refusing a delete while a
2738
+ * start was in flight, one blocking a concurrent start during the async
2739
+ * teardown — protected a shared `Map` that no longer exists. Cancelling the
2740
+ * run is what stops it, and the run store refuses to recreate a live id.
2677
2741
  */
2678
2742
  async autoloopDelete(runId) {
2679
2743
  // Refuse to tear down a run that is still coming up: its Planner session is
2680
- // mid-startSession, so deleting now would drop the registry row and orphan a
2681
- // session that finishes starting a moment later.
2682
- if (this._startingAutoloops.has(runId)) {
2744
+ // mid-startSession, so deleting now would drop the run and orphan a session
2745
+ // that finishes starting a moment later. `_autoloopReady` holds an entry for
2746
+ // exactly the window between "run created" and "engine up", which is the
2747
+ // window that used to need a dedicated `_startingAutoloops` Set.
2748
+ if ([...this._autoloopStarting.values()].includes(runId)) {
2683
2749
  throw new Error(`Autoloop with id '${runId}' is still starting`);
2684
2750
  }
2685
- // Fence the async teardown so a concurrent start/chat can't race on this id.
2686
- this._deletingAutoloops.add(runId);
2687
- try {
2688
- return await this._autoloopDeleteInner(runId);
2689
- }
2690
- finally {
2691
- this._deletingAutoloops.delete(runId);
2692
- }
2693
- }
2694
- async _autoloopDeleteInner(runId) {
2695
- const ctx = this.autoloops.get(runId);
2751
+ const ctx = this.kernel.handle(runId, LEGACY_NODE);
2696
2752
  let touched = false;
2697
2753
  if (ctx) {
2698
2754
  // Delete = "really gone". Call dispatcher.shutdown directly with
@@ -2714,7 +2770,7 @@ export class SessionManager {
2714
2770
  catch {
2715
2771
  /* runner may already be stopped */
2716
2772
  }
2717
- this.autoloops.delete(runId);
2773
+ this.kernel.cancel(runId);
2718
2774
  touched = true;
2719
2775
  }
2720
2776
  else {
@@ -2731,22 +2787,20 @@ export class SessionManager {
2731
2787
  this.persistedSessions.delete(`autoloop-${runId}-reviewer`);
2732
2788
  savePersistedSessions(this.persistedSessions, this.logger);
2733
2789
  }
2734
- try {
2735
- const removed = removeAutoloopFromRegistry(DEFAULT_AUTOLOOP_REGISTRY, runId);
2736
- if (removed > 0)
2737
- touched = true;
2738
- }
2739
- catch (err) {
2740
- this.logger.warn?.(`[autoloop/${runId}] registry scrub failed: ${err.message}`);
2790
+ // No registry to scrub: the run record IS the registry, and removing it is
2791
+ // the delete. The ledger directory under tasks/<runId>/ is deliberately left
2792
+ // alone — postmortem artifacts (chat history, push log, plan.md) outlive the
2793
+ // run, exactly as before.
2794
+ if (loadRun(runId)) {
2795
+ this.kernel.delete(runId);
2796
+ touched = true;
2741
2797
  }
2742
2798
  return touched;
2743
2799
  }
2744
- /** Used by embedded-server to attach SSE listeners. */
2800
+ /** Used by embedded-server to attach SSE listeners. Live runs only. */
2745
2801
  getAutoloop(runId) {
2746
- const ctx = this.autoloops.get(runId);
2747
- if (!ctx)
2748
- return undefined;
2749
- return { runner: ctx.runner, dispatcher: ctx.dispatcher };
2802
+ const handle = this.kernel.handle(runId, LEGACY_NODE);
2803
+ return handle ? { runner: handle.runner, dispatcher: handle.dispatcher } : undefined;
2750
2804
  }
2751
2805
  _cleanupIdleSessions() {
2752
2806
  const ttlMs = this.pluginConfig.sessionTtlMinutes * 60_000;