@enderfga/claw-orchestrator 4.7.0 → 4.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/README.md +8 -9
  2. package/configs/autoloop-coder-prompt.md +4 -4
  3. package/configs/autoloop-planner-prompt.md +10 -10
  4. package/configs/autoloop-reviewer-prompt.md +3 -4
  5. package/dist/src/autoloop/dispatcher.d.ts +59 -10
  6. package/dist/src/autoloop/dispatcher.js +261 -40
  7. package/dist/src/autoloop/dispatcher.js.map +1 -1
  8. package/dist/src/autoloop/planner-tools.d.ts +4 -3
  9. package/dist/src/autoloop/planner-tools.js +35 -2
  10. package/dist/src/autoloop/planner-tools.js.map +1 -1
  11. package/dist/src/autoloop/types.d.ts +2 -0
  12. package/dist/src/autoloop/types.js.map +1 -1
  13. package/dist/src/dashboard/index.html +185 -94
  14. package/dist/src/embedded-server.js +67 -9
  15. package/dist/src/embedded-server.js.map +1 -1
  16. package/dist/src/index.js +84 -52
  17. package/dist/src/index.js.map +1 -1
  18. package/dist/src/persistent-agy-session.js +4 -1
  19. package/dist/src/persistent-agy-session.js.map +1 -1
  20. package/dist/src/persistent-codex-session.js +3 -0
  21. package/dist/src/persistent-codex-session.js.map +1 -1
  22. package/dist/src/persistent-cursor-session.d.ts +6 -0
  23. package/dist/src/persistent-cursor-session.js +98 -3
  24. package/dist/src/persistent-cursor-session.js.map +1 -1
  25. package/dist/src/persistent-custom-session.js +13 -0
  26. package/dist/src/persistent-custom-session.js.map +1 -1
  27. package/dist/src/persistent-gemini-session.d.ts +4 -0
  28. package/dist/src/persistent-gemini-session.js +55 -2
  29. package/dist/src/persistent-gemini-session.js.map +1 -1
  30. package/dist/src/persistent-opencode-session.d.ts +5 -5
  31. package/dist/src/persistent-opencode-session.js +58 -6
  32. package/dist/src/persistent-opencode-session.js.map +1 -1
  33. package/dist/src/persistent-session.js +6 -1
  34. package/dist/src/persistent-session.js.map +1 -1
  35. package/dist/src/session-manager.d.ts +44 -5
  36. package/dist/src/session-manager.js +190 -18
  37. package/dist/src/session-manager.js.map +1 -1
  38. package/dist/src/types.d.ts +3 -2
  39. package/package.json +1 -1
  40. package/skills/SKILL.md +18 -26
  41. package/skills/references/autoloop.md +49 -6
  42. package/skills/references/claude-cli-tracking.md +2 -1
  43. package/skills/references/cli.md +2 -2
  44. package/skills/references/council.md +1 -1
  45. package/skills/references/getting-started.md +4 -4
  46. package/skills/references/multi-engine.md +21 -34
  47. package/skills/references/openai-compat.md +1 -1
  48. package/skills/references/sessions.md +3 -3
  49. package/skills/references/tools.md +39 -20
package/README.md CHANGED
@@ -4,7 +4,7 @@
4
4
 
5
5
  # Claw Orchestrator
6
6
 
7
- > A runtime for coding agents. Wrap Claude Code, Codex, Gemini, Cursor Agent, OpenCode, or any custom CLI as persistent programmable sessions; coordinate them in multi-agent councils; run autonomous Planner / Coder / Reviewer loops; or hand a five-question interview to an Opus council that ships a deployed web app at `localhost:19000/forge/<slug>/`.
7
+ > A runtime for coding agents. Wrap Claude Code, Codex, Antigravity, Cursor Agent, OpenCode, or any custom CLI as persistent programmable sessions; coordinate them in multi-agent councils; run autonomous Planner / Coder / Reviewer loops; or hand a five-question interview to an Opus council that ships a deployed web app at `localhost:19000/forge/<slug>/`.
8
8
 
9
9
  [![npm version](https://img.shields.io/npm/v/@enderfga/claw-orchestrator.svg)](https://www.npmjs.com/package/@enderfga/claw-orchestrator)
10
10
  [![CI](https://github.com/Enderfga/claw-orchestrator/actions/workflows/ci.yml/badge.svg)](https://github.com/Enderfga/claw-orchestrator/actions/workflows/ci.yml)
@@ -27,11 +27,11 @@ https://github.com/user-attachments/assets/fbd2b0ea-28d8-4387-9894-c29cf15ba030
27
27
  | Capability | What it does | Reference |
28
28
  | --------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------- |
29
29
  | **Persistent Sessions** | Long-lived coding agents kept alive across requests, with full context, tool, model, and worktree control. | [`sessions.md`](./skills/references/sessions.md) |
30
- | **Multi-Engine Runtime** | One interface over Claude Code, Codex, Gemini, Antigravity (agy), Cursor Agent, OpenCode, and arbitrary custom CLIs. | [`multi-engine.md`](./skills/references/multi-engine.md) |
30
+ | **Multi-Engine Runtime** | One interface over Claude Code, Codex, Antigravity (agy), Cursor Agent, OpenCode, and arbitrary custom CLIs. | [`multi-engine.md`](./skills/references/multi-engine.md) |
31
31
  | **Multi-Agent Council** | Parallel agents in isolated git worktrees, voting on consensus until they agree. | [`council.md`](./skills/references/council.md) |
32
32
  | **Fan-out** | Run one task across N engine/model agents in parallel and collect their answers, with an optional synthesis pass — the cross-engine best-of-N / diverse-perspective primitive (no rounds or worktrees). | [`tools.md`](./skills/references/tools.md) |
33
33
  | **ultracode** | `session_start({ ultracode: true })` lets Claude orchestrate a dynamic JS workflow and fan out to subagents per task (Claude engine). | [`tools.md`](./skills/references/tools.md) |
34
- | **Autoloop** | Three-agent autonomous workspace iteration. Chat with the Planner; it spawns Coder + Reviewer into a self-iterating subloop and pushes you on regression, target-hit, or decision points. | [`autoloop.md`](./skills/references/autoloop.md) |
34
+ | **Autoloop** | Three-agent autonomous workspace iteration with independent engine/model selection for Planner, Coder, and Reviewer. Chat with the Planner; it spawns Coder + Reviewer into a self-iterating subloop and pushes you on regression, target-hit, or decision points. | [`autoloop.md`](./skills/references/autoloop.md) |
35
35
  | **Ultraapp** | A three-agent Opus council turns a five-question interview into a deployed web app — Tailwind UI, BYOK, file-queue runtime, smoke test, all live at `localhost:19000/forge/<slug>/`. | [`ultraapp.md`](./skills/references/ultraapp.md) |
36
36
  | **Embedded Dashboard** | Three-tab UI for Autoloop, Council, and Forge with sidebar lifecycle controls, per-run live event streaming, and cookie-based auth via a `/login` redirect. | [`dashboard.md`](./skills/references/dashboard.md) |
37
37
  | **OpenAI-Compatible Proxy** | `POST /v1/chat/completions` translates OpenAI requests into native Anthropic, OpenAI, and Google calls and streams responses back in OpenAI shape. Point any OpenAI-SDK client at the orchestrator without changing call sites. | [`openai-compat.md`](./skills/references/openai-compat.md) |
@@ -91,12 +91,11 @@ Register `clawo-mcp` with any MCP-compatible host: Hermes Agent, Claude Desktop,
91
91
 
92
92
  | Engine | CLI | Tested Version |
93
93
  | ------------ | ---------- | -------------- |
94
- | Claude Code | `claude` | 2.1.206 |
95
- | Codex | `codex` | 0.143.0 |
96
- | Gemini | `gemini` | 0.43.0 |
97
- | Antigravity | `agy` | 1.0.16 |
98
- | Cursor Agent | `agent` | 2026.03.30 |
99
- | OpenCode | `opencode` | 1.1.40 |
94
+ | Claude Code | `claude` | 2.1.207 |
95
+ | Codex | `codex` | 0.144.1 |
96
+ | Antigravity | `agy` | 1.1.1 |
97
+ | Cursor Agent | `agent` | 2026.07.09-a3815c0 |
98
+ | OpenCode | `opencode` | 1.17.15 |
100
99
  | Custom CLI | any | — |
101
100
 
102
101
  Any coding CLI that runs as a subprocess can be wired up as a custom engine — see [`multi-engine.md`](./skills/references/multi-engine.md#custom-engine-enginecustom).
@@ -46,10 +46,10 @@ make Reviewer audits unreliable.
46
46
 
47
47
  ## Your tools
48
48
 
49
- You are a Claude Code session with the workspace as cwd. You have the full
50
- file-editing palette: Read, Write, Edit, Glob, Grep, Bash. Use them freely
51
- on workspace code — that's your job. The role boundary is Rule 1: do not
52
- touch `plan.md`, `goal.json`, or `tasks/`.
49
+ You are a coding-agent session with the workspace as cwd. Use the selected
50
+ engine's file-reading, editing, search, and shell tools freely on workspace
51
+ code — that's your job. The role boundary is Rule 1: do not touch `plan.md`,
52
+ `goal.json`, or `tasks/`.
53
53
 
54
54
  You also have **autoloop control tools** via fenced JSON blocks:
55
55
 
@@ -41,15 +41,14 @@ a one-paragraph email is a Coder task.
41
41
  > 5. Ask: "Plan ready, spawn the Coder?"
42
42
  > 6. On approval, `spawn_subagents` with an initial directive.
43
43
 
44
- ### Rule 2 — You CANNOT use Write / Edit / MultiEdit / NotebookEdit
44
+ ### Rule 2 — You CANNOT edit workspace source files
45
45
 
46
- These tools have been stripped from your session. Trying to call them will
47
- error. This is by design: it physically prevents Rule 1 from being violated.
48
-
49
- The only way you can author files is the `write_plan` and `write_goal`
50
- autoloop tools, and those only write to `plan.md` and `goal.json`. There is
51
- no escape hatch. Bash heredocs that try to write content files are also
52
- out-of-bounds — they violate Rule 1 even though they're technically possible.
46
+ The orchestrator starts Planner sessions in the engine's native read-only /
47
+ plan mode. Claude additionally has Write / Edit / MultiEdit / NotebookEdit
48
+ removed. The only supported authoring path is the `write_plan` and
49
+ `write_goal` autoloop tools, which write only `plan.md` and `goal.json`.
50
+ Bash heredocs, `tee`, or output redirection that author content files violate
51
+ Rule 1 and must not be used.
53
52
 
54
53
  ### Rule 3 — Never `spawn_subagents` without explicit user approval
55
54
 
@@ -61,7 +60,8 @@ and wait for go / ok / 开干 / 干 / yes / similar. The only exception:
61
60
 
62
61
  ## Your tools
63
62
 
64
- You are a Claude Code session with the workspace as cwd. You have:
63
+ You are a read-only coding-agent session with the workspace as cwd. Depending
64
+ on the selected engine, equivalent tool names may differ. You have:
65
65
 
66
66
  | Tool | Purpose |
67
67
  |---|---|
@@ -91,7 +91,7 @@ turn. Anything outside the blocks is shown to the user as your chat reply.
91
91
  | `write_plan` | `content` (full plan.md body as string), `commit_message?` | Writes `plan.md` to the workspace and git-commits. Re-running replaces the whole file (no patches). |
92
92
  | `write_goal` | `content` (full goal.json body as string), `commit_message?` | Same, for `goal.json`. The orchestrator parses the content as JSON before writing; malformed JSON errors back to you. |
93
93
  | `notify_user` | `level` ('info'/'warn'/'decision'/'error'), `summary` (one line), `detail?`, `channel?` ('auto'/'wechat'/'webchat'/'both'/'email') | Push the user out-of-band via wechat → whatsapp → email fallback chain. Use sparingly: 5-min dedup applies. |
94
- | `spawn_subagents` | `coder_model?`, `reviewer_model?`, `initial_directive?: { goal, constraints?, success_criteria?, max_attempts? }` | Start the Coder + Reviewer subloop. Call this **only when the user has explicitly approved the plan** (Rule 3). Optionally include the first directive. |
94
+ | `spawn_subagents` | `coder_engine?`, `coder_model?`, `reviewer_engine?`, `reviewer_model?`, `initial_directive?: { goal, constraints?, success_criteria?, max_attempts? }` | Start the Coder + Reviewer subloop. Omitted fields inherit the run defaults; changing engine/model after that role's session has started is rejected. `custom` may only select a config preloaded by `autoloop_start` or resume; never include custom config, `env`, or `extra` in this block. Call this **only when the user has explicitly approved the plan** (Rule 3). |
95
95
  | `send_directive` | `goal`, `constraints?`, `success_criteria?`, `max_attempts?` | Send a fresh directive to Coder for the next iter. |
96
96
  | `pause_loop` | `reason` | Halt the Coder/Reviewer subloop at the next iter boundary (you can keep chatting). |
97
97
  | `resume_loop` | `{}` | Resume after a pause. |
@@ -35,10 +35,9 @@ orchestrator stalls if you skip it.
35
35
 
36
36
  ## Your tools
37
37
 
38
- Standard Claude Code palette in the sandbox cwd: Read, Glob, Grep, Bash.
39
- You technically have Write/Edit too, but Rule 3 confines you to the
40
- sandbox cwd. The orchestrator does not enforce this at the tool level —
41
- it enforces it by trusting you.
38
+ Use the selected engine's read, search, and shell tools inside the sandbox
39
+ cwd. You may edit only sandbox memory, scratch, and audit files. Rule 3
40
+ forbids writes outside that cwd; engine-specific tool names may differ.
42
41
 
43
42
  The current contents of `reviewer_memory.md` are **injected as a frozen
44
43
  snapshot in your system prompt** at session start. You do not need to
@@ -1,9 +1,7 @@
1
1
  /**
2
- * ClaudeAgentDispatcher — wires the v2 runner to real persistent Claude
3
- * sessions managed by SessionManager.
4
- *
5
- * S2 scope: Planner only (chat-mode, no subagents yet). Coder/Reviewer
6
- * delivery throws — S4 wires them in.
2
+ * ClaudeAgentDispatcher — wires the v2 runner to real persistent coding
3
+ * sessions managed by SessionManager. The historical class name is retained
4
+ * for compatibility; each Autoloop role may use a different engine.
7
5
  *
8
6
  * Naming convention:
9
7
  * autoloop-<run_id>-planner
@@ -21,6 +19,7 @@
21
19
  import { EventEmitter } from 'node:events';
22
20
  import type { SessionManager } from '../session-manager.js';
23
21
  import type { Logger } from '../logger.js';
22
+ import { type CustomEngineConfig, type EngineType } from '../types.js';
24
23
  import { type AnyAutoloopMessage } from './messages.js';
25
24
  import { type AgentDispatcher, type AutoloopState, type PushPolicy } from './types.js';
26
25
  import { type SpawnSubagentsArgs } from './planner-tools.js';
@@ -33,12 +32,18 @@ export interface ClaudeAgentDispatcherConfig {
33
32
  /** Override Coder/Reviewer prompt paths (defaults walk-up to configs/autoloop-{coder,reviewer}-prompt.md). */
34
33
  coderPromptPath?: string;
35
34
  reviewerPromptPath?: string;
36
- /** Model alias for Planner (default: 'opus'). */
35
+ /** Planner engine/model (default: claude/opus). */
36
+ plannerEngine?: EngineType;
37
37
  plannerModel?: string;
38
- /** Default Coder model (default: 'sonnet'). Can be overridden per spawn_subagents call. */
38
+ plannerCustomEngine?: CustomEngineConfig;
39
+ /** Coder defaults. Engine/model can be overridden per spawn_subagents call. */
40
+ coderEngine?: EngineType;
39
41
  coderModel?: string;
40
- /** Default Reviewer model (default: 'sonnet'). */
42
+ coderCustomEngine?: CustomEngineConfig;
43
+ /** Reviewer defaults. Engine/model can be overridden per spawn_subagents call. */
44
+ reviewerEngine?: EngineType;
41
45
  reviewerModel?: string;
46
+ reviewerCustomEngine?: CustomEngineConfig;
42
47
  /** Per-message wall-clock cap. Default 10 min. */
43
48
  sendTimeoutMs?: number;
44
49
  logger?: Logger;
@@ -63,6 +68,17 @@ export interface ClaudeAgentDispatcherConfig {
63
68
  pushPolicyRef?: PushPolicy;
64
69
  /** Called when Planner emits spawn_subagents. S4 implements; S3 records the intent. */
65
70
  onSpawnSubagents?: (args: SpawnSubagentsArgs) => Promise<void>;
71
+ /** Persist the effective non-secret role selection after a successful spawn. */
72
+ onRoleSelectionChanged?: (selection: {
73
+ coder: {
74
+ engine: EngineType;
75
+ model?: string;
76
+ };
77
+ reviewer: {
78
+ engine: EngineType;
79
+ model?: string;
80
+ };
81
+ }) => Promise<void> | void;
66
82
  }
67
83
  export declare class ClaudeAgentDispatcher extends EventEmitter implements AgentDispatcher {
68
84
  readonly config: ClaudeAgentDispatcherConfig;
@@ -76,8 +92,10 @@ export declare class ClaudeAgentDispatcher extends EventEmitter implements Agent
76
92
  private plannerSystemPrompt;
77
93
  private coderSystemPrompt;
78
94
  private reviewerSystemPrompt;
79
- private coderModel;
80
- private reviewerModel;
95
+ private reviewerSessionPrompt;
96
+ private plannerSelection;
97
+ private coderSelection;
98
+ private reviewerSelection;
81
99
  /** Where Reviewer reads from. Created lazily by stageReviewSandbox(). */
82
100
  private reviewerSandboxDir;
83
101
  private ledgerDir;
@@ -92,6 +110,37 @@ export declare class ClaudeAgentDispatcher extends EventEmitter implements Agent
92
110
  purge?: boolean;
93
111
  }): Promise<void>;
94
112
  deliver(env: AnyAutoloopMessage): Promise<AnyAutoloopMessage[]>;
113
+ private roleModel;
114
+ private validateSelection;
115
+ /**
116
+ * Stop a session we started during a failed spawn. Returns true only when the
117
+ * session is genuinely gone — the caller uses that to decide whether it may
118
+ * clear the role's `started` flag. Returning false keeps the role marked as
119
+ * started, which is the safe lie: a later engine change is then rejected
120
+ * instead of silently binding the run to a process that never went away.
121
+ */
122
+ private stopRolledBackSession;
123
+ /**
124
+ * Does this engine carry conversation across sends on its own?
125
+ *
126
+ * claude keeps one subprocess alive; codex / codex-app resume a thread; agy
127
+ * resumes a harvested `--conversation <uuid>`. Everything else (gemini,
128
+ * cursor, opencode, and non-persistent custom engines) spawns a FRESH process
129
+ * per send with zero memory of the last turn — for those the dispatcher must
130
+ * replay the transcript in-band, or the role is amnesiac and a chat-driven
131
+ * Planner can never remember the plan it just proposed (let alone whether the
132
+ * user approved it).
133
+ */
134
+ private hasNativeConversation;
135
+ /**
136
+ * Replayed transcript for engines without native conversation. Capped so a
137
+ * long run can't grow the prompt without bound: we keep the most recent
138
+ * turns within REPLAY_CHAR_BUDGET, oldest dropped first.
139
+ */
140
+ private transcripts;
141
+ private recordTurn;
142
+ private renderHistory;
143
+ private withRoleInstructions;
95
144
  /**
96
145
  * Start Coder + Reviewer sessions. Idempotent. Called in response to a
97
146
  * Planner spawn_subagents tool (the SessionManager wires this via
@@ -1,9 +1,7 @@
1
1
  /**
2
- * ClaudeAgentDispatcher — wires the v2 runner to real persistent Claude
3
- * sessions managed by SessionManager.
4
- *
5
- * S2 scope: Planner only (chat-mode, no subagents yet). Coder/Reviewer
6
- * delivery throws — S4 wires them in.
2
+ * ClaudeAgentDispatcher — wires the v2 runner to real persistent coding
3
+ * sessions managed by SessionManager. The historical class name is retained
4
+ * for compatibility; each Autoloop role may use a different engine.
7
5
  *
8
6
  * Naming convention:
9
7
  * autoloop-<run_id>-planner
@@ -22,12 +20,19 @@ import { EventEmitter } from 'node:events';
22
20
  import * as fs from 'node:fs';
23
21
  import * as path from 'node:path';
24
22
  import { fileURLToPath } from 'node:url';
23
+ import { ENGINE_TYPES } from '../types.js';
25
24
  import { nullLogger } from '../logger.js';
26
25
  import { spawn } from 'node:child_process';
27
26
  import { Msg } from './messages.js';
28
- import { LEDGER_SCHEMA_VERSION } from './types.js';
27
+ import { LEDGER_SCHEMA_VERSION, } from './types.js';
29
28
  import { applyPlannerToolCalls, parsePlannerReply, } from './planner-tools.js';
30
29
  import { extractIterComplete, extractReviewComplete, parseAgentReply } from './agent-tools.js';
30
+ /**
31
+ * Character budget for the replayed transcript handed to engines without native
32
+ * conversation (see hasNativeConversation). Oldest turns are dropped first, so a
33
+ * long run keeps the recent context instead of growing the prompt forever.
34
+ */
35
+ const REPLAY_CHAR_BUDGET = 24_000;
31
36
  /**
32
37
  * Files inside <ledger>/reviewer_sandbox/ that survive `stageReviewSandbox`.
33
38
  * Anything not listed is wiped between iters. `reviewer_memory.md` is also
@@ -70,8 +75,10 @@ export class ClaudeAgentDispatcher extends EventEmitter {
70
75
  plannerSystemPrompt;
71
76
  coderSystemPrompt;
72
77
  reviewerSystemPrompt;
73
- coderModel;
74
- reviewerModel;
78
+ reviewerSessionPrompt = null;
79
+ plannerSelection;
80
+ coderSelection;
81
+ reviewerSelection;
75
82
  /** Where Reviewer reads from. Created lazily by stageReviewSandbox(). */
76
83
  reviewerSandboxDir;
77
84
  ledgerDir;
@@ -86,8 +93,21 @@ export class ClaudeAgentDispatcher extends EventEmitter {
86
93
  this.plannerSystemPrompt = fs.readFileSync(promptPath, 'utf-8');
87
94
  this.coderSystemPrompt = fs.readFileSync(config.coderPromptPath ?? resolveDefaultCoderPrompt(), 'utf-8');
88
95
  this.reviewerSystemPrompt = fs.readFileSync(config.reviewerPromptPath ?? resolveDefaultReviewerPrompt(), 'utf-8');
89
- this.coderModel = config.coderModel ?? 'sonnet';
90
- this.reviewerModel = config.reviewerModel ?? 'sonnet';
96
+ this.plannerSelection = {
97
+ engine: config.plannerEngine ?? 'claude',
98
+ model: config.plannerModel,
99
+ customEngine: config.plannerCustomEngine,
100
+ };
101
+ this.coderSelection = {
102
+ engine: config.coderEngine ?? 'claude',
103
+ model: config.coderModel,
104
+ customEngine: config.coderCustomEngine,
105
+ };
106
+ this.reviewerSelection = {
107
+ engine: config.reviewerEngine ?? 'claude',
108
+ model: config.reviewerModel,
109
+ customEngine: config.reviewerCustomEngine,
110
+ };
91
111
  this.ledgerDir = path.join(config.workspace, 'tasks', config.runId);
92
112
  this.reviewerSandboxDir = path.join(this.ledgerDir, 'reviewer_sandbox');
93
113
  }
@@ -129,18 +149,198 @@ export class ClaudeAgentDispatcher extends EventEmitter {
129
149
  throw new Error(`[autoloop] unexpected dispatcher target: ${env.to}`);
130
150
  }
131
151
  }
152
+ roleModel(role, selection) {
153
+ if (selection.model !== undefined)
154
+ return selection.model;
155
+ if (selection.engine !== 'claude')
156
+ return undefined;
157
+ return role === 'planner' ? 'opus' : 'sonnet';
158
+ }
159
+ validateSelection(role, selection) {
160
+ const label = role[0].toUpperCase() + role.slice(1);
161
+ if (!ENGINE_TYPES.includes(selection.engine)) {
162
+ throw new Error(`${label} engine '${String(selection.engine)}' is not supported`);
163
+ }
164
+ if (selection.engine === 'custom' && !selection.customEngine) {
165
+ throw new Error(`${label} custom engine config is required`);
166
+ }
167
+ }
168
+ /**
169
+ * Stop a session we started during a failed spawn. Returns true only when the
170
+ * session is genuinely gone — the caller uses that to decide whether it may
171
+ * clear the role's `started` flag. Returning false keeps the role marked as
172
+ * started, which is the safe lie: a later engine change is then rejected
173
+ * instead of silently binding the run to a process that never went away.
174
+ */
175
+ async stopRolledBackSession(name) {
176
+ try {
177
+ await this.config.manager.stopSession(name);
178
+ return true;
179
+ }
180
+ catch (stopErr) {
181
+ this.logger.error?.(`[autoloop] rollback could not stop ${name}: ${stopErr.message} — ` +
182
+ `leaving it marked started so a later engine change is rejected rather than silently ignored`);
183
+ this.appendDecisionLog({
184
+ kind: 'phase_error',
185
+ actor: 'dispatcher',
186
+ payload: { agent: name, phase: 'rollback_stop', error: stopErr.message },
187
+ });
188
+ return false;
189
+ }
190
+ }
191
+ /**
192
+ * Does this engine carry conversation across sends on its own?
193
+ *
194
+ * claude keeps one subprocess alive; codex / codex-app resume a thread; agy
195
+ * resumes a harvested `--conversation <uuid>`. Everything else (gemini,
196
+ * cursor, opencode, and non-persistent custom engines) spawns a FRESH process
197
+ * per send with zero memory of the last turn — for those the dispatcher must
198
+ * replay the transcript in-band, or the role is amnesiac and a chat-driven
199
+ * Planner can never remember the plan it just proposed (let alone whether the
200
+ * user approved it).
201
+ */
202
+ hasNativeConversation(selection) {
203
+ switch (selection.engine) {
204
+ case 'claude':
205
+ case 'codex':
206
+ case 'codex-app':
207
+ case 'agy':
208
+ return true;
209
+ case 'custom':
210
+ // A persistent custom engine is a long-running stdin/stdout process, so
211
+ // it keeps context the same way claude does. One-shot ones do not.
212
+ return selection.customEngine?.persistent === true;
213
+ default:
214
+ return false;
215
+ }
216
+ }
217
+ /**
218
+ * Replayed transcript for engines without native conversation. Capped so a
219
+ * long run can't grow the prompt without bound: we keep the most recent
220
+ * turns within REPLAY_CHAR_BUDGET, oldest dropped first.
221
+ */
222
+ transcripts = {
223
+ planner: [],
224
+ coder: [],
225
+ reviewer: [],
226
+ };
227
+ recordTurn(role, who, text) {
228
+ if (!text)
229
+ return;
230
+ const log = this.transcripts[role];
231
+ log.push({ who, text });
232
+ let budget = REPLAY_CHAR_BUDGET;
233
+ let keepFrom = log.length;
234
+ for (let i = log.length - 1; i >= 0; i--) {
235
+ budget -= log[i].text.length;
236
+ if (budget < 0)
237
+ break;
238
+ keepFrom = i;
239
+ }
240
+ if (keepFrom > 0)
241
+ log.splice(0, keepFrom);
242
+ }
243
+ renderHistory(role, selection) {
244
+ if (this.hasNativeConversation(selection))
245
+ return null;
246
+ const log = this.transcripts[role];
247
+ if (log.length === 0)
248
+ return null;
249
+ const lines = log.map((entry) => `<${entry.who}>\n${entry.text}\n</${entry.who}>`);
250
+ return ['<conversation_history>', ...lines, '</conversation_history>'].join('\n');
251
+ }
252
+ withRoleInstructions(role, selection, systemPrompt, message) {
253
+ if (selection.engine === 'claude')
254
+ return message;
255
+ const parts = ['<autoloop_role_instructions>', systemPrompt.trim(), '</autoloop_role_instructions>', ''];
256
+ const history = this.renderHistory(role, selection);
257
+ if (history)
258
+ parts.push(history, '');
259
+ parts.push('<autoloop_message>', message, '</autoloop_message>');
260
+ return parts.join('\n');
261
+ }
132
262
  /**
133
263
  * Start Coder + Reviewer sessions. Idempotent. Called in response to a
134
264
  * Planner spawn_subagents tool (the SessionManager wires this via
135
265
  * onSpawnSubagents).
136
266
  */
137
267
  async spawnSubagents(args = {}) {
138
- if (args.coder_model)
139
- this.coderModel = args.coder_model;
140
- if (args.reviewer_model)
141
- this.reviewerModel = args.reviewer_model;
142
- await this.ensureCoder();
143
- await this.ensureReviewer();
268
+ const nextCoderEngine = args.coder_engine ?? this.coderSelection.engine;
269
+ const nextReviewerEngine = args.reviewer_engine ?? this.reviewerSelection.engine;
270
+ const nextCoder = {
271
+ ...this.coderSelection,
272
+ engine: nextCoderEngine,
273
+ model: args.coder_model !== undefined
274
+ ? args.coder_model
275
+ : nextCoderEngine !== this.coderSelection.engine
276
+ ? undefined
277
+ : this.coderSelection.model,
278
+ };
279
+ const nextReviewer = {
280
+ ...this.reviewerSelection,
281
+ engine: nextReviewerEngine,
282
+ model: args.reviewer_model !== undefined
283
+ ? args.reviewer_model
284
+ : nextReviewerEngine !== this.reviewerSelection.engine
285
+ ? undefined
286
+ : this.reviewerSelection.model,
287
+ };
288
+ this.validateSelection('coder', nextCoder);
289
+ this.validateSelection('reviewer', nextReviewer);
290
+ const coderChanged = nextCoder.engine !== this.coderSelection.engine ||
291
+ this.roleModel('coder', nextCoder) !== this.roleModel('coder', this.coderSelection);
292
+ const reviewerChanged = nextReviewer.engine !== this.reviewerSelection.engine ||
293
+ this.roleModel('reviewer', nextReviewer) !== this.roleModel('reviewer', this.reviewerSelection);
294
+ if (this.coderStarted && coderChanged) {
295
+ throw new Error('Cannot change Coder engine or model after its session has started');
296
+ }
297
+ if (this.reviewerStarted && reviewerChanged) {
298
+ throw new Error('Cannot change Reviewer engine or model after its session has started');
299
+ }
300
+ const previousCoder = this.coderSelection;
301
+ const previousReviewer = this.reviewerSelection;
302
+ const coderWasStarted = this.coderStarted;
303
+ const reviewerWasStarted = this.reviewerStarted;
304
+ this.coderSelection = nextCoder;
305
+ this.reviewerSelection = nextReviewer;
306
+ try {
307
+ await this.ensureCoder();
308
+ await this.ensureReviewer();
309
+ }
310
+ catch (err) {
311
+ // Roll back only what THIS call started. Crucially, `<role>Started` may be
312
+ // cleared only when the stop actually succeeded: SessionManager.startSession
313
+ // returns the EXISTING session for a name that is still live and ignores the
314
+ // new engine/model. So if we lied about the session being gone, the next
315
+ // spawn_subagents would sail past the "engine cannot change after start"
316
+ // guard, silently reuse the old engine's process, and still record the new
317
+ // engine in decisions.jsonl and the registry — the exact divergence that
318
+ // guard exists to prevent.
319
+ if (!coderWasStarted && this.coderStarted) {
320
+ this.coderStarted = !(await this.stopRolledBackSession(this.coderName));
321
+ }
322
+ if (!reviewerWasStarted && this.reviewerStarted) {
323
+ this.reviewerStarted = !(await this.stopRolledBackSession(this.reviewerName));
324
+ }
325
+ this.coderSelection = previousCoder;
326
+ this.reviewerSelection = previousReviewer;
327
+ throw err;
328
+ }
329
+ const effectiveSelection = {
330
+ coder: { engine: nextCoder.engine, model: nextCoder.model },
331
+ reviewer: { engine: nextReviewer.engine, model: nextReviewer.model },
332
+ };
333
+ this.appendDecisionLog({
334
+ kind: 'spawn_subagents',
335
+ actor: 'planner',
336
+ payload: {
337
+ coder_engine: nextCoder.engine,
338
+ coder_model: this.roleModel('coder', nextCoder),
339
+ reviewer_engine: nextReviewer.engine,
340
+ reviewer_model: this.roleModel('reviewer', nextReviewer),
341
+ },
342
+ });
343
+ await this.config.onRoleSelectionChanged?.(effectiveSelection);
144
344
  }
145
345
  /**
146
346
  * Reset a single subagent — stop its session, clear the started flag, and
@@ -171,8 +371,10 @@ export class ClaudeAgentDispatcher extends EventEmitter {
171
371
  this.plannerStarted = false;
172
372
  if (agent === 'coder')
173
373
  this.coderStarted = false;
174
- if (agent === 'reviewer')
374
+ if (agent === 'reviewer') {
175
375
  this.reviewerStarted = false;
376
+ this.reviewerSessionPrompt = null;
377
+ }
176
378
  if (opts.eagerRestart) {
177
379
  if (agent === 'planner')
178
380
  await this.ensurePlanner();
@@ -314,12 +516,15 @@ export class ClaudeAgentDispatcher extends EventEmitter {
314
516
  async ensurePlanner() {
315
517
  if (this.plannerStarted)
316
518
  return;
519
+ this.validateSelection('planner', this.plannerSelection);
317
520
  await this.config.manager.startSession({
318
521
  name: this.plannerName,
319
522
  cwd: this.config.workspace,
320
- engine: 'claude',
321
- model: this.config.plannerModel ?? 'opus',
322
- permissionMode: 'bypassPermissions',
523
+ engine: this.plannerSelection.engine,
524
+ model: this.roleModel('planner', this.plannerSelection),
525
+ customEngine: this.plannerSelection.engine === 'custom' ? this.plannerSelection.customEngine : undefined,
526
+ permissionMode: this.plannerSelection.engine === 'claude' ? 'bypassPermissions' : 'manual',
527
+ sandboxMode: this.plannerSelection.engine === 'claude' ? undefined : 'read-only',
323
528
  systemPrompt: this.plannerSystemPrompt,
324
529
  // Hard role boundary: Planner must NEVER author content files itself.
325
530
  // Its only writes are plan.md / goal.json via the write_plan /
@@ -355,7 +560,7 @@ export class ClaudeAgentDispatcher extends EventEmitter {
355
560
  // iter_done
356
561
  promptText = `[system] iter ${env.iter} done. verdict=${env.payload.verdict} metric=${env.payload.metric}`;
357
562
  }
358
- const result = (await this.config.manager.sendMessage(this.plannerName, promptText, {
563
+ const result = (await this.config.manager.sendMessage(this.plannerName, this.withRoleInstructions('planner', this.plannerSelection, this.plannerSystemPrompt, promptText), {
359
564
  timeout: this.config.sendTimeoutMs ?? 10 * 60_000,
360
565
  }));
361
566
  if (result.error) {
@@ -363,6 +568,11 @@ export class ClaudeAgentDispatcher extends EventEmitter {
363
568
  this.emit('planner_error', new Error(result.error));
364
569
  }
365
570
  const replyText = (result.output ?? '').trim();
571
+ // Feed the transcript that engines without native conversation replay next
572
+ // turn. Recorded AFTER the send so the current message isn't duplicated in
573
+ // its own history block.
574
+ this.recordTurn('planner', 'user', promptText);
575
+ this.recordTurn('planner', 'agent', replyText);
366
576
  // S3: parse autoloop-fenced tool calls out of the reply, apply effects,
367
577
  // and bubble emitted messages back into the runner queue.
368
578
  const parsed = parsePlannerReply(replyText);
@@ -371,16 +581,11 @@ export class ClaudeAgentDispatcher extends EventEmitter {
371
581
  }
372
582
  const effects = {
373
583
  spawnSubagents: async (args) => {
374
- this.appendDecisionLog({
375
- kind: 'spawn_subagents',
376
- actor: 'planner',
377
- payload: { args },
378
- });
379
584
  if (this.config.onSpawnSubagents) {
380
585
  await this.config.onSpawnSubagents(args);
381
586
  }
382
587
  else {
383
- this.logger.warn?.('[autoloop] spawn_subagents called but no handler installed (S4 not wired yet)');
588
+ this.logger.warn?.('[autoloop] spawn_subagents called but no handler is installed');
384
589
  }
385
590
  },
386
591
  updatePushPolicy: (delta) => {
@@ -470,11 +675,13 @@ export class ClaudeAgentDispatcher extends EventEmitter {
470
675
  async ensureCoder() {
471
676
  if (this.coderStarted)
472
677
  return;
678
+ this.validateSelection('coder', this.coderSelection);
473
679
  await this.config.manager.startSession({
474
680
  name: this.coderName,
475
681
  cwd: this.config.workspace,
476
- engine: 'claude',
477
- model: this.coderModel,
682
+ engine: this.coderSelection.engine,
683
+ model: this.roleModel('coder', this.coderSelection),
684
+ customEngine: this.coderSelection.engine === 'custom' ? this.coderSelection.customEngine : undefined,
478
685
  permissionMode: 'bypassPermissions',
479
686
  systemPrompt: this.coderSystemPrompt,
480
687
  });
@@ -526,7 +733,9 @@ export class ClaudeAgentDispatcher extends EventEmitter {
526
733
  text: `🔨 Coder iter ${env.iter} working…`,
527
734
  ts: new Date().toISOString(),
528
735
  });
529
- const result = await this.sendWithRecovery('coder', this.coderName, promptText);
736
+ const result = await this.sendWithRecovery('coder', this.coderName, this.withRoleInstructions('coder', this.coderSelection, this.coderSystemPrompt, promptText));
737
+ this.recordTurn('coder', 'user', promptText);
738
+ this.recordTurn('coder', 'agent', (result.output ?? '').trim());
530
739
  // A3: subprocess died (recovery retry exhausted). Surface as phase_error
531
740
  // rather than silently masquerading as a "clarification request"; the
532
741
  // runner's circuit breaker can then trip after enough consecutive failures.
@@ -645,16 +854,26 @@ export class ClaudeAgentDispatcher extends EventEmitter {
645
854
  async ensureReviewer() {
646
855
  if (this.reviewerStarted)
647
856
  return;
857
+ this.validateSelection('reviewer', this.reviewerSelection);
648
858
  fs.mkdirSync(this.reviewerSandboxDir, { recursive: true });
649
- await this.config.manager.startSession({
650
- name: this.reviewerName,
651
- cwd: this.reviewerSandboxDir,
652
- engine: 'claude',
653
- model: this.reviewerModel,
654
- permissionMode: 'bypassPermissions',
655
- systemPrompt: this.buildReviewerSystemPrompt(),
656
- });
657
- this.reviewerStarted = true;
859
+ const sessionPrompt = this.buildReviewerSystemPrompt();
860
+ this.reviewerSessionPrompt = sessionPrompt;
861
+ try {
862
+ await this.config.manager.startSession({
863
+ name: this.reviewerName,
864
+ cwd: this.reviewerSandboxDir,
865
+ engine: this.reviewerSelection.engine,
866
+ model: this.roleModel('reviewer', this.reviewerSelection),
867
+ customEngine: this.reviewerSelection.engine === 'custom' ? this.reviewerSelection.customEngine : undefined,
868
+ permissionMode: 'bypassPermissions',
869
+ systemPrompt: sessionPrompt,
870
+ });
871
+ this.reviewerStarted = true;
872
+ }
873
+ catch (err) {
874
+ this.reviewerSessionPrompt = null;
875
+ throw err;
876
+ }
658
877
  }
659
878
  /**
660
879
  * Stage the iter's artifacts into the Reviewer sandbox cwd. Reviewer is a
@@ -724,7 +943,9 @@ export class ClaudeAgentDispatcher extends EventEmitter {
724
943
  text: `🔍 Reviewer iter ${env.payload.iter} auditing…`,
725
944
  ts: new Date().toISOString(),
726
945
  });
727
- const result = await this.sendWithRecovery('reviewer', this.reviewerName, promptText);
946
+ const result = await this.sendWithRecovery('reviewer', this.reviewerName, this.withRoleInstructions('reviewer', this.reviewerSelection, this.reviewerSessionPrompt ?? this.reviewerSystemPrompt, promptText));
947
+ this.recordTurn('reviewer', 'user', promptText);
948
+ this.recordTurn('reviewer', 'agent', (result.output ?? '').trim());
728
949
  if (result.fatal) {
729
950
  this.appendDecisionLog({
730
951
  kind: 'phase_error',