@enderfga/claw-orchestrator 4.7.0 → 4.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -5
- package/configs/autoloop-coder-prompt.md +4 -4
- package/configs/autoloop-planner-prompt.md +10 -10
- package/configs/autoloop-reviewer-prompt.md +3 -4
- package/dist/src/autoloop/dispatcher.d.ts +59 -10
- package/dist/src/autoloop/dispatcher.js +261 -40
- package/dist/src/autoloop/dispatcher.js.map +1 -1
- package/dist/src/autoloop/planner-tools.d.ts +4 -3
- package/dist/src/autoloop/planner-tools.js +35 -2
- package/dist/src/autoloop/planner-tools.js.map +1 -1
- package/dist/src/autoloop/types.d.ts +2 -0
- package/dist/src/autoloop/types.js.map +1 -1
- package/dist/src/dashboard/index.html +185 -94
- package/dist/src/embedded-server.js +67 -9
- package/dist/src/embedded-server.js.map +1 -1
- package/dist/src/index.js +84 -52
- package/dist/src/index.js.map +1 -1
- package/dist/src/persistent-agy-session.js +4 -1
- package/dist/src/persistent-agy-session.js.map +1 -1
- package/dist/src/persistent-codex-session.js +3 -0
- package/dist/src/persistent-codex-session.js.map +1 -1
- package/dist/src/persistent-cursor-session.js +7 -2
- package/dist/src/persistent-cursor-session.js.map +1 -1
- package/dist/src/persistent-custom-session.js +13 -0
- package/dist/src/persistent-custom-session.js.map +1 -1
- package/dist/src/persistent-gemini-session.d.ts +4 -0
- package/dist/src/persistent-gemini-session.js +55 -2
- package/dist/src/persistent-gemini-session.js.map +1 -1
- package/dist/src/persistent-opencode-session.js +35 -1
- package/dist/src/persistent-opencode-session.js.map +1 -1
- package/dist/src/persistent-session.js +6 -1
- package/dist/src/persistent-session.js.map +1 -1
- package/dist/src/session-manager.d.ts +44 -5
- package/dist/src/session-manager.js +190 -18
- package/dist/src/session-manager.js.map +1 -1
- package/dist/src/types.d.ts +3 -2
- package/package.json +1 -1
- package/skills/SKILL.md +11 -10
- package/skills/references/autoloop.md +50 -6
- package/skills/references/claude-cli-tracking.md +2 -1
- package/skills/references/multi-engine.md +12 -9
- package/skills/references/tools.md +38 -19
package/README.md
CHANGED
|
@@ -31,7 +31,7 @@ https://github.com/user-attachments/assets/fbd2b0ea-28d8-4387-9894-c29cf15ba030
|
|
|
31
31
|
| **Multi-Agent Council** | Parallel agents in isolated git worktrees, voting on consensus until they agree. | [`council.md`](./skills/references/council.md) |
|
|
32
32
|
| **Fan-out** | Run one task across N engine/model agents in parallel and collect their answers, with an optional synthesis pass — the cross-engine best-of-N / diverse-perspective primitive (no rounds or worktrees). | [`tools.md`](./skills/references/tools.md) |
|
|
33
33
|
| **ultracode** | `session_start({ ultracode: true })` lets Claude orchestrate a dynamic JS workflow and fan out to subagents per task (Claude engine). | [`tools.md`](./skills/references/tools.md) |
|
|
34
|
-
| **Autoloop** | Three-agent autonomous workspace iteration. Chat with the Planner; it spawns Coder + Reviewer into a self-iterating subloop and pushes you on regression, target-hit, or decision points.
|
|
34
|
+
| **Autoloop** | Three-agent autonomous workspace iteration with independent engine/model selection for Planner, Coder, and Reviewer. Chat with the Planner; it spawns Coder + Reviewer into a self-iterating subloop and pushes you on regression, target-hit, or decision points. | [`autoloop.md`](./skills/references/autoloop.md) |
|
|
35
35
|
| **Ultraapp** | A three-agent Opus council turns a five-question interview into a deployed web app — Tailwind UI, BYOK, file-queue runtime, smoke test, all live at `localhost:19000/forge/<slug>/`. | [`ultraapp.md`](./skills/references/ultraapp.md) |
|
|
36
36
|
| **Embedded Dashboard** | Three-tab UI for Autoloop, Council, and Forge with sidebar lifecycle controls, per-run live event streaming, and cookie-based auth via a `/login` redirect. | [`dashboard.md`](./skills/references/dashboard.md) |
|
|
37
37
|
| **OpenAI-Compatible Proxy** | `POST /v1/chat/completions` translates OpenAI requests into native Anthropic, OpenAI, and Google calls and streams responses back in OpenAI shape. Point any OpenAI-SDK client at the orchestrator without changing call sites. | [`openai-compat.md`](./skills/references/openai-compat.md) |
|
|
@@ -91,11 +91,11 @@ Register `clawo-mcp` with any MCP-compatible host: Hermes Agent, Claude Desktop,
|
|
|
91
91
|
|
|
92
92
|
| Engine | CLI | Tested Version |
|
|
93
93
|
| ------------ | ---------- | -------------- |
|
|
94
|
-
| Claude Code | `claude` | 2.1.
|
|
95
|
-
| Codex | `codex` | 0.
|
|
94
|
+
| Claude Code | `claude` | 2.1.207 |
|
|
95
|
+
| Codex | `codex` | 0.144.1 |
|
|
96
96
|
| Gemini | `gemini` | 0.43.0 |
|
|
97
|
-
| Antigravity | `agy` | 1.
|
|
98
|
-
| Cursor Agent | `agent` | 2026.
|
|
97
|
+
| Antigravity | `agy` | 1.1.1 |
|
|
98
|
+
| Cursor Agent | `agent` | 2026.04.08-a41fba1 |
|
|
99
99
|
| OpenCode | `opencode` | 1.1.40 |
|
|
100
100
|
| Custom CLI | any | — |
|
|
101
101
|
|
|
@@ -46,10 +46,10 @@ make Reviewer audits unreliable.
|
|
|
46
46
|
|
|
47
47
|
## Your tools
|
|
48
48
|
|
|
49
|
-
You are a
|
|
50
|
-
file-
|
|
51
|
-
|
|
52
|
-
|
|
49
|
+
You are a coding-agent session with the workspace as cwd. Use the selected
|
|
50
|
+
engine's file-reading, editing, search, and shell tools freely on workspace
|
|
51
|
+
code — that's your job. The role boundary is Rule 1: do not touch `plan.md`,
|
|
52
|
+
`goal.json`, or `tasks/`.
|
|
53
53
|
|
|
54
54
|
You also have **autoloop control tools** via fenced JSON blocks:
|
|
55
55
|
|
|
@@ -41,15 +41,14 @@ a one-paragraph email is a Coder task.
|
|
|
41
41
|
> 5. Ask: "Plan ready, spawn the Coder?"
|
|
42
42
|
> 6. On approval, `spawn_subagents` with an initial directive.
|
|
43
43
|
|
|
44
|
-
### Rule 2 — You CANNOT
|
|
44
|
+
### Rule 2 — You CANNOT edit workspace source files
|
|
45
45
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
out-of-bounds — they violate Rule 1 even though they're technically possible.
|
|
46
|
+
The orchestrator starts Planner sessions in the engine's native read-only /
|
|
47
|
+
plan mode. Claude additionally has Write / Edit / MultiEdit / NotebookEdit
|
|
48
|
+
removed. The only supported authoring path is the `write_plan` and
|
|
49
|
+
`write_goal` autoloop tools, which write only `plan.md` and `goal.json`.
|
|
50
|
+
Bash heredocs, `tee`, or output redirection that author content files violate
|
|
51
|
+
Rule 1 and must not be used.
|
|
53
52
|
|
|
54
53
|
### Rule 3 — Never `spawn_subagents` without explicit user approval
|
|
55
54
|
|
|
@@ -61,7 +60,8 @@ and wait for go / ok / 开干 / 干 / yes / similar. The only exception:
|
|
|
61
60
|
|
|
62
61
|
## Your tools
|
|
63
62
|
|
|
64
|
-
You are a
|
|
63
|
+
You are a read-only coding-agent session with the workspace as cwd. Depending
|
|
64
|
+
on the selected engine, equivalent tool names may differ. You have:
|
|
65
65
|
|
|
66
66
|
| Tool | Purpose |
|
|
67
67
|
|---|---|
|
|
@@ -91,7 +91,7 @@ turn. Anything outside the blocks is shown to the user as your chat reply.
|
|
|
91
91
|
| `write_plan` | `content` (full plan.md body as string), `commit_message?` | Writes `plan.md` to the workspace and git-commits. Re-running replaces the whole file (no patches). |
|
|
92
92
|
| `write_goal` | `content` (full goal.json body as string), `commit_message?` | Same, for `goal.json`. The orchestrator parses the content as JSON before writing; malformed JSON errors back to you. |
|
|
93
93
|
| `notify_user` | `level` ('info'/'warn'/'decision'/'error'), `summary` (one line), `detail?`, `channel?` ('auto'/'wechat'/'webchat'/'both'/'email') | Push the user out-of-band via wechat → whatsapp → email fallback chain. Use sparingly: 5-min dedup applies. |
|
|
94
|
-
| `spawn_subagents` | `coder_model?`, `reviewer_model?`, `initial_directive?: { goal, constraints?, success_criteria?, max_attempts? }` | Start the Coder + Reviewer subloop. Call this **only when the user has explicitly approved the plan** (Rule 3).
|
|
94
|
+
| `spawn_subagents` | `coder_engine?`, `coder_model?`, `reviewer_engine?`, `reviewer_model?`, `initial_directive?: { goal, constraints?, success_criteria?, max_attempts? }` | Start the Coder + Reviewer subloop. Omitted fields inherit the run defaults; changing engine/model after that role's session has started is rejected. `custom` may only select a config preloaded by `autoloop_start` or resume; never include custom config, `env`, or `extra` in this block. Call this **only when the user has explicitly approved the plan** (Rule 3). |
|
|
95
95
|
| `send_directive` | `goal`, `constraints?`, `success_criteria?`, `max_attempts?` | Send a fresh directive to Coder for the next iter. |
|
|
96
96
|
| `pause_loop` | `reason` | Halt the Coder/Reviewer subloop at the next iter boundary (you can keep chatting). |
|
|
97
97
|
| `resume_loop` | `{}` | Resume after a pause. |
|
|
@@ -35,10 +35,9 @@ orchestrator stalls if you skip it.
|
|
|
35
35
|
|
|
36
36
|
## Your tools
|
|
37
37
|
|
|
38
|
-
|
|
39
|
-
You
|
|
40
|
-
|
|
41
|
-
it enforces it by trusting you.
|
|
38
|
+
Use the selected engine's read, search, and shell tools inside the sandbox
|
|
39
|
+
cwd. You may edit only sandbox memory, scratch, and audit files. Rule 3
|
|
40
|
+
forbids writes outside that cwd; engine-specific tool names may differ.
|
|
42
41
|
|
|
43
42
|
The current contents of `reviewer_memory.md` are **injected as a frozen
|
|
44
43
|
snapshot in your system prompt** at session start. You do not need to
|
|
@@ -1,9 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* ClaudeAgentDispatcher — wires the v2 runner to real persistent
|
|
3
|
-
* sessions managed by SessionManager.
|
|
4
|
-
*
|
|
5
|
-
* S2 scope: Planner only (chat-mode, no subagents yet). Coder/Reviewer
|
|
6
|
-
* delivery throws — S4 wires them in.
|
|
2
|
+
* ClaudeAgentDispatcher — wires the v2 runner to real persistent coding
|
|
3
|
+
* sessions managed by SessionManager. The historical class name is retained
|
|
4
|
+
* for compatibility; each Autoloop role may use a different engine.
|
|
7
5
|
*
|
|
8
6
|
* Naming convention:
|
|
9
7
|
* autoloop-<run_id>-planner
|
|
@@ -21,6 +19,7 @@
|
|
|
21
19
|
import { EventEmitter } from 'node:events';
|
|
22
20
|
import type { SessionManager } from '../session-manager.js';
|
|
23
21
|
import type { Logger } from '../logger.js';
|
|
22
|
+
import { type CustomEngineConfig, type EngineType } from '../types.js';
|
|
24
23
|
import { type AnyAutoloopMessage } from './messages.js';
|
|
25
24
|
import { type AgentDispatcher, type AutoloopState, type PushPolicy } from './types.js';
|
|
26
25
|
import { type SpawnSubagentsArgs } from './planner-tools.js';
|
|
@@ -33,12 +32,18 @@ export interface ClaudeAgentDispatcherConfig {
|
|
|
33
32
|
/** Override Coder/Reviewer prompt paths (defaults walk-up to configs/autoloop-{coder,reviewer}-prompt.md). */
|
|
34
33
|
coderPromptPath?: string;
|
|
35
34
|
reviewerPromptPath?: string;
|
|
36
|
-
/**
|
|
35
|
+
/** Planner engine/model (default: claude/opus). */
|
|
36
|
+
plannerEngine?: EngineType;
|
|
37
37
|
plannerModel?: string;
|
|
38
|
-
|
|
38
|
+
plannerCustomEngine?: CustomEngineConfig;
|
|
39
|
+
/** Coder defaults. Engine/model can be overridden per spawn_subagents call. */
|
|
40
|
+
coderEngine?: EngineType;
|
|
39
41
|
coderModel?: string;
|
|
40
|
-
|
|
42
|
+
coderCustomEngine?: CustomEngineConfig;
|
|
43
|
+
/** Reviewer defaults. Engine/model can be overridden per spawn_subagents call. */
|
|
44
|
+
reviewerEngine?: EngineType;
|
|
41
45
|
reviewerModel?: string;
|
|
46
|
+
reviewerCustomEngine?: CustomEngineConfig;
|
|
42
47
|
/** Per-message wall-clock cap. Default 10 min. */
|
|
43
48
|
sendTimeoutMs?: number;
|
|
44
49
|
logger?: Logger;
|
|
@@ -63,6 +68,17 @@ export interface ClaudeAgentDispatcherConfig {
|
|
|
63
68
|
pushPolicyRef?: PushPolicy;
|
|
64
69
|
/** Called when Planner emits spawn_subagents. S4 implements; S3 records the intent. */
|
|
65
70
|
onSpawnSubagents?: (args: SpawnSubagentsArgs) => Promise<void>;
|
|
71
|
+
/** Persist the effective non-secret role selection after a successful spawn. */
|
|
72
|
+
onRoleSelectionChanged?: (selection: {
|
|
73
|
+
coder: {
|
|
74
|
+
engine: EngineType;
|
|
75
|
+
model?: string;
|
|
76
|
+
};
|
|
77
|
+
reviewer: {
|
|
78
|
+
engine: EngineType;
|
|
79
|
+
model?: string;
|
|
80
|
+
};
|
|
81
|
+
}) => Promise<void> | void;
|
|
66
82
|
}
|
|
67
83
|
export declare class ClaudeAgentDispatcher extends EventEmitter implements AgentDispatcher {
|
|
68
84
|
readonly config: ClaudeAgentDispatcherConfig;
|
|
@@ -76,8 +92,10 @@ export declare class ClaudeAgentDispatcher extends EventEmitter implements Agent
|
|
|
76
92
|
private plannerSystemPrompt;
|
|
77
93
|
private coderSystemPrompt;
|
|
78
94
|
private reviewerSystemPrompt;
|
|
79
|
-
private
|
|
80
|
-
private
|
|
95
|
+
private reviewerSessionPrompt;
|
|
96
|
+
private plannerSelection;
|
|
97
|
+
private coderSelection;
|
|
98
|
+
private reviewerSelection;
|
|
81
99
|
/** Where Reviewer reads from. Created lazily by stageReviewSandbox(). */
|
|
82
100
|
private reviewerSandboxDir;
|
|
83
101
|
private ledgerDir;
|
|
@@ -92,6 +110,37 @@ export declare class ClaudeAgentDispatcher extends EventEmitter implements Agent
|
|
|
92
110
|
purge?: boolean;
|
|
93
111
|
}): Promise<void>;
|
|
94
112
|
deliver(env: AnyAutoloopMessage): Promise<AnyAutoloopMessage[]>;
|
|
113
|
+
private roleModel;
|
|
114
|
+
private validateSelection;
|
|
115
|
+
/**
|
|
116
|
+
* Stop a session we started during a failed spawn. Returns true only when the
|
|
117
|
+
* session is genuinely gone — the caller uses that to decide whether it may
|
|
118
|
+
* clear the role's `started` flag. Returning false keeps the role marked as
|
|
119
|
+
* started, which is the safe lie: a later engine change is then rejected
|
|
120
|
+
* instead of silently binding the run to a process that never went away.
|
|
121
|
+
*/
|
|
122
|
+
private stopRolledBackSession;
|
|
123
|
+
/**
|
|
124
|
+
* Does this engine carry conversation across sends on its own?
|
|
125
|
+
*
|
|
126
|
+
* claude keeps one subprocess alive; codex / codex-app resume a thread; agy
|
|
127
|
+
* resumes a harvested `--conversation <uuid>`. Everything else (gemini,
|
|
128
|
+
* cursor, opencode, and non-persistent custom engines) spawns a FRESH process
|
|
129
|
+
* per send with zero memory of the last turn — for those the dispatcher must
|
|
130
|
+
* replay the transcript in-band, or the role is amnesiac and a chat-driven
|
|
131
|
+
* Planner can never remember the plan it just proposed (let alone whether the
|
|
132
|
+
* user approved it).
|
|
133
|
+
*/
|
|
134
|
+
private hasNativeConversation;
|
|
135
|
+
/**
|
|
136
|
+
* Replayed transcript for engines without native conversation. Capped so a
|
|
137
|
+
* long run can't grow the prompt without bound: we keep the most recent
|
|
138
|
+
* turns within REPLAY_CHAR_BUDGET, oldest dropped first.
|
|
139
|
+
*/
|
|
140
|
+
private transcripts;
|
|
141
|
+
private recordTurn;
|
|
142
|
+
private renderHistory;
|
|
143
|
+
private withRoleInstructions;
|
|
95
144
|
/**
|
|
96
145
|
* Start Coder + Reviewer sessions. Idempotent. Called in response to a
|
|
97
146
|
* Planner spawn_subagents tool (the SessionManager wires this via
|
|
@@ -1,9 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* ClaudeAgentDispatcher — wires the v2 runner to real persistent
|
|
3
|
-
* sessions managed by SessionManager.
|
|
4
|
-
*
|
|
5
|
-
* S2 scope: Planner only (chat-mode, no subagents yet). Coder/Reviewer
|
|
6
|
-
* delivery throws — S4 wires them in.
|
|
2
|
+
* ClaudeAgentDispatcher — wires the v2 runner to real persistent coding
|
|
3
|
+
* sessions managed by SessionManager. The historical class name is retained
|
|
4
|
+
* for compatibility; each Autoloop role may use a different engine.
|
|
7
5
|
*
|
|
8
6
|
* Naming convention:
|
|
9
7
|
* autoloop-<run_id>-planner
|
|
@@ -22,12 +20,19 @@ import { EventEmitter } from 'node:events';
|
|
|
22
20
|
import * as fs from 'node:fs';
|
|
23
21
|
import * as path from 'node:path';
|
|
24
22
|
import { fileURLToPath } from 'node:url';
|
|
23
|
+
import { ENGINE_TYPES } from '../types.js';
|
|
25
24
|
import { nullLogger } from '../logger.js';
|
|
26
25
|
import { spawn } from 'node:child_process';
|
|
27
26
|
import { Msg } from './messages.js';
|
|
28
|
-
import { LEDGER_SCHEMA_VERSION } from './types.js';
|
|
27
|
+
import { LEDGER_SCHEMA_VERSION, } from './types.js';
|
|
29
28
|
import { applyPlannerToolCalls, parsePlannerReply, } from './planner-tools.js';
|
|
30
29
|
import { extractIterComplete, extractReviewComplete, parseAgentReply } from './agent-tools.js';
|
|
30
|
+
/**
|
|
31
|
+
* Character budget for the replayed transcript handed to engines without native
|
|
32
|
+
* conversation (see hasNativeConversation). Oldest turns are dropped first, so a
|
|
33
|
+
* long run keeps the recent context instead of growing the prompt forever.
|
|
34
|
+
*/
|
|
35
|
+
const REPLAY_CHAR_BUDGET = 24_000;
|
|
31
36
|
/**
|
|
32
37
|
* Files inside <ledger>/reviewer_sandbox/ that survive `stageReviewSandbox`.
|
|
33
38
|
* Anything not listed is wiped between iters. `reviewer_memory.md` is also
|
|
@@ -70,8 +75,10 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
70
75
|
plannerSystemPrompt;
|
|
71
76
|
coderSystemPrompt;
|
|
72
77
|
reviewerSystemPrompt;
|
|
73
|
-
|
|
74
|
-
|
|
78
|
+
reviewerSessionPrompt = null;
|
|
79
|
+
plannerSelection;
|
|
80
|
+
coderSelection;
|
|
81
|
+
reviewerSelection;
|
|
75
82
|
/** Where Reviewer reads from. Created lazily by stageReviewSandbox(). */
|
|
76
83
|
reviewerSandboxDir;
|
|
77
84
|
ledgerDir;
|
|
@@ -86,8 +93,21 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
86
93
|
this.plannerSystemPrompt = fs.readFileSync(promptPath, 'utf-8');
|
|
87
94
|
this.coderSystemPrompt = fs.readFileSync(config.coderPromptPath ?? resolveDefaultCoderPrompt(), 'utf-8');
|
|
88
95
|
this.reviewerSystemPrompt = fs.readFileSync(config.reviewerPromptPath ?? resolveDefaultReviewerPrompt(), 'utf-8');
|
|
89
|
-
this.
|
|
90
|
-
|
|
96
|
+
this.plannerSelection = {
|
|
97
|
+
engine: config.plannerEngine ?? 'claude',
|
|
98
|
+
model: config.plannerModel,
|
|
99
|
+
customEngine: config.plannerCustomEngine,
|
|
100
|
+
};
|
|
101
|
+
this.coderSelection = {
|
|
102
|
+
engine: config.coderEngine ?? 'claude',
|
|
103
|
+
model: config.coderModel,
|
|
104
|
+
customEngine: config.coderCustomEngine,
|
|
105
|
+
};
|
|
106
|
+
this.reviewerSelection = {
|
|
107
|
+
engine: config.reviewerEngine ?? 'claude',
|
|
108
|
+
model: config.reviewerModel,
|
|
109
|
+
customEngine: config.reviewerCustomEngine,
|
|
110
|
+
};
|
|
91
111
|
this.ledgerDir = path.join(config.workspace, 'tasks', config.runId);
|
|
92
112
|
this.reviewerSandboxDir = path.join(this.ledgerDir, 'reviewer_sandbox');
|
|
93
113
|
}
|
|
@@ -129,18 +149,198 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
129
149
|
throw new Error(`[autoloop] unexpected dispatcher target: ${env.to}`);
|
|
130
150
|
}
|
|
131
151
|
}
|
|
152
|
+
roleModel(role, selection) {
|
|
153
|
+
if (selection.model !== undefined)
|
|
154
|
+
return selection.model;
|
|
155
|
+
if (selection.engine !== 'claude')
|
|
156
|
+
return undefined;
|
|
157
|
+
return role === 'planner' ? 'opus' : 'sonnet';
|
|
158
|
+
}
|
|
159
|
+
validateSelection(role, selection) {
|
|
160
|
+
const label = role[0].toUpperCase() + role.slice(1);
|
|
161
|
+
if (!ENGINE_TYPES.includes(selection.engine)) {
|
|
162
|
+
throw new Error(`${label} engine '${String(selection.engine)}' is not supported`);
|
|
163
|
+
}
|
|
164
|
+
if (selection.engine === 'custom' && !selection.customEngine) {
|
|
165
|
+
throw new Error(`${label} custom engine config is required`);
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
/**
|
|
169
|
+
* Stop a session we started during a failed spawn. Returns true only when the
|
|
170
|
+
* session is genuinely gone — the caller uses that to decide whether it may
|
|
171
|
+
* clear the role's `started` flag. Returning false keeps the role marked as
|
|
172
|
+
* started, which is the safe lie: a later engine change is then rejected
|
|
173
|
+
* instead of silently binding the run to a process that never went away.
|
|
174
|
+
*/
|
|
175
|
+
async stopRolledBackSession(name) {
|
|
176
|
+
try {
|
|
177
|
+
await this.config.manager.stopSession(name);
|
|
178
|
+
return true;
|
|
179
|
+
}
|
|
180
|
+
catch (stopErr) {
|
|
181
|
+
this.logger.error?.(`[autoloop] rollback could not stop ${name}: ${stopErr.message} — ` +
|
|
182
|
+
`leaving it marked started so a later engine change is rejected rather than silently ignored`);
|
|
183
|
+
this.appendDecisionLog({
|
|
184
|
+
kind: 'phase_error',
|
|
185
|
+
actor: 'dispatcher',
|
|
186
|
+
payload: { agent: name, phase: 'rollback_stop', error: stopErr.message },
|
|
187
|
+
});
|
|
188
|
+
return false;
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
/**
|
|
192
|
+
* Does this engine carry conversation across sends on its own?
|
|
193
|
+
*
|
|
194
|
+
* claude keeps one subprocess alive; codex / codex-app resume a thread; agy
|
|
195
|
+
* resumes a harvested `--conversation <uuid>`. Everything else (gemini,
|
|
196
|
+
* cursor, opencode, and non-persistent custom engines) spawns a FRESH process
|
|
197
|
+
* per send with zero memory of the last turn — for those the dispatcher must
|
|
198
|
+
* replay the transcript in-band, or the role is amnesiac and a chat-driven
|
|
199
|
+
* Planner can never remember the plan it just proposed (let alone whether the
|
|
200
|
+
* user approved it).
|
|
201
|
+
*/
|
|
202
|
+
hasNativeConversation(selection) {
|
|
203
|
+
switch (selection.engine) {
|
|
204
|
+
case 'claude':
|
|
205
|
+
case 'codex':
|
|
206
|
+
case 'codex-app':
|
|
207
|
+
case 'agy':
|
|
208
|
+
return true;
|
|
209
|
+
case 'custom':
|
|
210
|
+
// A persistent custom engine is a long-running stdin/stdout process, so
|
|
211
|
+
// it keeps context the same way claude does. One-shot ones do not.
|
|
212
|
+
return selection.customEngine?.persistent === true;
|
|
213
|
+
default:
|
|
214
|
+
return false;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
/**
|
|
218
|
+
* Replayed transcript for engines without native conversation. Capped so a
|
|
219
|
+
* long run can't grow the prompt without bound: we keep the most recent
|
|
220
|
+
* turns within REPLAY_CHAR_BUDGET, oldest dropped first.
|
|
221
|
+
*/
|
|
222
|
+
transcripts = {
|
|
223
|
+
planner: [],
|
|
224
|
+
coder: [],
|
|
225
|
+
reviewer: [],
|
|
226
|
+
};
|
|
227
|
+
recordTurn(role, who, text) {
|
|
228
|
+
if (!text)
|
|
229
|
+
return;
|
|
230
|
+
const log = this.transcripts[role];
|
|
231
|
+
log.push({ who, text });
|
|
232
|
+
let budget = REPLAY_CHAR_BUDGET;
|
|
233
|
+
let keepFrom = log.length;
|
|
234
|
+
for (let i = log.length - 1; i >= 0; i--) {
|
|
235
|
+
budget -= log[i].text.length;
|
|
236
|
+
if (budget < 0)
|
|
237
|
+
break;
|
|
238
|
+
keepFrom = i;
|
|
239
|
+
}
|
|
240
|
+
if (keepFrom > 0)
|
|
241
|
+
log.splice(0, keepFrom);
|
|
242
|
+
}
|
|
243
|
+
renderHistory(role, selection) {
|
|
244
|
+
if (this.hasNativeConversation(selection))
|
|
245
|
+
return null;
|
|
246
|
+
const log = this.transcripts[role];
|
|
247
|
+
if (log.length === 0)
|
|
248
|
+
return null;
|
|
249
|
+
const lines = log.map((entry) => `<${entry.who}>\n${entry.text}\n</${entry.who}>`);
|
|
250
|
+
return ['<conversation_history>', ...lines, '</conversation_history>'].join('\n');
|
|
251
|
+
}
|
|
252
|
+
withRoleInstructions(role, selection, systemPrompt, message) {
|
|
253
|
+
if (selection.engine === 'claude')
|
|
254
|
+
return message;
|
|
255
|
+
const parts = ['<autoloop_role_instructions>', systemPrompt.trim(), '</autoloop_role_instructions>', ''];
|
|
256
|
+
const history = this.renderHistory(role, selection);
|
|
257
|
+
if (history)
|
|
258
|
+
parts.push(history, '');
|
|
259
|
+
parts.push('<autoloop_message>', message, '</autoloop_message>');
|
|
260
|
+
return parts.join('\n');
|
|
261
|
+
}
|
|
132
262
|
/**
|
|
133
263
|
* Start Coder + Reviewer sessions. Idempotent. Called in response to a
|
|
134
264
|
* Planner spawn_subagents tool (the SessionManager wires this via
|
|
135
265
|
* onSpawnSubagents).
|
|
136
266
|
*/
|
|
137
267
|
async spawnSubagents(args = {}) {
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
this.
|
|
142
|
-
|
|
143
|
-
|
|
268
|
+
const nextCoderEngine = args.coder_engine ?? this.coderSelection.engine;
|
|
269
|
+
const nextReviewerEngine = args.reviewer_engine ?? this.reviewerSelection.engine;
|
|
270
|
+
const nextCoder = {
|
|
271
|
+
...this.coderSelection,
|
|
272
|
+
engine: nextCoderEngine,
|
|
273
|
+
model: args.coder_model !== undefined
|
|
274
|
+
? args.coder_model
|
|
275
|
+
: nextCoderEngine !== this.coderSelection.engine
|
|
276
|
+
? undefined
|
|
277
|
+
: this.coderSelection.model,
|
|
278
|
+
};
|
|
279
|
+
const nextReviewer = {
|
|
280
|
+
...this.reviewerSelection,
|
|
281
|
+
engine: nextReviewerEngine,
|
|
282
|
+
model: args.reviewer_model !== undefined
|
|
283
|
+
? args.reviewer_model
|
|
284
|
+
: nextReviewerEngine !== this.reviewerSelection.engine
|
|
285
|
+
? undefined
|
|
286
|
+
: this.reviewerSelection.model,
|
|
287
|
+
};
|
|
288
|
+
this.validateSelection('coder', nextCoder);
|
|
289
|
+
this.validateSelection('reviewer', nextReviewer);
|
|
290
|
+
const coderChanged = nextCoder.engine !== this.coderSelection.engine ||
|
|
291
|
+
this.roleModel('coder', nextCoder) !== this.roleModel('coder', this.coderSelection);
|
|
292
|
+
const reviewerChanged = nextReviewer.engine !== this.reviewerSelection.engine ||
|
|
293
|
+
this.roleModel('reviewer', nextReviewer) !== this.roleModel('reviewer', this.reviewerSelection);
|
|
294
|
+
if (this.coderStarted && coderChanged) {
|
|
295
|
+
throw new Error('Cannot change Coder engine or model after its session has started');
|
|
296
|
+
}
|
|
297
|
+
if (this.reviewerStarted && reviewerChanged) {
|
|
298
|
+
throw new Error('Cannot change Reviewer engine or model after its session has started');
|
|
299
|
+
}
|
|
300
|
+
const previousCoder = this.coderSelection;
|
|
301
|
+
const previousReviewer = this.reviewerSelection;
|
|
302
|
+
const coderWasStarted = this.coderStarted;
|
|
303
|
+
const reviewerWasStarted = this.reviewerStarted;
|
|
304
|
+
this.coderSelection = nextCoder;
|
|
305
|
+
this.reviewerSelection = nextReviewer;
|
|
306
|
+
try {
|
|
307
|
+
await this.ensureCoder();
|
|
308
|
+
await this.ensureReviewer();
|
|
309
|
+
}
|
|
310
|
+
catch (err) {
|
|
311
|
+
// Roll back only what THIS call started. Crucially, `<role>Started` may be
|
|
312
|
+
// cleared only when the stop actually succeeded: SessionManager.startSession
|
|
313
|
+
// returns the EXISTING session for a name that is still live and ignores the
|
|
314
|
+
// new engine/model. So if we lied about the session being gone, the next
|
|
315
|
+
// spawn_subagents would sail past the "engine cannot change after start"
|
|
316
|
+
// guard, silently reuse the old engine's process, and still record the new
|
|
317
|
+
// engine in decisions.jsonl and the registry — the exact divergence that
|
|
318
|
+
// guard exists to prevent.
|
|
319
|
+
if (!coderWasStarted && this.coderStarted) {
|
|
320
|
+
this.coderStarted = !(await this.stopRolledBackSession(this.coderName));
|
|
321
|
+
}
|
|
322
|
+
if (!reviewerWasStarted && this.reviewerStarted) {
|
|
323
|
+
this.reviewerStarted = !(await this.stopRolledBackSession(this.reviewerName));
|
|
324
|
+
}
|
|
325
|
+
this.coderSelection = previousCoder;
|
|
326
|
+
this.reviewerSelection = previousReviewer;
|
|
327
|
+
throw err;
|
|
328
|
+
}
|
|
329
|
+
const effectiveSelection = {
|
|
330
|
+
coder: { engine: nextCoder.engine, model: nextCoder.model },
|
|
331
|
+
reviewer: { engine: nextReviewer.engine, model: nextReviewer.model },
|
|
332
|
+
};
|
|
333
|
+
this.appendDecisionLog({
|
|
334
|
+
kind: 'spawn_subagents',
|
|
335
|
+
actor: 'planner',
|
|
336
|
+
payload: {
|
|
337
|
+
coder_engine: nextCoder.engine,
|
|
338
|
+
coder_model: this.roleModel('coder', nextCoder),
|
|
339
|
+
reviewer_engine: nextReviewer.engine,
|
|
340
|
+
reviewer_model: this.roleModel('reviewer', nextReviewer),
|
|
341
|
+
},
|
|
342
|
+
});
|
|
343
|
+
await this.config.onRoleSelectionChanged?.(effectiveSelection);
|
|
144
344
|
}
|
|
145
345
|
/**
|
|
146
346
|
* Reset a single subagent — stop its session, clear the started flag, and
|
|
@@ -171,8 +371,10 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
171
371
|
this.plannerStarted = false;
|
|
172
372
|
if (agent === 'coder')
|
|
173
373
|
this.coderStarted = false;
|
|
174
|
-
if (agent === 'reviewer')
|
|
374
|
+
if (agent === 'reviewer') {
|
|
175
375
|
this.reviewerStarted = false;
|
|
376
|
+
this.reviewerSessionPrompt = null;
|
|
377
|
+
}
|
|
176
378
|
if (opts.eagerRestart) {
|
|
177
379
|
if (agent === 'planner')
|
|
178
380
|
await this.ensurePlanner();
|
|
@@ -314,12 +516,15 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
314
516
|
async ensurePlanner() {
|
|
315
517
|
if (this.plannerStarted)
|
|
316
518
|
return;
|
|
519
|
+
this.validateSelection('planner', this.plannerSelection);
|
|
317
520
|
await this.config.manager.startSession({
|
|
318
521
|
name: this.plannerName,
|
|
319
522
|
cwd: this.config.workspace,
|
|
320
|
-
engine:
|
|
321
|
-
model: this.
|
|
322
|
-
|
|
523
|
+
engine: this.plannerSelection.engine,
|
|
524
|
+
model: this.roleModel('planner', this.plannerSelection),
|
|
525
|
+
customEngine: this.plannerSelection.engine === 'custom' ? this.plannerSelection.customEngine : undefined,
|
|
526
|
+
permissionMode: this.plannerSelection.engine === 'claude' ? 'bypassPermissions' : 'manual',
|
|
527
|
+
sandboxMode: this.plannerSelection.engine === 'claude' ? undefined : 'read-only',
|
|
323
528
|
systemPrompt: this.plannerSystemPrompt,
|
|
324
529
|
// Hard role boundary: Planner must NEVER author content files itself.
|
|
325
530
|
// Its only writes are plan.md / goal.json via the write_plan /
|
|
@@ -355,7 +560,7 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
355
560
|
// iter_done
|
|
356
561
|
promptText = `[system] iter ${env.iter} done. verdict=${env.payload.verdict} metric=${env.payload.metric}`;
|
|
357
562
|
}
|
|
358
|
-
const result = (await this.config.manager.sendMessage(this.plannerName, promptText, {
|
|
563
|
+
const result = (await this.config.manager.sendMessage(this.plannerName, this.withRoleInstructions('planner', this.plannerSelection, this.plannerSystemPrompt, promptText), {
|
|
359
564
|
timeout: this.config.sendTimeoutMs ?? 10 * 60_000,
|
|
360
565
|
}));
|
|
361
566
|
if (result.error) {
|
|
@@ -363,6 +568,11 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
363
568
|
this.emit('planner_error', new Error(result.error));
|
|
364
569
|
}
|
|
365
570
|
const replyText = (result.output ?? '').trim();
|
|
571
|
+
// Feed the transcript that engines without native conversation replay next
|
|
572
|
+
// turn. Recorded AFTER the send so the current message isn't duplicated in
|
|
573
|
+
// its own history block.
|
|
574
|
+
this.recordTurn('planner', 'user', promptText);
|
|
575
|
+
this.recordTurn('planner', 'agent', replyText);
|
|
366
576
|
// S3: parse autoloop-fenced tool calls out of the reply, apply effects,
|
|
367
577
|
// and bubble emitted messages back into the runner queue.
|
|
368
578
|
const parsed = parsePlannerReply(replyText);
|
|
@@ -371,16 +581,11 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
371
581
|
}
|
|
372
582
|
const effects = {
|
|
373
583
|
spawnSubagents: async (args) => {
|
|
374
|
-
this.appendDecisionLog({
|
|
375
|
-
kind: 'spawn_subagents',
|
|
376
|
-
actor: 'planner',
|
|
377
|
-
payload: { args },
|
|
378
|
-
});
|
|
379
584
|
if (this.config.onSpawnSubagents) {
|
|
380
585
|
await this.config.onSpawnSubagents(args);
|
|
381
586
|
}
|
|
382
587
|
else {
|
|
383
|
-
this.logger.warn?.('[autoloop] spawn_subagents called but no handler installed
|
|
588
|
+
this.logger.warn?.('[autoloop] spawn_subagents called but no handler is installed');
|
|
384
589
|
}
|
|
385
590
|
},
|
|
386
591
|
updatePushPolicy: (delta) => {
|
|
@@ -470,11 +675,13 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
470
675
|
async ensureCoder() {
|
|
471
676
|
if (this.coderStarted)
|
|
472
677
|
return;
|
|
678
|
+
this.validateSelection('coder', this.coderSelection);
|
|
473
679
|
await this.config.manager.startSession({
|
|
474
680
|
name: this.coderName,
|
|
475
681
|
cwd: this.config.workspace,
|
|
476
|
-
engine:
|
|
477
|
-
model: this.
|
|
682
|
+
engine: this.coderSelection.engine,
|
|
683
|
+
model: this.roleModel('coder', this.coderSelection),
|
|
684
|
+
customEngine: this.coderSelection.engine === 'custom' ? this.coderSelection.customEngine : undefined,
|
|
478
685
|
permissionMode: 'bypassPermissions',
|
|
479
686
|
systemPrompt: this.coderSystemPrompt,
|
|
480
687
|
});
|
|
@@ -526,7 +733,9 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
526
733
|
text: `🔨 Coder iter ${env.iter} working…`,
|
|
527
734
|
ts: new Date().toISOString(),
|
|
528
735
|
});
|
|
529
|
-
const result = await this.sendWithRecovery('coder', this.coderName, promptText);
|
|
736
|
+
const result = await this.sendWithRecovery('coder', this.coderName, this.withRoleInstructions('coder', this.coderSelection, this.coderSystemPrompt, promptText));
|
|
737
|
+
this.recordTurn('coder', 'user', promptText);
|
|
738
|
+
this.recordTurn('coder', 'agent', (result.output ?? '').trim());
|
|
530
739
|
// A3: subprocess died (recovery retry exhausted). Surface as phase_error
|
|
531
740
|
// rather than silently masquerading as a "clarification request"; the
|
|
532
741
|
// runner's circuit breaker can then trip after enough consecutive failures.
|
|
@@ -645,16 +854,26 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
645
854
|
async ensureReviewer() {
|
|
646
855
|
if (this.reviewerStarted)
|
|
647
856
|
return;
|
|
857
|
+
this.validateSelection('reviewer', this.reviewerSelection);
|
|
648
858
|
fs.mkdirSync(this.reviewerSandboxDir, { recursive: true });
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
859
|
+
const sessionPrompt = this.buildReviewerSystemPrompt();
|
|
860
|
+
this.reviewerSessionPrompt = sessionPrompt;
|
|
861
|
+
try {
|
|
862
|
+
await this.config.manager.startSession({
|
|
863
|
+
name: this.reviewerName,
|
|
864
|
+
cwd: this.reviewerSandboxDir,
|
|
865
|
+
engine: this.reviewerSelection.engine,
|
|
866
|
+
model: this.roleModel('reviewer', this.reviewerSelection),
|
|
867
|
+
customEngine: this.reviewerSelection.engine === 'custom' ? this.reviewerSelection.customEngine : undefined,
|
|
868
|
+
permissionMode: 'bypassPermissions',
|
|
869
|
+
systemPrompt: sessionPrompt,
|
|
870
|
+
});
|
|
871
|
+
this.reviewerStarted = true;
|
|
872
|
+
}
|
|
873
|
+
catch (err) {
|
|
874
|
+
this.reviewerSessionPrompt = null;
|
|
875
|
+
throw err;
|
|
876
|
+
}
|
|
658
877
|
}
|
|
659
878
|
/**
|
|
660
879
|
* Stage the iter's artifacts into the Reviewer sandbox cwd. Reviewer is a
|
|
@@ -724,7 +943,9 @@ export class ClaudeAgentDispatcher extends EventEmitter {
|
|
|
724
943
|
text: `🔍 Reviewer iter ${env.payload.iter} auditing…`,
|
|
725
944
|
ts: new Date().toISOString(),
|
|
726
945
|
});
|
|
727
|
-
const result = await this.sendWithRecovery('reviewer', this.reviewerName, promptText);
|
|
946
|
+
const result = await this.sendWithRecovery('reviewer', this.reviewerName, this.withRoleInstructions('reviewer', this.reviewerSelection, this.reviewerSessionPrompt ?? this.reviewerSystemPrompt, promptText));
|
|
947
|
+
this.recordTurn('reviewer', 'user', promptText);
|
|
948
|
+
this.recordTurn('reviewer', 'agent', (result.output ?? '').trim());
|
|
728
949
|
if (result.fatal) {
|
|
729
950
|
this.appendDecisionLog({
|
|
730
951
|
kind: 'phase_error',
|