@andromarces/agent-loops 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +461 -0
- package/docs/orchestrator-instructions.md +166 -0
- package/package.json +55 -0
- package/src/agents/agy.mjs +46 -0
- package/src/agents/claude.mjs +85 -0
- package/src/agents/codex.mjs +73 -0
- package/src/agents/copilot.mjs +86 -0
- package/src/agents/index.mjs +37 -0
- package/src/agents/opencode.mjs +152 -0
- package/src/cli.mjs +338 -0
- package/src/contracts/orchestrator-action.mjs +74 -0
- package/src/entrypoints/copilot.mjs +57 -0
- package/src/hook/copilot-parent-guard.mjs +35 -0
- package/src/hook/decision.mjs +27 -0
- package/src/hook/parent-guard.mjs +40 -0
- package/src/lib/args.mjs +43 -0
- package/src/lib/entrypoint.mjs +17 -0
- package/src/lib/exec.mjs +100 -0
- package/src/lib/json.mjs +53 -0
- package/src/lib/log.mjs +44 -0
- package/src/lib/report.mjs +89 -0
- package/src/lib/runstate.mjs +207 -0
- package/src/lib/snapshot.mjs +231 -0
- package/src/orchestrator.mjs +54 -0
- package/src/prompts/orchestrator.mjs +73 -0
- package/src/prompts/report.mjs +10 -0
- package/src/prompts/reviewer.mjs +17 -0
- package/src/prompts/worker.mjs +21 -0
- package/src/role.mjs +661 -0
- package/src/runtime.mjs +185 -0
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
# Orchestrator instructions (harness-neutral)
|
|
2
|
+
|
|
3
|
+
You are the parent orchestrator for an `agent-loop role` run. Per-harness entry
|
|
4
|
+
points (the Claude Code and Codex CLI skills, the OpenCode plugin command)
|
|
5
|
+
include this file instead of copying it. The headless prompt in
|
|
6
|
+
`src/prompts/orchestrator.mjs` states the same role rules in JSON-action form;
|
|
7
|
+
this file is the source for shared rules.
|
|
8
|
+
|
|
9
|
+
## Role
|
|
10
|
+
|
|
11
|
+
- Delegate the task, track results, handle blockers, and report completion.
|
|
12
|
+
- Never implement changes, never review code yourself, never run tests, never
|
|
13
|
+
open child transcripts.
|
|
14
|
+
- Read only the JSON envelope the subcommand prints on stdout. Child stderr
|
|
15
|
+
logs and child response text beyond the envelope are not input.
|
|
16
|
+
|
|
17
|
+
## Inputs to collect from the invocation
|
|
18
|
+
|
|
19
|
+
Collect these before the first dispatch:
|
|
20
|
+
|
|
21
|
+
- the task
|
|
22
|
+
- worker CLI, model, and effort (`--worker`, `--worker-model`, `--worker-effort`)
|
|
23
|
+
- reviewer CLI, model, and effort (`--reviewer`, `--reviewer-model`, `--reviewer-effort`)
|
|
24
|
+
- mode: `work-first`, `review-first`, or `review-only` (`--mode`)
|
|
25
|
+
- maximum steps (`--max-steps`)
|
|
26
|
+
|
|
27
|
+
## Resolving the CLI
|
|
28
|
+
|
|
29
|
+
The command blocks below call `agent-loop` directly. Install the CLI globally:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
npm install -g @andromarces/agent-loops
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
From a clone, register the `bin` field globally instead. Add the pnpm global
|
|
36
|
+
bin directory to PATH first:
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
pnpm setup # restart the shell afterwards
|
|
40
|
+
pnpm add -g .
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
pnpm v11 removed `pnpm link --global` and keeps global bins under `PNPM_HOME`;
|
|
44
|
+
`pnpm add -g .` fails with `ERR_PNPM_GLOBAL_BIN_DIR_NOT_IN_PATH` until
|
|
45
|
+
`pnpm setup` puts that directory on PATH. Without a global install, replace
|
|
46
|
+
`agent-loop` with `node "<repo>/src/cli.mjs"` and quote the repository path so
|
|
47
|
+
a path with spaces works, or run `pnpm agent-loop` from the repository root.
|
|
48
|
+
|
|
49
|
+
## Starting a run
|
|
50
|
+
|
|
51
|
+
The first dispatch carries the init flags, including `--parent-session` when
|
|
52
|
+
the harness provides a session id:
|
|
53
|
+
|
|
54
|
+
```bash
|
|
55
|
+
printf '%s' "<first child prompt>" | agent-loop role dispatch \
|
|
56
|
+
--role <first role> \
|
|
57
|
+
--cwd "<work tree>" \
|
|
58
|
+
--task "<task>" \
|
|
59
|
+
--mode "<mode>" \
|
|
60
|
+
--parent-session "<parent session id>" \
|
|
61
|
+
--worker <cli> --worker-model <model> --worker-effort <level> \
|
|
62
|
+
--reviewer <cli> --reviewer-model <model> --reviewer-effort <level> \
|
|
63
|
+
--max-steps <count>
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
- The first role matches the mode: `worker` in `work-first` and `review-first`,
|
|
67
|
+
`reviewer` in `review-only`.
|
|
68
|
+
- `--worker` is required in `work-first` and `review-first` and optional in
|
|
69
|
+
`review-only`, which never dispatches the worker. `--worker-model` and
|
|
70
|
+
`--worker-effort` are always optional, and require `--worker`. An OpenCode
|
|
71
|
+
`--<role>-effort` also requires an explicit `--<role>-model`.
|
|
72
|
+
- Always pass `--cwd`. It defaults to the current directory, which for an
|
|
73
|
+
interactive parent is normally not the target work tree.
|
|
74
|
+
- Pass the child prompt on stdin. No prompt files.
|
|
75
|
+
- A second task in the same session starts a new run with a new init call, and
|
|
76
|
+
only after the previous run is terminal. The subcommand archives the previous
|
|
77
|
+
state file and rejects init over a non-terminal run.
|
|
78
|
+
- Later dispatches read the configuration from the state file. Do not repeat
|
|
79
|
+
`--task` on a later dispatch: `--task` keys init detection, so a dispatch
|
|
80
|
+
that carries it while a run is active fails instead of continuing the run.
|
|
81
|
+
Repeating any other init flag with its current value is accepted; changing
|
|
82
|
+
one is rejected, so omit changed flags and never invent new values.
|
|
83
|
+
|
|
84
|
+
## Dispatch
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
printf '%s' "<prompt>" | agent-loop role dispatch --role worker --cwd "<work tree>"
|
|
88
|
+
printf '%s' "<prompt>" | agent-loop role dispatch --role reviewer --cwd "<work tree>"
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Read the JSON envelope on stdout. Example reviewer envelope:
|
|
92
|
+
|
|
93
|
+
```json
|
|
94
|
+
{
|
|
95
|
+
"role": "reviewer",
|
|
96
|
+
"status": "ok",
|
|
97
|
+
"report": { "conclusion": "...", "why": "...", "blockers": "..." },
|
|
98
|
+
"verdict": "accept"
|
|
99
|
+
}
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
- `status: "error"` carries `error`. Treat it as not accepted.
|
|
103
|
+
- A reviewer envelope carries `verdict`: `accept`, `reject`, or `unknown` when
|
|
104
|
+
the required `Verdict:` line is missing or malformed. Treat `unknown` as not
|
|
105
|
+
accepted. Process success never implies acceptance.
|
|
106
|
+
- When the closing block cannot be parsed, `report` is null and `raw` carries
|
|
107
|
+
the tail of the response. Treat a missing report as not accepted.
|
|
108
|
+
|
|
109
|
+
## Loop policy
|
|
110
|
+
|
|
111
|
+
- `work-first`: worker, reviewer, worker corrections, reviewer, until
|
|
112
|
+
`verdict: accept`.
|
|
113
|
+
- `review-first`: reviewer first, then worker corrections and another review
|
|
114
|
+
if needed.
|
|
115
|
+
- `review-only`: reviewer, then report. Findings alone never authorize edits;
|
|
116
|
+
the subcommand rejects worker dispatch in this mode.
|
|
117
|
+
|
|
118
|
+
## Completion
|
|
119
|
+
|
|
120
|
+
Completion is completion of the requested work, not code acceptance.
|
|
121
|
+
|
|
122
|
+
- `work-first` and `review-first`: call `finish` after the reviewer returns
|
|
123
|
+
`verdict: accept` on the latest changed state and the report names the
|
|
124
|
+
checks that passed.
|
|
125
|
+
- `review-only`: call `finish` after the reviewer report, whatever the
|
|
126
|
+
verdict. The summary records the verdict in `verified` and the findings in
|
|
127
|
+
`open`.
|
|
128
|
+
|
|
129
|
+
## Finish output
|
|
130
|
+
|
|
131
|
+
`finish` reads the summary as JSON on stdin and is accepted only from the
|
|
132
|
+
`active` lifecycle. All five keys are required and carry the same keys as the
|
|
133
|
+
orchestrator action contract (`validateAction`):
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
printf '%s' '{"changed":"...","verified":"...","deferred":"...","notDone":"...","open":"..."}' \
|
|
137
|
+
| agent-loop role finish --cwd "<work tree>"
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
## Blockers and terminal states
|
|
141
|
+
|
|
142
|
+
- Blocker, `interrupted` lifecycle, or step limit with work remaining: call
|
|
143
|
+
`agent-loop role abort --cwd "<work tree>" --reason "<explanation>"` and
|
|
144
|
+
report the unresolved condition. Never repeat an uncertain turn without a
|
|
145
|
+
maintainer decision.
|
|
146
|
+
- `halted` lifecycle (reviewer mutation or snapshot error): the run is already
|
|
147
|
+
terminal and the subcommand rejects further operations, including `abort`.
|
|
148
|
+
Report the failure and the modified paths from the envelope, then stop.
|
|
149
|
+
- Every run ends in exactly one terminal lifecycle: `finished`, `aborted`, or
|
|
150
|
+
`halted`. The parent-edit guard (#57, Claude Code, Codex CLI, and OpenCode)
|
|
151
|
+
releases on any of them; any run without `--parent-session` keeps its parent
|
|
152
|
+
unguarded.
|
|
153
|
+
|
|
154
|
+
## Recovery after compaction or restart
|
|
155
|
+
|
|
156
|
+
Read the state file at
|
|
157
|
+
`<runs root>/<sha256 of the --cwd, shortened>/state.json`, where the runs root
|
|
158
|
+
is `<os tmpdir>/agent-loops/runs` by default and the `AGENT_LOOP_RUNS_ROOT`
|
|
159
|
+
environment variable overrides it. The directory name is the first 12 hex
|
|
160
|
+
characters of the sha256 of the resolved `--cwd`, with the Windows drive letter
|
|
161
|
+
lowercased before hashing, so `c:\repo` and `C:\repo` share one directory.
|
|
162
|
+
This state-file read is the one exception to the stdout-envelope rule. Honor
|
|
163
|
+
the state file's `mode` and `lifecycle`, and continue from `stepsUsed` and
|
|
164
|
+
`lastResult`. A resumed review-only task keeps its prohibition on worker
|
|
165
|
+
dispatch. From `interrupted`, the parent aborts; only a maintainer may decide
|
|
166
|
+
to resume with `dispatch --resume-interrupted`.
|
package/package.json
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@andromarces/agent-loops",
|
|
3
|
+
"version": "0.2.0",
|
|
4
|
+
"private": false,
|
|
5
|
+
"description": "Run a task loop across several CLI coding agents.",
|
|
6
|
+
"homepage": "https://github.com/andromarces/agent-loops#readme",
|
|
7
|
+
"bugs": {
|
|
8
|
+
"url": "https://github.com/andromarces/agent-loops/issues"
|
|
9
|
+
},
|
|
10
|
+
"license": "MIT",
|
|
11
|
+
"repository": {
|
|
12
|
+
"type": "git",
|
|
13
|
+
"url": "git+https://github.com/andromarces/agent-loops.git"
|
|
14
|
+
},
|
|
15
|
+
"bin": {
|
|
16
|
+
"agent-loop": "./src/cli.mjs",
|
|
17
|
+
"agent-loop-copilot": "./src/entrypoints/copilot.mjs",
|
|
18
|
+
"agent-loops": "./src/cli.mjs"
|
|
19
|
+
},
|
|
20
|
+
"files": [
|
|
21
|
+
"src",
|
|
22
|
+
"docs/orchestrator-instructions.md",
|
|
23
|
+
"LICENSE"
|
|
24
|
+
],
|
|
25
|
+
"type": "module",
|
|
26
|
+
"publishConfig": {
|
|
27
|
+
"access": "public"
|
|
28
|
+
},
|
|
29
|
+
"scripts": {
|
|
30
|
+
"agent-loop": "node ./src/cli.mjs",
|
|
31
|
+
"fmt": "oxfmt",
|
|
32
|
+
"fmt:check": "oxfmt --check",
|
|
33
|
+
"lint": "oxlint",
|
|
34
|
+
"test": "vitest run",
|
|
35
|
+
"prepare": "node .husky/install.mjs",
|
|
36
|
+
"test:watch": "vitest"
|
|
37
|
+
},
|
|
38
|
+
"dependencies": {
|
|
39
|
+
"execa": "^10.0.1"
|
|
40
|
+
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"husky": "^9.1.7",
|
|
43
|
+
"lint-staged": "^17.5.1",
|
|
44
|
+
"oxfmt": "^0.70.0",
|
|
45
|
+
"oxlint": "^1.85.0",
|
|
46
|
+
"vitest": "^5.0.1"
|
|
47
|
+
},
|
|
48
|
+
"lint-staged": {
|
|
49
|
+
"*.{js,mjs,cjs,ts,mts,cts,jsx,tsx}": "oxfmt"
|
|
50
|
+
},
|
|
51
|
+
"engines": {
|
|
52
|
+
"node": ">=22"
|
|
53
|
+
},
|
|
54
|
+
"packageManager": "pnpm@12.4.1"
|
|
55
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { parseJson } from "../lib/json.mjs";
|
|
2
|
+
import { exec } from "../lib/exec.mjs";
|
|
3
|
+
|
|
4
|
+
export async function runAgy(state, prompt, options = {}) {
|
|
5
|
+
const { cwd, readOnly, timeout, signal, role } = options;
|
|
6
|
+
// --input-format text reads the prompt from stdin; -p is omitted because it consumes the next arg as the prompt value.
|
|
7
|
+
const args = ["--input-format", "text", "--output-format", "json"];
|
|
8
|
+
|
|
9
|
+
if (readOnly) {
|
|
10
|
+
args.push("--mode", "plan");
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
if (state.model) {
|
|
14
|
+
args.push("--model", state.model);
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
if (state.effort) {
|
|
18
|
+
args.push("--effort", state.effort);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
if (state.sessionId) {
|
|
22
|
+
args.push("--conversation", state.sessionId);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
const { stdout } = await exec("agy", args, { cwd, input: prompt, timeout, signal, role });
|
|
26
|
+
const result = parseJson(stdout, "Antigravity CLI");
|
|
27
|
+
|
|
28
|
+
if (!result.conversation_id) {
|
|
29
|
+
throw new Error("Antigravity did not return a conversation_id.");
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
state.sessionId = result.conversation_id;
|
|
33
|
+
setUsage(state, result);
|
|
34
|
+
|
|
35
|
+
return String(result.response ?? "").trim();
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function setUsage(state, result) {
|
|
39
|
+
const usage = {};
|
|
40
|
+
if (result?.usage) usage.mainLoop = result.usage;
|
|
41
|
+
if (Object.keys(usage).length > 0) {
|
|
42
|
+
state.usage = usage;
|
|
43
|
+
} else {
|
|
44
|
+
delete state.usage;
|
|
45
|
+
}
|
|
46
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import { parseJson } from "../lib/json.mjs";
|
|
2
|
+
import { exec } from "../lib/exec.mjs";
|
|
3
|
+
|
|
4
|
+
export async function runClaude(state, prompt, options = {}) {
|
|
5
|
+
const { cwd, readOnly, timeout, signal, role } = options;
|
|
6
|
+
const args = ["-p"];
|
|
7
|
+
const execOptions = { cwd, input: prompt, timeout, signal, role };
|
|
8
|
+
|
|
9
|
+
if (state.sessionId) {
|
|
10
|
+
args.push("--resume", state.sessionId);
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
if (readOnly) {
|
|
14
|
+
args.push("--permission-mode", "plan");
|
|
15
|
+
// Plan mode is a write guard here, not a planning workflow. Without this variable it
|
|
16
|
+
// delegates research to the built-in Explore and Plan subagents, which inherit the role
|
|
17
|
+
// model (Explore capped at Opus on the Claude API). Requires Claude Code v2.1.198+.
|
|
18
|
+
execOptions.env = { CLAUDE_CODE_DISABLE_EXPLORE_PLAN_AGENTS: "1" };
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
if (state.model) {
|
|
22
|
+
args.push("--model", state.model);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
if (state.effort) {
|
|
26
|
+
args.push("--effort", state.effort);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
args.push("--output-format", "json");
|
|
30
|
+
|
|
31
|
+
let stdout;
|
|
32
|
+
try {
|
|
33
|
+
({ stdout } = await exec("claude", args, execOptions));
|
|
34
|
+
} catch (err) {
|
|
35
|
+
// A non-zero exit can still carry a result event with usage. Expose it, then rethrow.
|
|
36
|
+
let failed;
|
|
37
|
+
try {
|
|
38
|
+
failed = JSON.parse(err?.stdout ?? "");
|
|
39
|
+
} catch {
|
|
40
|
+
failed = undefined;
|
|
41
|
+
}
|
|
42
|
+
setUsage(state, findResultEvent(failed));
|
|
43
|
+
throw err;
|
|
44
|
+
}
|
|
45
|
+
const parsed = parseJson(stdout, "Claude Code");
|
|
46
|
+
|
|
47
|
+
const sessionId = Array.isArray(parsed)
|
|
48
|
+
? parsed.map((event) => event?.session_id).find(Boolean)
|
|
49
|
+
: parsed.session_id;
|
|
50
|
+
|
|
51
|
+
if (!sessionId) {
|
|
52
|
+
throw new Error("Claude Code did not return a session_id.");
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
state.sessionId = sessionId;
|
|
56
|
+
const resultEvent = findResultEvent(parsed);
|
|
57
|
+
setUsage(state, resultEvent);
|
|
58
|
+
|
|
59
|
+
return String(resultEvent?.result ?? "").trim();
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function findResultEvent(parsed) {
|
|
63
|
+
if (Array.isArray(parsed)) {
|
|
64
|
+
return parsed.find((event) => event?.type === "result");
|
|
65
|
+
}
|
|
66
|
+
return parsed && typeof parsed === "object" ? parsed : undefined;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* Sets `state.usage` from a result event, or removes it when the event carries no usage.
|
|
71
|
+
* `usage` covers the top-level loop only; `modelUsage` and `total_cost_usd` include subagents.
|
|
72
|
+
*/
|
|
73
|
+
function setUsage(state, resultEvent) {
|
|
74
|
+
const usage = {};
|
|
75
|
+
if (resultEvent?.modelUsage) usage.models = resultEvent.modelUsage;
|
|
76
|
+
if (resultEvent?.usage) usage.mainLoop = resultEvent.usage;
|
|
77
|
+
if (typeof resultEvent?.total_cost_usd === "number") {
|
|
78
|
+
usage.totalCostUsd = resultEvent.total_cost_usd;
|
|
79
|
+
}
|
|
80
|
+
if (Object.keys(usage).length > 0) {
|
|
81
|
+
state.usage = usage;
|
|
82
|
+
} else {
|
|
83
|
+
delete state.usage;
|
|
84
|
+
}
|
|
85
|
+
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
import { parseJsonLines } from "../lib/json.mjs";
|
|
2
|
+
import { exec } from "../lib/exec.mjs";
|
|
3
|
+
|
|
4
|
+
export async function runCodex(state, prompt, options = {}) {
|
|
5
|
+
const { cwd, readOnly, timeout, signal, role } = options;
|
|
6
|
+
const configArgs = [];
|
|
7
|
+
|
|
8
|
+
if (readOnly) {
|
|
9
|
+
configArgs.push("-c", 'sandbox_mode="read-only"');
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
const modelArgs = [];
|
|
13
|
+
if (state.model) {
|
|
14
|
+
modelArgs.push("-m", state.model);
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
if (state.effort) {
|
|
18
|
+
modelArgs.push("-c", `model_reasoning_effort=${state.effort}`);
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
let args;
|
|
22
|
+
if (state.sessionId) {
|
|
23
|
+
args = ["exec", "resume", state.sessionId, ...configArgs, "--json", ...modelArgs, "-"];
|
|
24
|
+
} else {
|
|
25
|
+
args = ["exec", ...configArgs, "--json", ...modelArgs];
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
const { stdout } = await exec("codex", args, { cwd, input: prompt, timeout, signal, role });
|
|
29
|
+
const events = parseJsonLines(stdout);
|
|
30
|
+
|
|
31
|
+
const started = events.find((event) => event.type === "thread.started");
|
|
32
|
+
const returnedId = started?.thread_id;
|
|
33
|
+
|
|
34
|
+
if (!returnedId) {
|
|
35
|
+
throw new Error("Codex did not return a thread ID.");
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
if (state.sessionId && returnedId !== state.sessionId) {
|
|
39
|
+
throw new Error(
|
|
40
|
+
[
|
|
41
|
+
"Codex did not resume the expected thread.",
|
|
42
|
+
`Expected: ${state.sessionId}`,
|
|
43
|
+
`Received: ${returnedId}`,
|
|
44
|
+
].join("\n"),
|
|
45
|
+
);
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
state.sessionId = returnedId;
|
|
49
|
+
setUsage(
|
|
50
|
+
state,
|
|
51
|
+
events.find((event) => event.type === "turn.completed"),
|
|
52
|
+
);
|
|
53
|
+
|
|
54
|
+
const messages = events
|
|
55
|
+
.filter((event) => event.type === "item.completed" && event.item?.type === "agent_message")
|
|
56
|
+
.map((event) => event.item.text)
|
|
57
|
+
.filter(Boolean);
|
|
58
|
+
|
|
59
|
+
if (messages.length === 0) {
|
|
60
|
+
throw new Error("Codex did not return an agent message.");
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
return String(messages.at(-1)).trim();
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** Sets top-level turn usage, or removes stale usage when Codex omits it. */
|
|
67
|
+
function setUsage(state, completedTurn) {
|
|
68
|
+
if (completedTurn?.usage) {
|
|
69
|
+
state.usage = { mainLoop: completedTurn.usage };
|
|
70
|
+
} else {
|
|
71
|
+
delete state.usage;
|
|
72
|
+
}
|
|
73
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
2
|
+
import { parseJsonLines } from "../lib/json.mjs";
|
|
3
|
+
import { exec } from "../lib/exec.mjs";
|
|
4
|
+
|
|
5
|
+
export async function runCopilot(state, prompt, options = {}) {
|
|
6
|
+
const { cwd, readOnly, timeout, signal, role } = options;
|
|
7
|
+
const requestedSessionId = state.sessionId;
|
|
8
|
+
|
|
9
|
+
if (!state.sessionId) {
|
|
10
|
+
state.sessionId = randomUUID();
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
const args = ["--session-id", state.sessionId, "-s", "--no-ask-user", "--output-format", "json"];
|
|
14
|
+
|
|
15
|
+
if (readOnly) {
|
|
16
|
+
args.push("--deny-tool", "write");
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
if (state.model) {
|
|
20
|
+
args.push("--model", state.model);
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
if (state.effort) {
|
|
24
|
+
args.push("--reasoning-effort", state.effort);
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
let stdout;
|
|
28
|
+
try {
|
|
29
|
+
({ stdout } = await exec("copilot", args, { cwd, input: prompt, timeout, signal, role }));
|
|
30
|
+
} catch (error) {
|
|
31
|
+
const failed = parseJsonLines(error?.stdout ?? "");
|
|
32
|
+
setUsage(state, findResultEvent(failed));
|
|
33
|
+
throw error;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
const events = parseJsonLines(stdout);
|
|
37
|
+
const resultEvent = findResultEvent(events);
|
|
38
|
+
setUsage(state, resultEvent);
|
|
39
|
+
|
|
40
|
+
const returnedId = resultEvent?.sessionId ?? resultEvent?.session_id;
|
|
41
|
+
if (!returnedId) {
|
|
42
|
+
throw new Error("Copilot did not return a session ID.");
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
if (requestedSessionId && requestedSessionId !== returnedId) {
|
|
46
|
+
throw new Error(
|
|
47
|
+
[
|
|
48
|
+
"Copilot did not resume the expected session.",
|
|
49
|
+
`Expected: ${requestedSessionId}`,
|
|
50
|
+
`Received: ${returnedId}`,
|
|
51
|
+
].join("\n"),
|
|
52
|
+
);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
state.sessionId = returnedId;
|
|
56
|
+
|
|
57
|
+
const message = events
|
|
58
|
+
.filter((event) => event.type === "assistant.message")
|
|
59
|
+
.map((event) => readAssistantMessage(event))
|
|
60
|
+
.filter(Boolean)
|
|
61
|
+
.at(-1);
|
|
62
|
+
|
|
63
|
+
if (!message) {
|
|
64
|
+
throw new Error("Copilot did not return response text.");
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
return String(message).trim();
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
function findResultEvent(events) {
|
|
71
|
+
return events.filter((event) => event?.type === "result").at(-1);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function readAssistantMessage(event) {
|
|
75
|
+
const content = event?.data?.content;
|
|
76
|
+
return typeof content === "string" ? content : "";
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function setUsage(state, resultEvent) {
|
|
80
|
+
if (resultEvent && resultEvent.usage && typeof resultEvent.usage === "object") {
|
|
81
|
+
state.usage = { mainLoop: resultEvent.usage };
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
delete state.usage;
|
|
86
|
+
}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { runClaude } from "./claude.mjs";
|
|
2
|
+
import { runCodex } from "./codex.mjs";
|
|
3
|
+
import { runAgy } from "./agy.mjs";
|
|
4
|
+
import { runOpenCode } from "./opencode.mjs";
|
|
5
|
+
import { runCopilot } from "./copilot.mjs";
|
|
6
|
+
|
|
7
|
+
export function normalizeAgent(kind) {
|
|
8
|
+
return kind === "antigravity" ? "agy" : kind;
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Adapter registry. Each adapter implements `run(state, prompt, options) -> string`:
|
|
13
|
+
* it runs one turn of the CLI and returns the response text, and it mutates
|
|
14
|
+
* `state.sessionId` to hold the persistent session id used for resume.
|
|
15
|
+
* An adapter may set `state.usage` for the turn it just completed; the runtime
|
|
16
|
+
* consumes and removes it after every call. Adapters that expose no usage leave it unset.
|
|
17
|
+
*/
|
|
18
|
+
export const defaultAgents = {
|
|
19
|
+
claude: { run: runClaude },
|
|
20
|
+
codex: { run: runCodex },
|
|
21
|
+
agy: { run: runAgy },
|
|
22
|
+
opencode: { run: runOpenCode },
|
|
23
|
+
copilot: { run: runCopilot },
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
export const supportedAgents = new Set([...Object.keys(defaultAgents), "antigravity"]);
|
|
27
|
+
|
|
28
|
+
export async function runAgent(state, prompt, options = {}, agents = defaultAgents) {
|
|
29
|
+
const kind = normalizeAgent(state.kind);
|
|
30
|
+
const adapter = agents[kind];
|
|
31
|
+
|
|
32
|
+
if (!adapter || typeof adapter.run !== "function") {
|
|
33
|
+
throw new Error(`Unsupported agent: ${state.kind}`);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
return adapter.run(state, prompt, options);
|
|
37
|
+
}
|