@andromarces/agent-loops 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,166 @@
1
+ # Orchestrator instructions (harness-neutral)
2
+
3
+ You are the parent orchestrator for an `agent-loop role` run. Per-harness entry
4
+ points (the Claude Code and Codex CLI skills, the OpenCode plugin command)
5
+ include this file instead of copying it. The headless prompt in
6
+ `src/prompts/orchestrator.mjs` states the same role rules in JSON-action form;
7
+ this file is the source for shared rules.
8
+
9
+ ## Role
10
+
11
+ - Delegate the task, track results, handle blockers, and report completion.
12
+ - Never implement changes, never review code yourself, never run tests, never
13
+ open child transcripts.
14
+ - Read only the JSON envelope the subcommand prints on stdout. Child stderr
15
+ logs and child response text beyond the envelope are not input.
16
+
17
+ ## Inputs to collect from the invocation
18
+
19
+ Collect these before the first dispatch:
20
+
21
+ - the task
22
+ - worker CLI, model, and effort (`--worker`, `--worker-model`, `--worker-effort`)
23
+ - reviewer CLI, model, and effort (`--reviewer`, `--reviewer-model`, `--reviewer-effort`)
24
+ - mode: `work-first`, `review-first`, or `review-only` (`--mode`)
25
+ - maximum steps (`--max-steps`)
26
+
27
+ ## Resolving the CLI
28
+
29
+ The command blocks below call `agent-loop` directly. Install the CLI globally:
30
+
31
+ ```bash
32
+ npm install -g @andromarces/agent-loops
33
+ ```
34
+
35
+ From a clone, register the `bin` field globally instead. Add the pnpm global
36
+ bin directory to PATH first:
37
+
38
+ ```bash
39
+ pnpm setup # restart the shell afterwards
40
+ pnpm add -g .
41
+ ```
42
+
43
+ pnpm v11 removed `pnpm link --global` and keeps global bins under `PNPM_HOME`;
44
+ `pnpm add -g .` fails with `ERR_PNPM_GLOBAL_BIN_DIR_NOT_IN_PATH` until
45
+ `pnpm setup` puts that directory on PATH. Without a global install, replace
46
+ `agent-loop` with `node "<repo>/src/cli.mjs"` and quote the repository path so
47
+ a path with spaces works, or run `pnpm agent-loop` from the repository root.
48
+
49
+ ## Starting a run
50
+
51
+ The first dispatch carries the init flags, including `--parent-session` when
52
+ the harness provides a session id:
53
+
54
+ ```bash
55
+ printf '%s' "<first child prompt>" | agent-loop role dispatch \
56
+ --role <first role> \
57
+ --cwd "<work tree>" \
58
+ --task "<task>" \
59
+ --mode "<mode>" \
60
+ --parent-session "<parent session id>" \
61
+ --worker <cli> --worker-model <model> --worker-effort <level> \
62
+ --reviewer <cli> --reviewer-model <model> --reviewer-effort <level> \
63
+ --max-steps <count>
64
+ ```
65
+
66
+ - The first role matches the mode: `worker` in `work-first` and `review-first`,
67
+ `reviewer` in `review-only`.
68
+ - `--worker` is required in `work-first` and `review-first` and optional in
69
+ `review-only`, which never dispatches the worker. `--worker-model` and
70
+ `--worker-effort` are always optional, and require `--worker`. An OpenCode
71
+ `--<role>-effort` also requires an explicit `--<role>-model`.
72
+ - Always pass `--cwd`. It defaults to the current directory, which for an
73
+ interactive parent is normally not the target work tree.
74
+ - Pass the child prompt on stdin. No prompt files.
75
+ - A second task in the same session starts a new run with a new init call, and
76
+ only after the previous run is terminal. The subcommand archives the previous
77
+ state file and rejects init over a non-terminal run.
78
+ - Later dispatches read the configuration from the state file. Do not repeat
79
+ `--task` on a later dispatch: `--task` keys init detection, so a dispatch
80
+ that carries it while a run is active fails instead of continuing the run.
81
+ Repeating any other init flag with its current value is accepted; changing
82
+ one is rejected, so omit changed flags and never invent new values.
83
+
84
+ ## Dispatch
85
+
86
+ ```bash
87
+ printf '%s' "<prompt>" | agent-loop role dispatch --role worker --cwd "<work tree>"
88
+ printf '%s' "<prompt>" | agent-loop role dispatch --role reviewer --cwd "<work tree>"
89
+ ```
90
+
91
+ Read the JSON envelope on stdout. Example reviewer envelope:
92
+
93
+ ```json
94
+ {
95
+ "role": "reviewer",
96
+ "status": "ok",
97
+ "report": { "conclusion": "...", "why": "...", "blockers": "..." },
98
+ "verdict": "accept"
99
+ }
100
+ ```
101
+
102
+ - `status: "error"` carries `error`. Treat it as not accepted.
103
+ - A reviewer envelope carries `verdict`: `accept`, `reject`, or `unknown` when
104
+ the required `Verdict:` line is missing or malformed. Treat `unknown` as not
105
+ accepted. Process success never implies acceptance.
106
+ - When the closing block cannot be parsed, `report` is null and `raw` carries
107
+ the tail of the response. Treat a missing report as not accepted.
108
+
109
+ ## Loop policy
110
+
111
+ - `work-first`: worker, reviewer, worker corrections, reviewer, until
112
+ `verdict: accept`.
113
+ - `review-first`: reviewer first, then worker corrections and another review
114
+ if needed.
115
+ - `review-only`: reviewer, then report. Findings alone never authorize edits;
116
+ the subcommand rejects worker dispatch in this mode.
117
+
118
+ ## Completion
119
+
120
+ Completion is completion of the requested work, not code acceptance.
121
+
122
+ - `work-first` and `review-first`: call `finish` after the reviewer returns
123
+ `verdict: accept` on the latest changed state and the report names the
124
+ checks that passed.
125
+ - `review-only`: call `finish` after the reviewer report, whatever the
126
+ verdict. The summary records the verdict in `verified` and the findings in
127
+ `open`.
128
+
129
+ ## Finish output
130
+
131
+ `finish` reads the summary as JSON on stdin and is accepted only from the
132
+ `active` lifecycle. All five keys are required and carry the same keys as the
133
+ orchestrator action contract (`validateAction`):
134
+
135
+ ```bash
136
+ printf '%s' '{"changed":"...","verified":"...","deferred":"...","notDone":"...","open":"..."}' \
137
+ | agent-loop role finish --cwd "<work tree>"
138
+ ```
139
+
140
+ ## Blockers and terminal states
141
+
142
+ - Blocker, `interrupted` lifecycle, or step limit with work remaining: call
143
+ `agent-loop role abort --cwd "<work tree>" --reason "<explanation>"` and
144
+ report the unresolved condition. Never repeat an uncertain turn without a
145
+ maintainer decision.
146
+ - `halted` lifecycle (reviewer mutation or snapshot error): the run is already
147
+ terminal and the subcommand rejects further operations, including `abort`.
148
+ Report the failure and the modified paths from the envelope, then stop.
149
+ - Every run ends in exactly one terminal lifecycle: `finished`, `aborted`, or
150
+ `halted`. The parent-edit guard (#57, Claude Code, Codex CLI, and OpenCode)
151
+ releases on any of them; any run without `--parent-session` keeps its parent
152
+ unguarded.
153
+
154
+ ## Recovery after compaction or restart
155
+
156
+ Read the state file at
157
+ `<runs root>/<sha256 of the --cwd, shortened>/state.json`, where the runs root
158
+ is `<os tmpdir>/agent-loops/runs` by default and the `AGENT_LOOP_RUNS_ROOT`
159
+ environment variable overrides it. The directory name is the first 12 hex
160
+ characters of the sha256 of the resolved `--cwd`, with the Windows drive letter
161
+ lowercased before hashing, so `c:\repo` and `C:\repo` share one directory.
162
+ This state-file read is the one exception to the stdout-envelope rule. Honor
163
+ the state file's `mode` and `lifecycle`, and continue from `stepsUsed` and
164
+ `lastResult`. A resumed review-only task keeps its prohibition on worker
165
+ dispatch. From `interrupted`, the parent aborts; only a maintainer may decide
166
+ to resume with `dispatch --resume-interrupted`.
package/package.json ADDED
@@ -0,0 +1,55 @@
1
+ {
2
+ "name": "@andromarces/agent-loops",
3
+ "version": "0.2.0",
4
+ "private": false,
5
+ "description": "Run a task loop across several CLI coding agents.",
6
+ "homepage": "https://github.com/andromarces/agent-loops#readme",
7
+ "bugs": {
8
+ "url": "https://github.com/andromarces/agent-loops/issues"
9
+ },
10
+ "license": "MIT",
11
+ "repository": {
12
+ "type": "git",
13
+ "url": "git+https://github.com/andromarces/agent-loops.git"
14
+ },
15
+ "bin": {
16
+ "agent-loop": "./src/cli.mjs",
17
+ "agent-loop-copilot": "./src/entrypoints/copilot.mjs",
18
+ "agent-loops": "./src/cli.mjs"
19
+ },
20
+ "files": [
21
+ "src",
22
+ "docs/orchestrator-instructions.md",
23
+ "LICENSE"
24
+ ],
25
+ "type": "module",
26
+ "publishConfig": {
27
+ "access": "public"
28
+ },
29
+ "scripts": {
30
+ "agent-loop": "node ./src/cli.mjs",
31
+ "fmt": "oxfmt",
32
+ "fmt:check": "oxfmt --check",
33
+ "lint": "oxlint",
34
+ "test": "vitest run",
35
+ "prepare": "node .husky/install.mjs",
36
+ "test:watch": "vitest"
37
+ },
38
+ "dependencies": {
39
+ "execa": "^10.0.1"
40
+ },
41
+ "devDependencies": {
42
+ "husky": "^9.1.7",
43
+ "lint-staged": "^17.5.1",
44
+ "oxfmt": "^0.70.0",
45
+ "oxlint": "^1.85.0",
46
+ "vitest": "^5.0.1"
47
+ },
48
+ "lint-staged": {
49
+ "*.{js,mjs,cjs,ts,mts,cts,jsx,tsx}": "oxfmt"
50
+ },
51
+ "engines": {
52
+ "node": ">=22"
53
+ },
54
+ "packageManager": "pnpm@12.4.1"
55
+ }
@@ -0,0 +1,46 @@
1
+ import { parseJson } from "../lib/json.mjs";
2
+ import { exec } from "../lib/exec.mjs";
3
+
4
+ export async function runAgy(state, prompt, options = {}) {
5
+ const { cwd, readOnly, timeout, signal, role } = options;
6
+ // --input-format text reads the prompt from stdin; -p is omitted because it consumes the next arg as the prompt value.
7
+ const args = ["--input-format", "text", "--output-format", "json"];
8
+
9
+ if (readOnly) {
10
+ args.push("--mode", "plan");
11
+ }
12
+
13
+ if (state.model) {
14
+ args.push("--model", state.model);
15
+ }
16
+
17
+ if (state.effort) {
18
+ args.push("--effort", state.effort);
19
+ }
20
+
21
+ if (state.sessionId) {
22
+ args.push("--conversation", state.sessionId);
23
+ }
24
+
25
+ const { stdout } = await exec("agy", args, { cwd, input: prompt, timeout, signal, role });
26
+ const result = parseJson(stdout, "Antigravity CLI");
27
+
28
+ if (!result.conversation_id) {
29
+ throw new Error("Antigravity did not return a conversation_id.");
30
+ }
31
+
32
+ state.sessionId = result.conversation_id;
33
+ setUsage(state, result);
34
+
35
+ return String(result.response ?? "").trim();
36
+ }
37
+
38
+ function setUsage(state, result) {
39
+ const usage = {};
40
+ if (result?.usage) usage.mainLoop = result.usage;
41
+ if (Object.keys(usage).length > 0) {
42
+ state.usage = usage;
43
+ } else {
44
+ delete state.usage;
45
+ }
46
+ }
@@ -0,0 +1,85 @@
1
+ import { parseJson } from "../lib/json.mjs";
2
+ import { exec } from "../lib/exec.mjs";
3
+
4
+ export async function runClaude(state, prompt, options = {}) {
5
+ const { cwd, readOnly, timeout, signal, role } = options;
6
+ const args = ["-p"];
7
+ const execOptions = { cwd, input: prompt, timeout, signal, role };
8
+
9
+ if (state.sessionId) {
10
+ args.push("--resume", state.sessionId);
11
+ }
12
+
13
+ if (readOnly) {
14
+ args.push("--permission-mode", "plan");
15
+ // Plan mode is a write guard here, not a planning workflow. Without this variable it
16
+ // delegates research to the built-in Explore and Plan subagents, which inherit the role
17
+ // model (Explore capped at Opus on the Claude API). Requires Claude Code v2.1.198+.
18
+ execOptions.env = { CLAUDE_CODE_DISABLE_EXPLORE_PLAN_AGENTS: "1" };
19
+ }
20
+
21
+ if (state.model) {
22
+ args.push("--model", state.model);
23
+ }
24
+
25
+ if (state.effort) {
26
+ args.push("--effort", state.effort);
27
+ }
28
+
29
+ args.push("--output-format", "json");
30
+
31
+ let stdout;
32
+ try {
33
+ ({ stdout } = await exec("claude", args, execOptions));
34
+ } catch (err) {
35
+ // A non-zero exit can still carry a result event with usage. Expose it, then rethrow.
36
+ let failed;
37
+ try {
38
+ failed = JSON.parse(err?.stdout ?? "");
39
+ } catch {
40
+ failed = undefined;
41
+ }
42
+ setUsage(state, findResultEvent(failed));
43
+ throw err;
44
+ }
45
+ const parsed = parseJson(stdout, "Claude Code");
46
+
47
+ const sessionId = Array.isArray(parsed)
48
+ ? parsed.map((event) => event?.session_id).find(Boolean)
49
+ : parsed.session_id;
50
+
51
+ if (!sessionId) {
52
+ throw new Error("Claude Code did not return a session_id.");
53
+ }
54
+
55
+ state.sessionId = sessionId;
56
+ const resultEvent = findResultEvent(parsed);
57
+ setUsage(state, resultEvent);
58
+
59
+ return String(resultEvent?.result ?? "").trim();
60
+ }
61
+
62
+ function findResultEvent(parsed) {
63
+ if (Array.isArray(parsed)) {
64
+ return parsed.find((event) => event?.type === "result");
65
+ }
66
+ return parsed && typeof parsed === "object" ? parsed : undefined;
67
+ }
68
+
69
+ /**
70
+ * Sets `state.usage` from a result event, or removes it when the event carries no usage.
71
+ * `usage` covers the top-level loop only; `modelUsage` and `total_cost_usd` include subagents.
72
+ */
73
+ function setUsage(state, resultEvent) {
74
+ const usage = {};
75
+ if (resultEvent?.modelUsage) usage.models = resultEvent.modelUsage;
76
+ if (resultEvent?.usage) usage.mainLoop = resultEvent.usage;
77
+ if (typeof resultEvent?.total_cost_usd === "number") {
78
+ usage.totalCostUsd = resultEvent.total_cost_usd;
79
+ }
80
+ if (Object.keys(usage).length > 0) {
81
+ state.usage = usage;
82
+ } else {
83
+ delete state.usage;
84
+ }
85
+ }
@@ -0,0 +1,73 @@
1
+ import { parseJsonLines } from "../lib/json.mjs";
2
+ import { exec } from "../lib/exec.mjs";
3
+
4
+ export async function runCodex(state, prompt, options = {}) {
5
+ const { cwd, readOnly, timeout, signal, role } = options;
6
+ const configArgs = [];
7
+
8
+ if (readOnly) {
9
+ configArgs.push("-c", 'sandbox_mode="read-only"');
10
+ }
11
+
12
+ const modelArgs = [];
13
+ if (state.model) {
14
+ modelArgs.push("-m", state.model);
15
+ }
16
+
17
+ if (state.effort) {
18
+ modelArgs.push("-c", `model_reasoning_effort=${state.effort}`);
19
+ }
20
+
21
+ let args;
22
+ if (state.sessionId) {
23
+ args = ["exec", "resume", state.sessionId, ...configArgs, "--json", ...modelArgs, "-"];
24
+ } else {
25
+ args = ["exec", ...configArgs, "--json", ...modelArgs];
26
+ }
27
+
28
+ const { stdout } = await exec("codex", args, { cwd, input: prompt, timeout, signal, role });
29
+ const events = parseJsonLines(stdout);
30
+
31
+ const started = events.find((event) => event.type === "thread.started");
32
+ const returnedId = started?.thread_id;
33
+
34
+ if (!returnedId) {
35
+ throw new Error("Codex did not return a thread ID.");
36
+ }
37
+
38
+ if (state.sessionId && returnedId !== state.sessionId) {
39
+ throw new Error(
40
+ [
41
+ "Codex did not resume the expected thread.",
42
+ `Expected: ${state.sessionId}`,
43
+ `Received: ${returnedId}`,
44
+ ].join("\n"),
45
+ );
46
+ }
47
+
48
+ state.sessionId = returnedId;
49
+ setUsage(
50
+ state,
51
+ events.find((event) => event.type === "turn.completed"),
52
+ );
53
+
54
+ const messages = events
55
+ .filter((event) => event.type === "item.completed" && event.item?.type === "agent_message")
56
+ .map((event) => event.item.text)
57
+ .filter(Boolean);
58
+
59
+ if (messages.length === 0) {
60
+ throw new Error("Codex did not return an agent message.");
61
+ }
62
+
63
+ return String(messages.at(-1)).trim();
64
+ }
65
+
66
+ /** Sets top-level turn usage, or removes stale usage when Codex omits it. */
67
+ function setUsage(state, completedTurn) {
68
+ if (completedTurn?.usage) {
69
+ state.usage = { mainLoop: completedTurn.usage };
70
+ } else {
71
+ delete state.usage;
72
+ }
73
+ }
@@ -0,0 +1,86 @@
1
+ import { randomUUID } from "node:crypto";
2
+ import { parseJsonLines } from "../lib/json.mjs";
3
+ import { exec } from "../lib/exec.mjs";
4
+
5
+ export async function runCopilot(state, prompt, options = {}) {
6
+ const { cwd, readOnly, timeout, signal, role } = options;
7
+ const requestedSessionId = state.sessionId;
8
+
9
+ if (!state.sessionId) {
10
+ state.sessionId = randomUUID();
11
+ }
12
+
13
+ const args = ["--session-id", state.sessionId, "-s", "--no-ask-user", "--output-format", "json"];
14
+
15
+ if (readOnly) {
16
+ args.push("--deny-tool", "write");
17
+ }
18
+
19
+ if (state.model) {
20
+ args.push("--model", state.model);
21
+ }
22
+
23
+ if (state.effort) {
24
+ args.push("--reasoning-effort", state.effort);
25
+ }
26
+
27
+ let stdout;
28
+ try {
29
+ ({ stdout } = await exec("copilot", args, { cwd, input: prompt, timeout, signal, role }));
30
+ } catch (error) {
31
+ const failed = parseJsonLines(error?.stdout ?? "");
32
+ setUsage(state, findResultEvent(failed));
33
+ throw error;
34
+ }
35
+
36
+ const events = parseJsonLines(stdout);
37
+ const resultEvent = findResultEvent(events);
38
+ setUsage(state, resultEvent);
39
+
40
+ const returnedId = resultEvent?.sessionId ?? resultEvent?.session_id;
41
+ if (!returnedId) {
42
+ throw new Error("Copilot did not return a session ID.");
43
+ }
44
+
45
+ if (requestedSessionId && requestedSessionId !== returnedId) {
46
+ throw new Error(
47
+ [
48
+ "Copilot did not resume the expected session.",
49
+ `Expected: ${requestedSessionId}`,
50
+ `Received: ${returnedId}`,
51
+ ].join("\n"),
52
+ );
53
+ }
54
+
55
+ state.sessionId = returnedId;
56
+
57
+ const message = events
58
+ .filter((event) => event.type === "assistant.message")
59
+ .map((event) => readAssistantMessage(event))
60
+ .filter(Boolean)
61
+ .at(-1);
62
+
63
+ if (!message) {
64
+ throw new Error("Copilot did not return response text.");
65
+ }
66
+
67
+ return String(message).trim();
68
+ }
69
+
70
+ function findResultEvent(events) {
71
+ return events.filter((event) => event?.type === "result").at(-1);
72
+ }
73
+
74
+ function readAssistantMessage(event) {
75
+ const content = event?.data?.content;
76
+ return typeof content === "string" ? content : "";
77
+ }
78
+
79
+ function setUsage(state, resultEvent) {
80
+ if (resultEvent && resultEvent.usage && typeof resultEvent.usage === "object") {
81
+ state.usage = { mainLoop: resultEvent.usage };
82
+ return;
83
+ }
84
+
85
+ delete state.usage;
86
+ }
@@ -0,0 +1,37 @@
1
+ import { runClaude } from "./claude.mjs";
2
+ import { runCodex } from "./codex.mjs";
3
+ import { runAgy } from "./agy.mjs";
4
+ import { runOpenCode } from "./opencode.mjs";
5
+ import { runCopilot } from "./copilot.mjs";
6
+
7
+ export function normalizeAgent(kind) {
8
+ return kind === "antigravity" ? "agy" : kind;
9
+ }
10
+
11
+ /**
12
+ * Adapter registry. Each adapter implements `run(state, prompt, options) -> string`:
13
+ * it runs one turn of the CLI and returns the response text, and it mutates
14
+ * `state.sessionId` to hold the persistent session id used for resume.
15
+ * An adapter may set `state.usage` for the turn it just completed; the runtime
16
+ * consumes and removes it after every call. Adapters that expose no usage leave it unset.
17
+ */
18
+ export const defaultAgents = {
19
+ claude: { run: runClaude },
20
+ codex: { run: runCodex },
21
+ agy: { run: runAgy },
22
+ opencode: { run: runOpenCode },
23
+ copilot: { run: runCopilot },
24
+ };
25
+
26
+ export const supportedAgents = new Set([...Object.keys(defaultAgents), "antigravity"]);
27
+
28
+ export async function runAgent(state, prompt, options = {}, agents = defaultAgents) {
29
+ const kind = normalizeAgent(state.kind);
30
+ const adapter = agents[kind];
31
+
32
+ if (!adapter || typeof adapter.run !== "function") {
33
+ throw new Error(`Unsupported agent: ${state.kind}`);
34
+ }
35
+
36
+ return adapter.run(state, prompt, options);
37
+ }