@andromarces/agent-loops 0.2.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +194 -114
  2. package/docs/orchestrator-instructions.md +25 -22
  3. package/package.json +2 -2
  4. package/src/agents/agy.mjs +2 -11
  5. package/src/agents/codex.mjs +3 -20
  6. package/src/agents/copilot.mjs +8 -16
  7. package/src/agents/opencode.mjs +2 -7
  8. package/src/agents/shared.mjs +29 -0
  9. package/src/cli.mjs +46 -25
  10. package/src/entrypoints/copilot.mjs +6 -1
  11. package/src/hook/antigravity-parent-guard.mjs +28 -0
  12. package/src/hook/copilot-parent-guard.mjs +5 -30
  13. package/src/hook/decision.mjs +59 -6
  14. package/src/hook/opencode-plugin.mjs +92 -0
  15. package/src/hook/parent-guard.mjs +8 -33
  16. package/src/install/commands.mjs +268 -0
  17. package/src/install/fsutil.mjs +159 -0
  18. package/src/install/harnesses.mjs +170 -0
  19. package/src/install/installer.mjs +688 -0
  20. package/src/install/manifest.mjs +222 -0
  21. package/src/install/settings.mjs +217 -0
  22. package/src/install/templates/antigravity/agent-loop-antigravity-parent-guard.mjs +14 -0
  23. package/src/install/templates/antigravity/hooks.json +16 -0
  24. package/src/install/templates/antigravity/skills/agent-loop/SKILL.md +24 -0
  25. package/src/install/templates/claude/skills/agent-loop/SKILL.md +33 -0
  26. package/src/install/templates/codex/skills/agent-loop/SKILL.md +30 -0
  27. package/src/install/templates/codex/skills/agent-loop/agents/openai.yaml +2 -0
  28. package/src/install/templates/copilot/hooks/parent-guard.json +15 -0
  29. package/src/install/templates/opencode/plugins/parent-guard.ts +11 -0
  30. package/src/lib/args.mjs +23 -0
  31. package/src/lib/hash.mjs +9 -0
  32. package/src/lib/log.mjs +18 -3
  33. package/src/lib/process-ancestry.mjs +104 -0
  34. package/src/lib/runstate.mjs +133 -38
  35. package/src/lib/snapshot.mjs +3 -7
  36. package/src/role.mjs +71 -87
  37. package/src/runtime.mjs +9 -9
@@ -1,6 +1,7 @@
1
1
  import { parseJsonLines } from "../lib/json.mjs";
2
2
  import { exec } from "../lib/exec.mjs";
3
3
  import { logInfo } from "../lib/log.mjs";
4
+ import { resumeMismatchError } from "./shared.mjs";
4
5
 
5
6
  // The built-in plan agent can launch explore and general subagents through the `subagent`
6
7
  // action. They inherit the session model, so a read-only turn spends the role model budget
@@ -59,13 +60,7 @@ export async function runOpenCode(state, prompt, options = {}) {
59
60
  }
60
61
 
61
62
  if (state.sessionId && state.sessionId !== sessionId) {
62
- throw new Error(
63
- [
64
- `opencode did not resume the expected session.`,
65
- `Expected: ${state.sessionId}`,
66
- `Received: ${sessionId}`,
67
- ].join("\n"),
68
- );
63
+ throw resumeMismatchError("opencode", "session", state.sessionId, sessionId);
69
64
  }
70
65
 
71
66
  state.sessionId = sessionId;
@@ -0,0 +1,29 @@
1
+ /**
2
+ * Sets `state.usage.mainLoop` from one turn's usage, or removes `state.usage` when
3
+ * the CLI reported none, so a turn that omits usage leaves no stale value behind.
4
+ * Any truthy value counts as usage; a caller with a narrower rule passes a filtered value.
5
+ */
6
+ export function setMainLoopUsage(state, usage) {
7
+ if (usage) {
8
+ state.usage = { mainLoop: usage };
9
+ } else {
10
+ delete state.usage;
11
+ }
12
+ }
13
+
14
+ /**
15
+ * Builds the error for a resumed CLI that returned a different session id.
16
+ * @param {string} agent display name of the adapter, for example `Codex`
17
+ * @param {string} idLabel name of the id in that CLI, for example `thread` or `session`
18
+ * @param {string} expected session id the caller asked the CLI to resume
19
+ * @param {string} received session id the CLI returned
20
+ */
21
+ export function resumeMismatchError(agent, idLabel, expected, received) {
22
+ return new Error(
23
+ [
24
+ `${agent} did not resume the expected ${idLabel}.`,
25
+ `Expected: ${expected}`,
26
+ `Received: ${received}`,
27
+ ].join("\n"),
28
+ );
29
+ }
package/src/cli.mjs CHANGED
@@ -4,25 +4,34 @@ import { writeFile } from "node:fs/promises";
4
4
  import { resolve } from "node:path";
5
5
  import { defaultAgents, normalizeAgent, supportedAgents } from "./agents/index.mjs";
6
6
  import {
7
+ DEFAULT_MAX_STEPS,
8
+ DEFAULT_TIMEOUT,
9
+ ROLE_KINDS as ROLES,
7
10
  assertOpenCodeOptions,
8
11
  readArgValue,
9
12
  readNonNegativeInt,
10
13
  readPositiveInt,
14
+ roleFlags,
11
15
  } from "./lib/args.mjs";
12
16
  import { isEntryPoint } from "./lib/entrypoint.mjs";
17
+ import {
18
+ runHarnessCheckCommand,
19
+ runInstallCommand,
20
+ runUninstallCommand,
21
+ } from "./install/commands.mjs";
13
22
  import { setVerbose } from "./lib/log.mjs";
14
23
  import { assertGitWorkTree } from "./lib/snapshot.mjs";
15
24
  import { runLoop } from "./runtime.mjs";
16
25
  import { main as runRoleMain } from "./role.mjs";
17
26
 
18
- const ROLES = ["orchestrator", "worker", "reviewer"];
27
+ const ROLE_FLAGS = roleFlags(ROLES);
19
28
 
20
29
  export function parseArgs(argv) {
21
30
  const options = {
22
31
  cwd: process.cwd(),
23
32
  task: null,
24
- maxSteps: 20,
25
- timeout: 3600,
33
+ maxSteps: DEFAULT_MAX_STEPS,
34
+ timeout: DEFAULT_TIMEOUT,
26
35
  transcript: null,
27
36
  verbose: false,
28
37
  };
@@ -37,21 +46,10 @@ export function parseArgs(argv) {
37
46
  for (let i = 0; i < argv.length; i++) {
38
47
  const arg = argv[i];
39
48
 
40
- let matched = false;
41
- for (const role of ROLES) {
42
- if (arg === `--${role}`) {
43
- options[role] = readValue(arg, ++i);
44
- matched = true;
45
- } else if (arg === `--${role}-model`) {
46
- options[`${role}Model`] = readValue(arg, ++i);
47
- matched = true;
48
- } else if (arg === `--${role}-effort`) {
49
- options[`${role}Effort`] = readValue(arg, ++i);
50
- matched = true;
51
- }
52
- if (matched) break;
49
+ if (Object.hasOwn(ROLE_FLAGS, arg)) {
50
+ options[ROLE_FLAGS[arg]] = readValue(arg, ++i);
51
+ continue;
53
52
  }
54
- if (matched) continue;
55
53
 
56
54
  switch (arg) {
57
55
  case "--cwd":
@@ -133,6 +131,15 @@ Subcommands:
133
131
  agent-loop role Run a single worker or reviewer turn, or finish/abort
134
132
  a run, from a lifecycle state file (see below). One JSON
135
133
  object on stdout; logs on stderr.
134
+ agent-loop install Install harness entry points and parent guards at user
135
+ scope. Interactive, or --harness <list> --yes.
136
+ agent-loop uninstall Remove the installed entry points and guards. Restores
137
+ files that install changed.
138
+ agent-loop harness-check Exit 0 only when the nearest harness process above the
139
+ shell matches the named harness, 3 when another harness
140
+ is nearest, and 1 when the check cannot run or finds no
141
+ harness ancestor. Used by the Claude, Codex, and
142
+ Antigravity skills.
136
143
 
137
144
  Role operations:
138
145
 
@@ -144,12 +151,16 @@ Role flags:
144
151
 
145
152
  --role worker|reviewer Role to dispatch. Required for dispatch.
146
153
  --cwd <directory> Target work tree. Defaults to the current directory.
147
- --task / --mode / --parent-session / --worker* / --reviewer* / --max-steps / --timeout
154
+ --parent-session <id> Required on the init call. The harness session id the
155
+ parent-edit guard matches; later calls reject a changed
156
+ value. The headless form (no subcommand) is the explicit
157
+ unguarded path.
158
+ --task / --mode / --worker* / --reviewer* / --max-steps / --timeout
148
159
  First (init) call only. Later calls read these from the
149
160
  state file and reject any attempt to change them.
150
161
  --prompt-file <path> Prompt source. Default is stdin.
151
162
  --transcript <file> Append invocation and result events (JSON lines).
152
- --resume-interrupted Explicitly continue after an uncertain previous turn.
163
+ --resume-interrupted Explicitly continue after an uncertain previous turn.
153
164
 
154
165
  Options:
155
166
 
@@ -183,12 +194,7 @@ Environment:
183
194
 
184
195
  Agents:
185
196
 
186
- claude
187
- codex
188
- agy
189
- antigravity
190
- opencode
191
- copilot
197
+ ${[...supportedAgents].join("\n ")}
192
198
  `.trim(),
193
199
  );
194
200
  }
@@ -209,6 +215,21 @@ export async function main(argv = process.argv.slice(2), agents = defaultAgents)
209
215
  return;
210
216
  }
211
217
 
218
+ if (argv[0] === "install") {
219
+ await runInstallCommand(argv.slice(1));
220
+ return;
221
+ }
222
+
223
+ if (argv[0] === "uninstall") {
224
+ await runUninstallCommand(argv.slice(1));
225
+ return;
226
+ }
227
+
228
+ if (argv[0] === "harness-check") {
229
+ await runHarnessCheckCommand(argv.slice(1));
230
+ return;
231
+ }
232
+
212
233
  let options;
213
234
  try {
214
235
  options = parseArgs(argv);
@@ -13,13 +13,18 @@ import { isEntryPoint } from "../lib/entrypoint.mjs";
13
13
  const INSTRUCTIONS_DIR = fileURLToPath(new URL("../../docs", import.meta.url));
14
14
  const INSTRUCTIONS_PATH = join(INSTRUCTIONS_DIR, "orchestrator-instructions.md");
15
15
 
16
+ // The prompt is one line. execa rejects CR or LF in an argument on Windows when
17
+ // it spawns a batch shim through cmd.exe, and npm installs the copilot CLI as a
18
+ // .cmd shim on Windows.
16
19
  const INSTRUCTIONS = (sessionId, task) =>
17
20
  [
18
21
  `Read \`${INSTRUCTIONS_PATH}\` and follow it for this request.`,
19
22
  "The task and role settings are:",
20
23
  task,
21
24
  `This Copilot CLI session id is \`${sessionId}\`. Pass it as \`--parent-session\` on the init dispatch call.`,
22
- ].join("\n\n");
25
+ ]
26
+ .join(" ")
27
+ .replace(/[\r\n]+/g, " ");
23
28
 
24
29
  export function buildCopilotInvocation(task, sessionId) {
25
30
  const normalizedTask = String(task ?? "").trim();
@@ -0,0 +1,28 @@
1
+ // Antigravity CLI PreToolUse hook for the parent-edit guard (#88). Antigravity
2
+ // names the session `conversationId` and the tool `toolCall.name`. It blocks a
3
+ // tool call when a hook prints `{}`, prints an empty decision, or exits
4
+ // non-zero, so every allow path here prints nothing and exits 0: the normal
5
+ // permission flow stays intact. One deny object prints only for a guarded tool
6
+ // call from the registered parent while its run is non-terminal. Fail-open by
7
+ // design: an unparseable payload, an absent record, a terminal lifecycle, and a
8
+ // lookup error all deny nothing.
9
+ import { runParentGuard } from "./decision.mjs";
10
+
11
+ // The guarded tools: every registered file-edit tool, plus the subagent start
12
+ // and message tools. `manage_subagents` stays allowed, so the parent can still
13
+ // list and terminate a child. The `hooks.json` matcher names the same set.
14
+ const ANTIGRAVITY_GUARDED_TOOLS = new Set([
15
+ "write_to_file",
16
+ "replace_file_content",
17
+ "multi_replace_file_content",
18
+ "sed_file",
19
+ "notebook_edit",
20
+ "invoke_subagent",
21
+ "send_message",
22
+ ]);
23
+
24
+ await runParentGuard((reason) => ({ decision: "deny", reason }), {
25
+ readSessionId: (input) =>
26
+ typeof input?.conversationId === "string" ? input.conversationId : null,
27
+ isGuarded: (input) => ANTIGRAVITY_GUARDED_TOOLS.has(input?.toolCall?.name),
28
+ });
@@ -2,34 +2,9 @@
2
2
  // PascalCase event payload uses `session_id`, and its command-hook decision is
3
3
  // a flat permissionDecision object rather than Claude's hookSpecificOutput.
4
4
  // Fail-open by design: malformed input and lookup failures deny nothing.
5
- import { decideParentGuard } from "./decision.mjs";
5
+ import { runParentGuard } from "./decision.mjs";
6
6
 
7
- async function readHookInput() {
8
- const chunks = [];
9
- for await (const chunk of process.stdin) {
10
- chunks.push(chunk);
11
- }
12
- try {
13
- return JSON.parse(Buffer.concat(chunks).toString("utf8"));
14
- } catch {
15
- return null;
16
- }
17
- }
18
-
19
- const input = await readHookInput();
20
- const sessionId = typeof input?.session_id === "string" ? input.session_id : null;
21
- if (sessionId) {
22
- try {
23
- const verdict = await decideParentGuard(sessionId);
24
- if (verdict.decision === "deny") {
25
- console.log(
26
- JSON.stringify({
27
- permissionDecision: "deny",
28
- permissionDecisionReason: verdict.reason,
29
- }),
30
- );
31
- }
32
- } catch {
33
- // A failed lookup denies nothing: the guard never blocks on its own error.
34
- }
35
- }
7
+ await runParentGuard((reason) => ({
8
+ permissionDecision: "deny",
9
+ permissionDecisionReason: reason,
10
+ }));
@@ -1,9 +1,12 @@
1
- // Decision logic for the #57 parent guard. The Claude Code PreToolUse hook
2
- // denies file-edit tools only when the hook session id matches `parentSession`
3
- // in the state file registered for that session; every other session, every
4
- // terminal lifecycle, and every absent or corrupt record allows the call.
5
- // Fail-open by design: the guard is optional defense for the prompt-only
6
- // parent rule, so unknown records never block a tool call.
1
+ // Decision logic and the shared command-hook flow for the #57 parent guard.
2
+ // `decideParentGuard` denies file-edit tools only when the hook session id
3
+ // matches `parentSession` in the state file registered for that session; every
4
+ // other session, every terminal lifecycle, and every absent or corrupt record
5
+ // allows the call. `runParentGuard` drives that decision from the stdin payload
6
+ // for the Claude Code, Codex, Copilot, and Antigravity command hooks, each of
7
+ // which supplies its own input mapping and deny shape. Fail-open by design: the
8
+ // guard is optional defense for the prompt-only parent rule, so unknown records
9
+ // never block a tool call.
7
10
  import { TERMINAL_LIFECYCLES, readStateForSession } from "../lib/runstate.mjs";
8
11
 
9
12
  export const GUARD_DENY_REASON =
@@ -25,3 +28,53 @@ export async function decideParentGuard(sessionId, { lookup = readStateForSessio
25
28
  }
26
29
  return { decision: "deny", reason: GUARD_DENY_REASON };
27
30
  }
31
+
32
+ async function readHookInput() {
33
+ const chunks = [];
34
+ for await (const chunk of process.stdin) {
35
+ chunks.push(chunk);
36
+ }
37
+ try {
38
+ return JSON.parse(Buffer.concat(chunks).toString("utf8"));
39
+ } catch {
40
+ return null;
41
+ }
42
+ }
43
+
44
+ /**
45
+ * Runs the shared command-hook flow: read the hook JSON from stdin, extract the
46
+ * harness's session id, evaluate the guard, and print one deny payload from
47
+ * `formatDeny`. Fails open: a missing session id, unparseable input, a tool the
48
+ * harness does not guard, and a lookup error all exit without output, so the
49
+ * normal permission flow stays intact.
50
+ * @param {(reason: string) => object} formatDeny builds the harness's deny payload
51
+ * @param {{
52
+ * readSessionId?: (input: unknown) => string | null,
53
+ * isGuarded?: (input: unknown) => boolean,
54
+ * }} [options] maps the harness input; defaults to the Claude/Codex/Copilot
55
+ * `session_id` field and every tool
56
+ */
57
+ export async function runParentGuard(
58
+ formatDeny,
59
+ {
60
+ readSessionId = (input) => (typeof input?.session_id === "string" ? input.session_id : null),
61
+ isGuarded = () => true,
62
+ } = {},
63
+ ) {
64
+ const input = await readHookInput();
65
+ if (!input || !isGuarded(input)) {
66
+ return;
67
+ }
68
+ const sessionId = readSessionId(input);
69
+ if (!sessionId) {
70
+ return;
71
+ }
72
+ try {
73
+ const verdict = await decideParentGuard(sessionId);
74
+ if (verdict.decision === "deny") {
75
+ console.log(JSON.stringify(formatDeny(verdict.reason)));
76
+ }
77
+ } catch {
78
+ // A failed lookup denies nothing: the guard never blocks on its own error.
79
+ }
80
+ }
@@ -0,0 +1,92 @@
1
+ // OpenCode parent-guard plugin factory (#75, packaged by #139). The shipped
2
+ // entry point is the rendered template at
3
+ // `src/install/templates/opencode/plugins/parent-guard.ts`, which imports this
4
+ // factory and the shared decision logic by absolute URL. Keeping the behavior
5
+ // here lets tests drive the same object the rendered plugin exports.
6
+ //
7
+ // OpenCode ships no session placeholder for Markdown command templates, so the
8
+ // plugin channel supplies the session id: a plugin command reads
9
+ // `CommandInvocation.sessionID` and passes it to the parent, and a permission
10
+ // hook reads `PermissionEvaluation.sessionID` and denies file edits from the
11
+ // registered parent while its run is non-terminal. Both paths reuse
12
+ // `decideParentGuard`, so the OpenCode guard matches it exactly. Fail-open by
13
+ // design: every other session, every terminal lifecycle, and every absent or
14
+ // corrupt record allows the call.
15
+ //
16
+ // The shared Claude and Codex skills set `metadata.opencode/autoinvoke: false`,
17
+ // so OpenCode drops them from the model's skill list and the model cannot
18
+ // auto-invoke one for an /agent-loop request. This command is then the only
19
+ // `/agent-loop` entry point that carries the OpenCode session id (#150).
20
+ //
21
+ // The guard resolves the runs root from this server process's environment, so
22
+ // an out-of-process `AGENT_LOOP_RUNS_ROOT` override (test-only) would desync
23
+ // it from the `agent-loop` CLI that wrote the index.
24
+
25
+ // The permission actions that carry a file edit. A set, not a single constant,
26
+ // so a further edit tool is one entry. A live probe against OpenCode
27
+ // v0.0.0-dev-19933 showed the built-in `edit`, `write`, and `apply_patch` tools
28
+ // all raise the `edit` action, so the set covers every built-in file-edit tool.
29
+ // `shell` is a different action and stays allowed. A tool served by an MCP
30
+ // server raises its own action name and passes the guard.
31
+ export const EDIT_ACTIONS = new Set(["edit"]);
32
+
33
+ /**
34
+ * Builds the OpenCode plugin.
35
+ * @param {{
36
+ * decideParentGuard: (sessionId: string) => Promise<{ decision: string, reason?: string }>,
37
+ * instructionsPath: string,
38
+ * }} deps absolute path to the installed orchestrator instructions
39
+ */
40
+ export function createParentGuardPlugin({ decideParentGuard, instructionsPath }) {
41
+ const COMMAND_NAME = "agent-loop";
42
+
43
+ const INSTRUCTIONS = (sessionID, task) =>
44
+ [
45
+ `Read \`${instructionsPath}\` and follow it for this request.`,
46
+ "The task and role settings are:",
47
+ task,
48
+ `This OpenCode session id is \`${sessionID}\`. Pass it as \`--parent-session\` on the init dispatch call.`,
49
+ ].join("\n\n");
50
+
51
+ return {
52
+ id: "agent-loop.parent-guard",
53
+ async setup(ctx) {
54
+ // Registers the entry point. The stored Markdown template cannot receive
55
+ // the parent session id, so the plugin command injects it into the
56
+ // orchestrator prompt instead.
57
+ await ctx.command.transform((editor) => {
58
+ editor.add({
59
+ name: COMMAND_NAME,
60
+ description:
61
+ "Run a delegated agent-loop role orchestration through the agent-loop CLI (rules in the installed orchestrator instructions).",
62
+ execute: async ({ sessionID, prompt, delivery }) => {
63
+ await ctx.session.prompt({
64
+ ...prompt,
65
+ sessionID,
66
+ text: INSTRUCTIONS(sessionID, prompt.text),
67
+ delivery,
68
+ });
69
+ },
70
+ });
71
+ });
72
+
73
+ // Denies a file edit from the registered parent while the run is
74
+ // non-terminal. A failed lookup denies nothing.
75
+ await ctx.permission.hook("evaluate", async (event) => {
76
+ if (!EDIT_ACTIONS.has(event.action) || typeof event.sessionID !== "string") {
77
+ return;
78
+ }
79
+ let verdict;
80
+ try {
81
+ verdict = await decideParentGuard(event.sessionID);
82
+ } catch {
83
+ return;
84
+ }
85
+ if (verdict.decision === "deny") {
86
+ event.effect = "deny";
87
+ event.message = verdict.reason;
88
+ }
89
+ });
90
+ },
91
+ };
92
+ }
@@ -4,37 +4,12 @@
4
4
  // non-terminal run. Every other outcome exits 0 with no output — unparseable
5
5
  // input, a malformed session id, or an unexpected lookup error all leave the
6
6
  // normal permission flow intact; the hook never leaks an error to the session.
7
- import { decideParentGuard } from "./decision.mjs";
7
+ import { runParentGuard } from "./decision.mjs";
8
8
 
9
- async function readHookInput() {
10
- const chunks = [];
11
- for await (const chunk of process.stdin) {
12
- chunks.push(chunk);
13
- }
14
- try {
15
- return JSON.parse(Buffer.concat(chunks).toString("utf8"));
16
- } catch {
17
- return null;
18
- }
19
- }
20
-
21
- const input = await readHookInput();
22
- const sessionId = typeof input?.session_id === "string" ? input.session_id : null;
23
- if (sessionId) {
24
- try {
25
- const verdict = await decideParentGuard(sessionId);
26
- if (verdict.decision === "deny") {
27
- console.log(
28
- JSON.stringify({
29
- hookSpecificOutput: {
30
- hookEventName: "PreToolUse",
31
- permissionDecision: "deny",
32
- permissionDecisionReason: verdict.reason,
33
- },
34
- }),
35
- );
36
- }
37
- } catch {
38
- // A failed lookup denies nothing: the guard never blocks on its own error.
39
- }
40
- }
9
+ await runParentGuard((reason) => ({
10
+ hookSpecificOutput: {
11
+ hookEventName: "PreToolUse",
12
+ permissionDecision: "deny",
13
+ permissionDecisionReason: reason,
14
+ },
15
+ }));