@andromarces/agent-loops 0.2.1 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +194 -114
  2. package/docs/orchestrator-instructions.md +25 -22
  3. package/package.json +2 -2
  4. package/src/agents/agy.mjs +2 -11
  5. package/src/agents/codex.mjs +3 -20
  6. package/src/agents/copilot.mjs +8 -16
  7. package/src/agents/opencode.mjs +2 -7
  8. package/src/agents/shared.mjs +29 -0
  9. package/src/cli.mjs +46 -25
  10. package/src/entrypoints/copilot.mjs +6 -1
  11. package/src/hook/antigravity-parent-guard.mjs +28 -0
  12. package/src/hook/copilot-parent-guard.mjs +5 -30
  13. package/src/hook/decision.mjs +59 -6
  14. package/src/hook/opencode-plugin.mjs +92 -0
  15. package/src/hook/parent-guard.mjs +8 -33
  16. package/src/install/commands.mjs +268 -0
  17. package/src/install/fsutil.mjs +159 -0
  18. package/src/install/harnesses.mjs +170 -0
  19. package/src/install/installer.mjs +688 -0
  20. package/src/install/manifest.mjs +222 -0
  21. package/src/install/settings.mjs +217 -0
  22. package/src/install/templates/antigravity/agent-loop-antigravity-parent-guard.mjs +14 -0
  23. package/src/install/templates/antigravity/hooks.json +16 -0
  24. package/src/install/templates/antigravity/skills/agent-loop/SKILL.md +24 -0
  25. package/src/install/templates/claude/skills/agent-loop/SKILL.md +33 -0
  26. package/src/install/templates/codex/skills/agent-loop/SKILL.md +30 -0
  27. package/src/install/templates/codex/skills/agent-loop/agents/openai.yaml +2 -0
  28. package/src/install/templates/copilot/hooks/parent-guard.json +15 -0
  29. package/src/install/templates/opencode/plugins/parent-guard.ts +11 -0
  30. package/src/lib/args.mjs +23 -0
  31. package/src/lib/hash.mjs +9 -0
  32. package/src/lib/log.mjs +18 -3
  33. package/src/lib/process-ancestry.mjs +104 -0
  34. package/src/lib/runstate.mjs +133 -38
  35. package/src/lib/snapshot.mjs +3 -7
  36. package/src/role.mjs +71 -87
  37. package/src/runtime.mjs +9 -9
package/README.md CHANGED
@@ -79,6 +79,64 @@ npm install -g @andromarces/agent-loops
79
79
  pnpm add -g @andromarces/agent-loops
80
80
  ```
81
81
 
82
+ Set up the harness entry points and parent guards at user scope:
83
+
84
+ ```bash
85
+ agent-loop install
86
+ ```
87
+
88
+ `install` detects the harness CLIs on `PATH`, preselects them, and writes the
89
+ entry point and guard for each selected harness. It is interactive; pass
90
+ `--harness <list> --yes` for scripts, and `--dry-run` to print the planned
91
+ writes without changing anything. A second run with the same package makes no
92
+ change. `agent-loop uninstall` removes only the files and settings entries the
93
+ install recorded, and restores a file that install changed when the file is
94
+ otherwise unchanged. Install applies a harness guard before its entry point, and
95
+ if a write fails part way through it records the writes that completed, so
96
+ `uninstall` still removes or restores them. It leaves a target in place when it
97
+ cannot remove or restore it safely; see [Uninstall skips](#uninstall-skips).
98
+
99
+ `install` must run from a global install or a linked clone. It writes the package
100
+ location as an absolute path into every entry point and guard, and `npx` and
101
+ `pnpm dlx` place the package in a cache directory that npm or pnpm can delete.
102
+ When it detects its own package root inside that cache, `install` refuses with a
103
+ message that asks for a global install first, so it never writes a path that can
104
+ disappear. `uninstall` reads only the manifest and the harness files, so it still
105
+ removes them after the package is gone.
106
+
107
+ ```bash
108
+ agent-loop install --harness claude,codex --yes
109
+ agent-loop uninstall
110
+ ```
111
+
112
+ ### Uninstall skips
113
+
114
+ `agent-loop uninstall` prints one line per target as
115
+ `<harness>: <action> <path> (<detail>)`. A target that it cannot remove or
116
+ restore safely reports `skip`, stays on disk, and loses its manifest record:
117
+ uninstall deletes the harness record, and removes the manifest when no harness
118
+ remains. Uninstall never retries a skipped target, and a backup file can remain
119
+ on disk with no record. Recover a skipped target by hand.
120
+
121
+ `install` writes a backup at `<file>.agent-loops-backup` when the target existed
122
+ before install and install changed it, and uninstall reads that file to restore
123
+ the pre-install bytes. Install writes no backup when the target did not exist
124
+ before install, and none when it already held the installed bytes.
125
+
126
+ | Reported detail | Target | What stays on disk | Manual recovery |
127
+ | -------------------------------------------------- | -------- | -------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
128
+ | `owned file changed since install; left unchanged` | file | The file, with your later edits, and a backup when present | Restore `<file>.agent-loops-backup` over the file when that backup exists and you want the pre-install content, or delete the file when you do not want it. Delete the backup by hand when it stays behind. |
129
+ | `backup missing; left unchanged` | file | The file, unchanged since install, with no backup | Install wrote no backup when the file already held the installed bytes, so the file can be your original and needs no change. Otherwise restore the original from version control or another copy, or delete the file when you do not want the installed entry point. |
130
+ | `settings do not parse; left unchanged` | settings | The settings file, still invalid, and a backup when present | The record is gone, so no later `agent-loop uninstall` retries it. Repair the syntax and remove the recorded entry by hand, or restore `<file>.agent-loops-backup` when that backup exists and your later edits can be discarded. Then delete the backup. |
131
+ | `recorded entry not found; left unchanged` | settings | The settings file, with no recorded entry, and a backup when present | The recorded entry is already gone. Remove another agent-loop entry by hand when one is present, then delete the backup when it exists. |
132
+ | `backup missing; left unchanged` | settings | The settings file, still holding the recorded entry | Remove the recorded entry by hand, or restore the original from version control or another copy. The backup is gone, so none remains to delete. |
133
+
134
+ When a settings file changed after install but still parses and holds the
135
+ recorded entry, uninstall reports `remove-entry` instead: it removes only that
136
+ entry, keeps your edits, and leaves the result not byte-identical. The backup at
137
+ `<file>.agent-loops-backup` stays behind, so delete it by hand when you no
138
+ longer need the pre-install copy.
139
+
82
140
  Run it without installing:
83
141
 
84
142
  ```bash
@@ -86,6 +144,9 @@ npx @andromarces/agent-loops --help
86
144
  pnpm dlx @andromarces/agent-loops --help
87
145
  ```
88
146
 
147
+ `--help` writes no config, so it runs from the cache. `install` writes config
148
+ that points at the package, so it refuses from the cache; see [Install](#install).
149
+
89
150
  Install from a Git URL instead of the registry:
90
151
 
91
152
  ```bash
@@ -102,28 +163,39 @@ The `bin` script keeps its `#!/usr/bin/env node` shebang and executable bit, so
102
163
 
103
164
  ### From a clone (development)
104
165
 
105
- Development uses pnpm and the repository Git hooks:
166
+ Development uses pnpm and the repository Git hooks. User-scope integrations
167
+ point at the package location, so a clone registers its own copy with `npm link`
168
+ before `agent-loop install`:
106
169
 
107
170
  ```bash
108
171
  git clone <repository-url>
109
172
  cd agent-loops
110
173
  pnpm install
111
- pnpm agent-loop role ...
174
+ npm link
175
+ agent-loop install
112
176
  ```
113
177
 
114
- `pnpm agent-loop` runs the CLI entry from the repository root. To expose the `agent-loop` command on `PATH`, add the pnpm global bin directory to `PATH`, then register the `bin` field globally from the repository root:
115
-
116
- ```bash
117
- pnpm setup # restart the shell afterwards
118
- pnpm add -g .
119
- ```
178
+ `npm link` writes its bin shim to the global prefix bin directory, so that
179
+ directory must be on `PATH` for `agent-loop` to resolve. It is the
180
+ `npm prefix -g` directory on Windows and `$(npm prefix -g)/bin` on macOS and
181
+ Linux.
120
182
 
121
- pnpm v11 removed `pnpm link --global` and keeps global bins under `PNPM_HOME`; `pnpm add -g .` fails with `ERR_PNPM_GLOBAL_BIN_DIR_NOT_IN_PATH` until `pnpm setup` puts that directory on `PATH`. Without a global install, call the CLI entry directly and quote the repository path so a path with spaces works:
183
+ Entries rendered from a clone point at that clone. After the clone moves, run
184
+ `npm link --force` from the new location, then `agent-loop install` again. A
185
+ plain `npm link` fails with `EEXIST` on Windows when a shim already exists;
186
+ `--force` overwrites the shim.
187
+ `pnpm link` is not a supported path: pnpm 12 `link` has no global mode. Without a
188
+ link, call the CLI entry directly and quote the repository path so a path with
189
+ spaces works:
122
190
 
123
191
  ```bash
124
- node "<repo>/src/cli.mjs" role ...
192
+ node "<repo>/src/cli.mjs" install --harness claude --yes
125
193
  ```
126
194
 
195
+ `pnpm agent-loop` still runs the CLI entry from the repository root. The
196
+ repository Git hooks run formatting and linting on commit and push; they need
197
+ `pnpm install` to have completed, because its `prepare` script installs husky.
198
+
127
199
  ## Usage
128
200
 
129
201
  ```bash
@@ -183,11 +255,16 @@ agent-loop role dispatch --role reviewer --cwd /path/to/work-tree --prompt-file
183
255
  Operations: `dispatch` (default), `finish`, `abort`.
184
256
 
185
257
  - The run state lives at a fixed path derived from the resolved `--cwd` (`<os tmpdir>/agent-loops/runs/<sha256 of cwd, shortened>/state.json`, with `state.lock` beside it). There is no `--state` flag; `AGENT_LOOP_RUNS_ROOT` overrides the root for tests only.
258
+ - The init call requires `--parent-session`: the CLI refuses an init without it,
259
+ and refuses an unexpanded placeholder such as `${CLAUDE_SESSION_ID}` or
260
+ `%CODEX_THREAD_ID%`, before any state is written. A run through `role` is
261
+ therefore always guarded; the headless `agent-loop` command is the explicit
262
+ unguarded path.
186
263
  - The init call writes a session index entry at `<root>/sessions/<parent-session>` pointing at the state file, so a parent guard hook (#57) can look the run up by session id even when `--cwd` is a different work tree. A later init call from the same session overwrites the entry.
187
264
  - Prompts come from stdin by default, or `--prompt-file`. `finish` reads the five-key summary as JSON on stdin; `abort` takes `--reason`.
188
265
  - The state file records `task`, `mode`, `cwd`, `parentSession`, `maxSteps`, `timeout`, `stepsUsed`, `lifecycle`, `roles.{worker,reviewer}.{kind,model,effort,sessionId}` (`roles.worker` is null in `review-only`), `lastDispatch`, and `lastResult`, plus `summary` or `reason` when terminal and `resumeDecision` when a maintainer resumed an interrupted run. Updates are atomic (temp file plus rename); exclusive access uses `state.lock` with a stale-lock check on the owner pid.
189
266
  - Lifecycle values: `active`, `dispatched`, `interrupted`, `halted`, `finished`, `aborted` (terminal: `halted`, `finished`, `aborted`). A turn interrupted between the CLI start and the state write leaves `dispatched` with a dead lock owner; the first call after the crash marks it `interrupted`, exits non-zero, and never repeats the turn, even with `--resume-interrupted`. From `interrupted`, only `abort` or an explicit `dispatch --resume-interrupted` is accepted.
190
- - The lock is fail-closed on ambiguity: a contender that finds a lock it cannot read (created moments ago, content not yet written) exits non-zero and never removes it; only an unparseable lock older than a grace window, or one whose recorded pid is dead, is treated as stale.
267
+ - The lock is fail-closed on ambiguity: a contender that finds a lock it cannot read (created moments ago, content not yet written) exits non-zero and never removes it; only an unparseable lock older than a grace window, or one whose recorded pid is dead, is treated as stale. Lock creation is an atomic hard link, with an exclusive-create fallback on filesystems that have no hard links (FAT/exFAT, some network mounts); a lock temp file left by a crashed process is pruned on the next acquisition.
191
268
  - The reviewer turn runs under the same `withMutationCheck` as the headless loop: a detected mutation or snapshot error is fatal, keeps the charged step, and sets `halted`. No further dispatch is possible; the next run needs a new init call, which archives the halted file as `state.<timestamp>.json`.
192
269
  - `mode: review-only` rejects `--role worker` as a hard guard and does not require `--worker` at init. `finish` is completion of the requested work, not code acceptance: it is accepted from `active` in any mode, and the five-key summary carries the reviewer verdict and unresolved findings.
193
270
  - The reviewer is required to end with one explicit `Verdict:` line (`accept` or `reject`, parsed case-insensitively) inside its closing block. The verdict word alone, the word closed by a sentence period (`reject.`), or the word followed by a separator and a trailing clause (`reject — the state does not pass`) parses to that word, unless the clause names either verdict as a whole word. Any other malformed value, including a missing line or a line that names both verdicts, yields `verdict: unknown`; process success never implies acceptance.
@@ -215,113 +292,114 @@ includes it rather than copying it. Role activation never goes into `AGENTS.md`
215
292
  or `CLAUDE.md`, because dispatched children read those files; activation
216
293
  happens only through explicit invocation.
217
294
 
218
- | Harness | Entry point | Invocation |
219
- | --------------- | ------------------------------------ | --------------------------------------------- |
220
- | Claude Code | `.claude/skills/agent-loop/SKILL.md` | `/agent-loop <task and role settings>` |
221
- | OpenCode | `.opencode/plugins/parent-guard.ts` | `/agent-loop <task and role settings>` |
222
- | Codex CLI | `.agents/skills/agent-loop/SKILL.md` | `$agent-loop <task and role settings>` |
223
- | Copilot CLI | `src/entrypoints/copilot.mjs` | `agent-loop-copilot <task and role settings>` |
224
- | Antigravity CLI | universal fallback (below) | first prompt references the file |
225
-
295
+ | Harness | Installed entry point | Invocation |
296
+ | --------------- | ------------------------------------------------------ | --------------------------------------------- |
297
+ | Claude Code | `~/.claude/skills/agent-loop/SKILL.md` | `/agent-loop <task and role settings>` |
298
+ | OpenCode | `~/.config/opencode/plugins/parent-guard.ts` | `/agent-loop <task and role settings>` |
299
+ | Codex CLI | `~/.agents/skills/agent-loop/SKILL.md` | `$agent-loop <task and role settings>` |
300
+ | Copilot CLI | `agent-loop-copilot` (packaged bin) | `agent-loop-copilot <task and role settings>` |
301
+ | Antigravity CLI | `~/.gemini/antigravity-cli/skills/agent-loop/SKILL.md` | `/agent-loop <task and role settings>` |
302
+
303
+ `agent-loop install` writes these files from the templates under
304
+ `src/install/templates/`, and each rendered entry point points at the installed
305
+ `docs/orchestrator-instructions.md` by absolute path, so it works outside a
306
+ clone. Each skill renders the CLI by absolute path and runs every `agent-loop`
307
+ command in the instructions with that resolved invocation, so a Git Bash session
308
+ that cannot resolve the `agent-loop` command still starts a guarded run (#151).
309
+ `harness-check` exits 0 for a match, 3 when another harness is the nearest
310
+ ancestor, and 1 when the check cannot run or finds no harness ancestor, so the
311
+ skill separates a refusal from a check that did not decide. See
312
+ [Parent guard details](docs/parent-guard.md) for every guard target.
313
+
314
+ - The `~/.agents/skills` directory is shared. Copilot CLI and OpenCode also
315
+ discover personal skills there, so the installed Codex skill appears in both.
316
+ The Codex selection owns that directory; install and uninstall for Copilot
317
+ never touch it. The Codex skill sets `metadata.opencode/autoinvoke: false`, so
318
+ OpenCode drops it from the model's skill list and the model cannot auto-invoke
319
+ it for an `/agent-loop` request; the OpenCode plugin command then owns
320
+ `/agent-loop`. The skill also refuses to start a run when the nearest harness
321
+ process is not Codex, so a Copilot or OpenCode session that inherits
322
+ `CODEX_THREAD_ID` cannot start a run through it.
226
323
  - The Claude Code skill sets `disable-model-invocation: true`, so only the
227
324
  maintainer activates it with `/agent-loop`, and it omits `context: fork`, so
228
- the skill runs in the current session. Its body passes `${CLAUDE_SESSION_ID}`
229
- as `--parent-session` on the init dispatch call.
230
- - The OpenCode entry point is a local plugin (`.opencode/plugins/parent-guard.ts`),
231
- loaded automatically from `.opencode/plugins/`. Stored command templates expose
232
- no session id, so the plugin registers the `/agent-loop` command itself: its
233
- executor reads `CommandInvocation.sessionID` and carries that id into the
234
- orchestrator prompt, and the init dispatch call passes it as
235
- `--parent-session`.
236
- - The Codex CLI skill (`.agents/skills/agent-loop/SKILL.md`) activates only through
237
- `$agent-loop`. It passes `CODEX_THREAD_ID` as `--parent-session`. Use
238
- `$env:CODEX_THREAD_ID` in PowerShell and `$CODEX_THREAD_ID` in POSIX shells.
239
- In PowerShell, pass `--cwd $worktree` after setting
240
- `$worktree = (Get-Location).Path`.
241
- The CLI rejects an empty `--parent-session` before it initializes a run.
242
- This environment variable is an undocumented dependency and can change on
243
- upgrade. When it is absent, do not start a guarded run. Run `/hooks` once to
244
- review and trust the repository hook. Do not use
245
- `--dangerously-bypass-hook-trust` for normal use.
246
- - The Copilot CLI entry point is `src/entrypoints/copilot.mjs`, exposed as
247
- `agent-loop-copilot`. It mints a UUID, starts
248
- `copilot --session-id <uuid> --interactive <prompt>`, and includes the
249
- instruction file, task, and same id for `--parent-session` in the first
250
- prompt. The current CLI documentation exposes no custom command-template
325
+ the skill runs in the current session. OpenCode also discovers
326
+ `~/.claude/skills`, so a foreign OpenCode session lists this same copy to the
327
+ model. The skill sets `metadata.opencode/autoinvoke: false`, so OpenCode drops
328
+ it from the model's skill list and the model cannot auto-invoke it for an
329
+ `/agent-loop` request; the OpenCode plugin command then owns `/agent-loop`.
330
+ Before the init dispatch call the skill runs the installed CLI by absolute path
331
+ with `harness-check claude`; exit 3 stops the run because another harness owns
332
+ the session, and any other non-zero exit stops the run because the check could
333
+ not run or found no harness ancestor. Both cover the literal
334
+ `${CLAUDE_SESSION_ID}` that OpenCode leaves unexpanded. The resolved invocation
335
+ then replaces `agent-loop` for the init dispatch and every later command, so the
336
+ run does not need the `agent-loop` command on PATH. Its body passes
337
+ `${CLAUDE_SESSION_ID}` as `--parent-session` on the init dispatch call.
338
+ - The OpenCode entry point is a user plugin at
339
+ `~/.config/opencode/plugins/parent-guard.ts`, loaded automatically from that
340
+ directory. Stored command templates expose no session id, so the plugin
341
+ registers the `/agent-loop` command itself: its executor reads
342
+ `CommandInvocation.sessionID` and carries that id into the orchestrator
343
+ prompt, and the init dispatch call passes it as `--parent-session`. The shared
344
+ Claude and Codex skills set `metadata.opencode/autoinvoke: false`, so OpenCode
345
+ drops them from the model's skill list and the plugin command is the only
346
+ `/agent-loop` entry point.
347
+ - The Codex CLI skill (`~/.agents/skills/agent-loop/SKILL.md`) activates only
348
+ through `$agent-loop`. Before the init call it runs the installed CLI by
349
+ absolute path with `harness-check codex`, which uses the same exit codes. Exit
350
+ 3 means another harness is the nearest ancestor; any other non-zero exit means
351
+ the check could not run or found no harness ancestor. Either stops the run: an
352
+ environment check cannot
353
+ decide it, because a nested harness inherits `CODEX_THREAD_ID`. The resolved
354
+ invocation replaces `agent-loop` for every command in the instructions. The
355
+ skill passes `CODEX_THREAD_ID` as `--parent-session`. Use `$env:CODEX_THREAD_ID` in
356
+ PowerShell and `$CODEX_THREAD_ID` in POSIX shells. In PowerShell, pass
357
+ `--cwd $worktree` after setting `$worktree = (Get-Location).Path`. The CLI
358
+ requires `--parent-session` and rejects an empty or unexpanded value before it
359
+ initializes a run. This
360
+ environment variable is an undocumented dependency and can change on upgrade.
361
+ When it is absent, do not start a guarded run. Run `/hooks` once to review and
362
+ trust the installed user hook; a changed hook command needs a new trust step.
363
+ Do not use `--dangerously-bypass-hook-trust` for normal use.
364
+ - The Copilot CLI entry point is the packaged `src/entrypoints/copilot.mjs`,
365
+ exposed as `agent-loop-copilot`. It mints a UUID, starts
366
+ `copilot --session-id <uuid> --add-dir <docs dir> --interactive <prompt>`, and
367
+ includes the instruction file, task, and same id for `--parent-session` in the
368
+ first prompt. The current CLI documentation exposes no custom command-template
251
369
  session-id placeholder, so the launcher is the native entry point.
252
- - Universal fallback (Antigravity): reference
253
- `docs/orchestrator-instructions.md` in the first prompt and follow it. A
254
- native entry point is added only after that harness documents an explicit
255
- extension mechanism.
256
- - On Copilot CLI, `.github/hooks/parent-guard.json` registers PascalCase `PreToolUse`, so the payload carries
257
- `session_id` and `tool_name`; the hook prints the flat `permissionDecision` object that Copilot CLI consumes.
258
- Repository hooks require a trusted folder. GitHub documents the `.github/hooks/*.json` path for Windows, macOS, and
259
- Linux, and a live probe on Windows with Copilot CLI 1.0.87-0 confirmed the hook loaded and reported `tool_name:
260
- Write`; no macOS runtime was available for this change. Copilot also loads `.claude/settings.json` as repository
261
- settings and sets `CLAUDE_PROJECT_DIR` to the repository root, so the Claude hook runs the same guard there and
262
- Copilot honors the Claude `hookSpecificOutput.permissionDecision` response. Every Copilot `Edit`/`Write` therefore
263
- runs two parent guards: the Claude-compat one and the `.github/hooks` one. Both deny the registered parent and both
264
- stay silent for any other session. The `.github/hooks` hook stays as the documented, version-stable path.
370
+ - The Antigravity CLI entry point is a skill at
371
+ `~/.gemini/antigravity-cli/skills/agent-loop/SKILL.md`, activated as
372
+ `/agent-loop` in an interactive session. It runs the installed CLI by absolute
373
+ path with `harness-check antigravity` before the init call, and it replaces
374
+ `agent-loop` with that resolved invocation for every command in the
375
+ instructions. Then it passes
376
+ `ANTIGRAVITY_CONVERSATION_ID` as `--parent-session`; use
377
+ `$env:ANTIGRAVITY_CONVERSATION_ID` in PowerShell. This environment variable is
378
+ an undocumented dependency and can change on upgrade, and child processes
379
+ inherit it. When it is absent, or when the harness check stops the run, do not
380
+ start a run.
381
+ - Universal fallback: reference `docs/orchestrator-instructions.md` in the first
382
+ prompt and follow it when the harness entry point is not installed. The
383
+ fallback must still pass the harness session id as `--parent-session`; the CLI
384
+ refuses an init without it. The headless `agent-loop` command is the explicit
385
+ unguarded path.
386
+ - On Copilot CLI, the installed `~/.copilot/hooks/parent-guard.json` registers
387
+ PascalCase `PreToolUse`, so the payload carries `session_id` and `tool_name`,
388
+ and the hook prints the flat `permissionDecision` object that Copilot CLI
389
+ consumes. A live probe on Windows with Copilot CLI 1.0.87-0 confirmed the
390
+ repository form of this hook loaded and reported `tool_name: Write`; no macOS
391
+ runtime was available for that change. Copilot reads the shared subset of a
392
+ _repository_ `.claude/settings.json`, which does not cover a Claude user
393
+ guard, so Copilot gets its own user hook file; install never relies on Copilot
394
+ reading the Claude user settings.
265
395
  - The parent-edit guard (#57) reads `--parent-session` from the state index
266
- (see Parent guard below). For any run without `--parent-session`, the parent
267
- stays unguarded.
396
+ (see [Parent guard details](docs/parent-guard.md)). Every `role` init requires
397
+ `--parent-session`; a legacy state written without one keeps its parent
398
+ unguarded.
268
399
 
269
400
  ## Parent guard: hard read-only for the parent session
270
401
 
271
- The parent rule ("the orchestrator never edits files") is prompt-only, so a drifting parent session can still edit. Four harnesses add a hard guard for the file-edit tools: Claude Code through a `PreToolUse` hook in `.claude/settings.json`, Codex CLI through a `PreToolUse` hook in `.codex/hooks.json`, OpenCode through a `permission` `evaluate` plugin hook, and GitHub Copilot CLI through a repository `PreToolUse` hook in `.github/hooks/parent-guard.json`. All share the decision logic in `src/hook/decision.mjs`, so they deny and release under the same rule.
272
-
273
- - Matchers: Claude Code uses
274
- `Edit|Write|MultiEdit|NotebookEdit`. Codex CLI uses
275
- `apply_patch`, which its hook input reports as `tool_name: "apply_patch"`.
276
- Copilot CLI uses `Edit|Write`, which covers the current built-in edit and create
277
- tools. `Bash` stays allowed because the parent needs it to run `agent-loop role`;
278
- a shell-based edit bypasses all guards. Full enforcement needs a harness that
279
- exposes only orchestration tools.
280
- - The hook (`src/hook/parent-guard.mjs`) reads the hook input on stdin and uses only `session_id`. It resolves the state file through the session index written by the init call, never through the hook `cwd`, so a parent whose run targets a different `--cwd` stays guarded wherever it edits. No environment variable is required at parent start-up; the harness entry points (#56) pass their session id as `--parent-session` on init.
281
- - Deny only when the hook `session_id` equals `parentSession` in the registered state and the lifecycle is non-terminal (`active`, `dispatched`, `interrupted`). The guard releases only on `finish`, `abort`, or a `halted` state; during `interrupted` it stays engaged, and `dispatch --resume-interrupted` keeps it engaged because the resumed run is non-terminal again.
282
- - Everything else allows: a worker dispatched by `role` in the same cwd (a different session id), a second interactive session in the same cwd, a state without `parentSession`, and a missing or corrupt index entry or state file. The guard fails open by design: it supplements the prompt-only rule, so an unknown record never blocks a tool call.
283
- - Without a state file the hook does one absent-file read, prints nothing, and exits 0; the normal permission flow applies. The deny reason names orchestrator mode and points at `role dispatch` / `finish` / `abort`.
284
- - The Claude hook is registered as a shell-form `command` with no `args`, the form both Claude Code and Copilot's Claude-compatible settings loader execute through a shell. An exec-form `command` (`node` + script `args`) fails under Copilot, which ignores the Claude `args` field and runs `node` with the hook payload on stdin; `node` then exits 1 and Copilot fail-closes the tool call. With the shell form, Copilot imports the guard, which exits 0 with no output for every session that is not the registered parent.
285
- - On OpenCode, `.opencode/plugins/parent-guard.ts` registers a `permission` `evaluate` hook. It reads `PermissionEvaluation.sessionID`, resolves the state through the same session index, and sets `effect: "deny"` with the same reason under the same rule. A live probe against OpenCode v0.0.0-dev-19933 showed the `edit`, `write`, and `apply_patch` tools all raise the `edit` action, so the guard's action set (one set entry, `edit`) covers every built-in file-edit tool; a tool served by an MCP server raises its own action name and passes the guard. `shell` raises a different action and stays allowed, as `Bash` does on Claude Code. The guard exists only while the plugin is loaded, so a session that disables it stays unguarded.
286
- - The plugin runs inside the OpenCode server process, so it resolves `AGENT_LOOP_RUNS_ROOT` from that process's environment; the Claude Code hook inherits the parent shell's environment instead. The override is test-only, but using it outside tests would point the plugin and the `agent-loop` CLI at different roots and disable the guard silently.
287
-
288
- ### Pre-tool hook availability by harness
289
-
290
- Surveyed 2026-09-21 against current vendor docs, current binaries, and the installed Copilot CLI 1.0.87-0 binary. A session-keyed guard needs both a pre-tool hook and a documented way for the parent to learn its own session id at init time; the guard ships only where both exist.
291
-
292
- | Harness | Pre-tool hook | Guard |
293
- | ------------------ | -------------------------------------------------------------------------------------------------------------------------------------- | ------------------------- |
294
- | Claude Code | `PreToolUse`, deny supported, `session_id` in input | Implemented (this repo) |
295
- | Codex CLI | `PreToolUse`, deny supported, `session_id` in input; entry point depends on undocumented `CODEX_THREAD_ID` | Implemented (best effort) |
296
- | Antigravity CLI | `PreToolUse` hooks (workspace or global `hooks.json`), `conversationId` in input | Not implemented |
297
- | GitHub Copilot CLI | PascalCase `PreToolUse`, deny supported, `session_id` and `tool_name` in input; repository `.github/hooks/*.json` loads in current CLI | Implemented (this repo) |
298
- | OpenCode | `permission` `evaluate` plugin hook can set `deny`; event carries `PermissionEvaluation.sessionID` | Implemented (this repo) |
299
-
300
- Codex CLI now uses its `PreToolUse` hook with the session id in hook input. Its
301
- entry point relies on `CODEX_THREAD_ID` from the shell environment, which is not
302
- documented and can break on upgrade. The guard fails open when the variable is
303
- absent or unusable. Antigravity still lacks a documented parent-session channel.
304
- Copilot uses the documented session-keyed launcher and repository hook described
305
- above. OpenCode has both: a command reads `CommandInvocation.sessionID`, and the
306
- permission hook reads `PermissionEvaluation.sessionID`.
307
-
308
- The guard path blocked `apply_patch` on Codex CLI 0.156.0-alpha.14 for this
309
- Windows check. This is a known-good runtime, not a stable minimum version.
310
- Codex hook denial was not enforced in CLI 0.133.0 or Desktop 0.138.0-alpha.7.
311
- See [openai/codex#27833](https://github.com/openai/codex/issues/27833). Verify
312
- that the installed Codex version blocks `apply_patch` before relying on this
313
- guard. The hook is a best-effort guardrail, not a complete enforcement boundary.
314
-
315
- The full skill-to-edit check passed on Codex CLI 0.157.0-alpha.1 on Windows: the
316
- `$agent-loop` skill registered the Codex thread as the parent, `apply_patch`
317
- returned `GUARD_DENY_REASON`, `role abort` set the lifecycle to `aborted`, and
318
- `apply_patch` then succeeded. The probe ran with
319
- `sandbox_mode = "danger-full-access"`; the `role finish` release path is covered
320
- by `tests/hook/parent-guard.test.mjs`. The `workspace-write` leg stays
321
- unverified on Windows: Codex's unelevated Windows sandbox blocks the child
322
- `git` spawn (`EPERM`) before run initialization
323
- ([openai/codex#37415](https://github.com/openai/codex/issues/37415)). Run that
324
- leg on macOS or Linux, or under the elevated Windows sandbox.
402
+ The parent rule ("the orchestrator never edits files") is prompt-only, so a drifting parent session can still edit. Five harnesses add a hard guard for the file-edit tools, and all share the decision logic in `src/hook/decision.mjs`. See [Parent guard details](docs/parent-guard.md) for the per-harness matchers, the session-field mapping, and the pre-tool hook availability survey.
325
403
 
326
404
  ## Reviewer safety
327
405
 
@@ -360,7 +438,7 @@ Known limits:
360
438
 
361
439
  - Ignored files (matching `.gitignore`) are not tracked.
362
440
  - Mutations reverted within the same turn are not detected.
363
- - Only paths within `--cwd` are monitored.
441
+ - A change anywhere in the repository that contains `--cwd` aborts a read-only turn, even outside `--cwd`.
364
442
 
365
443
  ## Transcript
366
444
 
@@ -449,6 +527,8 @@ npm stage view <stage-id>
449
527
  npm stage approve <stage-id>
450
528
  ```
451
529
 
530
+ A manual dispatch must run on the release tag, for example `gh workflow run release.yml --ref v<version>`. A dispatch on a branch is rejected, because npm records provenance from the run ref, not the checked-out commit.
531
+
452
532
  The version goes live only after approval. Staged publishing needs npm 11.15.0 or later and 2FA on the account.
453
533
 
454
534
  ## Manual smoke test
@@ -480,5 +560,5 @@ Features considered for future development once the hybrid loop stabilizes:
480
560
  - Per-role extra CLI arguments and flags
481
561
  - GitHub pull request mode
482
562
  - Configurable validation commands and automated gates
483
- - Persistent controller state and session resume across process restarts
484
- - Streaming transcript logs and usage metadata
563
+ - Persistent headless-loop state and session resume across process restarts
564
+ - A streamed headless transcript
@@ -1,10 +1,10 @@
1
1
  # Orchestrator instructions (harness-neutral)
2
2
 
3
3
  You are the parent orchestrator for an `agent-loop role` run. Per-harness entry
4
- points (the Claude Code and Codex CLI skills, the OpenCode plugin command)
5
- include this file instead of copying it. The headless prompt in
6
- `src/prompts/orchestrator.mjs` states the same role rules in JSON-action form;
7
- this file is the source for shared rules.
4
+ points (the Claude Code, Codex CLI, and Antigravity CLI skills, the OpenCode
5
+ plugin command, and the Copilot launcher) include this file instead of copying
6
+ it. The headless prompt in `src/prompts/orchestrator.mjs` states the same role
7
+ rules in JSON-action form; this file is the source for shared rules.
8
8
 
9
9
  ## Role
10
10
 
@@ -32,24 +32,26 @@ The command blocks below call `agent-loop` directly. Install the CLI globally:
32
32
  npm install -g @andromarces/agent-loops
33
33
  ```
34
34
 
35
- From a clone, register the `bin` field globally instead. Add the pnpm global
36
- bin directory to PATH first:
37
-
38
- ```bash
39
- pnpm setup # restart the shell afterwards
40
- pnpm add -g .
41
- ```
42
-
43
- pnpm v11 removed `pnpm link --global` and keeps global bins under `PNPM_HOME`;
44
- `pnpm add -g .` fails with `ERR_PNPM_GLOBAL_BIN_DIR_NOT_IN_PATH` until
45
- `pnpm setup` puts that directory on PATH. Without a global install, replace
46
- `agent-loop` with `node "<repo>/src/cli.mjs"` and quote the repository path so
47
- a path with spaces works, or run `pnpm agent-loop` from the repository root.
35
+ Then run `agent-loop install` once to write the harness entry points and guards
36
+ at user scope.
37
+
38
+ From a clone, run `npm link` in the clone, then `agent-loop install`. `npm link`
39
+ writes the bin shim to the global prefix bin directory and points the rendered
40
+ entries at the clone. That directory must be on PATH for `agent-loop` to
41
+ resolve: it is the `npm prefix -g` directory on Windows and
42
+ `$(npm prefix -g)/bin` on macOS and Linux. After the clone moves, run
43
+ `npm link --force` from the new location, then `agent-loop install` again; a
44
+ plain `npm link` fails with `EEXIST` on Windows when a shim already exists.
45
+ `pnpm link` is not a supported path: pnpm 12 `link` has no global mode. Without a
46
+ link, replace `agent-loop` with `node "<repo>/src/cli.mjs"` and quote the
47
+ repository path so a path with spaces works, or run `pnpm agent-loop` from the
48
+ repository root.
48
49
 
49
50
  ## Starting a run
50
51
 
51
- The first dispatch carries the init flags, including `--parent-session` when
52
- the harness provides a session id:
52
+ The first dispatch carries the init flags, including the required
53
+ `--parent-session`. The subcommand refuses an init without it, so an interactive
54
+ run is never left unguarded:
53
55
 
54
56
  ```bash
55
57
  printf '%s' "<first child prompt>" | agent-loop role dispatch \
@@ -147,9 +149,10 @@ printf '%s' '{"changed":"...","verified":"...","deferred":"...","notDone":"...",
147
149
  terminal and the subcommand rejects further operations, including `abort`.
148
150
  Report the failure and the modified paths from the envelope, then stop.
149
151
  - Every run ends in exactly one terminal lifecycle: `finished`, `aborted`, or
150
- `halted`. The parent-edit guard (#57, Claude Code, Codex CLI, and OpenCode)
151
- releases on any of them; any run without `--parent-session` keeps its parent
152
- unguarded.
152
+ `halted`. The parent-edit guard (#57, Claude Code, Codex CLI, Copilot CLI,
153
+ OpenCode, and Antigravity CLI) releases on any of them. Every `agent-loop role`
154
+ init requires `--parent-session`, so the guard always has a parent to match.
155
+ The headless `agent-loop` command is the explicit unguarded path.
153
156
 
154
157
  ## Recovery after compaction or restart
155
158
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@andromarces/agent-loops",
3
- "version": "0.2.1",
3
+ "version": "0.3.0",
4
4
  "private": false,
5
5
  "description": "Run a task loop across several CLI coding agents.",
6
6
  "homepage": "https://github.com/andromarces/agent-loops#readme",
@@ -51,5 +51,5 @@
51
51
  "engines": {
52
52
  "node": ">=22"
53
53
  },
54
- "packageManager": "pnpm@12.4.1"
54
+ "packageManager": "pnpm@12.6.0"
55
55
  }
@@ -1,5 +1,6 @@
1
1
  import { parseJson } from "../lib/json.mjs";
2
2
  import { exec } from "../lib/exec.mjs";
3
+ import { setMainLoopUsage } from "./shared.mjs";
3
4
 
4
5
  export async function runAgy(state, prompt, options = {}) {
5
6
  const { cwd, readOnly, timeout, signal, role } = options;
@@ -30,17 +31,7 @@ export async function runAgy(state, prompt, options = {}) {
30
31
  }
31
32
 
32
33
  state.sessionId = result.conversation_id;
33
- setUsage(state, result);
34
+ setMainLoopUsage(state, result?.usage);
34
35
 
35
36
  return String(result.response ?? "").trim();
36
37
  }
37
-
38
- function setUsage(state, result) {
39
- const usage = {};
40
- if (result?.usage) usage.mainLoop = result.usage;
41
- if (Object.keys(usage).length > 0) {
42
- state.usage = usage;
43
- } else {
44
- delete state.usage;
45
- }
46
- }
@@ -1,5 +1,6 @@
1
1
  import { parseJsonLines } from "../lib/json.mjs";
2
2
  import { exec } from "../lib/exec.mjs";
3
+ import { resumeMismatchError, setMainLoopUsage } from "./shared.mjs";
3
4
 
4
5
  export async function runCodex(state, prompt, options = {}) {
5
6
  const { cwd, readOnly, timeout, signal, role } = options;
@@ -36,20 +37,11 @@ export async function runCodex(state, prompt, options = {}) {
36
37
  }
37
38
 
38
39
  if (state.sessionId && returnedId !== state.sessionId) {
39
- throw new Error(
40
- [
41
- "Codex did not resume the expected thread.",
42
- `Expected: ${state.sessionId}`,
43
- `Received: ${returnedId}`,
44
- ].join("\n"),
45
- );
40
+ throw resumeMismatchError("Codex", "thread", state.sessionId, returnedId);
46
41
  }
47
42
 
48
43
  state.sessionId = returnedId;
49
- setUsage(
50
- state,
51
- events.find((event) => event.type === "turn.completed"),
52
- );
44
+ setMainLoopUsage(state, events.find((event) => event.type === "turn.completed")?.usage);
53
45
 
54
46
  const messages = events
55
47
  .filter((event) => event.type === "item.completed" && event.item?.type === "agent_message")
@@ -62,12 +54,3 @@ export async function runCodex(state, prompt, options = {}) {
62
54
 
63
55
  return String(messages.at(-1)).trim();
64
56
  }
65
-
66
- /** Sets top-level turn usage, or removes stale usage when Codex omits it. */
67
- function setUsage(state, completedTurn) {
68
- if (completedTurn?.usage) {
69
- state.usage = { mainLoop: completedTurn.usage };
70
- } else {
71
- delete state.usage;
72
- }
73
- }
@@ -1,6 +1,7 @@
1
1
  import { randomUUID } from "node:crypto";
2
2
  import { parseJsonLines } from "../lib/json.mjs";
3
3
  import { exec } from "../lib/exec.mjs";
4
+ import { resumeMismatchError, setMainLoopUsage } from "./shared.mjs";
4
5
 
5
6
  export async function runCopilot(state, prompt, options = {}) {
6
7
  const { cwd, readOnly, timeout, signal, role } = options;
@@ -29,13 +30,13 @@ export async function runCopilot(state, prompt, options = {}) {
29
30
  ({ stdout } = await exec("copilot", args, { cwd, input: prompt, timeout, signal, role }));
30
31
  } catch (error) {
31
32
  const failed = parseJsonLines(error?.stdout ?? "");
32
- setUsage(state, findResultEvent(failed));
33
+ setMainLoopUsage(state, objectUsage(findResultEvent(failed)));
33
34
  throw error;
34
35
  }
35
36
 
36
37
  const events = parseJsonLines(stdout);
37
38
  const resultEvent = findResultEvent(events);
38
- setUsage(state, resultEvent);
39
+ setMainLoopUsage(state, objectUsage(resultEvent));
39
40
 
40
41
  const returnedId = resultEvent?.sessionId ?? resultEvent?.session_id;
41
42
  if (!returnedId) {
@@ -43,13 +44,7 @@ export async function runCopilot(state, prompt, options = {}) {
43
44
  }
44
45
 
45
46
  if (requestedSessionId && requestedSessionId !== returnedId) {
46
- throw new Error(
47
- [
48
- "Copilot did not resume the expected session.",
49
- `Expected: ${requestedSessionId}`,
50
- `Received: ${returnedId}`,
51
- ].join("\n"),
52
- );
47
+ throw resumeMismatchError("Copilot", "session", requestedSessionId, returnedId);
53
48
  }
54
49
 
55
50
  state.sessionId = returnedId;
@@ -76,11 +71,8 @@ function readAssistantMessage(event) {
76
71
  return typeof content === "string" ? content : "";
77
72
  }
78
73
 
79
- function setUsage(state, resultEvent) {
80
- if (resultEvent && resultEvent.usage && typeof resultEvent.usage === "object") {
81
- state.usage = { mainLoop: resultEvent.usage };
82
- return;
83
- }
84
-
85
- delete state.usage;
74
+ /** Copilot reports usage as an object; any other shape counts as absent. */
75
+ function objectUsage(resultEvent) {
76
+ const usage = resultEvent?.usage;
77
+ return usage && typeof usage === "object" ? usage : undefined;
86
78
  }