@andromarces/agent-loops 0.2.2 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +192 -114
- package/docs/orchestrator-instructions.md +25 -22
- package/package.json +2 -2
- package/src/agents/agy.mjs +2 -11
- package/src/agents/codex.mjs +3 -20
- package/src/agents/copilot.mjs +8 -16
- package/src/agents/opencode.mjs +2 -7
- package/src/agents/shared.mjs +29 -0
- package/src/cli.mjs +46 -25
- package/src/hook/antigravity-parent-guard.mjs +28 -0
- package/src/hook/copilot-parent-guard.mjs +5 -30
- package/src/hook/decision.mjs +59 -6
- package/src/hook/opencode-plugin.mjs +92 -0
- package/src/hook/parent-guard.mjs +8 -33
- package/src/install/commands.mjs +268 -0
- package/src/install/fsutil.mjs +159 -0
- package/src/install/harnesses.mjs +170 -0
- package/src/install/installer.mjs +688 -0
- package/src/install/manifest.mjs +222 -0
- package/src/install/settings.mjs +217 -0
- package/src/install/templates/antigravity/agent-loop-antigravity-parent-guard.mjs +14 -0
- package/src/install/templates/antigravity/hooks.json +16 -0
- package/src/install/templates/antigravity/skills/agent-loop/SKILL.md +24 -0
- package/src/install/templates/claude/skills/agent-loop/SKILL.md +33 -0
- package/src/install/templates/codex/skills/agent-loop/SKILL.md +30 -0
- package/src/install/templates/codex/skills/agent-loop/agents/openai.yaml +2 -0
- package/src/install/templates/copilot/hooks/parent-guard.json +15 -0
- package/src/install/templates/opencode/plugins/parent-guard.ts +11 -0
- package/src/lib/args.mjs +23 -0
- package/src/lib/hash.mjs +9 -0
- package/src/lib/log.mjs +18 -3
- package/src/lib/process-ancestry.mjs +104 -0
- package/src/lib/runstate.mjs +133 -38
- package/src/lib/snapshot.mjs +3 -7
- package/src/role.mjs +71 -87
- package/src/runtime.mjs +9 -9
package/README.md
CHANGED
|
@@ -79,6 +79,64 @@ npm install -g @andromarces/agent-loops
|
|
|
79
79
|
pnpm add -g @andromarces/agent-loops
|
|
80
80
|
```
|
|
81
81
|
|
|
82
|
+
Set up the harness entry points and parent guards at user scope:
|
|
83
|
+
|
|
84
|
+
```bash
|
|
85
|
+
agent-loop install
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
`install` detects the harness CLIs on `PATH`, preselects them, and writes the
|
|
89
|
+
entry point and guard for each selected harness. It is interactive; pass
|
|
90
|
+
`--harness <list> --yes` for scripts, and `--dry-run` to print the planned
|
|
91
|
+
writes without changing anything. A second run with the same package makes no
|
|
92
|
+
change. `agent-loop uninstall` removes only the files and settings entries the
|
|
93
|
+
install recorded, and restores a file that install changed when the file is
|
|
94
|
+
otherwise unchanged. Install applies a harness guard before its entry point, and
|
|
95
|
+
if a write fails part way through it records the writes that completed, so
|
|
96
|
+
`uninstall` still removes or restores them. It leaves a target in place when it
|
|
97
|
+
cannot remove or restore it safely; see [Uninstall skips](#uninstall-skips).
|
|
98
|
+
|
|
99
|
+
`install` must run from a global install or a linked clone. It writes the package
|
|
100
|
+
location as an absolute path into every entry point and guard, and `npx` and
|
|
101
|
+
`pnpm dlx` place the package in a cache directory that npm or pnpm can delete.
|
|
102
|
+
When it detects its own package root inside that cache, `install` refuses with a
|
|
103
|
+
message that asks for a global install first, so it never writes a path that can
|
|
104
|
+
disappear. `uninstall` reads only the manifest and the harness files, so it still
|
|
105
|
+
removes them after the package is gone.
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
agent-loop install --harness claude,codex --yes
|
|
109
|
+
agent-loop uninstall
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
### Uninstall skips
|
|
113
|
+
|
|
114
|
+
`agent-loop uninstall` prints one line per target as
|
|
115
|
+
`<harness>: <action> <path> (<detail>)`. A target that it cannot remove or
|
|
116
|
+
restore safely reports `skip`, stays on disk, and loses its manifest record:
|
|
117
|
+
uninstall deletes the harness record, and removes the manifest when no harness
|
|
118
|
+
remains. Uninstall never retries a skipped target, and a backup file can remain
|
|
119
|
+
on disk with no record. Recover a skipped target by hand.
|
|
120
|
+
|
|
121
|
+
`install` writes a backup at `<file>.agent-loops-backup` when the target existed
|
|
122
|
+
before install and install changed it, and uninstall reads that file to restore
|
|
123
|
+
the pre-install bytes. Install writes no backup when the target did not exist
|
|
124
|
+
before install, and none when it already held the installed bytes.
|
|
125
|
+
|
|
126
|
+
| Reported detail | Target | What stays on disk | Manual recovery |
|
|
127
|
+
| -------------------------------------------------- | -------- | -------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
128
|
+
| `owned file changed since install; left unchanged` | file | The file, with your later edits, and a backup when present | Restore `<file>.agent-loops-backup` over the file when that backup exists and you want the pre-install content, or delete the file when you do not want it. Delete the backup by hand when it stays behind. |
|
|
129
|
+
| `backup missing; left unchanged` | file | The file, unchanged since install, with no backup | Install wrote no backup when the file already held the installed bytes, so the file can be your original and needs no change. Otherwise restore the original from version control or another copy, or delete the file when you do not want the installed entry point. |
|
|
130
|
+
| `settings do not parse; left unchanged` | settings | The settings file, still invalid, and a backup when present | The record is gone, so no later `agent-loop uninstall` retries it. Repair the syntax and remove the recorded entry by hand, or restore `<file>.agent-loops-backup` when that backup exists and your later edits can be discarded. Then delete the backup. |
|
|
131
|
+
| `recorded entry not found; left unchanged` | settings | The settings file, with no recorded entry, and a backup when present | The recorded entry is already gone. Remove another agent-loop entry by hand when one is present, then delete the backup when it exists. |
|
|
132
|
+
| `backup missing; left unchanged` | settings | The settings file, still holding the recorded entry | Remove the recorded entry by hand, or restore the original from version control or another copy. The backup is gone, so none remains to delete. |
|
|
133
|
+
|
|
134
|
+
When a settings file changed after install but still parses and holds the
|
|
135
|
+
recorded entry, uninstall reports `remove-entry` instead: it removes only that
|
|
136
|
+
entry, keeps your edits, and leaves the result not byte-identical. The backup at
|
|
137
|
+
`<file>.agent-loops-backup` stays behind, so delete it by hand when you no
|
|
138
|
+
longer need the pre-install copy.
|
|
139
|
+
|
|
82
140
|
Run it without installing:
|
|
83
141
|
|
|
84
142
|
```bash
|
|
@@ -86,6 +144,9 @@ npx @andromarces/agent-loops --help
|
|
|
86
144
|
pnpm dlx @andromarces/agent-loops --help
|
|
87
145
|
```
|
|
88
146
|
|
|
147
|
+
`--help` writes no config, so it runs from the cache. `install` writes config
|
|
148
|
+
that points at the package, so it refuses from the cache; see [Install](#install).
|
|
149
|
+
|
|
89
150
|
Install from a Git URL instead of the registry:
|
|
90
151
|
|
|
91
152
|
```bash
|
|
@@ -102,28 +163,39 @@ The `bin` script keeps its `#!/usr/bin/env node` shebang and executable bit, so
|
|
|
102
163
|
|
|
103
164
|
### From a clone (development)
|
|
104
165
|
|
|
105
|
-
Development uses pnpm and the repository Git hooks
|
|
166
|
+
Development uses pnpm and the repository Git hooks. User-scope integrations
|
|
167
|
+
point at the package location, so a clone registers its own copy with `npm link`
|
|
168
|
+
before `agent-loop install`:
|
|
106
169
|
|
|
107
170
|
```bash
|
|
108
171
|
git clone <repository-url>
|
|
109
172
|
cd agent-loops
|
|
110
173
|
pnpm install
|
|
111
|
-
|
|
174
|
+
npm link
|
|
175
|
+
agent-loop install
|
|
112
176
|
```
|
|
113
177
|
|
|
114
|
-
`
|
|
178
|
+
`npm link` writes its bin shim to the global prefix bin directory, so that
|
|
179
|
+
directory must be on `PATH` for `agent-loop` to resolve. It is the
|
|
180
|
+
`npm prefix -g` directory on Windows and `$(npm prefix -g)/bin` on macOS and
|
|
181
|
+
Linux.
|
|
115
182
|
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
183
|
+
Entries rendered from a clone point at that clone. After the clone moves, run
|
|
184
|
+
`npm link --force` from the new location, then `agent-loop install` again. A
|
|
185
|
+
plain `npm link` fails with `EEXIST` on Windows when a shim already exists;
|
|
186
|
+
`--force` overwrites the shim.
|
|
187
|
+
`pnpm link` is not a supported path: pnpm 12 `link` has no global mode. Without a
|
|
188
|
+
link, call the CLI entry directly and quote the repository path so a path with
|
|
189
|
+
spaces works:
|
|
122
190
|
|
|
123
191
|
```bash
|
|
124
|
-
node "<repo>/src/cli.mjs"
|
|
192
|
+
node "<repo>/src/cli.mjs" install --harness claude --yes
|
|
125
193
|
```
|
|
126
194
|
|
|
195
|
+
`pnpm agent-loop` still runs the CLI entry from the repository root. The
|
|
196
|
+
repository Git hooks run formatting and linting on commit and push; they need
|
|
197
|
+
`pnpm install` to have completed, because its `prepare` script installs husky.
|
|
198
|
+
|
|
127
199
|
## Usage
|
|
128
200
|
|
|
129
201
|
```bash
|
|
@@ -183,11 +255,16 @@ agent-loop role dispatch --role reviewer --cwd /path/to/work-tree --prompt-file
|
|
|
183
255
|
Operations: `dispatch` (default), `finish`, `abort`.
|
|
184
256
|
|
|
185
257
|
- The run state lives at a fixed path derived from the resolved `--cwd` (`<os tmpdir>/agent-loops/runs/<sha256 of cwd, shortened>/state.json`, with `state.lock` beside it). There is no `--state` flag; `AGENT_LOOP_RUNS_ROOT` overrides the root for tests only.
|
|
258
|
+
- The init call requires `--parent-session`: the CLI refuses an init without it,
|
|
259
|
+
and refuses an unexpanded placeholder such as `${CLAUDE_SESSION_ID}` or
|
|
260
|
+
`%CODEX_THREAD_ID%`, before any state is written. A run through `role` is
|
|
261
|
+
therefore always guarded; the headless `agent-loop` command is the explicit
|
|
262
|
+
unguarded path.
|
|
186
263
|
- The init call writes a session index entry at `<root>/sessions/<parent-session>` pointing at the state file, so a parent guard hook (#57) can look the run up by session id even when `--cwd` is a different work tree. A later init call from the same session overwrites the entry.
|
|
187
264
|
- Prompts come from stdin by default, or `--prompt-file`. `finish` reads the five-key summary as JSON on stdin; `abort` takes `--reason`.
|
|
188
265
|
- The state file records `task`, `mode`, `cwd`, `parentSession`, `maxSteps`, `timeout`, `stepsUsed`, `lifecycle`, `roles.{worker,reviewer}.{kind,model,effort,sessionId}` (`roles.worker` is null in `review-only`), `lastDispatch`, and `lastResult`, plus `summary` or `reason` when terminal and `resumeDecision` when a maintainer resumed an interrupted run. Updates are atomic (temp file plus rename); exclusive access uses `state.lock` with a stale-lock check on the owner pid.
|
|
189
266
|
- Lifecycle values: `active`, `dispatched`, `interrupted`, `halted`, `finished`, `aborted` (terminal: `halted`, `finished`, `aborted`). A turn interrupted between the CLI start and the state write leaves `dispatched` with a dead lock owner; the first call after the crash marks it `interrupted`, exits non-zero, and never repeats the turn, even with `--resume-interrupted`. From `interrupted`, only `abort` or an explicit `dispatch --resume-interrupted` is accepted.
|
|
190
|
-
- The lock is fail-closed on ambiguity: a contender that finds a lock it cannot read (created moments ago, content not yet written) exits non-zero and never removes it; only an unparseable lock older than a grace window, or one whose recorded pid is dead, is treated as stale.
|
|
267
|
+
- The lock is fail-closed on ambiguity: a contender that finds a lock it cannot read (created moments ago, content not yet written) exits non-zero and never removes it; only an unparseable lock older than a grace window, or one whose recorded pid is dead, is treated as stale. Lock creation is an atomic hard link, with an exclusive-create fallback on filesystems that have no hard links (FAT/exFAT, some network mounts); a lock temp file left by a crashed process is pruned on the next acquisition.
|
|
191
268
|
- The reviewer turn runs under the same `withMutationCheck` as the headless loop: a detected mutation or snapshot error is fatal, keeps the charged step, and sets `halted`. No further dispatch is possible; the next run needs a new init call, which archives the halted file as `state.<timestamp>.json`.
|
|
192
269
|
- `mode: review-only` rejects `--role worker` as a hard guard and does not require `--worker` at init. `finish` is completion of the requested work, not code acceptance: it is accepted from `active` in any mode, and the five-key summary carries the reviewer verdict and unresolved findings.
|
|
193
270
|
- The reviewer is required to end with one explicit `Verdict:` line (`accept` or `reject`, parsed case-insensitively) inside its closing block. The verdict word alone, the word closed by a sentence period (`reject.`), or the word followed by a separator and a trailing clause (`reject — the state does not pass`) parses to that word, unless the clause names either verdict as a whole word. Any other malformed value, including a missing line or a line that names both verdicts, yields `verdict: unknown`; process success never implies acceptance.
|
|
@@ -215,113 +292,114 @@ includes it rather than copying it. Role activation never goes into `AGENTS.md`
|
|
|
215
292
|
or `CLAUDE.md`, because dispatched children read those files; activation
|
|
216
293
|
happens only through explicit invocation.
|
|
217
294
|
|
|
218
|
-
| Harness |
|
|
219
|
-
| --------------- |
|
|
220
|
-
| Claude Code |
|
|
221
|
-
| OpenCode |
|
|
222
|
-
| Codex CLI |
|
|
223
|
-
| Copilot CLI | `
|
|
224
|
-
| Antigravity CLI |
|
|
225
|
-
|
|
295
|
+
| Harness | Installed entry point | Invocation |
|
|
296
|
+
| --------------- | ------------------------------------------------------ | --------------------------------------------- |
|
|
297
|
+
| Claude Code | `~/.claude/skills/agent-loop/SKILL.md` | `/agent-loop <task and role settings>` |
|
|
298
|
+
| OpenCode | `~/.config/opencode/plugins/parent-guard.ts` | `/agent-loop <task and role settings>` |
|
|
299
|
+
| Codex CLI | `~/.agents/skills/agent-loop/SKILL.md` | `$agent-loop <task and role settings>` |
|
|
300
|
+
| Copilot CLI | `agent-loop-copilot` (packaged bin) | `agent-loop-copilot <task and role settings>` |
|
|
301
|
+
| Antigravity CLI | `~/.gemini/antigravity-cli/skills/agent-loop/SKILL.md` | `/agent-loop <task and role settings>` |
|
|
302
|
+
|
|
303
|
+
`agent-loop install` writes these files from the templates under
|
|
304
|
+
`src/install/templates/`, and each rendered entry point points at the installed
|
|
305
|
+
`docs/orchestrator-instructions.md` by absolute path, so it works outside a
|
|
306
|
+
clone. Each skill renders the CLI by absolute path and runs every `agent-loop`
|
|
307
|
+
command in the instructions with that resolved invocation, so a Git Bash session
|
|
308
|
+
that cannot resolve the `agent-loop` command still starts a guarded run (#151).
|
|
309
|
+
`harness-check` exits 0 for a match, 3 when another harness is the nearest
|
|
310
|
+
ancestor, and 1 when the check cannot run or finds no harness ancestor, so the
|
|
311
|
+
skill separates a refusal from a check that did not decide. See
|
|
312
|
+
[Parent guard details](docs/parent-guard.md) for every guard target.
|
|
313
|
+
|
|
314
|
+
- The `~/.agents/skills` directory is shared. Copilot CLI and OpenCode also
|
|
315
|
+
discover personal skills there, so the installed Codex skill appears in both.
|
|
316
|
+
The Codex selection owns that directory; install and uninstall for Copilot
|
|
317
|
+
never touch it. The Codex skill sets `metadata.opencode/autoinvoke: false`, so
|
|
318
|
+
OpenCode drops it from the model's skill list and the model cannot auto-invoke
|
|
319
|
+
it for an `/agent-loop` request; the OpenCode plugin command then owns
|
|
320
|
+
`/agent-loop`. The skill also refuses to start a run when the nearest harness
|
|
321
|
+
process is not Codex, so a Copilot or OpenCode session that inherits
|
|
322
|
+
`CODEX_THREAD_ID` cannot start a run through it.
|
|
226
323
|
- The Claude Code skill sets `disable-model-invocation: true`, so only the
|
|
227
324
|
maintainer activates it with `/agent-loop`, and it omits `context: fork`, so
|
|
228
|
-
the skill runs in the current session.
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
`$
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
`$
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
325
|
+
the skill runs in the current session. OpenCode also discovers
|
|
326
|
+
`~/.claude/skills`, so a foreign OpenCode session lists this same copy to the
|
|
327
|
+
model. The skill sets `metadata.opencode/autoinvoke: false`, so OpenCode drops
|
|
328
|
+
it from the model's skill list and the model cannot auto-invoke it for an
|
|
329
|
+
`/agent-loop` request; the OpenCode plugin command then owns `/agent-loop`.
|
|
330
|
+
Before the init dispatch call the skill runs the installed CLI by absolute path
|
|
331
|
+
with `harness-check claude`; exit 3 stops the run because another harness owns
|
|
332
|
+
the session, and any other non-zero exit stops the run because the check could
|
|
333
|
+
not run or found no harness ancestor. Both cover the literal
|
|
334
|
+
`${CLAUDE_SESSION_ID}` that OpenCode leaves unexpanded. The resolved invocation
|
|
335
|
+
then replaces `agent-loop` for the init dispatch and every later command, so the
|
|
336
|
+
run does not need the `agent-loop` command on PATH. Its body passes
|
|
337
|
+
`${CLAUDE_SESSION_ID}` as `--parent-session` on the init dispatch call.
|
|
338
|
+
- The OpenCode entry point is a user plugin at
|
|
339
|
+
`~/.config/opencode/plugins/parent-guard.ts`, loaded automatically from that
|
|
340
|
+
directory. Stored command templates expose no session id, so the plugin
|
|
341
|
+
registers the `/agent-loop` command itself: its executor reads
|
|
342
|
+
`CommandInvocation.sessionID` and carries that id into the orchestrator
|
|
343
|
+
prompt, and the init dispatch call passes it as `--parent-session`. The shared
|
|
344
|
+
Claude and Codex skills set `metadata.opencode/autoinvoke: false`, so OpenCode
|
|
345
|
+
drops them from the model's skill list and the plugin command is the only
|
|
346
|
+
`/agent-loop` entry point.
|
|
347
|
+
- The Codex CLI skill (`~/.agents/skills/agent-loop/SKILL.md`) activates only
|
|
348
|
+
through `$agent-loop`. Before the init call it runs the installed CLI by
|
|
349
|
+
absolute path with `harness-check codex`, which uses the same exit codes. Exit
|
|
350
|
+
3 means another harness is the nearest ancestor; any other non-zero exit means
|
|
351
|
+
the check could not run or found no harness ancestor. Either stops the run: an
|
|
352
|
+
environment check cannot
|
|
353
|
+
decide it, because a nested harness inherits `CODEX_THREAD_ID`. The resolved
|
|
354
|
+
invocation replaces `agent-loop` for every command in the instructions. The
|
|
355
|
+
skill passes `CODEX_THREAD_ID` as `--parent-session`. Use `$env:CODEX_THREAD_ID` in
|
|
356
|
+
PowerShell and `$CODEX_THREAD_ID` in POSIX shells. In PowerShell, pass
|
|
357
|
+
`--cwd $worktree` after setting `$worktree = (Get-Location).Path`. The CLI
|
|
358
|
+
requires `--parent-session` and rejects an empty or unexpanded value before it
|
|
359
|
+
initializes a run. This
|
|
360
|
+
environment variable is an undocumented dependency and can change on upgrade.
|
|
361
|
+
When it is absent, do not start a guarded run. Run `/hooks` once to review and
|
|
362
|
+
trust the installed user hook; a changed hook command needs a new trust step.
|
|
363
|
+
Do not use `--dangerously-bypass-hook-trust` for normal use.
|
|
364
|
+
- The Copilot CLI entry point is the packaged `src/entrypoints/copilot.mjs`,
|
|
365
|
+
exposed as `agent-loop-copilot`. It mints a UUID, starts
|
|
366
|
+
`copilot --session-id <uuid> --add-dir <docs dir> --interactive <prompt>`, and
|
|
367
|
+
includes the instruction file, task, and same id for `--parent-session` in the
|
|
368
|
+
first prompt. The current CLI documentation exposes no custom command-template
|
|
251
369
|
session-id placeholder, so the launcher is the native entry point.
|
|
252
|
-
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
370
|
+
- The Antigravity CLI entry point is a skill at
|
|
371
|
+
`~/.gemini/antigravity-cli/skills/agent-loop/SKILL.md`, activated as
|
|
372
|
+
`/agent-loop` in an interactive session. It runs the installed CLI by absolute
|
|
373
|
+
path with `harness-check antigravity` before the init call, and it replaces
|
|
374
|
+
`agent-loop` with that resolved invocation for every command in the
|
|
375
|
+
instructions. Then it passes
|
|
376
|
+
`ANTIGRAVITY_CONVERSATION_ID` as `--parent-session`; use
|
|
377
|
+
`$env:ANTIGRAVITY_CONVERSATION_ID` in PowerShell. This environment variable is
|
|
378
|
+
an undocumented dependency and can change on upgrade, and child processes
|
|
379
|
+
inherit it. When it is absent, or when the harness check stops the run, do not
|
|
380
|
+
start a run.
|
|
381
|
+
- Universal fallback: reference `docs/orchestrator-instructions.md` in the first
|
|
382
|
+
prompt and follow it when the harness entry point is not installed. The
|
|
383
|
+
fallback must still pass the harness session id as `--parent-session`; the CLI
|
|
384
|
+
refuses an init without it. The headless `agent-loop` command is the explicit
|
|
385
|
+
unguarded path.
|
|
386
|
+
- On Copilot CLI, the installed `~/.copilot/hooks/parent-guard.json` registers
|
|
387
|
+
PascalCase `PreToolUse`, so the payload carries `session_id` and `tool_name`,
|
|
388
|
+
and the hook prints the flat `permissionDecision` object that Copilot CLI
|
|
389
|
+
consumes. A live probe on Windows with Copilot CLI 1.0.87-0 confirmed the
|
|
390
|
+
repository form of this hook loaded and reported `tool_name: Write`; no macOS
|
|
391
|
+
runtime was available for that change. Copilot reads the shared subset of a
|
|
392
|
+
_repository_ `.claude/settings.json`, which does not cover a Claude user
|
|
393
|
+
guard, so Copilot gets its own user hook file; install never relies on Copilot
|
|
394
|
+
reading the Claude user settings.
|
|
265
395
|
- The parent-edit guard (#57) reads `--parent-session` from the state index
|
|
266
|
-
(see Parent guard
|
|
267
|
-
|
|
396
|
+
(see [Parent guard details](docs/parent-guard.md)). Every `role` init requires
|
|
397
|
+
`--parent-session`; a legacy state written without one keeps its parent
|
|
398
|
+
unguarded.
|
|
268
399
|
|
|
269
400
|
## Parent guard: hard read-only for the parent session
|
|
270
401
|
|
|
271
|
-
The parent rule ("the orchestrator never edits files") is prompt-only, so a drifting parent session can still edit.
|
|
272
|
-
|
|
273
|
-
- Matchers: Claude Code uses
|
|
274
|
-
`Edit|Write|MultiEdit|NotebookEdit`. Codex CLI uses
|
|
275
|
-
`apply_patch`, which its hook input reports as `tool_name: "apply_patch"`.
|
|
276
|
-
Copilot CLI uses `Edit|Write`, which covers the current built-in edit and create
|
|
277
|
-
tools. `Bash` stays allowed because the parent needs it to run `agent-loop role`;
|
|
278
|
-
a shell-based edit bypasses all guards. Full enforcement needs a harness that
|
|
279
|
-
exposes only orchestration tools.
|
|
280
|
-
- The hook (`src/hook/parent-guard.mjs`) reads the hook input on stdin and uses only `session_id`. It resolves the state file through the session index written by the init call, never through the hook `cwd`, so a parent whose run targets a different `--cwd` stays guarded wherever it edits. No environment variable is required at parent start-up; the harness entry points (#56) pass their session id as `--parent-session` on init.
|
|
281
|
-
- Deny only when the hook `session_id` equals `parentSession` in the registered state and the lifecycle is non-terminal (`active`, `dispatched`, `interrupted`). The guard releases only on `finish`, `abort`, or a `halted` state; during `interrupted` it stays engaged, and `dispatch --resume-interrupted` keeps it engaged because the resumed run is non-terminal again.
|
|
282
|
-
- Everything else allows: a worker dispatched by `role` in the same cwd (a different session id), a second interactive session in the same cwd, a state without `parentSession`, and a missing or corrupt index entry or state file. The guard fails open by design: it supplements the prompt-only rule, so an unknown record never blocks a tool call.
|
|
283
|
-
- Without a state file the hook does one absent-file read, prints nothing, and exits 0; the normal permission flow applies. The deny reason names orchestrator mode and points at `role dispatch` / `finish` / `abort`.
|
|
284
|
-
- The Claude hook is registered as a shell-form `command` with no `args`, the form both Claude Code and Copilot's Claude-compatible settings loader execute through a shell. An exec-form `command` (`node` + script `args`) fails under Copilot, which ignores the Claude `args` field and runs `node` with the hook payload on stdin; `node` then exits 1 and Copilot fail-closes the tool call. With the shell form, Copilot imports the guard, which exits 0 with no output for every session that is not the registered parent.
|
|
285
|
-
- On OpenCode, `.opencode/plugins/parent-guard.ts` registers a `permission` `evaluate` hook. It reads `PermissionEvaluation.sessionID`, resolves the state through the same session index, and sets `effect: "deny"` with the same reason under the same rule. A live probe against OpenCode v0.0.0-dev-19933 showed the `edit`, `write`, and `apply_patch` tools all raise the `edit` action, so the guard's action set (one set entry, `edit`) covers every built-in file-edit tool; a tool served by an MCP server raises its own action name and passes the guard. `shell` raises a different action and stays allowed, as `Bash` does on Claude Code. The guard exists only while the plugin is loaded, so a session that disables it stays unguarded.
|
|
286
|
-
- The plugin runs inside the OpenCode server process, so it resolves `AGENT_LOOP_RUNS_ROOT` from that process's environment; the Claude Code hook inherits the parent shell's environment instead. The override is test-only, but using it outside tests would point the plugin and the `agent-loop` CLI at different roots and disable the guard silently.
|
|
287
|
-
|
|
288
|
-
### Pre-tool hook availability by harness
|
|
289
|
-
|
|
290
|
-
Surveyed 2026-09-21 against current vendor docs, current binaries, and the installed Copilot CLI 1.0.87-0 binary. A session-keyed guard needs both a pre-tool hook and a documented way for the parent to learn its own session id at init time; the guard ships only where both exist.
|
|
291
|
-
|
|
292
|
-
| Harness | Pre-tool hook | Guard |
|
|
293
|
-
| ------------------ | -------------------------------------------------------------------------------------------------------------------------------------- | ------------------------- |
|
|
294
|
-
| Claude Code | `PreToolUse`, deny supported, `session_id` in input | Implemented (this repo) |
|
|
295
|
-
| Codex CLI | `PreToolUse`, deny supported, `session_id` in input; entry point depends on undocumented `CODEX_THREAD_ID` | Implemented (best effort) |
|
|
296
|
-
| Antigravity CLI | `PreToolUse` hooks (workspace or global `hooks.json`), `conversationId` in input | Not implemented |
|
|
297
|
-
| GitHub Copilot CLI | PascalCase `PreToolUse`, deny supported, `session_id` and `tool_name` in input; repository `.github/hooks/*.json` loads in current CLI | Implemented (this repo) |
|
|
298
|
-
| OpenCode | `permission` `evaluate` plugin hook can set `deny`; event carries `PermissionEvaluation.sessionID` | Implemented (this repo) |
|
|
299
|
-
|
|
300
|
-
Codex CLI now uses its `PreToolUse` hook with the session id in hook input. Its
|
|
301
|
-
entry point relies on `CODEX_THREAD_ID` from the shell environment, which is not
|
|
302
|
-
documented and can break on upgrade. The guard fails open when the variable is
|
|
303
|
-
absent or unusable. Antigravity still lacks a documented parent-session channel.
|
|
304
|
-
Copilot uses the documented session-keyed launcher and repository hook described
|
|
305
|
-
above. OpenCode has both: a command reads `CommandInvocation.sessionID`, and the
|
|
306
|
-
permission hook reads `PermissionEvaluation.sessionID`.
|
|
307
|
-
|
|
308
|
-
The guard path blocked `apply_patch` on Codex CLI 0.156.0-alpha.14 for this
|
|
309
|
-
Windows check. This is a known-good runtime, not a stable minimum version.
|
|
310
|
-
Codex hook denial was not enforced in CLI 0.133.0 or Desktop 0.138.0-alpha.7.
|
|
311
|
-
See [openai/codex#27833](https://github.com/openai/codex/issues/27833). Verify
|
|
312
|
-
that the installed Codex version blocks `apply_patch` before relying on this
|
|
313
|
-
guard. The hook is a best-effort guardrail, not a complete enforcement boundary.
|
|
314
|
-
|
|
315
|
-
The full skill-to-edit check passed on Codex CLI 0.157.0-alpha.1 on Windows: the
|
|
316
|
-
`$agent-loop` skill registered the Codex thread as the parent, `apply_patch`
|
|
317
|
-
returned `GUARD_DENY_REASON`, `role abort` set the lifecycle to `aborted`, and
|
|
318
|
-
`apply_patch` then succeeded. The probe ran with
|
|
319
|
-
`sandbox_mode = "danger-full-access"`; the `role finish` release path is covered
|
|
320
|
-
by `tests/hook/parent-guard.test.mjs`. The `workspace-write` leg stays
|
|
321
|
-
unverified on Windows: Codex's unelevated Windows sandbox blocks the child
|
|
322
|
-
`git` spawn (`EPERM`) before run initialization
|
|
323
|
-
([openai/codex#37415](https://github.com/openai/codex/issues/37415)). Run that
|
|
324
|
-
leg on macOS or Linux, or under the elevated Windows sandbox.
|
|
402
|
+
The parent rule ("the orchestrator never edits files") is prompt-only, so a drifting parent session can still edit. Five harnesses add a hard guard for the file-edit tools, and all share the decision logic in `src/hook/decision.mjs`. See [Parent guard details](docs/parent-guard.md) for the per-harness matchers, the session-field mapping, and the pre-tool hook availability survey.
|
|
325
403
|
|
|
326
404
|
## Reviewer safety
|
|
327
405
|
|
|
@@ -360,7 +438,7 @@ Known limits:
|
|
|
360
438
|
|
|
361
439
|
- Ignored files (matching `.gitignore`) are not tracked.
|
|
362
440
|
- Mutations reverted within the same turn are not detected.
|
|
363
|
-
-
|
|
441
|
+
- A change anywhere in the repository that contains `--cwd` aborts a read-only turn, even outside `--cwd`.
|
|
364
442
|
|
|
365
443
|
## Transcript
|
|
366
444
|
|
|
@@ -482,5 +560,5 @@ Features considered for future development once the hybrid loop stabilizes:
|
|
|
482
560
|
- Per-role extra CLI arguments and flags
|
|
483
561
|
- GitHub pull request mode
|
|
484
562
|
- Configurable validation commands and automated gates
|
|
485
|
-
- Persistent
|
|
486
|
-
-
|
|
563
|
+
- Persistent headless-loop state and session resume across process restarts
|
|
564
|
+
- A streamed headless transcript
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
# Orchestrator instructions (harness-neutral)
|
|
2
2
|
|
|
3
3
|
You are the parent orchestrator for an `agent-loop role` run. Per-harness entry
|
|
4
|
-
points (the Claude Code and
|
|
5
|
-
include this file instead of copying
|
|
6
|
-
`src/prompts/orchestrator.mjs` states the same role
|
|
7
|
-
this file is the source for shared rules.
|
|
4
|
+
points (the Claude Code, Codex CLI, and Antigravity CLI skills, the OpenCode
|
|
5
|
+
plugin command, and the Copilot launcher) include this file instead of copying
|
|
6
|
+
it. The headless prompt in `src/prompts/orchestrator.mjs` states the same role
|
|
7
|
+
rules in JSON-action form; this file is the source for shared rules.
|
|
8
8
|
|
|
9
9
|
## Role
|
|
10
10
|
|
|
@@ -32,24 +32,26 @@ The command blocks below call `agent-loop` directly. Install the CLI globally:
|
|
|
32
32
|
npm install -g @andromarces/agent-loops
|
|
33
33
|
```
|
|
34
34
|
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
`
|
|
45
|
-
`pnpm
|
|
46
|
-
`agent-loop` with `node "<repo>/src/cli.mjs"` and quote the
|
|
47
|
-
a path with spaces works, or run `pnpm agent-loop` from the
|
|
35
|
+
Then run `agent-loop install` once to write the harness entry points and guards
|
|
36
|
+
at user scope.
|
|
37
|
+
|
|
38
|
+
From a clone, run `npm link` in the clone, then `agent-loop install`. `npm link`
|
|
39
|
+
writes the bin shim to the global prefix bin directory and points the rendered
|
|
40
|
+
entries at the clone. That directory must be on PATH for `agent-loop` to
|
|
41
|
+
resolve: it is the `npm prefix -g` directory on Windows and
|
|
42
|
+
`$(npm prefix -g)/bin` on macOS and Linux. After the clone moves, run
|
|
43
|
+
`npm link --force` from the new location, then `agent-loop install` again; a
|
|
44
|
+
plain `npm link` fails with `EEXIST` on Windows when a shim already exists.
|
|
45
|
+
`pnpm link` is not a supported path: pnpm 12 `link` has no global mode. Without a
|
|
46
|
+
link, replace `agent-loop` with `node "<repo>/src/cli.mjs"` and quote the
|
|
47
|
+
repository path so a path with spaces works, or run `pnpm agent-loop` from the
|
|
48
|
+
repository root.
|
|
48
49
|
|
|
49
50
|
## Starting a run
|
|
50
51
|
|
|
51
|
-
The first dispatch carries the init flags, including
|
|
52
|
-
|
|
52
|
+
The first dispatch carries the init flags, including the required
|
|
53
|
+
`--parent-session`. The subcommand refuses an init without it, so an interactive
|
|
54
|
+
run is never left unguarded:
|
|
53
55
|
|
|
54
56
|
```bash
|
|
55
57
|
printf '%s' "<first child prompt>" | agent-loop role dispatch \
|
|
@@ -147,9 +149,10 @@ printf '%s' '{"changed":"...","verified":"...","deferred":"...","notDone":"...",
|
|
|
147
149
|
terminal and the subcommand rejects further operations, including `abort`.
|
|
148
150
|
Report the failure and the modified paths from the envelope, then stop.
|
|
149
151
|
- Every run ends in exactly one terminal lifecycle: `finished`, `aborted`, or
|
|
150
|
-
`halted`. The parent-edit guard (#57, Claude Code, Codex CLI,
|
|
151
|
-
releases on any of them
|
|
152
|
-
|
|
152
|
+
`halted`. The parent-edit guard (#57, Claude Code, Codex CLI, Copilot CLI,
|
|
153
|
+
OpenCode, and Antigravity CLI) releases on any of them. Every `agent-loop role`
|
|
154
|
+
init requires `--parent-session`, so the guard always has a parent to match.
|
|
155
|
+
The headless `agent-loop` command is the explicit unguarded path.
|
|
153
156
|
|
|
154
157
|
## Recovery after compaction or restart
|
|
155
158
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@andromarces/agent-loops",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.3.0",
|
|
4
4
|
"private": false,
|
|
5
5
|
"description": "Run a task loop across several CLI coding agents.",
|
|
6
6
|
"homepage": "https://github.com/andromarces/agent-loops#readme",
|
|
@@ -51,5 +51,5 @@
|
|
|
51
51
|
"engines": {
|
|
52
52
|
"node": ">=22"
|
|
53
53
|
},
|
|
54
|
-
"packageManager": "pnpm@12.
|
|
54
|
+
"packageManager": "pnpm@12.6.0"
|
|
55
55
|
}
|
package/src/agents/agy.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { parseJson } from "../lib/json.mjs";
|
|
2
2
|
import { exec } from "../lib/exec.mjs";
|
|
3
|
+
import { setMainLoopUsage } from "./shared.mjs";
|
|
3
4
|
|
|
4
5
|
export async function runAgy(state, prompt, options = {}) {
|
|
5
6
|
const { cwd, readOnly, timeout, signal, role } = options;
|
|
@@ -30,17 +31,7 @@ export async function runAgy(state, prompt, options = {}) {
|
|
|
30
31
|
}
|
|
31
32
|
|
|
32
33
|
state.sessionId = result.conversation_id;
|
|
33
|
-
|
|
34
|
+
setMainLoopUsage(state, result?.usage);
|
|
34
35
|
|
|
35
36
|
return String(result.response ?? "").trim();
|
|
36
37
|
}
|
|
37
|
-
|
|
38
|
-
function setUsage(state, result) {
|
|
39
|
-
const usage = {};
|
|
40
|
-
if (result?.usage) usage.mainLoop = result.usage;
|
|
41
|
-
if (Object.keys(usage).length > 0) {
|
|
42
|
-
state.usage = usage;
|
|
43
|
-
} else {
|
|
44
|
-
delete state.usage;
|
|
45
|
-
}
|
|
46
|
-
}
|
package/src/agents/codex.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { parseJsonLines } from "../lib/json.mjs";
|
|
2
2
|
import { exec } from "../lib/exec.mjs";
|
|
3
|
+
import { resumeMismatchError, setMainLoopUsage } from "./shared.mjs";
|
|
3
4
|
|
|
4
5
|
export async function runCodex(state, prompt, options = {}) {
|
|
5
6
|
const { cwd, readOnly, timeout, signal, role } = options;
|
|
@@ -36,20 +37,11 @@ export async function runCodex(state, prompt, options = {}) {
|
|
|
36
37
|
}
|
|
37
38
|
|
|
38
39
|
if (state.sessionId && returnedId !== state.sessionId) {
|
|
39
|
-
throw
|
|
40
|
-
[
|
|
41
|
-
"Codex did not resume the expected thread.",
|
|
42
|
-
`Expected: ${state.sessionId}`,
|
|
43
|
-
`Received: ${returnedId}`,
|
|
44
|
-
].join("\n"),
|
|
45
|
-
);
|
|
40
|
+
throw resumeMismatchError("Codex", "thread", state.sessionId, returnedId);
|
|
46
41
|
}
|
|
47
42
|
|
|
48
43
|
state.sessionId = returnedId;
|
|
49
|
-
|
|
50
|
-
state,
|
|
51
|
-
events.find((event) => event.type === "turn.completed"),
|
|
52
|
-
);
|
|
44
|
+
setMainLoopUsage(state, events.find((event) => event.type === "turn.completed")?.usage);
|
|
53
45
|
|
|
54
46
|
const messages = events
|
|
55
47
|
.filter((event) => event.type === "item.completed" && event.item?.type === "agent_message")
|
|
@@ -62,12 +54,3 @@ export async function runCodex(state, prompt, options = {}) {
|
|
|
62
54
|
|
|
63
55
|
return String(messages.at(-1)).trim();
|
|
64
56
|
}
|
|
65
|
-
|
|
66
|
-
/** Sets top-level turn usage, or removes stale usage when Codex omits it. */
|
|
67
|
-
function setUsage(state, completedTurn) {
|
|
68
|
-
if (completedTurn?.usage) {
|
|
69
|
-
state.usage = { mainLoop: completedTurn.usage };
|
|
70
|
-
} else {
|
|
71
|
-
delete state.usage;
|
|
72
|
-
}
|
|
73
|
-
}
|
package/src/agents/copilot.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { randomUUID } from "node:crypto";
|
|
2
2
|
import { parseJsonLines } from "../lib/json.mjs";
|
|
3
3
|
import { exec } from "../lib/exec.mjs";
|
|
4
|
+
import { resumeMismatchError, setMainLoopUsage } from "./shared.mjs";
|
|
4
5
|
|
|
5
6
|
export async function runCopilot(state, prompt, options = {}) {
|
|
6
7
|
const { cwd, readOnly, timeout, signal, role } = options;
|
|
@@ -29,13 +30,13 @@ export async function runCopilot(state, prompt, options = {}) {
|
|
|
29
30
|
({ stdout } = await exec("copilot", args, { cwd, input: prompt, timeout, signal, role }));
|
|
30
31
|
} catch (error) {
|
|
31
32
|
const failed = parseJsonLines(error?.stdout ?? "");
|
|
32
|
-
|
|
33
|
+
setMainLoopUsage(state, objectUsage(findResultEvent(failed)));
|
|
33
34
|
throw error;
|
|
34
35
|
}
|
|
35
36
|
|
|
36
37
|
const events = parseJsonLines(stdout);
|
|
37
38
|
const resultEvent = findResultEvent(events);
|
|
38
|
-
|
|
39
|
+
setMainLoopUsage(state, objectUsage(resultEvent));
|
|
39
40
|
|
|
40
41
|
const returnedId = resultEvent?.sessionId ?? resultEvent?.session_id;
|
|
41
42
|
if (!returnedId) {
|
|
@@ -43,13 +44,7 @@ export async function runCopilot(state, prompt, options = {}) {
|
|
|
43
44
|
}
|
|
44
45
|
|
|
45
46
|
if (requestedSessionId && requestedSessionId !== returnedId) {
|
|
46
|
-
throw
|
|
47
|
-
[
|
|
48
|
-
"Copilot did not resume the expected session.",
|
|
49
|
-
`Expected: ${requestedSessionId}`,
|
|
50
|
-
`Received: ${returnedId}`,
|
|
51
|
-
].join("\n"),
|
|
52
|
-
);
|
|
47
|
+
throw resumeMismatchError("Copilot", "session", requestedSessionId, returnedId);
|
|
53
48
|
}
|
|
54
49
|
|
|
55
50
|
state.sessionId = returnedId;
|
|
@@ -76,11 +71,8 @@ function readAssistantMessage(event) {
|
|
|
76
71
|
return typeof content === "string" ? content : "";
|
|
77
72
|
}
|
|
78
73
|
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
delete state.usage;
|
|
74
|
+
/** Copilot reports usage as an object; any other shape counts as absent. */
|
|
75
|
+
function objectUsage(resultEvent) {
|
|
76
|
+
const usage = resultEvent?.usage;
|
|
77
|
+
return usage && typeof usage === "object" ? usage : undefined;
|
|
86
78
|
}
|
package/src/agents/opencode.mjs
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { parseJsonLines } from "../lib/json.mjs";
|
|
2
2
|
import { exec } from "../lib/exec.mjs";
|
|
3
3
|
import { logInfo } from "../lib/log.mjs";
|
|
4
|
+
import { resumeMismatchError } from "./shared.mjs";
|
|
4
5
|
|
|
5
6
|
// The built-in plan agent can launch explore and general subagents through the `subagent`
|
|
6
7
|
// action. They inherit the session model, so a read-only turn spends the role model budget
|
|
@@ -59,13 +60,7 @@ export async function runOpenCode(state, prompt, options = {}) {
|
|
|
59
60
|
}
|
|
60
61
|
|
|
61
62
|
if (state.sessionId && state.sessionId !== sessionId) {
|
|
62
|
-
throw
|
|
63
|
-
[
|
|
64
|
-
`opencode did not resume the expected session.`,
|
|
65
|
-
`Expected: ${state.sessionId}`,
|
|
66
|
-
`Received: ${sessionId}`,
|
|
67
|
-
].join("\n"),
|
|
68
|
-
);
|
|
63
|
+
throw resumeMismatchError("opencode", "session", state.sessionId, sessionId);
|
|
69
64
|
}
|
|
70
65
|
|
|
71
66
|
state.sessionId = sessionId;
|