agent-loop-tool 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_loop_tool-0.1.0/LICENSE +21 -0
- agent_loop_tool-0.1.0/PKG-INFO +241 -0
- agent_loop_tool-0.1.0/README.md +222 -0
- agent_loop_tool-0.1.0/pyproject.toml +35 -0
- agent_loop_tool-0.1.0/setup.cfg +4 -0
- agent_loop_tool-0.1.0/src/agent_loop/__init__.py +1 -0
- agent_loop_tool-0.1.0/src/agent_loop/audit.py +128 -0
- agent_loop_tool-0.1.0/src/agent_loop/cli.py +354 -0
- agent_loop_tool-0.1.0/src/agent_loop/git_ops.py +127 -0
- agent_loop_tool-0.1.0/src/agent_loop/orchestrator.py +314 -0
- agent_loop_tool-0.1.0/src/agent_loop/plan_init.py +357 -0
- agent_loop_tool-0.1.0/src/agent_loop/providers/__init__.py +30 -0
- agent_loop_tool-0.1.0/src/agent_loop/providers/antigravity.py +19 -0
- agent_loop_tool-0.1.0/src/agent_loop/providers/base.py +147 -0
- agent_loop_tool-0.1.0/src/agent_loop/providers/claude.py +19 -0
- agent_loop_tool-0.1.0/src/agent_loop/providers/codex.py +21 -0
- agent_loop_tool-0.1.0/src/agent_loop/providers/grok.py +18 -0
- agent_loop_tool-0.1.0/src/agent_loop/safety.py +105 -0
- agent_loop_tool-0.1.0/src/agent_loop/state.py +247 -0
- agent_loop_tool-0.1.0/src/agent_loop_tool.egg-info/PKG-INFO +241 -0
- agent_loop_tool-0.1.0/src/agent_loop_tool.egg-info/SOURCES.txt +32 -0
- agent_loop_tool-0.1.0/src/agent_loop_tool.egg-info/dependency_links.txt +1 -0
- agent_loop_tool-0.1.0/src/agent_loop_tool.egg-info/entry_points.txt +2 -0
- agent_loop_tool-0.1.0/src/agent_loop_tool.egg-info/top_level.txt +1 -0
- agent_loop_tool-0.1.0/tests/test_audit.py +164 -0
- agent_loop_tool-0.1.0/tests/test_cli.py +806 -0
- agent_loop_tool-0.1.0/tests/test_examples.py +30 -0
- agent_loop_tool-0.1.0/tests/test_git_integration.py +492 -0
- agent_loop_tool-0.1.0/tests/test_git_ops.py +69 -0
- agent_loop_tool-0.1.0/tests/test_orchestrator.py +445 -0
- agent_loop_tool-0.1.0/tests/test_plan_init.py +641 -0
- agent_loop_tool-0.1.0/tests/test_providers.py +394 -0
- agent_loop_tool-0.1.0/tests/test_safety.py +30 -0
- agent_loop_tool-0.1.0/tests/test_state.py +381 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Dwaraka Ramana Turlapati
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agent-loop-tool
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Checkpoint-gated developer/reviewer loop across Claude Code, Codex, Grok, and Antigravity CLIs.
|
|
5
|
+
Author: Dwaraka Ramana Turlapati
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/dcurioustech/agent-loop
|
|
8
|
+
Project-URL: Issues, https://github.com/dcurioustech/agent-loop/issues
|
|
9
|
+
Keywords: ai,agents,claude,codex,code-review,cli
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Software Development
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Dynamic: license-file
|
|
19
|
+
|
|
20
|
+
# agent-loop
|
|
21
|
+
|
|
22
|
+
A checkpoint-gated developer/reviewer loop. One coding-agent CLI plays **developer** and writes the code for the current checkpoint; another plays **reviewer** and approves or sends it back. The loop commits per checkpoint, refuses to run on `main`/`master`, and halts after a configurable number of failed review attempts.
|
|
23
|
+
|
|
24
|
+
Originally extracted from a Flutter project's `run_loop.sh`. Project-agnostic: build / test / lint commands are declared per project in `plan_checkpoints.json`.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
The PyPI distribution is `agent-loop-tool`; the installed command is
|
|
29
|
+
`agent-loop`. Until the first PyPI release, install from the repository
|
|
30
|
+
(requires repository access):
|
|
31
|
+
|
|
32
|
+
```console
|
|
33
|
+
pipx install git+https://github.com/dcurioustech/agent-loop.git
|
|
34
|
+
# or
|
|
35
|
+
uv tool install git+https://github.com/dcurioustech/agent-loop.git
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
After the first PyPI release, use `pipx install agent-loop-tool` or
|
|
39
|
+
`uv tool install agent-loop-tool`.
|
|
40
|
+
|
|
41
|
+
Requires at least one of these CLIs on PATH, depending on the roles you pick:
|
|
42
|
+
|
|
43
|
+
| Provider | Install |
|
|
44
|
+
|----------|----------------------------------------|
|
|
45
|
+
| claude | <https://docs.claude.com/claude-code> |
|
|
46
|
+
| codex | <https://github.com/openai/codex> |
|
|
47
|
+
| grok | <https://docs.x.ai/docs/grok-cli> |
|
|
48
|
+
| antigravity | <https://antigravity.google/docs/cli-overview> (binary: `agy`) |
|
|
49
|
+
|
|
50
|
+
## Usage
|
|
51
|
+
|
|
52
|
+
In a consumer repo containing `plan_checkpoints.json`:
|
|
53
|
+
|
|
54
|
+
```
|
|
55
|
+
agent-loop validate # schema-check the state file
|
|
56
|
+
agent-loop status # one-line summary per checkpoint
|
|
57
|
+
agent-loop run # developer=claude, reviewer=codex (defaults)
|
|
58
|
+
agent-loop run --developer codex --reviewer claude
|
|
59
|
+
agent-loop run --developer grok --reviewer antigravity
|
|
60
|
+
agent-loop run --developer antigravity --reviewer claude
|
|
61
|
+
agent-loop run --developer-model claude-opus-4-8 --reviewer-model gpt-5-codex
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
### Choosing the model
|
|
65
|
+
|
|
66
|
+
`--developer` / `--reviewer` pick the CLI; the *model* is resolved per role with this
|
|
67
|
+
precedence (first match wins):
|
|
68
|
+
|
|
69
|
+
1. `--developer-model` / `--reviewer-model` on the command line
|
|
70
|
+
2. a `models` block in `plan_checkpoints.json` (see schema below)
|
|
71
|
+
3. whatever default the CLI itself resolves (its config file / env / built-in)
|
|
72
|
+
|
|
73
|
+
So if you set nothing, each CLI keeps using its own default model — the loop never
|
|
74
|
+
overrides it. Model names are provider-specific, so a value pinned for one role only
|
|
75
|
+
makes sense for the CLI you assigned to that role. Under the hood the model is passed as
|
|
76
|
+
`--model <name>` (claude/grok) or `-m <name>` (codex/antigravity).
|
|
77
|
+
|
|
78
|
+
Safety envs (per provider, off by default):
|
|
79
|
+
|
|
80
|
+
```
|
|
81
|
+
ALLOW_AUTO_MODE_CLAUDE=1 # claude --dangerously-skip-permissions
|
|
82
|
+
ALLOW_AUTO_MODE_CODEX=1 # codex exec --dangerously-bypass-approvals-and-sandbox
|
|
83
|
+
ALLOW_AUTO_MODE_GROK=1 # grok --always-approve
|
|
84
|
+
ALLOW_AUTO_MODE_ANTIGRAVITY=1 # agy --dangerously-skip-permissions
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
The loop refuses to start unless the assigned provider's auto mode gate is set, so unattended runs cannot stall on a permission prompt.
|
|
88
|
+
|
|
89
|
+
## Run logs and the audit trail
|
|
90
|
+
|
|
91
|
+
Every run writes a log to `loop_<YYYYMMDD>_<HHMMSS>.log`. Where that lands is resolved in this order:
|
|
92
|
+
|
|
93
|
+
1. `--log-dir <path>` — explicit CLI flag, wins over everything.
|
|
94
|
+
2. `LOG_DIR=<path>` — environment override.
|
|
95
|
+
3. `<repo-root>/logs` — the default, resolved from the git root regardless of your current working directory (falls back to a relative `logs/` outside a git repo).
|
|
96
|
+
|
|
97
|
+
### Logs are committed as audit artifacts
|
|
98
|
+
|
|
99
|
+
Logs under the repository are **tracked and committed**, not ignored. The loop commits them alongside code at each checkpoint (`built`, `revision`, `approved`) and on every exit path (`audit: halted`, `audit: run failed`, `audit: completed run`), so a run's history survives in git even when it fails.
|
|
100
|
+
|
|
101
|
+
Two consequences worth knowing:
|
|
102
|
+
|
|
103
|
+
- **Each commit holds a partial log.** The log is still being appended to while the loop commits it, so a checkpoint commit captures the log *as of that moment*. The closing `audit:` commit flushes the tail. Bytes written after that land in the next run's first commit — partial by design, never lost.
|
|
104
|
+
- **A modified log does not block the next run.** The preflight worktree check ignores changes under the log directory (and only there); real source changes still refuse to start the loop.
|
|
105
|
+
|
|
106
|
+
Point `--log-dir` outside the repository and the loop warns that logs will not be committed, then runs normally.
|
|
107
|
+
|
|
108
|
+
### What the log contains
|
|
109
|
+
|
|
110
|
+
Agent stdout and stderr are streamed into the log as well as to your terminal, so the log can hold the agents' actual working output, not just the loop's own bookkeeping — subject to `--audit-level`, below. Alongside it the loop emits structured, one-line JSON events that are easy to grep or parse:
|
|
111
|
+
|
|
112
|
+
| Event | Emitted when |
|
|
113
|
+
| --- | --- |
|
|
114
|
+
| `AGENT_TRACE` | A developer/reviewer invocation starts, its prompt, and its result (`action` is `start`, `prompt`, or `finish`) |
|
|
115
|
+
| `REVIEW_COMMENT` | The reviewer returns, carrying its notes and resulting status |
|
|
116
|
+
| `APPROVAL_COMMENT` | A checkpoint is approved, or skipped because it already was |
|
|
117
|
+
| `COMMIT_STATEMENT` | A commit is made (with hash) or found unnecessary |
|
|
118
|
+
| `RUN_OUTCOME` | The run ends: `completed`, `halted`, or `failed` |
|
|
119
|
+
|
|
120
|
+
```console
|
|
121
|
+
$ grep RUN_OUTCOME logs/loop_20260816_101500.log
|
|
122
|
+
[agent-loop] RUN_OUTCOME {"branch": "feature-x", "event": "RUN_OUTCOME", "outcome": "completed"}
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
### `--audit-level`: how much of that ends up in git
|
|
126
|
+
|
|
127
|
+
Because logs are committed, anything an agent prints — or writes into a checkpoint's `review_notes` — becomes part of permanent git history the moment its commit is made. `--audit-level` controls how much of that actually reaches the log:
|
|
128
|
+
|
|
129
|
+
| Level | Raw agent stdout/stderr | Prompts, review notes, comments | Structured events |
|
|
130
|
+
| --- | --- | --- | --- |
|
|
131
|
+
| `full` | logged as-is | logged as-is | logged |
|
|
132
|
+
| `redacted` | scrubbed for known secret shapes first | scrubbed first | logged |
|
|
133
|
+
| `off` (default) | not logged at all | replaced with a `<suppressed: N chars>` placeholder | logged |
|
|
134
|
+
|
|
135
|
+
Resolution order: `--audit-level <level>` flag, then `AGENT_LOOP_AUDIT_LEVEL` env var, then `off`.
|
|
136
|
+
|
|
137
|
+
```console
|
|
138
|
+
agent-loop run --audit-level full # everything, unfiltered
|
|
139
|
+
agent-loop run --audit-level redacted # best-effort secret scrub
|
|
140
|
+
agent-loop run # off — structured log events only (default)
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
At every level, agents still receive the real, unredacted prompt — `--audit-level` only changes what gets *logged*, never what an agent is told to do. The state file retains `review_notes` as written by the agents and is committed with checkpoint changes, regardless of audit level.
|
|
144
|
+
|
|
145
|
+
**`redacted` is a best-effort net, not a guarantee.** It catches known secret *shapes* — AWS access keys, GitHub/Slack/OpenAI tokens, bearer tokens, PEM private-key blocks, `key: value`-style assignments — via regex over live, line-streamed subprocess output. It cannot catch a project's own custom secret formats, and a secret split across two flushed writes can slip through. Treat committed logs as something a human should skim before pushing, not as pre-cleared for a public remote. `off` suppresses raw content in the log, but does not redact the state file.
|
|
146
|
+
|
|
147
|
+
The `--log-dir` worktree exemption above only ever tolerates changes to log *files*; it has no bearing on what those files contain — that's entirely `--audit-level`'s job.
|
|
148
|
+
|
|
149
|
+
## Generating a plan (`agent-loop init`)
|
|
150
|
+
|
|
151
|
+
`agent-loop init` turns a feature description into a schema-valid `plan_checkpoints.json`
|
|
152
|
+
by asking a coding-agent CLI to break it into ordered, independently reviewable
|
|
153
|
+
checkpoints. It runs the provider once, non-interactively, captures its output, and
|
|
154
|
+
only writes the state file after the result parses as JSON and passes the same
|
|
155
|
+
validation `load_state` applies — nothing is written on a provider error, a timeout,
|
|
156
|
+
malformed output, or a schema failure.
|
|
157
|
+
|
|
158
|
+
Plain-English input:
|
|
159
|
+
|
|
160
|
+
```
|
|
161
|
+
agent-loop init "Add a login page with email/password auth and a logout button"
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Markdown input (e.g. an existing design doc):
|
|
165
|
+
|
|
166
|
+
```
|
|
167
|
+
agent-loop init --feature-file docs/feature.md
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Either form accepts:
|
|
171
|
+
|
|
172
|
+
| Flag | Default |
|
|
173
|
+
|----------------|-----------------------------------------------------------------------|
|
|
174
|
+
| `--state` | `plan_checkpoints.json` |
|
|
175
|
+
| `--branch` | sanitized `feature/<slug>` derived from the feature text (or the `--feature-file` filename) |
|
|
176
|
+
| `--plan-file` | `docs/implementation_plan.md` for plain-English input, or the `--feature-file` path for Markdown input |
|
|
177
|
+
| `--provider` | `claude` — any provider from `agent-loop run`'s table can generate the plan |
|
|
178
|
+
| `--model` | the provider CLI's own default |
|
|
179
|
+
| `--timeout` | `1800` seconds |
|
|
180
|
+
| `--force` | off — refuses to overwrite an existing `--state` file |
|
|
181
|
+
|
|
182
|
+
```
|
|
183
|
+
agent-loop init "Add CSV export to the reports page" \
|
|
184
|
+
--provider codex --model gpt-5-codex --timeout 600 --branch feature/csv-export
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
By default `init` refuses to touch an existing state file so you don't accidentally
|
|
188
|
+
clobber checkpoint progress; pass `--force` to regenerate and overwrite it. A target
|
|
189
|
+
branch of `main`/`master` is rejected before the provider is ever invoked, same as
|
|
190
|
+
`agent-loop run`. Generated checkpoints always start `pending` with `attempts: 0` and
|
|
191
|
+
empty `review_notes`, regardless of what the provider returned for those fields.
|
|
192
|
+
|
|
193
|
+
A freshly generated file is immediately usable:
|
|
194
|
+
|
|
195
|
+
```
|
|
196
|
+
agent-loop init "Add a login page" && agent-loop validate && agent-loop status
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
## Bootstrapping a new consumer repo
|
|
200
|
+
|
|
201
|
+
Download the example as your starting point and edit `branch`, `plan_file`, and the project block to match your stack:
|
|
202
|
+
|
|
203
|
+
```
|
|
204
|
+
curl -L https://raw.githubusercontent.com/dcurioustech/agent-loop/main/examples/plan_checkpoints.example.json \
|
|
205
|
+
-o plan_checkpoints.json
|
|
206
|
+
agent-loop validate
|
|
207
|
+
```
|
|
208
|
+
|
|
209
|
+
The example covers all three checkpoint statuses (`pending` / `built` / `approved`) and shows a per-checkpoint `test_cmd` override on top of the project-level default.
|
|
210
|
+
|
|
211
|
+
## `plan_checkpoints.json` schema
|
|
212
|
+
|
|
213
|
+
```jsonc
|
|
214
|
+
{
|
|
215
|
+
"plan_file": "docs/implementation_plan.md",
|
|
216
|
+
"branch": "feature-branch",
|
|
217
|
+
"project": {
|
|
218
|
+
"build_cmd": "flutter build web --release",
|
|
219
|
+
"test_cmd": "flutter test",
|
|
220
|
+
"lint_cmd": "flutter analyze",
|
|
221
|
+
"verify_in_review": true
|
|
222
|
+
},
|
|
223
|
+
"models": { // optional; per-role model, overridden by CLI flags
|
|
224
|
+
"developer": "claude-opus-4-8",
|
|
225
|
+
"reviewer": "gpt-5-codex"
|
|
226
|
+
},
|
|
227
|
+
"checkpoints": [
|
|
228
|
+
{
|
|
229
|
+
"id": "phase0",
|
|
230
|
+
"name": "...",
|
|
231
|
+
"status": "pending", // pending | built | approved
|
|
232
|
+
"scope": "...",
|
|
233
|
+
"exit_criteria": ["..."], // non-empty; each entry a non-empty string
|
|
234
|
+
"attempts": 0, // non-negative integer
|
|
235
|
+
"review_notes": ""
|
|
236
|
+
}
|
|
237
|
+
]
|
|
238
|
+
}
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
Per-checkpoint `build_cmd` / `test_cmd` / `lint_cmd` overrides are also supported. The reviewer prompt includes these commands so the agent knows how to verify.
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
# agent-loop
|
|
2
|
+
|
|
3
|
+
A checkpoint-gated developer/reviewer loop. One coding-agent CLI plays **developer** and writes the code for the current checkpoint; another plays **reviewer** and approves or sends it back. The loop commits per checkpoint, refuses to run on `main`/`master`, and halts after a configurable number of failed review attempts.
|
|
4
|
+
|
|
5
|
+
Originally extracted from a Flutter project's `run_loop.sh`. Project-agnostic: build / test / lint commands are declared per project in `plan_checkpoints.json`.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
The PyPI distribution is `agent-loop-tool`; the installed command is
|
|
10
|
+
`agent-loop`. Until the first PyPI release, install from the repository
|
|
11
|
+
(requires repository access):
|
|
12
|
+
|
|
13
|
+
```console
|
|
14
|
+
pipx install git+https://github.com/dcurioustech/agent-loop.git
|
|
15
|
+
# or
|
|
16
|
+
uv tool install git+https://github.com/dcurioustech/agent-loop.git
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
After the first PyPI release, use `pipx install agent-loop-tool` or
|
|
20
|
+
`uv tool install agent-loop-tool`.
|
|
21
|
+
|
|
22
|
+
Requires at least one of these CLIs on PATH, depending on the roles you pick:
|
|
23
|
+
|
|
24
|
+
| Provider | Install |
|
|
25
|
+
|----------|----------------------------------------|
|
|
26
|
+
| claude | <https://docs.claude.com/claude-code> |
|
|
27
|
+
| codex | <https://github.com/openai/codex> |
|
|
28
|
+
| grok | <https://docs.x.ai/docs/grok-cli> |
|
|
29
|
+
| antigravity | <https://antigravity.google/docs/cli-overview> (binary: `agy`) |
|
|
30
|
+
|
|
31
|
+
## Usage
|
|
32
|
+
|
|
33
|
+
In a consumer repo containing `plan_checkpoints.json`:
|
|
34
|
+
|
|
35
|
+
```
|
|
36
|
+
agent-loop validate # schema-check the state file
|
|
37
|
+
agent-loop status # one-line summary per checkpoint
|
|
38
|
+
agent-loop run # developer=claude, reviewer=codex (defaults)
|
|
39
|
+
agent-loop run --developer codex --reviewer claude
|
|
40
|
+
agent-loop run --developer grok --reviewer antigravity
|
|
41
|
+
agent-loop run --developer antigravity --reviewer claude
|
|
42
|
+
agent-loop run --developer-model claude-opus-4-8 --reviewer-model gpt-5-codex
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
### Choosing the model
|
|
46
|
+
|
|
47
|
+
`--developer` / `--reviewer` pick the CLI; the *model* is resolved per role with this
|
|
48
|
+
precedence (first match wins):
|
|
49
|
+
|
|
50
|
+
1. `--developer-model` / `--reviewer-model` on the command line
|
|
51
|
+
2. a `models` block in `plan_checkpoints.json` (see schema below)
|
|
52
|
+
3. whatever default the CLI itself resolves (its config file / env / built-in)
|
|
53
|
+
|
|
54
|
+
So if you set nothing, each CLI keeps using its own default model — the loop never
|
|
55
|
+
overrides it. Model names are provider-specific, so a value pinned for one role only
|
|
56
|
+
makes sense for the CLI you assigned to that role. Under the hood the model is passed as
|
|
57
|
+
`--model <name>` (claude/grok) or `-m <name>` (codex/antigravity).
|
|
58
|
+
|
|
59
|
+
Safety envs (per provider, off by default):
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
ALLOW_AUTO_MODE_CLAUDE=1 # claude --dangerously-skip-permissions
|
|
63
|
+
ALLOW_AUTO_MODE_CODEX=1 # codex exec --dangerously-bypass-approvals-and-sandbox
|
|
64
|
+
ALLOW_AUTO_MODE_GROK=1 # grok --always-approve
|
|
65
|
+
ALLOW_AUTO_MODE_ANTIGRAVITY=1 # agy --dangerously-skip-permissions
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
The loop refuses to start unless the assigned provider's auto mode gate is set, so unattended runs cannot stall on a permission prompt.
|
|
69
|
+
|
|
70
|
+
## Run logs and the audit trail
|
|
71
|
+
|
|
72
|
+
Every run writes a log to `loop_<YYYYMMDD>_<HHMMSS>.log`. Where that lands is resolved in this order:
|
|
73
|
+
|
|
74
|
+
1. `--log-dir <path>` — explicit CLI flag, wins over everything.
|
|
75
|
+
2. `LOG_DIR=<path>` — environment override.
|
|
76
|
+
3. `<repo-root>/logs` — the default, resolved from the git root regardless of your current working directory (falls back to a relative `logs/` outside a git repo).
|
|
77
|
+
|
|
78
|
+
### Logs are committed as audit artifacts
|
|
79
|
+
|
|
80
|
+
Logs under the repository are **tracked and committed**, not ignored. The loop commits them alongside code at each checkpoint (`built`, `revision`, `approved`) and on every exit path (`audit: halted`, `audit: run failed`, `audit: completed run`), so a run's history survives in git even when it fails.
|
|
81
|
+
|
|
82
|
+
Two consequences worth knowing:
|
|
83
|
+
|
|
84
|
+
- **Each commit holds a partial log.** The log is still being appended to while the loop commits it, so a checkpoint commit captures the log *as of that moment*. The closing `audit:` commit flushes the tail. Bytes written after that land in the next run's first commit — partial by design, never lost.
|
|
85
|
+
- **A modified log does not block the next run.** The preflight worktree check ignores changes under the log directory (and only there); real source changes still refuse to start the loop.
|
|
86
|
+
|
|
87
|
+
Point `--log-dir` outside the repository and the loop warns that logs will not be committed, then runs normally.
|
|
88
|
+
|
|
89
|
+
### What the log contains
|
|
90
|
+
|
|
91
|
+
Agent stdout and stderr are streamed into the log as well as to your terminal, so the log can hold the agents' actual working output, not just the loop's own bookkeeping — subject to `--audit-level`, below. Alongside it the loop emits structured, one-line JSON events that are easy to grep or parse:
|
|
92
|
+
|
|
93
|
+
| Event | Emitted when |
|
|
94
|
+
| --- | --- |
|
|
95
|
+
| `AGENT_TRACE` | A developer/reviewer invocation starts, its prompt, and its result (`action` is `start`, `prompt`, or `finish`) |
|
|
96
|
+
| `REVIEW_COMMENT` | The reviewer returns, carrying its notes and resulting status |
|
|
97
|
+
| `APPROVAL_COMMENT` | A checkpoint is approved, or skipped because it already was |
|
|
98
|
+
| `COMMIT_STATEMENT` | A commit is made (with hash) or found unnecessary |
|
|
99
|
+
| `RUN_OUTCOME` | The run ends: `completed`, `halted`, or `failed` |
|
|
100
|
+
|
|
101
|
+
```console
|
|
102
|
+
$ grep RUN_OUTCOME logs/loop_20260816_101500.log
|
|
103
|
+
[agent-loop] RUN_OUTCOME {"branch": "feature-x", "event": "RUN_OUTCOME", "outcome": "completed"}
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
### `--audit-level`: how much of that ends up in git
|
|
107
|
+
|
|
108
|
+
Because logs are committed, anything an agent prints — or writes into a checkpoint's `review_notes` — becomes part of permanent git history the moment its commit is made. `--audit-level` controls how much of that actually reaches the log:
|
|
109
|
+
|
|
110
|
+
| Level | Raw agent stdout/stderr | Prompts, review notes, comments | Structured events |
|
|
111
|
+
| --- | --- | --- | --- |
|
|
112
|
+
| `full` | logged as-is | logged as-is | logged |
|
|
113
|
+
| `redacted` | scrubbed for known secret shapes first | scrubbed first | logged |
|
|
114
|
+
| `off` (default) | not logged at all | replaced with a `<suppressed: N chars>` placeholder | logged |
|
|
115
|
+
|
|
116
|
+
Resolution order: `--audit-level <level>` flag, then `AGENT_LOOP_AUDIT_LEVEL` env var, then `off`.
|
|
117
|
+
|
|
118
|
+
```console
|
|
119
|
+
agent-loop run --audit-level full # everything, unfiltered
|
|
120
|
+
agent-loop run --audit-level redacted # best-effort secret scrub
|
|
121
|
+
agent-loop run # off — structured log events only (default)
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
At every level, agents still receive the real, unredacted prompt — `--audit-level` only changes what gets *logged*, never what an agent is told to do. The state file retains `review_notes` as written by the agents and is committed with checkpoint changes, regardless of audit level.
|
|
125
|
+
|
|
126
|
+
**`redacted` is a best-effort net, not a guarantee.** It catches known secret *shapes* — AWS access keys, GitHub/Slack/OpenAI tokens, bearer tokens, PEM private-key blocks, `key: value`-style assignments — via regex over live, line-streamed subprocess output. It cannot catch a project's own custom secret formats, and a secret split across two flushed writes can slip through. Treat committed logs as something a human should skim before pushing, not as pre-cleared for a public remote. `off` suppresses raw content in the log, but does not redact the state file.
|
|
127
|
+
|
|
128
|
+
The `--log-dir` worktree exemption above only ever tolerates changes to log *files*; it has no bearing on what those files contain — that's entirely `--audit-level`'s job.
|
|
129
|
+
|
|
130
|
+
## Generating a plan (`agent-loop init`)
|
|
131
|
+
|
|
132
|
+
`agent-loop init` turns a feature description into a schema-valid `plan_checkpoints.json`
|
|
133
|
+
by asking a coding-agent CLI to break it into ordered, independently reviewable
|
|
134
|
+
checkpoints. It runs the provider once, non-interactively, captures its output, and
|
|
135
|
+
only writes the state file after the result parses as JSON and passes the same
|
|
136
|
+
validation `load_state` applies — nothing is written on a provider error, a timeout,
|
|
137
|
+
malformed output, or a schema failure.
|
|
138
|
+
|
|
139
|
+
Plain-English input:
|
|
140
|
+
|
|
141
|
+
```
|
|
142
|
+
agent-loop init "Add a login page with email/password auth and a logout button"
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
Markdown input (e.g. an existing design doc):
|
|
146
|
+
|
|
147
|
+
```
|
|
148
|
+
agent-loop init --feature-file docs/feature.md
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Either form accepts:
|
|
152
|
+
|
|
153
|
+
| Flag | Default |
|
|
154
|
+
|----------------|-----------------------------------------------------------------------|
|
|
155
|
+
| `--state` | `plan_checkpoints.json` |
|
|
156
|
+
| `--branch` | sanitized `feature/<slug>` derived from the feature text (or the `--feature-file` filename) |
|
|
157
|
+
| `--plan-file` | `docs/implementation_plan.md` for plain-English input, or the `--feature-file` path for Markdown input |
|
|
158
|
+
| `--provider` | `claude` — any provider from `agent-loop run`'s table can generate the plan |
|
|
159
|
+
| `--model` | the provider CLI's own default |
|
|
160
|
+
| `--timeout` | `1800` seconds |
|
|
161
|
+
| `--force` | off — refuses to overwrite an existing `--state` file |
|
|
162
|
+
|
|
163
|
+
```
|
|
164
|
+
agent-loop init "Add CSV export to the reports page" \
|
|
165
|
+
--provider codex --model gpt-5-codex --timeout 600 --branch feature/csv-export
|
|
166
|
+
```
|
|
167
|
+
|
|
168
|
+
By default `init` refuses to touch an existing state file so you don't accidentally
|
|
169
|
+
clobber checkpoint progress; pass `--force` to regenerate and overwrite it. A target
|
|
170
|
+
branch of `main`/`master` is rejected before the provider is ever invoked, same as
|
|
171
|
+
`agent-loop run`. Generated checkpoints always start `pending` with `attempts: 0` and
|
|
172
|
+
empty `review_notes`, regardless of what the provider returned for those fields.
|
|
173
|
+
|
|
174
|
+
A freshly generated file is immediately usable:
|
|
175
|
+
|
|
176
|
+
```
|
|
177
|
+
agent-loop init "Add a login page" && agent-loop validate && agent-loop status
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
## Bootstrapping a new consumer repo
|
|
181
|
+
|
|
182
|
+
Download the example as your starting point and edit `branch`, `plan_file`, and the project block to match your stack:
|
|
183
|
+
|
|
184
|
+
```
|
|
185
|
+
curl -L https://raw.githubusercontent.com/dcurioustech/agent-loop/main/examples/plan_checkpoints.example.json \
|
|
186
|
+
-o plan_checkpoints.json
|
|
187
|
+
agent-loop validate
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
The example covers all three checkpoint statuses (`pending` / `built` / `approved`) and shows a per-checkpoint `test_cmd` override on top of the project-level default.
|
|
191
|
+
|
|
192
|
+
## `plan_checkpoints.json` schema
|
|
193
|
+
|
|
194
|
+
```jsonc
|
|
195
|
+
{
|
|
196
|
+
"plan_file": "docs/implementation_plan.md",
|
|
197
|
+
"branch": "feature-branch",
|
|
198
|
+
"project": {
|
|
199
|
+
"build_cmd": "flutter build web --release",
|
|
200
|
+
"test_cmd": "flutter test",
|
|
201
|
+
"lint_cmd": "flutter analyze",
|
|
202
|
+
"verify_in_review": true
|
|
203
|
+
},
|
|
204
|
+
"models": { // optional; per-role model, overridden by CLI flags
|
|
205
|
+
"developer": "claude-opus-4-8",
|
|
206
|
+
"reviewer": "gpt-5-codex"
|
|
207
|
+
},
|
|
208
|
+
"checkpoints": [
|
|
209
|
+
{
|
|
210
|
+
"id": "phase0",
|
|
211
|
+
"name": "...",
|
|
212
|
+
"status": "pending", // pending | built | approved
|
|
213
|
+
"scope": "...",
|
|
214
|
+
"exit_criteria": ["..."], // non-empty; each entry a non-empty string
|
|
215
|
+
"attempts": 0, // non-negative integer
|
|
216
|
+
"review_notes": ""
|
|
217
|
+
}
|
|
218
|
+
]
|
|
219
|
+
}
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
Per-checkpoint `build_cmd` / `test_cmd` / `lint_cmd` overrides are also supported. The reviewer prompt includes these commands so the agent knows how to verify.
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "agent-loop-tool"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Checkpoint-gated developer/reviewer loop across Claude Code, Codex, Grok, and Antigravity CLIs."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
authors = [{ name = "Dwaraka Ramana Turlapati" }]
|
|
12
|
+
license = "MIT"
|
|
13
|
+
license-files = ["LICENSE"]
|
|
14
|
+
keywords = ["ai", "agents", "claude", "codex", "code-review", "cli"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Development Status :: 3 - Alpha",
|
|
17
|
+
"Environment :: Console",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Topic :: Software Development",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://github.com/dcurioustech/agent-loop"
|
|
25
|
+
Issues = "https://github.com/dcurioustech/agent-loop/issues"
|
|
26
|
+
|
|
27
|
+
[project.scripts]
|
|
28
|
+
agent-loop = "agent_loop.cli:main"
|
|
29
|
+
|
|
30
|
+
[tool.setuptools.packages.find]
|
|
31
|
+
where = ["src"]
|
|
32
|
+
|
|
33
|
+
[tool.pytest.ini_options]
|
|
34
|
+
testpaths = ["tests"]
|
|
35
|
+
addopts = "-q"
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Controls how much of an agent's raw output reaches the committed log.
|
|
2
|
+
|
|
3
|
+
Run logs are git-committed audit artifacts (see `git_ops.commit_checkpoint_changes`),
|
|
4
|
+
which makes anything written to them effectively permanent — much harder to walk
|
|
5
|
+
back than a `/tmp` file. `--audit-level` trades completeness of that record
|
|
6
|
+
against exposure of whatever the developer/reviewer agents happen to print or
|
|
7
|
+
write while doing their work:
|
|
8
|
+
|
|
9
|
+
- ``full``: nothing is touched. Agent output and agent-authored free text
|
|
10
|
+
(prompts, review notes) are logged exactly as produced.
|
|
11
|
+
- ``redacted``: known secret *shapes* (AWS keys, GitHub/Slack/OpenAI tokens,
|
|
12
|
+
bearer tokens, PEM private key blocks, `key: value`-style
|
|
13
|
+
assignments) are scrubbed before anything is printed or logged.
|
|
14
|
+
This is a best-effort net, not a guarantee — it cannot catch a
|
|
15
|
+
project's own custom secret formats, and it runs on live,
|
|
16
|
+
streamed subprocess output rather than a byte buffer, so a
|
|
17
|
+
secret split across two flushed writes can slip through.
|
|
18
|
+
- ``off``: raw agent output and agent-authored free text are not printed
|
|
19
|
+
or logged at all. Only structured event metadata (which agent
|
|
20
|
+
ran, what was approved, what was committed) reaches the log.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
|
|
27
|
+
AUDIT_LEVELS = ("full", "redacted", "off")
|
|
28
|
+
DEFAULT_AUDIT_LEVEL = "off"
|
|
29
|
+
|
|
30
|
+
_ENV_VAR = "AGENT_LOOP_AUDIT_LEVEL"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class InvalidAuditLevel(ValueError):
|
|
34
|
+
pass
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def default_audit_level() -> str:
|
|
38
|
+
"""The configured level: env override if set and valid, else the default."""
|
|
39
|
+
configured = os.environ.get(_ENV_VAR)
|
|
40
|
+
if not configured:
|
|
41
|
+
return DEFAULT_AUDIT_LEVEL
|
|
42
|
+
validate(configured)
|
|
43
|
+
return configured
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def validate(level: str) -> None:
|
|
47
|
+
if level not in AUDIT_LEVELS:
|
|
48
|
+
raise InvalidAuditLevel(
|
|
49
|
+
f"Invalid audit level {level!r}; must be one of {', '.join(AUDIT_LEVELS)}."
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_PRIVATE_KEY_BLOCK = re.compile(
|
|
54
|
+
r"-----BEGIN [A-Z ]*PRIVATE KEY-----.*?-----END [A-Z ]*PRIVATE KEY-----",
|
|
55
|
+
re.DOTALL,
|
|
56
|
+
)
|
|
57
|
+
_PRIVATE_KEY_BEGIN = re.compile(r"-----BEGIN [A-Z ]*PRIVATE KEY-----")
|
|
58
|
+
_PRIVATE_KEY_END = re.compile(r"-----END [A-Z ]*PRIVATE KEY-----")
|
|
59
|
+
|
|
60
|
+
# High-confidence secret shapes only — anything looser produces enough false
|
|
61
|
+
# positives to make "redacted" logs unreadable without meaningfully raising
|
|
62
|
+
# recall. checked in order; each substitution runs on the previous pass's output.
|
|
63
|
+
_LINE_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
|
|
64
|
+
("aws-access-key-id", re.compile(r"\bAKIA[0-9A-Z]{16}\b")),
|
|
65
|
+
("github-token", re.compile(r"\bgh[pousr]_[A-Za-z0-9]{36,}\b")),
|
|
66
|
+
("slack-token", re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b")),
|
|
67
|
+
("openai-key", re.compile(r"\bsk-[A-Za-z0-9]{20,}\b")),
|
|
68
|
+
("bearer-token", re.compile(r"(?i)\bBearer\s+[A-Za-z0-9\-_.]{20,}")),
|
|
69
|
+
(
|
|
70
|
+
"assigned-secret",
|
|
71
|
+
re.compile(
|
|
72
|
+
r"""(?ix)
|
|
73
|
+
\b(api[_-]?key|secret|token|password|passwd)\b
|
|
74
|
+
\s*[:=]\s*
|
|
75
|
+
['"]?[^\s'"]{6,}['"]?
|
|
76
|
+
"""
|
|
77
|
+
),
|
|
78
|
+
),
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def redact(text: str) -> str:
|
|
83
|
+
"""Best-effort scrub of common secret shapes from a complete string.
|
|
84
|
+
|
|
85
|
+
Suitable for whole values already assembled in memory — prompts, review
|
|
86
|
+
notes, checkpoint comments. For output arriving line-by-line from a live
|
|
87
|
+
subprocess, use `StreamRedactor` instead so a PEM block split across
|
|
88
|
+
lines is still caught.
|
|
89
|
+
"""
|
|
90
|
+
text = _PRIVATE_KEY_BLOCK.sub("[REDACTED:private-key-block]", text)
|
|
91
|
+
for name, pattern in _LINE_PATTERNS:
|
|
92
|
+
text = pattern.sub(f"[REDACTED:{name}]", text)
|
|
93
|
+
return text
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class StreamRedactor:
|
|
97
|
+
"""Applies `redact` to output arriving one line at a time.
|
|
98
|
+
|
|
99
|
+
Holds just enough state to span a PEM private-key block across multiple
|
|
100
|
+
`feed_line` calls, which a single-call `redact(text)` on each line in
|
|
101
|
+
isolation cannot do.
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
def __init__(self) -> None:
|
|
105
|
+
self._in_private_key = False
|
|
106
|
+
|
|
107
|
+
def feed_line(self, line: str) -> str:
|
|
108
|
+
if self._in_private_key:
|
|
109
|
+
if _PRIVATE_KEY_END.search(line):
|
|
110
|
+
self._in_private_key = False
|
|
111
|
+
return ""
|
|
112
|
+
|
|
113
|
+
if _PRIVATE_KEY_BLOCK.search(line):
|
|
114
|
+
return redact(line) # BEGIN and END both landed on one line
|
|
115
|
+
if _PRIVATE_KEY_BEGIN.search(line):
|
|
116
|
+
self._in_private_key = True
|
|
117
|
+
return "[REDACTED:private-key-block]\n"
|
|
118
|
+
|
|
119
|
+
return redact(line)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def prepare_text(text: str, audit_level: str) -> str:
|
|
123
|
+
"""Apply an audit level to a complete piece of agent-authored text."""
|
|
124
|
+
if audit_level == "off":
|
|
125
|
+
return f"<suppressed by audit-level=off: {len(text)} chars>"
|
|
126
|
+
if audit_level == "redacted":
|
|
127
|
+
return redact(text)
|
|
128
|
+
return text
|