agent-loop-tool 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_loop/__init__.py +1 -0
- agent_loop/audit.py +128 -0
- agent_loop/cli.py +354 -0
- agent_loop/git_ops.py +127 -0
- agent_loop/orchestrator.py +314 -0
- agent_loop/plan_init.py +357 -0
- agent_loop/providers/__init__.py +30 -0
- agent_loop/providers/antigravity.py +19 -0
- agent_loop/providers/base.py +147 -0
- agent_loop/providers/claude.py +19 -0
- agent_loop/providers/codex.py +21 -0
- agent_loop/providers/grok.py +18 -0
- agent_loop/safety.py +105 -0
- agent_loop/state.py +247 -0
- agent_loop_tool-0.1.0.dist-info/METADATA +241 -0
- agent_loop_tool-0.1.0.dist-info/RECORD +20 -0
- agent_loop_tool-0.1.0.dist-info/WHEEL +5 -0
- agent_loop_tool-0.1.0.dist-info/entry_points.txt +2 -0
- agent_loop_tool-0.1.0.dist-info/licenses/LICENSE +21 -0
- agent_loop_tool-0.1.0.dist-info/top_level.txt +1 -0
agent_loop/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
agent_loop/audit.py
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Controls how much of an agent's raw output reaches the committed log.
|
|
2
|
+
|
|
3
|
+
Run logs are git-committed audit artifacts (see `git_ops.commit_checkpoint_changes`),
|
|
4
|
+
which makes anything written to them effectively permanent — much harder to walk
|
|
5
|
+
back than a `/tmp` file. `--audit-level` trades completeness of that record
|
|
6
|
+
against exposure of whatever the developer/reviewer agents happen to print or
|
|
7
|
+
write while doing their work:
|
|
8
|
+
|
|
9
|
+
- ``full``: nothing is touched. Agent output and agent-authored free text
|
|
10
|
+
(prompts, review notes) are logged exactly as produced.
|
|
11
|
+
- ``redacted``: known secret *shapes* (AWS keys, GitHub/Slack/OpenAI tokens,
|
|
12
|
+
bearer tokens, PEM private key blocks, `key: value`-style
|
|
13
|
+
assignments) are scrubbed before anything is printed or logged.
|
|
14
|
+
This is a best-effort net, not a guarantee — it cannot catch a
|
|
15
|
+
project's own custom secret formats, and it runs on live,
|
|
16
|
+
streamed subprocess output rather than a byte buffer, so a
|
|
17
|
+
secret split across two flushed writes can slip through.
|
|
18
|
+
- ``off``: raw agent output and agent-authored free text are not printed
|
|
19
|
+
or logged at all. Only structured event metadata (which agent
|
|
20
|
+
ran, what was approved, what was committed) reaches the log.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
|
|
27
|
+
AUDIT_LEVELS = ("full", "redacted", "off")
|
|
28
|
+
DEFAULT_AUDIT_LEVEL = "off"
|
|
29
|
+
|
|
30
|
+
_ENV_VAR = "AGENT_LOOP_AUDIT_LEVEL"
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class InvalidAuditLevel(ValueError):
|
|
34
|
+
pass
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def default_audit_level() -> str:
|
|
38
|
+
"""The configured level: env override if set and valid, else the default."""
|
|
39
|
+
configured = os.environ.get(_ENV_VAR)
|
|
40
|
+
if not configured:
|
|
41
|
+
return DEFAULT_AUDIT_LEVEL
|
|
42
|
+
validate(configured)
|
|
43
|
+
return configured
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def validate(level: str) -> None:
|
|
47
|
+
if level not in AUDIT_LEVELS:
|
|
48
|
+
raise InvalidAuditLevel(
|
|
49
|
+
f"Invalid audit level {level!r}; must be one of {', '.join(AUDIT_LEVELS)}."
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
_PRIVATE_KEY_BLOCK = re.compile(
|
|
54
|
+
r"-----BEGIN [A-Z ]*PRIVATE KEY-----.*?-----END [A-Z ]*PRIVATE KEY-----",
|
|
55
|
+
re.DOTALL,
|
|
56
|
+
)
|
|
57
|
+
_PRIVATE_KEY_BEGIN = re.compile(r"-----BEGIN [A-Z ]*PRIVATE KEY-----")
|
|
58
|
+
_PRIVATE_KEY_END = re.compile(r"-----END [A-Z ]*PRIVATE KEY-----")
|
|
59
|
+
|
|
60
|
+
# High-confidence secret shapes only — anything looser produces enough false
|
|
61
|
+
# positives to make "redacted" logs unreadable without meaningfully raising
|
|
62
|
+
# recall. checked in order; each substitution runs on the previous pass's output.
|
|
63
|
+
_LINE_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
|
|
64
|
+
("aws-access-key-id", re.compile(r"\bAKIA[0-9A-Z]{16}\b")),
|
|
65
|
+
("github-token", re.compile(r"\bgh[pousr]_[A-Za-z0-9]{36,}\b")),
|
|
66
|
+
("slack-token", re.compile(r"\bxox[baprs]-[A-Za-z0-9-]{10,}\b")),
|
|
67
|
+
("openai-key", re.compile(r"\bsk-[A-Za-z0-9]{20,}\b")),
|
|
68
|
+
("bearer-token", re.compile(r"(?i)\bBearer\s+[A-Za-z0-9\-_.]{20,}")),
|
|
69
|
+
(
|
|
70
|
+
"assigned-secret",
|
|
71
|
+
re.compile(
|
|
72
|
+
r"""(?ix)
|
|
73
|
+
\b(api[_-]?key|secret|token|password|passwd)\b
|
|
74
|
+
\s*[:=]\s*
|
|
75
|
+
['"]?[^\s'"]{6,}['"]?
|
|
76
|
+
"""
|
|
77
|
+
),
|
|
78
|
+
),
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def redact(text: str) -> str:
|
|
83
|
+
"""Best-effort scrub of common secret shapes from a complete string.
|
|
84
|
+
|
|
85
|
+
Suitable for whole values already assembled in memory — prompts, review
|
|
86
|
+
notes, checkpoint comments. For output arriving line-by-line from a live
|
|
87
|
+
subprocess, use `StreamRedactor` instead so a PEM block split across
|
|
88
|
+
lines is still caught.
|
|
89
|
+
"""
|
|
90
|
+
text = _PRIVATE_KEY_BLOCK.sub("[REDACTED:private-key-block]", text)
|
|
91
|
+
for name, pattern in _LINE_PATTERNS:
|
|
92
|
+
text = pattern.sub(f"[REDACTED:{name}]", text)
|
|
93
|
+
return text
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
class StreamRedactor:
|
|
97
|
+
"""Applies `redact` to output arriving one line at a time.
|
|
98
|
+
|
|
99
|
+
Holds just enough state to span a PEM private-key block across multiple
|
|
100
|
+
`feed_line` calls, which a single-call `redact(text)` on each line in
|
|
101
|
+
isolation cannot do.
|
|
102
|
+
"""
|
|
103
|
+
|
|
104
|
+
def __init__(self) -> None:
|
|
105
|
+
self._in_private_key = False
|
|
106
|
+
|
|
107
|
+
def feed_line(self, line: str) -> str:
|
|
108
|
+
if self._in_private_key:
|
|
109
|
+
if _PRIVATE_KEY_END.search(line):
|
|
110
|
+
self._in_private_key = False
|
|
111
|
+
return ""
|
|
112
|
+
|
|
113
|
+
if _PRIVATE_KEY_BLOCK.search(line):
|
|
114
|
+
return redact(line) # BEGIN and END both landed on one line
|
|
115
|
+
if _PRIVATE_KEY_BEGIN.search(line):
|
|
116
|
+
self._in_private_key = True
|
|
117
|
+
return "[REDACTED:private-key-block]\n"
|
|
118
|
+
|
|
119
|
+
return redact(line)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def prepare_text(text: str, audit_level: str) -> str:
|
|
123
|
+
"""Apply an audit level to a complete piece of agent-authored text."""
|
|
124
|
+
if audit_level == "off":
|
|
125
|
+
return f"<suppressed by audit-level=off: {len(text)} chars>"
|
|
126
|
+
if audit_level == "redacted":
|
|
127
|
+
return redact(text)
|
|
128
|
+
return text
|
agent_loop/cli.py
ADDED
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
"""agent-loop command-line interface."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Optional
|
|
8
|
+
|
|
9
|
+
from . import audit, plan_init, safety
|
|
10
|
+
from .orchestrator import LoopHalted, run_loop
|
|
11
|
+
from .providers import get_provider, known_provider_names
|
|
12
|
+
from .providers.base import ProviderError
|
|
13
|
+
from .state import PROTECTED_BRANCHES, InvalidPlanState, load_state
|
|
14
|
+
|
|
15
|
+
DEFAULT_STATE = Path("plan_checkpoints.json")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _add_state_arg(p: argparse.ArgumentParser) -> None:
|
|
19
|
+
p.add_argument(
|
|
20
|
+
"--state",
|
|
21
|
+
type=Path,
|
|
22
|
+
default=DEFAULT_STATE,
|
|
23
|
+
help=f"Path to plan_checkpoints.json (default: {DEFAULT_STATE})",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def positive_int(value: str) -> int:
|
|
28
|
+
"""argparse type for counts and timeouts: a zero/negative one can never succeed.
|
|
29
|
+
|
|
30
|
+
Named without a leading underscore because argparse builds its rejection
|
|
31
|
+
message from `__name__` ("invalid positive_int value: 'abc'").
|
|
32
|
+
"""
|
|
33
|
+
parsed = int(value)
|
|
34
|
+
if parsed <= 0:
|
|
35
|
+
raise argparse.ArgumentTypeError("must be greater than zero")
|
|
36
|
+
return parsed
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _build_parser() -> argparse.ArgumentParser:
|
|
40
|
+
parser = argparse.ArgumentParser(
|
|
41
|
+
prog="agent-loop",
|
|
42
|
+
description="Checkpoint-gated developer/reviewer loop across coding-agent CLIs.",
|
|
43
|
+
)
|
|
44
|
+
sub = parser.add_subparsers(dest="cmd")
|
|
45
|
+
|
|
46
|
+
run = sub.add_parser("run", help="Run the developer/reviewer loop.")
|
|
47
|
+
_add_state_arg(run)
|
|
48
|
+
run.add_argument(
|
|
49
|
+
"--developer",
|
|
50
|
+
choices=known_provider_names(),
|
|
51
|
+
default="claude",
|
|
52
|
+
help="CLI to play the developer role (default: claude).",
|
|
53
|
+
)
|
|
54
|
+
run.add_argument(
|
|
55
|
+
"--reviewer",
|
|
56
|
+
choices=known_provider_names(),
|
|
57
|
+
default="codex",
|
|
58
|
+
help="CLI to play the reviewer role (default: codex).",
|
|
59
|
+
)
|
|
60
|
+
run.add_argument(
|
|
61
|
+
"--developer-model",
|
|
62
|
+
default=None,
|
|
63
|
+
help="Model for the developer CLI. Overrides plan 'models.developer'; "
|
|
64
|
+
"if neither is set, the CLI picks its own default.",
|
|
65
|
+
)
|
|
66
|
+
run.add_argument(
|
|
67
|
+
"--reviewer-model",
|
|
68
|
+
default=None,
|
|
69
|
+
help="Model for the reviewer CLI. Overrides plan 'models.reviewer'; "
|
|
70
|
+
"if neither is set, the CLI picks its own default.",
|
|
71
|
+
)
|
|
72
|
+
run.add_argument("--max-review-attempts", type=positive_int, default=3)
|
|
73
|
+
run.add_argument(
|
|
74
|
+
"--timeout",
|
|
75
|
+
type=positive_int,
|
|
76
|
+
default=1800,
|
|
77
|
+
help="Per-agent-call timeout in seconds.",
|
|
78
|
+
)
|
|
79
|
+
run.add_argument("--log-dir", type=Path, default=None)
|
|
80
|
+
run.add_argument(
|
|
81
|
+
"--audit-level",
|
|
82
|
+
choices=audit.AUDIT_LEVELS,
|
|
83
|
+
default=None,
|
|
84
|
+
help="How much of the agents' output is committed to the run log: "
|
|
85
|
+
"'full' logs it as-is, 'redacted' scrubs known secret shapes first, "
|
|
86
|
+
"'off' logs only structured events (no raw agent output or notes). "
|
|
87
|
+
"Defaults to $AGENT_LOOP_AUDIT_LEVEL, or 'off' if that is unset — "
|
|
88
|
+
"logs are git-committed audit artifacts, so this leans safe by "
|
|
89
|
+
"default. Neither 'redacted' nor 'off' is a compliance guarantee; "
|
|
90
|
+
"see README for what each level actually does.",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
validate = sub.add_parser("validate", help="Schema-check the state file.")
|
|
94
|
+
_add_state_arg(validate)
|
|
95
|
+
|
|
96
|
+
status = sub.add_parser("status", help="One-line summary of each checkpoint.")
|
|
97
|
+
_add_state_arg(status)
|
|
98
|
+
|
|
99
|
+
init = sub.add_parser(
|
|
100
|
+
"init",
|
|
101
|
+
help="Generate plan_checkpoints.json from a feature description or Markdown file.",
|
|
102
|
+
)
|
|
103
|
+
init.add_argument(
|
|
104
|
+
"feature",
|
|
105
|
+
nargs="?",
|
|
106
|
+
default=None,
|
|
107
|
+
help="Plain-English feature description (mutually exclusive with --feature-file).",
|
|
108
|
+
)
|
|
109
|
+
init.add_argument(
|
|
110
|
+
"--feature-file",
|
|
111
|
+
type=Path,
|
|
112
|
+
default=None,
|
|
113
|
+
help="Path to a Markdown file describing the feature.",
|
|
114
|
+
)
|
|
115
|
+
_add_state_arg(init)
|
|
116
|
+
init.add_argument(
|
|
117
|
+
"--branch",
|
|
118
|
+
default=None,
|
|
119
|
+
help="Target branch (default: sanitized feature/<slug>).",
|
|
120
|
+
)
|
|
121
|
+
init.add_argument(
|
|
122
|
+
"--plan-file",
|
|
123
|
+
default=None,
|
|
124
|
+
help="'plan_file' recorded in the state (default: the --feature-file path, or "
|
|
125
|
+
f"{plan_init.DEFAULT_PLAN_FILE} for plain-English input).",
|
|
126
|
+
)
|
|
127
|
+
init.add_argument(
|
|
128
|
+
"--provider",
|
|
129
|
+
choices=known_provider_names(),
|
|
130
|
+
default="claude",
|
|
131
|
+
help="CLI used to generate the plan (default: claude).",
|
|
132
|
+
)
|
|
133
|
+
init.add_argument("--model", default=None, help="Model for the planning CLI.")
|
|
134
|
+
init.add_argument(
|
|
135
|
+
"--timeout",
|
|
136
|
+
type=positive_int,
|
|
137
|
+
default=1800,
|
|
138
|
+
help="Provider call timeout in seconds.",
|
|
139
|
+
)
|
|
140
|
+
init.add_argument(
|
|
141
|
+
"--force", action="store_true", help="Overwrite an existing state file."
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
return parser
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
# ---------------------------------------------------------------------------
|
|
148
|
+
# Command implementations
|
|
149
|
+
# ---------------------------------------------------------------------------
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _cmd_validate(args) -> int:
|
|
153
|
+
try:
|
|
154
|
+
state = load_state(args.state)
|
|
155
|
+
except InvalidPlanState as e:
|
|
156
|
+
print(f"error: {e}", file=sys.stderr)
|
|
157
|
+
return 2
|
|
158
|
+
|
|
159
|
+
print(f"OK {args.state} ({len(state.checkpoints)} checkpoints, branch={state.branch})")
|
|
160
|
+
for cp in state.checkpoints:
|
|
161
|
+
print(f" {cp['id']:10s} {cp['status']:9s} {cp['name']}")
|
|
162
|
+
return 0
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _cmd_status(args) -> int:
|
|
166
|
+
try:
|
|
167
|
+
state = load_state(args.state)
|
|
168
|
+
except InvalidPlanState as e:
|
|
169
|
+
print(f"error: {e}", file=sys.stderr)
|
|
170
|
+
return 2
|
|
171
|
+
|
|
172
|
+
print(f"branch={state.branch} plan={state.plan_file} state={args.state}")
|
|
173
|
+
for cp in state.checkpoints:
|
|
174
|
+
attempts = cp.get("attempts", 0)
|
|
175
|
+
print(f" {cp['id']:10s} {cp['status']:9s} attempts={attempts} {cp['name']}")
|
|
176
|
+
return 0
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _cmd_run(args) -> int:
|
|
180
|
+
try:
|
|
181
|
+
state = load_state(args.state)
|
|
182
|
+
except InvalidPlanState as e:
|
|
183
|
+
print(f"error: {e}", file=sys.stderr)
|
|
184
|
+
return 2
|
|
185
|
+
|
|
186
|
+
# Precedence per role: CLI flag > plan 'models.<role>' > the CLI's default.
|
|
187
|
+
dev_model = args.developer_model or state.model_for("developer")
|
|
188
|
+
rev_model = args.reviewer_model or state.model_for("reviewer")
|
|
189
|
+
|
|
190
|
+
try:
|
|
191
|
+
dev = get_provider(args.developer, dev_model)
|
|
192
|
+
rev = get_provider(args.reviewer, rev_model)
|
|
193
|
+
except ProviderError as e:
|
|
194
|
+
print(f"error: {e}", file=sys.stderr)
|
|
195
|
+
return 2
|
|
196
|
+
|
|
197
|
+
try:
|
|
198
|
+
dev.preflight()
|
|
199
|
+
rev.preflight()
|
|
200
|
+
safety.require_auto_mode_gates([dev, rev])
|
|
201
|
+
except (ProviderError, safety.AutoModeGateError) as e:
|
|
202
|
+
print(f"error: {e}", file=sys.stderr)
|
|
203
|
+
return 2
|
|
204
|
+
|
|
205
|
+
try:
|
|
206
|
+
audit_level = args.audit_level or audit.default_audit_level()
|
|
207
|
+
except audit.InvalidAuditLevel as e:
|
|
208
|
+
print(f"error: {e}", file=sys.stderr)
|
|
209
|
+
return 2
|
|
210
|
+
|
|
211
|
+
log_dir = args.log_dir or safety.default_log_dir()
|
|
212
|
+
log_path = safety.open_log_file(log_dir)
|
|
213
|
+
safety.tee_stdout_to(log_path)
|
|
214
|
+
def _role(p) -> str:
|
|
215
|
+
return f"{p.name}({p.model})" if p.model else p.name
|
|
216
|
+
|
|
217
|
+
print(
|
|
218
|
+
f"### agent-loop | developer={_role(dev)} reviewer={_role(rev)} "
|
|
219
|
+
f"log={log_path} audit-level={audit_level}"
|
|
220
|
+
)
|
|
221
|
+
|
|
222
|
+
try:
|
|
223
|
+
with safety.lockfile(safety.default_lock_path()):
|
|
224
|
+
run_loop(
|
|
225
|
+
state_path=args.state,
|
|
226
|
+
developer=dev,
|
|
227
|
+
reviewer=rev,
|
|
228
|
+
max_review_attempts=args.max_review_attempts,
|
|
229
|
+
timeout=args.timeout,
|
|
230
|
+
log_dir=log_dir,
|
|
231
|
+
audit_level=audit_level,
|
|
232
|
+
)
|
|
233
|
+
except safety.LockHeld as e:
|
|
234
|
+
print(f"error: {e}", file=sys.stderr)
|
|
235
|
+
return 2
|
|
236
|
+
except LoopHalted as e:
|
|
237
|
+
print(f"halted: {e}", file=sys.stderr)
|
|
238
|
+
return 1
|
|
239
|
+
except InvalidPlanState as e:
|
|
240
|
+
print(f"error: {e}", file=sys.stderr)
|
|
241
|
+
return 2
|
|
242
|
+
return 0
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _cmd_init(args) -> int:
|
|
246
|
+
# Which source was *supplied*, not which one turned out to be usable: an
|
|
247
|
+
# empty FEATURE alongside --feature-file is ambiguous input, not a
|
|
248
|
+
# single-source invocation.
|
|
249
|
+
have_text = args.feature is not None
|
|
250
|
+
have_file = args.feature_file is not None
|
|
251
|
+
|
|
252
|
+
if have_text and have_file:
|
|
253
|
+
print(
|
|
254
|
+
"error: provide a feature description or --feature-file, not both",
|
|
255
|
+
file=sys.stderr,
|
|
256
|
+
)
|
|
257
|
+
return 2
|
|
258
|
+
if not have_text and not have_file:
|
|
259
|
+
print(
|
|
260
|
+
"error: provide a feature description or --feature-file",
|
|
261
|
+
file=sys.stderr,
|
|
262
|
+
)
|
|
263
|
+
return 2
|
|
264
|
+
|
|
265
|
+
feature_text = (args.feature or "").strip()
|
|
266
|
+
if have_text and not feature_text:
|
|
267
|
+
print("error: feature description is empty", file=sys.stderr)
|
|
268
|
+
return 2
|
|
269
|
+
|
|
270
|
+
if have_file:
|
|
271
|
+
if not args.feature_file.is_file():
|
|
272
|
+
print(f"error: feature file not found: {args.feature_file}", file=sys.stderr)
|
|
273
|
+
return 2
|
|
274
|
+
feature_text = args.feature_file.read_text().strip()
|
|
275
|
+
if not feature_text:
|
|
276
|
+
print(f"error: feature file is empty: {args.feature_file}", file=sys.stderr)
|
|
277
|
+
return 2
|
|
278
|
+
slug_source = args.feature_file.stem
|
|
279
|
+
default_plan_file = str(args.feature_file)
|
|
280
|
+
else:
|
|
281
|
+
slug_source = feature_text
|
|
282
|
+
default_plan_file = plan_init.DEFAULT_PLAN_FILE
|
|
283
|
+
|
|
284
|
+
branch = args.branch or f"feature/{plan_init.slugify(slug_source)}"
|
|
285
|
+
plan_file = args.plan_file or default_plan_file
|
|
286
|
+
|
|
287
|
+
if branch in PROTECTED_BRANCHES:
|
|
288
|
+
print(f"error: refusing protected target branch: {branch}", file=sys.stderr)
|
|
289
|
+
return 2
|
|
290
|
+
|
|
291
|
+
if args.state.exists() and not args.force:
|
|
292
|
+
print(
|
|
293
|
+
f"error: {args.state} already exists; use --force to overwrite",
|
|
294
|
+
file=sys.stderr,
|
|
295
|
+
)
|
|
296
|
+
return 2
|
|
297
|
+
|
|
298
|
+
try:
|
|
299
|
+
provider = get_provider(args.provider, args.model)
|
|
300
|
+
provider.preflight()
|
|
301
|
+
except ProviderError as e:
|
|
302
|
+
print(f"error: {e}", file=sys.stderr)
|
|
303
|
+
return 2
|
|
304
|
+
|
|
305
|
+
try:
|
|
306
|
+
data = plan_init.generate_plan_json(provider, feature_text, args.timeout)
|
|
307
|
+
checkpoints = plan_init.normalize_checkpoints(data)
|
|
308
|
+
payload = plan_init.build_generated_payload(plan_file, branch, checkpoints)
|
|
309
|
+
plan_init.validate_generated_payload(payload)
|
|
310
|
+
except (
|
|
311
|
+
plan_init.ProviderExecutionError,
|
|
312
|
+
plan_init.PlanOutputParseError,
|
|
313
|
+
InvalidPlanState,
|
|
314
|
+
) as e:
|
|
315
|
+
print(f"error: {e}", file=sys.stderr)
|
|
316
|
+
return 2
|
|
317
|
+
|
|
318
|
+
try:
|
|
319
|
+
plan_init.write_generated_payload(args.state, payload, force=args.force)
|
|
320
|
+
except plan_init.StateFileExistsError:
|
|
321
|
+
# Only reachable if the file appeared after the check above.
|
|
322
|
+
print(
|
|
323
|
+
f"error: {args.state} already exists; use --force to overwrite",
|
|
324
|
+
file=sys.stderr,
|
|
325
|
+
)
|
|
326
|
+
return 2
|
|
327
|
+
except OSError as e:
|
|
328
|
+
print(f"error: unable to write {args.state}: {e}", file=sys.stderr)
|
|
329
|
+
return 2
|
|
330
|
+
|
|
331
|
+
print(
|
|
332
|
+
f"wrote {args.state} (branch={branch}, plan_file={plan_file}, "
|
|
333
|
+
f"{len(checkpoints)} checkpoints)"
|
|
334
|
+
)
|
|
335
|
+
return 0
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def main(argv: Optional[list[str]] = None) -> int:
|
|
339
|
+
parser = _build_parser()
|
|
340
|
+
args = parser.parse_args(argv)
|
|
341
|
+
if args.cmd == "run":
|
|
342
|
+
return _cmd_run(args)
|
|
343
|
+
if args.cmd == "validate":
|
|
344
|
+
return _cmd_validate(args)
|
|
345
|
+
if args.cmd == "status":
|
|
346
|
+
return _cmd_status(args)
|
|
347
|
+
if args.cmd == "init":
|
|
348
|
+
return _cmd_init(args)
|
|
349
|
+
parser.print_help(sys.stderr)
|
|
350
|
+
raise SystemExit(2)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
if __name__ == "__main__": # pragma: no cover
|
|
354
|
+
raise SystemExit(main())
|
agent_loop/git_ops.py
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""Git operations the orchestrator relies on."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import subprocess
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from .state import PROTECTED_BRANCHES
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class GitError(RuntimeError):
|
|
12
|
+
pass
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _log_commit_statement(*, message: str, status: str, commit_hash: str | None) -> None:
|
|
16
|
+
"""Print a commit outcome that is mirrored into the active loop log."""
|
|
17
|
+
fields = {
|
|
18
|
+
"commit_hash": commit_hash,
|
|
19
|
+
"event": "COMMIT_STATEMENT",
|
|
20
|
+
"message": message,
|
|
21
|
+
"status": status,
|
|
22
|
+
}
|
|
23
|
+
print(
|
|
24
|
+
f"[agent-loop] COMMIT_STATEMENT {json.dumps(fields, sort_keys=True)}",
|
|
25
|
+
flush=True,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _run(args: list[str], check: bool = True) -> subprocess.CompletedProcess:
|
|
30
|
+
return subprocess.run(args, capture_output=True, text=True, check=check)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def repo_root() -> Path:
|
|
34
|
+
res = _run(["git", "rev-parse", "--show-toplevel"], check=False)
|
|
35
|
+
if res.returncode != 0:
|
|
36
|
+
raise GitError("Not a git repository")
|
|
37
|
+
return Path(res.stdout.strip())
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def current_branch() -> str:
|
|
41
|
+
return _run(["git", "rev-parse", "--abbrev-ref", "HEAD"]).stdout.strip()
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def ensure_branch(branch: str) -> None:
|
|
45
|
+
if branch in PROTECTED_BRANCHES:
|
|
46
|
+
raise GitError(f"Refusing protected branch: {branch}")
|
|
47
|
+
repo_root() # raises if not a repo
|
|
48
|
+
if current_branch() == branch:
|
|
49
|
+
return
|
|
50
|
+
exists = _run(["git", "rev-parse", "--verify", branch], check=False).returncode == 0
|
|
51
|
+
if exists:
|
|
52
|
+
_run(["git", "checkout", branch])
|
|
53
|
+
else:
|
|
54
|
+
_run(["git", "checkout", "-b", branch])
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _relative_log_prefix(log_dir: Path | None) -> str | None:
|
|
58
|
+
"""Return the repo-relative ``logs/`` prefix, or None if not inside the repo.
|
|
59
|
+
|
|
60
|
+
Raises GitError if log_dir is the repository root itself, as that would
|
|
61
|
+
bypass all security checks by matching every file in the working tree.
|
|
62
|
+
"""
|
|
63
|
+
if log_dir is None:
|
|
64
|
+
return None
|
|
65
|
+
try:
|
|
66
|
+
rel = log_dir.resolve().relative_to(repo_root().resolve())
|
|
67
|
+
except (ValueError, GitError):
|
|
68
|
+
return None
|
|
69
|
+
if str(rel) == ".":
|
|
70
|
+
raise GitError(
|
|
71
|
+
"Log directory cannot be the repository root itself. "
|
|
72
|
+
"Use a subdirectory like 'logs/' instead."
|
|
73
|
+
)
|
|
74
|
+
return f"{rel}/"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def require_clean_worktree(log_dir: Path | None = None) -> None:
|
|
78
|
+
"""Refuse to start on a dirty worktree, tolerating the loop's own log files.
|
|
79
|
+
|
|
80
|
+
Logs are committed as audit artifacts, so a log file from a previous run is
|
|
81
|
+
both *tracked* and *modified* by the time the next run starts: the tee keeps
|
|
82
|
+
appending after the final commit of the run that created it. Untracked logs
|
|
83
|
+
(a brand new run) and modified logs (that trailing tail) are therefore both
|
|
84
|
+
expected, and neither should block the loop.
|
|
85
|
+
"""
|
|
86
|
+
res = _run(["git", "status", "--porcelain", "--untracked-files=all"])
|
|
87
|
+
lines = [ln for ln in res.stdout.splitlines() if ln.strip()]
|
|
88
|
+
|
|
89
|
+
prefix = _relative_log_prefix(log_dir)
|
|
90
|
+
if prefix is not None:
|
|
91
|
+
# Porcelain v1 status codes are two columns followed by a space, so the
|
|
92
|
+
# path starts at index 3. Paths containing spaces or other special
|
|
93
|
+
# characters come back double-quoted.
|
|
94
|
+
lines = [ln for ln in lines if not ln[3:].lstrip('"').startswith(prefix)]
|
|
95
|
+
|
|
96
|
+
if lines:
|
|
97
|
+
raise GitError(
|
|
98
|
+
"Worktree must be clean before running the loop.\n"
|
|
99
|
+
"Outstanding changes:\n" + "\n".join(lines)
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
#: Log directories already warned about, so a multi-checkpoint run says it once.
|
|
104
|
+
_WARNED_EXTERNAL_LOG_DIRS: set[str] = set()
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def commit_checkpoint_changes(message: str, log_dir: Path) -> None:
|
|
108
|
+
if _relative_log_prefix(log_dir) is None:
|
|
109
|
+
key = str(log_dir)
|
|
110
|
+
if key not in _WARNED_EXTERNAL_LOG_DIRS:
|
|
111
|
+
_WARNED_EXTERNAL_LOG_DIRS.add(key)
|
|
112
|
+
print(
|
|
113
|
+
f"[agent-loop] WARNING: log directory {log_dir} is outside the "
|
|
114
|
+
"repository; run logs will not be committed as audit artifacts.",
|
|
115
|
+
flush=True,
|
|
116
|
+
)
|
|
117
|
+
|
|
118
|
+
_run(["git", "add", "-A"])
|
|
119
|
+
|
|
120
|
+
cached = _run(["git", "diff", "--cached", "--quiet"], check=False)
|
|
121
|
+
if cached.returncode == 0:
|
|
122
|
+
_log_commit_statement(message=message, status="no_changes", commit_hash=None)
|
|
123
|
+
return
|
|
124
|
+
|
|
125
|
+
_run(["git", "commit", "-m", message, "--quiet"])
|
|
126
|
+
commit_hash = _run(["git", "rev-parse", "HEAD"]).stdout.strip()
|
|
127
|
+
_log_commit_statement(message=message, status="committed", commit_hash=commit_hash)
|