giro 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- giro/__init__.py +8 -0
- giro/cli.py +126 -0
- giro/config.py +168 -0
- giro/drivers.py +213 -0
- giro/envelope.py +177 -0
- giro/gates.py +139 -0
- giro/loops.py +323 -0
- giro/states.py +48 -0
- giro/store.py +243 -0
- giro/workspace.py +82 -0
- giro-0.0.1.dist-info/METADATA +100 -0
- giro-0.0.1.dist-info/RECORD +15 -0
- giro-0.0.1.dist-info/WHEEL +4 -0
- giro-0.0.1.dist-info/entry_points.txt +2 -0
- giro-0.0.1.dist-info/licenses/LICENSE +21 -0
giro/__init__.py
ADDED
giro/cli.py
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""The giro CLI — the product's front door.
|
|
2
|
+
|
|
3
|
+
Exit codes are part of the contract: 0 = proof, 2 = needs-human, 1 = error.
|
|
4
|
+
A CI job or a chat agent reads them the same way.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from . import __version__
|
|
14
|
+
from .config import CONFIG_NAME, INIT_TEMPLATE, ConfigError, load_config
|
|
15
|
+
from .drivers import DriverError, build_driver
|
|
16
|
+
from .envelope import EnvelopeError
|
|
17
|
+
from .gates import build_context, run_gates
|
|
18
|
+
from .loops import Engine
|
|
19
|
+
from .store import Store, StoreError
|
|
20
|
+
from .workspace import Workspace, WorkspaceError
|
|
21
|
+
|
|
22
|
+
EXIT_PROOF = 0
|
|
23
|
+
EXIT_ERROR = 1
|
|
24
|
+
EXIT_NEEDS_HUMAN = 2
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _build_engine(root: Path) -> Engine:
|
|
28
|
+
cfg = load_config(root)
|
|
29
|
+
drivers = {}
|
|
30
|
+
for role in ("implementer", "judge", "planner"):
|
|
31
|
+
entry = cfg.roster_for(role)
|
|
32
|
+
drivers[role] = build_driver(entry.driver, args=entry.args)
|
|
33
|
+
return Engine(cfg=cfg, store=Store(root), workspace=Workspace(root), drivers=drivers)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def cmd_init(root: Path) -> int:
|
|
37
|
+
path = root / CONFIG_NAME
|
|
38
|
+
if path.exists():
|
|
39
|
+
print(f"{CONFIG_NAME} already exists — edit it directly.")
|
|
40
|
+
return EXIT_ERROR
|
|
41
|
+
path.write_text(INIT_TEMPLATE, encoding="utf-8")
|
|
42
|
+
(root / "docs" / "specs").mkdir(parents=True, exist_ok=True)
|
|
43
|
+
print(f"Wrote {CONFIG_NAME} and docs/specs/.")
|
|
44
|
+
print("Next: set your [verify] gates, then write a Spec and run `giro implement <slug>`.")
|
|
45
|
+
return EXIT_PROOF
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def cmd_status(root: Path) -> int:
|
|
49
|
+
store = Store(root)
|
|
50
|
+
specs = store.list_specs()
|
|
51
|
+
if not specs:
|
|
52
|
+
print("No specs under docs/specs/.")
|
|
53
|
+
return EXIT_PROOF
|
|
54
|
+
for spec in specs:
|
|
55
|
+
print(f"{spec.slug} [{spec.state}] {spec.title}")
|
|
56
|
+
for issue in store.load_issues(spec.slug):
|
|
57
|
+
deps = f" blocked_by={','.join(issue.blocked_by)}" if issue.blocked_by else ""
|
|
58
|
+
print(
|
|
59
|
+
f" {issue.id} [{issue.state}] attempts={issue.attempts}{deps} {issue.title}"
|
|
60
|
+
)
|
|
61
|
+
return EXIT_PROOF
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def cmd_verify(root: Path) -> int:
|
|
65
|
+
engine = _build_engine(root)
|
|
66
|
+
ctx = build_context(
|
|
67
|
+
engine.cfg, root, "Manual `giro verify` run: judge the working tree.",
|
|
68
|
+
engine.drivers["judge"],
|
|
69
|
+
)
|
|
70
|
+
green, findings = run_gates(engine.cfg.verify_gates, ctx)
|
|
71
|
+
for finding in findings:
|
|
72
|
+
print(f"[{finding.gate}] {finding.summary}")
|
|
73
|
+
if finding.detail:
|
|
74
|
+
print(f" {finding.detail[:500]}")
|
|
75
|
+
print("all green" if green else f"{len(findings)} finding(s)")
|
|
76
|
+
return EXIT_PROOF if green else EXIT_NEEDS_HUMAN
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def cmd_implement(root: Path, target: str) -> int:
|
|
80
|
+
engine = _build_engine(root)
|
|
81
|
+
report = engine.implement(target)
|
|
82
|
+
print(f"{report.target}: {report.outcome} — {report.detail}")
|
|
83
|
+
for ref, state in sorted(report.issues.items()):
|
|
84
|
+
print(f" {ref}: {state}")
|
|
85
|
+
if report.outcome in ("done", "all-done"):
|
|
86
|
+
return EXIT_PROOF
|
|
87
|
+
return EXIT_NEEDS_HUMAN
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def main(argv: list[str] | None = None) -> int:
|
|
91
|
+
parser = argparse.ArgumentParser(
|
|
92
|
+
prog="giro",
|
|
93
|
+
description="Guarded loops: Specs become verified work — proof or escalation.",
|
|
94
|
+
)
|
|
95
|
+
parser.add_argument("--version", action="version", version=f"giro {__version__}")
|
|
96
|
+
parser.add_argument(
|
|
97
|
+
"-C", dest="root", default=".", help="project root (default: current directory)"
|
|
98
|
+
)
|
|
99
|
+
sub = parser.add_subparsers(dest="command")
|
|
100
|
+
sub.add_parser("init", help="write a starter giro.toml")
|
|
101
|
+
sub.add_parser("status", help="show spec and issue states")
|
|
102
|
+
sub.add_parser("verify", help="run the [verify] gate set once")
|
|
103
|
+
p_impl = sub.add_parser("implement", help="run the guarded loop for a spec or issue")
|
|
104
|
+
p_impl.add_argument("target", help="spec slug, issue id, or slug/issue-id")
|
|
105
|
+
|
|
106
|
+
args = parser.parse_args(argv)
|
|
107
|
+
root = Path(args.root).resolve()
|
|
108
|
+
|
|
109
|
+
try:
|
|
110
|
+
if args.command == "init":
|
|
111
|
+
return cmd_init(root)
|
|
112
|
+
if args.command == "status":
|
|
113
|
+
return cmd_status(root)
|
|
114
|
+
if args.command == "verify":
|
|
115
|
+
return cmd_verify(root)
|
|
116
|
+
if args.command == "implement":
|
|
117
|
+
return cmd_implement(root, args.target)
|
|
118
|
+
parser.print_help()
|
|
119
|
+
return EXIT_PROOF
|
|
120
|
+
except (ConfigError, StoreError, WorkspaceError, DriverError, EnvelopeError) as exc:
|
|
121
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
122
|
+
return EXIT_ERROR
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
if __name__ == "__main__":
|
|
126
|
+
sys.exit(main())
|
giro/config.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
"""giro.toml — gates, roster, and budgets, fixed at setup.
|
|
2
|
+
|
|
3
|
+
The loops read this; they never invent it.
|
|
4
|
+
"""
|
|
5
|
+
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import tomllib
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
GATE_TYPES = frozenset({"command", "prompt", "skill"})
|
|
13
|
+
ROLES = ("implementer", "judge", "planner")
|
|
14
|
+
|
|
15
|
+
CONFIG_NAME = "giro.toml"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class ConfigError(Exception):
|
|
19
|
+
"""giro.toml is missing or invalid."""
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class GateSpec:
|
|
24
|
+
name: str
|
|
25
|
+
type: str # "command" | "prompt" | "skill"
|
|
26
|
+
run: str = "" # command gates
|
|
27
|
+
rubric: str = "" # prompt gates
|
|
28
|
+
skill: str = "" # skill gates
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class RosterEntry:
|
|
33
|
+
driver: str = "claude"
|
|
34
|
+
model: str = ""
|
|
35
|
+
args: list[str] = field(default_factory=list)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass
|
|
39
|
+
class Config:
|
|
40
|
+
root: Path
|
|
41
|
+
verify_gates: list[GateSpec]
|
|
42
|
+
validate_gates: list[GateSpec]
|
|
43
|
+
roster: dict[str, RosterEntry]
|
|
44
|
+
concurrency: int = 1
|
|
45
|
+
issue_attempts: int = 3
|
|
46
|
+
validate_cycles: int = 3
|
|
47
|
+
gate_timeout: int = 600
|
|
48
|
+
|
|
49
|
+
def roster_for(self, role: str) -> RosterEntry:
|
|
50
|
+
"""Resolve a role, falling back to the implementer entry."""
|
|
51
|
+
if role in self.roster:
|
|
52
|
+
return self.roster[role]
|
|
53
|
+
if "implementer" in self.roster:
|
|
54
|
+
return self.roster["implementer"]
|
|
55
|
+
return RosterEntry()
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _parse_gates(section: object, where: str) -> list[GateSpec]:
|
|
59
|
+
if section is None:
|
|
60
|
+
return []
|
|
61
|
+
if not isinstance(section, dict) or not isinstance(section.get("gates"), list):
|
|
62
|
+
raise ConfigError(f"[{where}] must contain a 'gates' array")
|
|
63
|
+
gates: list[GateSpec] = []
|
|
64
|
+
for i, raw in enumerate(section["gates"]):
|
|
65
|
+
if not isinstance(raw, dict):
|
|
66
|
+
raise ConfigError(f"[{where}] gate #{i + 1} must be a table")
|
|
67
|
+
name = raw.get("name") or f"{where}-{i + 1}"
|
|
68
|
+
gtype = raw.get("type", "")
|
|
69
|
+
if gtype not in GATE_TYPES:
|
|
70
|
+
raise ConfigError(
|
|
71
|
+
f"[{where}] gate {name!r}: type must be one of {sorted(GATE_TYPES)}"
|
|
72
|
+
)
|
|
73
|
+
gate = GateSpec(
|
|
74
|
+
name=str(name),
|
|
75
|
+
type=gtype,
|
|
76
|
+
run=str(raw.get("run", "")),
|
|
77
|
+
rubric=str(raw.get("rubric", "")),
|
|
78
|
+
skill=str(raw.get("skill", "")),
|
|
79
|
+
)
|
|
80
|
+
required = {"command": "run", "prompt": "rubric", "skill": "skill"}[gtype]
|
|
81
|
+
if not getattr(gate, required):
|
|
82
|
+
raise ConfigError(f"[{where}] gate {name!r}: {gtype} gates require {required!r}")
|
|
83
|
+
gates.append(gate)
|
|
84
|
+
return gates
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _parse_roster(raw: object) -> dict[str, RosterEntry]:
|
|
88
|
+
roster: dict[str, RosterEntry] = {}
|
|
89
|
+
if raw is None:
|
|
90
|
+
return roster
|
|
91
|
+
if not isinstance(raw, dict):
|
|
92
|
+
raise ConfigError("[runner.roster] must be a table of role entries")
|
|
93
|
+
for role, entry in raw.items():
|
|
94
|
+
if not isinstance(entry, dict):
|
|
95
|
+
raise ConfigError(f"[runner.roster] {role!r} must be a table")
|
|
96
|
+
args = entry.get("args", [])
|
|
97
|
+
if not isinstance(args, list) or not all(isinstance(a, str) for a in args):
|
|
98
|
+
raise ConfigError(f"[runner.roster] {role!r}: 'args' must be a list of strings")
|
|
99
|
+
roster[role] = RosterEntry(
|
|
100
|
+
driver=str(entry.get("driver", "claude")),
|
|
101
|
+
model=str(entry.get("model", "")),
|
|
102
|
+
args=list(args),
|
|
103
|
+
)
|
|
104
|
+
return roster
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def load_config(root: Path) -> Config:
|
|
108
|
+
path = root / CONFIG_NAME
|
|
109
|
+
if not path.is_file():
|
|
110
|
+
raise ConfigError(f"{CONFIG_NAME} not found in {root} — run `giro init` first")
|
|
111
|
+
try:
|
|
112
|
+
data = tomllib.loads(path.read_text(encoding="utf-8"))
|
|
113
|
+
except tomllib.TOMLDecodeError as exc:
|
|
114
|
+
raise ConfigError(f"{CONFIG_NAME} is not valid TOML: {exc}") from exc
|
|
115
|
+
|
|
116
|
+
runner = data.get("runner", {})
|
|
117
|
+
budget = data.get("budget", {})
|
|
118
|
+
if not isinstance(runner, dict) or not isinstance(budget, dict):
|
|
119
|
+
raise ConfigError("[runner] and [budget] must be tables")
|
|
120
|
+
|
|
121
|
+
verify_gates = _parse_gates(data.get("verify"), "verify")
|
|
122
|
+
if not verify_gates:
|
|
123
|
+
raise ConfigError("[verify] must define at least one gate")
|
|
124
|
+
|
|
125
|
+
def _int(section: dict, key: str, default: int) -> int:
|
|
126
|
+
value = section.get(key, default)
|
|
127
|
+
if not isinstance(value, int) or isinstance(value, bool) or value < 1:
|
|
128
|
+
raise ConfigError(f"{key!r} must be a positive integer")
|
|
129
|
+
return value
|
|
130
|
+
|
|
131
|
+
return Config(
|
|
132
|
+
root=root,
|
|
133
|
+
verify_gates=verify_gates,
|
|
134
|
+
validate_gates=_parse_gates(data.get("validate"), "validate"),
|
|
135
|
+
roster=_parse_roster(runner.get("roster")),
|
|
136
|
+
concurrency=_int(runner, "concurrency", 1),
|
|
137
|
+
issue_attempts=_int(budget, "issue_attempts", 3),
|
|
138
|
+
validate_cycles=_int(budget, "validate_cycles", 3),
|
|
139
|
+
gate_timeout=_int(budget, "gate_timeout", 600),
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
INIT_TEMPLATE = """\
|
|
144
|
+
# giro configuration — gates, roster, budgets. See docs/design.md in the giro repo.
|
|
145
|
+
|
|
146
|
+
[verify] # Issue-level gate set: runs after every worker attempt
|
|
147
|
+
gates = [
|
|
148
|
+
{ name = "test", type = "command", run = "make test" },
|
|
149
|
+
]
|
|
150
|
+
|
|
151
|
+
[validate] # Spec-level gate set: runs once every child Issue is terminal
|
|
152
|
+
gates = [
|
|
153
|
+
{ name = "spec-fit", type = "prompt", rubric = "Every acceptance criterion in the Spec is met." },
|
|
154
|
+
]
|
|
155
|
+
|
|
156
|
+
[runner]
|
|
157
|
+
concurrency = 1 # 1 = sequential fresh workers; >1 = isolated worktrees (roadmap)
|
|
158
|
+
|
|
159
|
+
[runner.roster] # drivers: claude | agy | codex | gemini; args pass through to the CLI
|
|
160
|
+
implementer = { driver = "claude", args = ["--dangerously-skip-permissions"] }
|
|
161
|
+
judge = { driver = "claude" }
|
|
162
|
+
planner = { driver = "claude" }
|
|
163
|
+
|
|
164
|
+
[budget] # the "never silent" bounds
|
|
165
|
+
issue_attempts = 3 # verify retries per Issue before needs-human
|
|
166
|
+
validate_cycles = 3 # gap -> re-wave loops before needs-human
|
|
167
|
+
gate_timeout = 600 # seconds per command gate
|
|
168
|
+
"""
|
giro/drivers.py
ADDED
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Drivers — how the engine spawns an LLM context and gets an envelope back.
|
|
2
|
+
|
|
3
|
+
A Driver's whole contract is: run this prompt in a fresh context at this
|
|
4
|
+
cwd, and return the parsed JSON envelope the context ended with. Everything
|
|
5
|
+
else (what the prompt says, what the schema demands, what happens to the
|
|
6
|
+
result) belongs to the engine.
|
|
7
|
+
|
|
8
|
+
Four agent CLIs ship as drivers — ``claude``, ``agy``, ``codex``,
|
|
9
|
+
``gemini`` — each a thin argv recipe over one shared subprocess runner.
|
|
10
|
+
Errors surface the CLI's own message (auth failures, quota limits), never a
|
|
11
|
+
bare exit code.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import json
|
|
17
|
+
import subprocess
|
|
18
|
+
from collections.abc import Callable
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Protocol
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class DriverError(Exception):
|
|
24
|
+
"""The context could not be run or produced no parseable JSON."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class Driver(Protocol):
|
|
28
|
+
def run(
|
|
29
|
+
self, prompt: str, schema: dict[str, Any], cwd: Path, model: str = ""
|
|
30
|
+
) -> dict[str, Any]: ...
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def extract_json(text: str) -> dict[str, Any]:
|
|
34
|
+
"""Extract the last complete JSON object from free text (fenced or bare)."""
|
|
35
|
+
decoder = json.JSONDecoder()
|
|
36
|
+
found: dict[str, Any] | None = None
|
|
37
|
+
idx = 0
|
|
38
|
+
while (start := text.find("{", idx)) != -1:
|
|
39
|
+
try:
|
|
40
|
+
obj, end = decoder.raw_decode(text[start:])
|
|
41
|
+
except json.JSONDecodeError:
|
|
42
|
+
idx = start + 1
|
|
43
|
+
continue
|
|
44
|
+
if isinstance(obj, dict):
|
|
45
|
+
found = obj
|
|
46
|
+
idx = start + end
|
|
47
|
+
if found is None:
|
|
48
|
+
raise DriverError("no JSON object found in context output")
|
|
49
|
+
return found
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _tail(text: str, limit: int = 2000) -> str:
|
|
53
|
+
text = text.strip()
|
|
54
|
+
return text if len(text) <= limit else "…" + text[-limit:]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
FakeResponse = dict[str, Any] | Exception | Callable[[str, Path], dict[str, Any]]
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class FakeDriver:
|
|
61
|
+
"""Scripted driver for tests: canned envelopes, optional side effects.
|
|
62
|
+
|
|
63
|
+
Each queued response may be a dict (returned as-is), an Exception
|
|
64
|
+
(raised), or a callable ``(prompt, cwd) -> dict`` for responses that
|
|
65
|
+
need to mutate the workspace the way a real worker would.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(self, responses: list[FakeResponse]):
|
|
69
|
+
self._responses = list(responses)
|
|
70
|
+
self.calls: list[dict[str, Any]] = []
|
|
71
|
+
|
|
72
|
+
def run(
|
|
73
|
+
self, prompt: str, schema: dict[str, Any], cwd: Path, model: str = ""
|
|
74
|
+
) -> dict[str, Any]:
|
|
75
|
+
self.calls.append({"prompt": prompt, "schema": schema, "cwd": cwd, "model": model})
|
|
76
|
+
if not self._responses:
|
|
77
|
+
raise DriverError("FakeDriver ran out of scripted responses")
|
|
78
|
+
response = self._responses.pop(0)
|
|
79
|
+
if isinstance(response, Exception):
|
|
80
|
+
raise response
|
|
81
|
+
if callable(response):
|
|
82
|
+
return response(prompt, cwd)
|
|
83
|
+
return response
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class SubprocessDriver:
|
|
87
|
+
"""Shared shell-out runner: build argv, spawn, extract the final JSON."""
|
|
88
|
+
|
|
89
|
+
name = "subprocess"
|
|
90
|
+
|
|
91
|
+
def __init__(self, args: list[str] | None = None, timeout: int = 3600):
|
|
92
|
+
self.args = list(args or [])
|
|
93
|
+
self.timeout = timeout
|
|
94
|
+
|
|
95
|
+
def build_argv(self, prompt: str, model: str) -> list[str]: # pragma: no cover
|
|
96
|
+
raise NotImplementedError
|
|
97
|
+
|
|
98
|
+
def postprocess(self, stdout: str) -> str:
|
|
99
|
+
"""Hook: reduce raw stdout to the text that carries the envelope."""
|
|
100
|
+
return stdout
|
|
101
|
+
|
|
102
|
+
def error_detail(self, proc: subprocess.CompletedProcess[str]) -> str:
|
|
103
|
+
"""Hook: the most useful message when the CLI fails."""
|
|
104
|
+
return _tail(proc.stderr) or _tail(proc.stdout)
|
|
105
|
+
|
|
106
|
+
def run(
|
|
107
|
+
self, prompt: str, schema: dict[str, Any], cwd: Path, model: str = ""
|
|
108
|
+
) -> dict[str, Any]:
|
|
109
|
+
argv = self.build_argv(prompt, model)
|
|
110
|
+
try:
|
|
111
|
+
proc = subprocess.run(
|
|
112
|
+
argv, cwd=cwd, capture_output=True, text=True, timeout=self.timeout
|
|
113
|
+
)
|
|
114
|
+
except FileNotFoundError as exc:
|
|
115
|
+
raise DriverError(f"`{argv[0]}` CLI not found on PATH") from exc
|
|
116
|
+
except subprocess.TimeoutExpired as exc:
|
|
117
|
+
raise DriverError(f"{self.name} context timed out after {self.timeout}s") from exc
|
|
118
|
+
if proc.returncode != 0:
|
|
119
|
+
raise DriverError(
|
|
120
|
+
f"{self.name} exited {proc.returncode}: {self.error_detail(proc)}"
|
|
121
|
+
)
|
|
122
|
+
return extract_json(self.postprocess(proc.stdout))
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
class ClaudeDriver(SubprocessDriver):
|
|
126
|
+
"""``claude -p`` — non-interactive Claude Code context."""
|
|
127
|
+
|
|
128
|
+
name = "claude"
|
|
129
|
+
|
|
130
|
+
def build_argv(self, prompt: str, model: str) -> list[str]:
|
|
131
|
+
argv = ["claude", "-p", prompt, "--output-format", "json", *self.args]
|
|
132
|
+
if model:
|
|
133
|
+
argv += ["--model", model]
|
|
134
|
+
return argv
|
|
135
|
+
|
|
136
|
+
def _outer(self, stdout: str) -> dict[str, Any] | None:
|
|
137
|
+
try:
|
|
138
|
+
outer = json.loads(stdout)
|
|
139
|
+
except json.JSONDecodeError:
|
|
140
|
+
return None
|
|
141
|
+
return outer if isinstance(outer, dict) else None
|
|
142
|
+
|
|
143
|
+
def postprocess(self, stdout: str) -> str:
|
|
144
|
+
outer = self._outer(stdout)
|
|
145
|
+
if outer is None:
|
|
146
|
+
return stdout
|
|
147
|
+
# claude reports its own failures (auth, quota) inside the result JSON
|
|
148
|
+
# with is_error=true — surface the message, don't parse it as work.
|
|
149
|
+
if outer.get("is_error"):
|
|
150
|
+
raise DriverError(f"claude: {outer.get('result', 'unknown error')}")
|
|
151
|
+
result = outer.get("result")
|
|
152
|
+
return result if isinstance(result, str) else stdout
|
|
153
|
+
|
|
154
|
+
def error_detail(self, proc: subprocess.CompletedProcess[str]) -> str:
|
|
155
|
+
outer = self._outer(proc.stdout)
|
|
156
|
+
if outer is not None and outer.get("result"):
|
|
157
|
+
return str(outer["result"])
|
|
158
|
+
return super().error_detail(proc)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
class AgyDriver(SubprocessDriver):
|
|
162
|
+
"""Antigravity's ``agy`` CLI. Model must be an exact roster name,
|
|
163
|
+
e.g. ``Gemini 3.6 Flash (Medium)``."""
|
|
164
|
+
|
|
165
|
+
name = "agy"
|
|
166
|
+
|
|
167
|
+
def build_argv(self, prompt: str, model: str) -> list[str]:
|
|
168
|
+
argv = ["agy", "-p", prompt, "--dangerously-skip-permissions",
|
|
169
|
+
"--print-timeout", f"{self.timeout}s", *self.args]
|
|
170
|
+
if model:
|
|
171
|
+
argv += ["--model", model]
|
|
172
|
+
return argv
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
class CodexDriver(SubprocessDriver):
|
|
176
|
+
"""``codex exec`` — the prompt is positional and LAST."""
|
|
177
|
+
|
|
178
|
+
name = "codex"
|
|
179
|
+
|
|
180
|
+
def build_argv(self, prompt: str, model: str) -> list[str]:
|
|
181
|
+
argv = ["codex", "exec", "--dangerously-bypass-approvals-and-sandbox",
|
|
182
|
+
"--skip-git-repo-check", "--color", "never", *self.args]
|
|
183
|
+
if model:
|
|
184
|
+
argv += ["--model", model]
|
|
185
|
+
return [*argv, prompt]
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
class GeminiDriver(SubprocessDriver):
|
|
189
|
+
"""``gemini -p`` — needs GEMINI_API_KEY or prior CLI auth."""
|
|
190
|
+
|
|
191
|
+
name = "gemini"
|
|
192
|
+
|
|
193
|
+
def build_argv(self, prompt: str, model: str) -> list[str]:
|
|
194
|
+
argv = ["gemini", "-p", prompt, "--yolo", "--skip-trust", *self.args]
|
|
195
|
+
if model:
|
|
196
|
+
argv += ["--model", model]
|
|
197
|
+
return argv
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
DRIVER_REGISTRY: dict[str, type[SubprocessDriver]] = {
|
|
201
|
+
"claude": ClaudeDriver,
|
|
202
|
+
"agy": AgyDriver,
|
|
203
|
+
"codex": CodexDriver,
|
|
204
|
+
"gemini": GeminiDriver,
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def build_driver(name: str, args: list[str] | None = None) -> Driver:
|
|
209
|
+
if name not in DRIVER_REGISTRY:
|
|
210
|
+
raise DriverError(
|
|
211
|
+
f"unknown driver {name!r}; available: {sorted(DRIVER_REGISTRY)}"
|
|
212
|
+
)
|
|
213
|
+
return DRIVER_REGISTRY[name](args=args)
|
giro/envelope.py
ADDED
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
"""Envelopes — the only things that cross back from an LLM context.
|
|
2
|
+
|
|
3
|
+
Every spawned context ends by emitting one JSON envelope. The engine
|
|
4
|
+
schema-checks it here; a malformed envelope is an error the caller turns
|
|
5
|
+
into a failed attempt or a failed gate (fail closed), never a shrug.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
WORKER_OUTCOMES = frozenset({"completed", "needs-human", "failed"})
|
|
14
|
+
VERDICTS = frozenset({"pass", "fail"})
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class EnvelopeError(Exception):
|
|
18
|
+
"""The context's final output did not match the required schema."""
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Finding:
|
|
23
|
+
summary: str
|
|
24
|
+
detail: str = ""
|
|
25
|
+
location: str = ""
|
|
26
|
+
gate: str = ""
|
|
27
|
+
|
|
28
|
+
def as_dict(self) -> dict[str, str]:
|
|
29
|
+
out = {"summary": self.summary}
|
|
30
|
+
if self.detail:
|
|
31
|
+
out["detail"] = self.detail
|
|
32
|
+
if self.location:
|
|
33
|
+
out["location"] = self.location
|
|
34
|
+
if self.gate:
|
|
35
|
+
out["gate"] = self.gate
|
|
36
|
+
return out
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass
|
|
40
|
+
class GateVerdict:
|
|
41
|
+
verdict: str # "pass" | "fail"
|
|
42
|
+
findings: list[Finding] = field(default_factory=list)
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def passed(self) -> bool:
|
|
46
|
+
return self.verdict == "pass"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
@dataclass
|
|
50
|
+
class WorkerResult:
|
|
51
|
+
outcome: str # "completed" | "needs-human" | "failed"
|
|
52
|
+
summary: str
|
|
53
|
+
notes: str = ""
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _require(data: Any, key: str, kind: type) -> Any:
|
|
57
|
+
if not isinstance(data, dict):
|
|
58
|
+
raise EnvelopeError(f"envelope must be a JSON object, got {type(data).__name__}")
|
|
59
|
+
if key not in data:
|
|
60
|
+
raise EnvelopeError(f"envelope missing required key {key!r}")
|
|
61
|
+
value = data[key]
|
|
62
|
+
if not isinstance(value, kind):
|
|
63
|
+
raise EnvelopeError(f"envelope key {key!r} must be {kind.__name__}")
|
|
64
|
+
return value
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def parse_findings(raw: Any) -> list[Finding]:
|
|
68
|
+
if not isinstance(raw, list):
|
|
69
|
+
raise EnvelopeError("'findings' must be a list")
|
|
70
|
+
findings: list[Finding] = []
|
|
71
|
+
for item in raw:
|
|
72
|
+
summary = _require(item, "summary", str)
|
|
73
|
+
findings.append(
|
|
74
|
+
Finding(
|
|
75
|
+
summary=summary,
|
|
76
|
+
detail=str(item.get("detail", "")),
|
|
77
|
+
location=str(item.get("location", "")),
|
|
78
|
+
)
|
|
79
|
+
)
|
|
80
|
+
return findings
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def parse_verdict(data: Any) -> GateVerdict:
|
|
84
|
+
verdict = _require(data, "verdict", str)
|
|
85
|
+
if verdict not in VERDICTS:
|
|
86
|
+
raise EnvelopeError(f"'verdict' must be one of {sorted(VERDICTS)}, got {verdict!r}")
|
|
87
|
+
findings = parse_findings(data.get("findings", []))
|
|
88
|
+
if verdict == "fail" and not findings:
|
|
89
|
+
findings = [Finding(summary="gate failed without findings")]
|
|
90
|
+
return GateVerdict(verdict=verdict, findings=findings)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def parse_worker(data: Any) -> WorkerResult:
|
|
94
|
+
outcome = _require(data, "outcome", str)
|
|
95
|
+
if outcome not in WORKER_OUTCOMES:
|
|
96
|
+
raise EnvelopeError(f"'outcome' must be one of {sorted(WORKER_OUTCOMES)}, got {outcome!r}")
|
|
97
|
+
summary = _require(data, "summary", str)
|
|
98
|
+
return WorkerResult(outcome=outcome, summary=summary, notes=str(data.get("notes", "")))
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
@dataclass
|
|
102
|
+
class PlannedIssue:
|
|
103
|
+
title: str
|
|
104
|
+
body: str
|
|
105
|
+
blocked_by: list[int] = field(default_factory=list) # 1-based indices into the plan
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def parse_plan(data: Any) -> list[PlannedIssue]:
|
|
109
|
+
raw = _require(data, "issues", list)
|
|
110
|
+
if not raw:
|
|
111
|
+
raise EnvelopeError("'issues' must contain at least one issue")
|
|
112
|
+
planned: list[PlannedIssue] = []
|
|
113
|
+
for item in raw:
|
|
114
|
+
title = _require(item, "title", str)
|
|
115
|
+
body = _require(item, "body", str)
|
|
116
|
+
blocked = item.get("blocked_by", [])
|
|
117
|
+
if not isinstance(blocked, list) or not all(isinstance(b, int) for b in blocked):
|
|
118
|
+
raise EnvelopeError("'blocked_by' must be a list of integers (1-based plan indices)")
|
|
119
|
+
planned.append(PlannedIssue(title=title, body=body, blocked_by=list(blocked)))
|
|
120
|
+
for i, issue in enumerate(planned, start=1):
|
|
121
|
+
for dep in issue.blocked_by:
|
|
122
|
+
if dep < 1 or dep > len(planned) or dep == i:
|
|
123
|
+
raise EnvelopeError(f"plan issue {i} has invalid dependency index {dep}")
|
|
124
|
+
return planned
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
# JSON Schemas handed to drivers that support structured output, and embedded
|
|
128
|
+
# in prompts for those that do not.
|
|
129
|
+
|
|
130
|
+
VERDICT_SCHEMA: dict[str, Any] = {
|
|
131
|
+
"type": "object",
|
|
132
|
+
"required": ["verdict"],
|
|
133
|
+
"properties": {
|
|
134
|
+
"verdict": {"enum": ["pass", "fail"]},
|
|
135
|
+
"findings": {
|
|
136
|
+
"type": "array",
|
|
137
|
+
"items": {
|
|
138
|
+
"type": "object",
|
|
139
|
+
"required": ["summary"],
|
|
140
|
+
"properties": {
|
|
141
|
+
"summary": {"type": "string"},
|
|
142
|
+
"detail": {"type": "string"},
|
|
143
|
+
"location": {"type": "string"},
|
|
144
|
+
},
|
|
145
|
+
},
|
|
146
|
+
},
|
|
147
|
+
},
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
WORKER_SCHEMA: dict[str, Any] = {
|
|
151
|
+
"type": "object",
|
|
152
|
+
"required": ["outcome", "summary"],
|
|
153
|
+
"properties": {
|
|
154
|
+
"outcome": {"enum": ["completed", "needs-human", "failed"]},
|
|
155
|
+
"summary": {"type": "string"},
|
|
156
|
+
"notes": {"type": "string"},
|
|
157
|
+
},
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
PLAN_SCHEMA: dict[str, Any] = {
|
|
161
|
+
"type": "object",
|
|
162
|
+
"required": ["issues"],
|
|
163
|
+
"properties": {
|
|
164
|
+
"issues": {
|
|
165
|
+
"type": "array",
|
|
166
|
+
"items": {
|
|
167
|
+
"type": "object",
|
|
168
|
+
"required": ["title", "body"],
|
|
169
|
+
"properties": {
|
|
170
|
+
"title": {"type": "string"},
|
|
171
|
+
"body": {"type": "string"},
|
|
172
|
+
"blocked_by": {"type": "array", "items": {"type": "integer"}},
|
|
173
|
+
},
|
|
174
|
+
},
|
|
175
|
+
}
|
|
176
|
+
},
|
|
177
|
+
}
|