giro 0.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- giro-0.0.1/.github/workflows/ci.yml +16 -0
- giro-0.0.1/.github/workflows/release.yml +25 -0
- giro-0.0.1/.gitignore +7 -0
- giro-0.0.1/LICENSE +21 -0
- giro-0.0.1/Makefile +9 -0
- giro-0.0.1/PKG-INFO +100 -0
- giro-0.0.1/README.md +83 -0
- giro-0.0.1/docs/design.md +112 -0
- giro-0.0.1/pyproject.toml +42 -0
- giro-0.0.1/skills/giro/SKILL.md +32 -0
- giro-0.0.1/skills/implement/SKILL.md +22 -0
- giro-0.0.1/src/giro/__init__.py +8 -0
- giro-0.0.1/src/giro/cli.py +126 -0
- giro-0.0.1/src/giro/config.py +168 -0
- giro-0.0.1/src/giro/drivers.py +213 -0
- giro-0.0.1/src/giro/envelope.py +177 -0
- giro-0.0.1/src/giro/gates.py +139 -0
- giro-0.0.1/src/giro/loops.py +323 -0
- giro-0.0.1/src/giro/states.py +48 -0
- giro-0.0.1/src/giro/store.py +243 -0
- giro-0.0.1/src/giro/workspace.py +82 -0
- giro-0.0.1/tests/__init__.py +0 -0
- giro-0.0.1/tests/conftest.py +83 -0
- giro-0.0.1/tests/test_drivers.py +118 -0
- giro-0.0.1/tests/test_envelope.py +71 -0
- giro-0.0.1/tests/test_gates.py +81 -0
- giro-0.0.1/tests/test_loops.py +189 -0
- giro-0.0.1/tests/test_states.py +32 -0
- giro-0.0.1/tests/test_store.py +53 -0
- giro-0.0.1/uv.lock +108 -0
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
check:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v4
|
|
13
|
+
- uses: astral-sh/setup-uv@v5
|
|
14
|
+
- run: uv sync
|
|
15
|
+
- run: uv run ruff check .
|
|
16
|
+
- run: uv run pytest -q
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
# Publishes to PyPI via trusted publishing (OIDC) — no tokens stored anywhere.
|
|
4
|
+
# One-time setup on pypi.org: Account -> Publishing -> add a pending publisher
|
|
5
|
+
# project: giro, owner: pierg, repository: giro,
|
|
6
|
+
# workflow: release.yml, environment: pypi
|
|
7
|
+
# Then push a v* tag or trigger this workflow manually.
|
|
8
|
+
|
|
9
|
+
on:
|
|
10
|
+
push:
|
|
11
|
+
tags: ["v*"]
|
|
12
|
+
workflow_dispatch:
|
|
13
|
+
|
|
14
|
+
jobs:
|
|
15
|
+
publish:
|
|
16
|
+
runs-on: ubuntu-latest
|
|
17
|
+
environment: pypi
|
|
18
|
+
permissions:
|
|
19
|
+
id-token: write
|
|
20
|
+
contents: read
|
|
21
|
+
steps:
|
|
22
|
+
- uses: actions/checkout@v4
|
|
23
|
+
- uses: astral-sh/setup-uv@v5
|
|
24
|
+
- run: uv build
|
|
25
|
+
- run: uv publish --trusted-publishing always
|
giro-0.0.1/.gitignore
ADDED
giro-0.0.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Piergiuseppe Mallozzi
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
giro-0.0.1/Makefile
ADDED
giro-0.0.1/PKG-INFO
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: giro
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: Guarded-loop engine for AI coding agents: Specs become verified work through bounded loops that end in proof or escalation, never a silent stop.
|
|
5
|
+
Project-URL: Repository, https://github.com/pierg/giro
|
|
6
|
+
Author: Piergiuseppe Mallozzi
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: agents,ai,automation,loops,tdd,verification
|
|
10
|
+
Classifier: Development Status :: 2 - Pre-Alpha
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
14
|
+
Classifier: Topic :: Software Development
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
|
|
18
|
+
# giro
|
|
19
|
+
|
|
20
|
+
**Guarded loops for AI coding agents.** giro turns a Spec into integrated, verified work through bounded loops that exit exactly two ways — **proof**, or a **hand raised to a human**. Nothing silent in between.
|
|
21
|
+
|
|
22
|
+
*Giro* — a lap, a loop, a tour in stages. The engine runs the laps; you write the route and take the finish.
|
|
23
|
+
|
|
24
|
+
> **Status: pre-alpha (M0).** The loop engine, state store, gate runner, and CLI exist and are fully tested against a fake driver. Live agent drivers are the next milestone. Successor to [pierg/orchestrion](https://github.com/pierg/orchestrion), rebuilt from scratch around a narrower thesis.
|
|
25
|
+
|
|
26
|
+
## The thesis
|
|
27
|
+
|
|
28
|
+
Most agent frameworks describe what an agent *should* do and hope it complies. giro inverts that:
|
|
29
|
+
|
|
30
|
+
> **Control never crosses an LLM. Content always does.**
|
|
31
|
+
|
|
32
|
+
The `giro` CLI is a deterministic loop engine. It owns every arrow — state transitions, budgets, gate execution, wave scheduling, escalation. LLM contexts are spawned *inside* the loop to do the creative work (implement, plan, judge), and each one ends by emitting a JSON envelope. A budget enforced by a `while` loop cannot be talked out of; an LLM "following instructions" can.
|
|
33
|
+
|
|
34
|
+
## One primitive, two levels
|
|
35
|
+
|
|
36
|
+
An **actor** makes an attempt. An independent **verifier** — a set of gates — judges it. All green: the loop exits with proof. Otherwise the findings feed the next attempt, until the **budget** runs out and the loop escalates to `needs-human`. It cannot just stop.
|
|
37
|
+
|
|
38
|
+
| Loop | Actor | Verifier | On fail |
|
|
39
|
+
|---|---|---|---|
|
|
40
|
+
| **Issue loop** | a fresh worker context | `[verify]` gates | retry, findings carried forward |
|
|
41
|
+
| **Spec loop** | a wave of Issue loops | `[validate]` gates | findings become gap Issues → next wave |
|
|
42
|
+
|
|
43
|
+
There is no third loop. Validate *is* the Spec loop's verifier; gap Issues *are* its feedback. The "lifecycle" is one feedback edge, not a controller.
|
|
44
|
+
|
|
45
|
+
## One verb
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
giro implement <target>
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
The router reads the target and picks the level. An Issue runs the Issue loop. A Spec runs the Spec loop — planned into Issues first if it has none. No second entrypoint, no mode flags. When a run ends `needs-human`, you answer (edit the Issue, decide the question) and run the same command again: re-invoking *is* the answer, and the budget resets.
|
|
52
|
+
|
|
53
|
+
Merging `giro/<spec>` into your main line stays your hand, always.
|
|
54
|
+
|
|
55
|
+
## Gates
|
|
56
|
+
|
|
57
|
+
A gate is anything that ends in `{ "verdict": "pass" | "fail", "findings": [...] }`.
|
|
58
|
+
|
|
59
|
+
- **command** — a shell command; exit 0 is pass. Deterministic and cheap.
|
|
60
|
+
- **prompt** — a rubric applied by a fresh, blind judge context.
|
|
61
|
+
- **skill** — a full SKILL.md procedure, same envelope at the end.
|
|
62
|
+
|
|
63
|
+
Judges start blind on every attempt; findings feed the next *actor*, never the next judge. A judge that can't produce the envelope gets one retry, then **fails closed**. All gates run to completion — the next attempt sees every finding, not just the first.
|
|
64
|
+
|
|
65
|
+
## State
|
|
66
|
+
|
|
67
|
+
Specs and Issues are markdown with tiny frontmatter, committed beside the code:
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
docs/specs/<slug>/SPEC.md state: draft | active | done | needs-human
|
|
71
|
+
docs/specs/<slug>/issues/NN-slug.md state: ready | in-progress | done | needs-human | wontfix
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
The store is the durable checkpoint: kill the engine at any point and `giro implement` resumes from exactly where the markdown says. Every attempt appends its story to the Issue file — the history is readable, not buried in a log.
|
|
75
|
+
|
|
76
|
+
## Quickstart
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
uv tool install giro # not yet published — for now: uv sync && uv run giro
|
|
80
|
+
cd your-project
|
|
81
|
+
giro init # writes giro.toml — set your [verify] gates
|
|
82
|
+
giro implement <spec> # go
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Exit codes are the contract: `0` proof, `2` needs-human, `1` error. CI and chat agents read them the same way.
|
|
86
|
+
|
|
87
|
+
## Layout
|
|
88
|
+
|
|
89
|
+
- [`docs/design.md`](docs/design.md) — the founding design: vocabulary, state machines, envelopes, loop contracts, decisions.
|
|
90
|
+
- [`skills/`](skills/) — the LLM touchpoints: what workers, judges, and chat doorways are told.
|
|
91
|
+
- [`src/giro/`](src/giro/) — the engine. `loops.py` is the heart.
|
|
92
|
+
|
|
93
|
+
## Roadmap
|
|
94
|
+
|
|
95
|
+
- **M0 — engine core** *(this)*: guarded loops, gates, store, CLI; proven end-to-end with a fake driver.
|
|
96
|
+
- **M1 — live driver**: `claude -p` worker/judge/planner contexts; first real Spec shipped by the machine.
|
|
97
|
+
- **M2 — parallel waves**: `concurrency > 1`, isolated worktrees, integration branch + re-verify.
|
|
98
|
+
- **M3 — the doorway**: `/giro` chat skill, escalation console, one-shot install (CLI + skills).
|
|
99
|
+
|
|
100
|
+
MIT — see [LICENSE](LICENSE).
|
giro-0.0.1/README.md
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# giro
|
|
2
|
+
|
|
3
|
+
**Guarded loops for AI coding agents.** giro turns a Spec into integrated, verified work through bounded loops that exit exactly two ways — **proof**, or a **hand raised to a human**. Nothing silent in between.
|
|
4
|
+
|
|
5
|
+
*Giro* — a lap, a loop, a tour in stages. The engine runs the laps; you write the route and take the finish.
|
|
6
|
+
|
|
7
|
+
> **Status: pre-alpha (M0).** The loop engine, state store, gate runner, and CLI exist and are fully tested against a fake driver. Live agent drivers are the next milestone. Successor to [pierg/orchestrion](https://github.com/pierg/orchestrion), rebuilt from scratch around a narrower thesis.
|
|
8
|
+
|
|
9
|
+
## The thesis
|
|
10
|
+
|
|
11
|
+
Most agent frameworks describe what an agent *should* do and hope it complies. giro inverts that:
|
|
12
|
+
|
|
13
|
+
> **Control never crosses an LLM. Content always does.**
|
|
14
|
+
|
|
15
|
+
The `giro` CLI is a deterministic loop engine. It owns every arrow — state transitions, budgets, gate execution, wave scheduling, escalation. LLM contexts are spawned *inside* the loop to do the creative work (implement, plan, judge), and each one ends by emitting a JSON envelope. A budget enforced by a `while` loop cannot be talked out of; an LLM "following instructions" can.
|
|
16
|
+
|
|
17
|
+
## One primitive, two levels
|
|
18
|
+
|
|
19
|
+
An **actor** makes an attempt. An independent **verifier** — a set of gates — judges it. All green: the loop exits with proof. Otherwise the findings feed the next attempt, until the **budget** runs out and the loop escalates to `needs-human`. It cannot just stop.
|
|
20
|
+
|
|
21
|
+
| Loop | Actor | Verifier | On fail |
|
|
22
|
+
|---|---|---|---|
|
|
23
|
+
| **Issue loop** | a fresh worker context | `[verify]` gates | retry, findings carried forward |
|
|
24
|
+
| **Spec loop** | a wave of Issue loops | `[validate]` gates | findings become gap Issues → next wave |
|
|
25
|
+
|
|
26
|
+
There is no third loop. Validate *is* the Spec loop's verifier; gap Issues *are* its feedback. The "lifecycle" is one feedback edge, not a controller.
|
|
27
|
+
|
|
28
|
+
## One verb
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
giro implement <target>
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
The router reads the target and picks the level. An Issue runs the Issue loop. A Spec runs the Spec loop — planned into Issues first if it has none. No second entrypoint, no mode flags. When a run ends `needs-human`, you answer (edit the Issue, decide the question) and run the same command again: re-invoking *is* the answer, and the budget resets.
|
|
35
|
+
|
|
36
|
+
Merging `giro/<spec>` into your main line stays your hand, always.
|
|
37
|
+
|
|
38
|
+
## Gates
|
|
39
|
+
|
|
40
|
+
A gate is anything that ends in `{ "verdict": "pass" | "fail", "findings": [...] }`.
|
|
41
|
+
|
|
42
|
+
- **command** — a shell command; exit 0 is pass. Deterministic and cheap.
|
|
43
|
+
- **prompt** — a rubric applied by a fresh, blind judge context.
|
|
44
|
+
- **skill** — a full SKILL.md procedure, same envelope at the end.
|
|
45
|
+
|
|
46
|
+
Judges start blind on every attempt; findings feed the next *actor*, never the next judge. A judge that can't produce the envelope gets one retry, then **fails closed**. All gates run to completion — the next attempt sees every finding, not just the first.
|
|
47
|
+
|
|
48
|
+
## State
|
|
49
|
+
|
|
50
|
+
Specs and Issues are markdown with tiny frontmatter, committed beside the code:
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
docs/specs/<slug>/SPEC.md state: draft | active | done | needs-human
|
|
54
|
+
docs/specs/<slug>/issues/NN-slug.md state: ready | in-progress | done | needs-human | wontfix
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
The store is the durable checkpoint: kill the engine at any point and `giro implement` resumes from exactly where the markdown says. Every attempt appends its story to the Issue file — the history is readable, not buried in a log.
|
|
58
|
+
|
|
59
|
+
## Quickstart
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
uv tool install giro # not yet published — for now: uv sync && uv run giro
|
|
63
|
+
cd your-project
|
|
64
|
+
giro init # writes giro.toml — set your [verify] gates
|
|
65
|
+
giro implement <spec> # go
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Exit codes are the contract: `0` proof, `2` needs-human, `1` error. CI and chat agents read them the same way.
|
|
69
|
+
|
|
70
|
+
## Layout
|
|
71
|
+
|
|
72
|
+
- [`docs/design.md`](docs/design.md) — the founding design: vocabulary, state machines, envelopes, loop contracts, decisions.
|
|
73
|
+
- [`skills/`](skills/) — the LLM touchpoints: what workers, judges, and chat doorways are told.
|
|
74
|
+
- [`src/giro/`](src/giro/) — the engine. `loops.py` is the heart.
|
|
75
|
+
|
|
76
|
+
## Roadmap
|
|
77
|
+
|
|
78
|
+
- **M0 — engine core** *(this)*: guarded loops, gates, store, CLI; proven end-to-end with a fake driver.
|
|
79
|
+
- **M1 — live driver**: `claude -p` worker/judge/planner contexts; first real Spec shipped by the machine.
|
|
80
|
+
- **M2 — parallel waves**: `concurrency > 1`, isolated worktrees, integration branch + re-verify.
|
|
81
|
+
- **M3 — the doorway**: `/giro` chat skill, escalation console, one-shot install (CLI + skills).
|
|
82
|
+
|
|
83
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# giro — founding design
|
|
2
|
+
|
|
3
|
+
This is the contract the engine is built against. The prose vision lives in the [README](../README.md); this document is the part you can hold the code to.
|
|
4
|
+
|
|
5
|
+
## Vocabulary
|
|
6
|
+
|
|
7
|
+
- **Spec** — forward-looking intent for one shippable feature. `docs/specs/<slug>/SPEC.md`.
|
|
8
|
+
- **Issue** — one bounded, demoable slice of a Spec. `docs/specs/<slug>/issues/NN-slug.md`.
|
|
9
|
+
- **Gate** — a verification unit ending in a verdict envelope. Kinds: `command`, `prompt`, `skill`.
|
|
10
|
+
- **Verify** — the Issue-level gate set (`[verify]` in giro.toml).
|
|
11
|
+
- **Validate** — the Spec-level gate set (`[validate]`). It is the Spec loop's verifier, not a separate stage.
|
|
12
|
+
- **Worker** — a fresh LLM context that implements one Issue. Edits files; never commits.
|
|
13
|
+
- **Judge** — a fresh, blind LLM context that applies one prompt/skill gate.
|
|
14
|
+
- **Planner** — a fresh LLM context that decomposes a Spec into Issues. Returns a plan; the engine writes the files.
|
|
15
|
+
- **Driver** — how a context is spawned: `claude`, `agy`, `codex`, `gemini` (plus `FakeDriver` in tests). One contract: prompt in, JSON envelope out; CLI failures surface the CLI's own message and fail the attempt, never a bare exit code.
|
|
16
|
+
- **Finding** — one structured problem `{summary, detail?, location?, gate}`. Findings feed the next actor.
|
|
17
|
+
- **Budget** — the bound that turns "loop until green" into "loop until green or ask."
|
|
18
|
+
- **needs-human** — the only non-proof exit. Both state machines land here when a budget is spent.
|
|
19
|
+
|
|
20
|
+
## The topology rule
|
|
21
|
+
|
|
22
|
+
**Control never crosses an LLM. Content always does.**
|
|
23
|
+
|
|
24
|
+
The engine (deterministic Python) owns: state transitions, attempt counting, gate execution, wave scheduling, commits, escalation. LLM contexts own: implementing, planning, judging. The engine's view of any context is identical — *spawn via driver, wait, parse envelope*. A context that cannot produce its envelope is a failed attempt or a failed gate (fail closed), never a shrug.
|
|
25
|
+
|
|
26
|
+
Corollaries:
|
|
27
|
+
|
|
28
|
+
- Workers never run `git commit`; the engine commits. The engine's own bookkeeping (claim, state moves) is committed **separately** from worker output, so "worker produced no changes" is detectable.
|
|
29
|
+
- Judges start blind on every attempt — no memory of prior rounds. Findings feed the next *actor*, never the next judge.
|
|
30
|
+
- Workers never edit `docs/specs/` — state belongs to the engine.
|
|
31
|
+
|
|
32
|
+
## State machines
|
|
33
|
+
|
|
34
|
+
Stored frontmatter is exactly what the loops branch on. `blocked` is computed from `blocked_by` at scheduling time, never stored.
|
|
35
|
+
|
|
36
|
+
### Issue — `state:` in frontmatter, plus `blocked_by: []`, `attempts: N`
|
|
37
|
+
|
|
38
|
+
| From | To | Actor | When |
|
|
39
|
+
|---|---|---|---|
|
|
40
|
+
| ready | in-progress | engine | attempt starts (claim committed first) |
|
|
41
|
+
| in-progress | done | engine | worker completed + all Verify gates green |
|
|
42
|
+
| in-progress | ready | engine | attempt failed, budget remains (findings carried) |
|
|
43
|
+
| in-progress | needs-human | engine | budget spent, or worker escalated |
|
|
44
|
+
| needs-human | ready | human | re-invoking `giro implement` on the issue *is* the answer; attempts reset |
|
|
45
|
+
| any | wontfix | human | by editing the file; the engine never sets it |
|
|
46
|
+
|
|
47
|
+
Terminal for scheduling: `done`, `wontfix`.
|
|
48
|
+
|
|
49
|
+
### Spec — `state:` in frontmatter, plus `base: <sha>`, `gap_cycles: N`
|
|
50
|
+
|
|
51
|
+
| From | To | Actor | When |
|
|
52
|
+
|---|---|---|---|
|
|
53
|
+
| draft | active | engine | first `giro implement <slug>` |
|
|
54
|
+
| active | done | engine | all Issues terminal + all Validate gates green |
|
|
55
|
+
| active | needs-human | engine | any Issue stuck needs-human, or validate budget spent |
|
|
56
|
+
| needs-human | active | human | re-invoking `giro implement` *is* the answer |
|
|
57
|
+
|
|
58
|
+
`done` means: **Validate passed; the human merges `giro/<slug>`.** There is no `shipped` state, no Feature record, no auto-merge. The machine's job ends where your branch review begins.
|
|
59
|
+
|
|
60
|
+
## Envelopes
|
|
61
|
+
|
|
62
|
+
Every context ends with exactly one JSON object. Schemas in [`envelope.py`](../src/giro/envelope.py):
|
|
63
|
+
|
|
64
|
+
```json
|
|
65
|
+
// worker // judge (gate verdict) // planner
|
|
66
|
+
{ "outcome": "completed | needs-human | failed",
|
|
67
|
+
"summary": "…", { "verdict": "pass | fail", { "issues": [ { "title", "body",
|
|
68
|
+
"notes": "…" } "findings": [ {"summary", "blocked_by": [1] } ] }
|
|
69
|
+
"detail", "location"} ] }
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Parsing is strict and fails closed: a malformed judge envelope gets one retry (with the validation error appended), then the gate fails with a `failed closed` finding. A malformed worker envelope consumes the attempt.
|
|
73
|
+
|
|
74
|
+
## The loops
|
|
75
|
+
|
|
76
|
+
Both loops instantiate one primitive — actor → verifier → proof | escalation — with these bindings:
|
|
77
|
+
|
|
78
|
+
| Clause | Issue loop | Spec loop |
|
|
79
|
+
|---|---|---|
|
|
80
|
+
| objective | Issue `done` | Spec `done` |
|
|
81
|
+
| actor | worker context | wave of Issue loops |
|
|
82
|
+
| verifier | `[verify]` gates | `[validate]` gates |
|
|
83
|
+
| findings feed | next worker attempt | gap Issues → next wave |
|
|
84
|
+
| budget | `issue_attempts` per Issue | `validate_cycles` gap cycles |
|
|
85
|
+
| escalation | Issue `needs-human` | Spec `needs-human` |
|
|
86
|
+
| memory | Issue frontmatter + attempt log | Spec frontmatter + gap Issues |
|
|
87
|
+
|
|
88
|
+
Gates run to completion — every gate runs, every finding is collected, so an attempt sees the whole picture. Command gates run first in config order; nothing short-circuits the evidence.
|
|
89
|
+
|
|
90
|
+
Waves: the frontier is every `ready` Issue whose `blocked_by` are all terminal. Issues in a wave are assumed scope-disjoint; collisions surface in Verify. At `concurrency = 1` (all of M0) the frontier runs sequentially — still one fresh context per attempt, never the invoking chat session.
|
|
91
|
+
|
|
92
|
+
## Decisions (locked)
|
|
93
|
+
|
|
94
|
+
1. **CLI owns the loop; skills are the touchpoints** (gnhf-shaped, skills readable). No LLM in the control path.
|
|
95
|
+
2. **One verb.** `giro implement <target>`, routed by what the target is. No mode flags, no second entrypoint.
|
|
96
|
+
3. **Empty Spec → planner context**, engine writes the Issue files (may be a single Issue). No synthetic in-memory units.
|
|
97
|
+
4. **`done` = Validate passed + human merges the branch.** No ship stage, no Feature records, no auto-merge to main.
|
|
98
|
+
5. **Local vs parallel is one code path** — `concurrency` decides; worktree isolation arrives at M2 without a topology change.
|
|
99
|
+
6. **Fail closed everywhere.** Malformed envelope = failed attempt/gate. Dirty tree = refuse to run.
|
|
100
|
+
7. **Re-invocation is the human answer.** `needs-human` + `giro implement` again → state resets, budget resets.
|
|
101
|
+
8. **Markdown is the memory.** Kill the engine anywhere; the store resumes it. Attempt history lives in the Issue file, human-readable.
|
|
102
|
+
|
|
103
|
+
## Non-goals
|
|
104
|
+
|
|
105
|
+
Bug intake/triage queues, tracker mirrors (GitHub/Linear sync), Feature documentation, merge-to-main automation, multi-repo orchestration. Some may return later; none are core.
|
|
106
|
+
|
|
107
|
+
## Milestones
|
|
108
|
+
|
|
109
|
+
- **M0 — engine core** (done): everything above, proven by `tests/` against `FakeDriver`.
|
|
110
|
+
- **M1 — live driver**: harden `ClaudeDriver` (permissions, sandboxing, structured output), first real Spec end-to-end.
|
|
111
|
+
- **M2 — parallel waves**: worktrees per worker, integration branch, integrated re-Verify after merge.
|
|
112
|
+
- **M3 — the doorway**: `/giro` chat skill + escalation console; one-shot install placing CLI + skills together.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "giro"
|
|
3
|
+
version = "0.0.1"
|
|
4
|
+
description = "Guarded-loop engine for AI coding agents: Specs become verified work through bounded loops that end in proof or escalation, never a silent stop."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = "MIT"
|
|
8
|
+
authors = [{ name = "Piergiuseppe Mallozzi" }]
|
|
9
|
+
keywords = ["agents", "ai", "automation", "verification", "tdd", "loops"]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 2 - Pre-Alpha",
|
|
12
|
+
"Environment :: Console",
|
|
13
|
+
"Intended Audience :: Developers",
|
|
14
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
15
|
+
"Topic :: Software Development",
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
[project.scripts]
|
|
19
|
+
giro = "giro.cli:main"
|
|
20
|
+
|
|
21
|
+
[project.urls]
|
|
22
|
+
Repository = "https://github.com/pierg/giro"
|
|
23
|
+
|
|
24
|
+
[build-system]
|
|
25
|
+
requires = ["hatchling"]
|
|
26
|
+
build-backend = "hatchling.build"
|
|
27
|
+
|
|
28
|
+
[tool.hatch.build.targets.wheel]
|
|
29
|
+
packages = ["src/giro"]
|
|
30
|
+
|
|
31
|
+
[dependency-groups]
|
|
32
|
+
dev = ["pytest>=8", "ruff>=0.6"]
|
|
33
|
+
|
|
34
|
+
[tool.ruff]
|
|
35
|
+
line-length = 100
|
|
36
|
+
src = ["src", "tests"]
|
|
37
|
+
|
|
38
|
+
[tool.ruff.lint]
|
|
39
|
+
select = ["E", "F", "W", "I", "UP", "B", "SIM"]
|
|
40
|
+
|
|
41
|
+
[tool.pytest.ini_options]
|
|
42
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: giro
|
|
3
|
+
description: Drive giro from a chat session — start guarded loops, relay progress, and help the human answer needs-human escalations. Use when the user asks to implement a Spec or Issue with giro, check giro status, or resolve a giro escalation.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# giro — the doorway
|
|
7
|
+
|
|
8
|
+
giro is a deterministic loop engine: it turns Specs and Issues under `docs/specs/` into verified work on a `giro/<slug>` branch, exiting only through proof or `needs-human`. You do not implement anything yourself — the engine spawns fresh worker contexts for that. Your job is the conversation around the machine.
|
|
9
|
+
|
|
10
|
+
## Commands
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
giro status # spec and issue states
|
|
14
|
+
giro implement <target> # spec slug, issue id, or slug/issue-id — starts or resumes the loop
|
|
15
|
+
giro verify # run the [verify] gate set once, on demand
|
|
16
|
+
giro init # first-time setup: writes giro.toml
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Exit codes: `0` proof, `2` needs-human, `1` error.
|
|
20
|
+
|
|
21
|
+
## How to behave
|
|
22
|
+
|
|
23
|
+
1. **Starting work.** Resolve what the user wants into one target (`giro status` helps), then run `giro implement <target>` and relay the report: outcome, per-issue states, and where the work landed (`giro/<slug>` branch).
|
|
24
|
+
2. **On `needs-human` (exit 2).** This is the escalation console — your main job. Open the Issue file, read its attempt log (the trailing `## Attempt N` / `## Budget exhausted` sections), and present the blocker plainly. Discuss it with the user. Their answer usually lands as an edit to the Issue body (sharper acceptance criteria, the decision made explicit). Then run the same `giro implement` again — **re-invoking is the answer**; the engine resets the budget.
|
|
25
|
+
3. **On proof (exit 0).** Tell the user what is on `giro/<slug>` and remind them the merge into their main line is theirs to make. Never merge it yourself.
|
|
26
|
+
4. **Planning.** If there is no Spec yet, help the user write `docs/specs/<slug>/SPEC.md` conversationally before reaching for the engine. A Spec with no Issues is fine — the engine plans it.
|
|
27
|
+
|
|
28
|
+
## Hard rules
|
|
29
|
+
|
|
30
|
+
- Never edit Issue/Spec `state:` frontmatter or attempt logs — state belongs to the engine; you edit *bodies* when the user answers an escalation.
|
|
31
|
+
- Never bypass the engine by implementing Issues in this chat context.
|
|
32
|
+
- Never merge `giro/<slug>` into the main branch.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: implement
|
|
3
|
+
description: Discipline for a giro worker context implementing one Issue. Injected into every worker task packet by the engine; edit this file to tune how workers in this project behave.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Implement — worker discipline
|
|
7
|
+
|
|
8
|
+
You are implementing one Issue. The engine chose it, will commit your work, will run the project's gates, and will decide what happens next. Your job is only the change itself.
|
|
9
|
+
|
|
10
|
+
## Discipline
|
|
11
|
+
|
|
12
|
+
1. **Read before writing.** Understand the Issue's acceptance criteria and the surrounding code before editing. Match the codebase's existing style, naming, and idiom.
|
|
13
|
+
2. **Test-first.** Write the smallest failing test that captures the Issue's behavior, watch it fail, make it pass, then clean up. Run only the tests you author — the project-wide gate is the engine's job, not yours.
|
|
14
|
+
3. **Stay on the slice.** Implement this Issue, nothing adjacent. If you notice unrelated problems, mention them in your envelope `notes` — do not fix them.
|
|
15
|
+
4. **Findings first.** If the task packet carries findings from a previous attempt, fix those before anything else — they are why the last attempt failed.
|
|
16
|
+
|
|
17
|
+
## Hard rules
|
|
18
|
+
|
|
19
|
+
- Never run `git commit`, `git push`, or anything under `.git` — the engine owns version control.
|
|
20
|
+
- Never edit files under `docs/specs/` — Spec and Issue state belongs to the engine.
|
|
21
|
+
- Never weaken, skip, or delete an existing test to get to green.
|
|
22
|
+
- If the Issue cannot be done as written — ambiguous, contradictory, or requiring a decision that is not yours — stop and return `needs-human` with the question in `summary`. A good question beats a wrong guess.
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""The giro CLI — the product's front door.
|
|
2
|
+
|
|
3
|
+
Exit codes are part of the contract: 0 = proof, 2 = needs-human, 1 = error.
|
|
4
|
+
A CI job or a chat agent reads them the same way.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import argparse
|
|
10
|
+
import sys
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
from . import __version__
|
|
14
|
+
from .config import CONFIG_NAME, INIT_TEMPLATE, ConfigError, load_config
|
|
15
|
+
from .drivers import DriverError, build_driver
|
|
16
|
+
from .envelope import EnvelopeError
|
|
17
|
+
from .gates import build_context, run_gates
|
|
18
|
+
from .loops import Engine
|
|
19
|
+
from .store import Store, StoreError
|
|
20
|
+
from .workspace import Workspace, WorkspaceError
|
|
21
|
+
|
|
22
|
+
EXIT_PROOF = 0
|
|
23
|
+
EXIT_ERROR = 1
|
|
24
|
+
EXIT_NEEDS_HUMAN = 2
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _build_engine(root: Path) -> Engine:
|
|
28
|
+
cfg = load_config(root)
|
|
29
|
+
drivers = {}
|
|
30
|
+
for role in ("implementer", "judge", "planner"):
|
|
31
|
+
entry = cfg.roster_for(role)
|
|
32
|
+
drivers[role] = build_driver(entry.driver, args=entry.args)
|
|
33
|
+
return Engine(cfg=cfg, store=Store(root), workspace=Workspace(root), drivers=drivers)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def cmd_init(root: Path) -> int:
|
|
37
|
+
path = root / CONFIG_NAME
|
|
38
|
+
if path.exists():
|
|
39
|
+
print(f"{CONFIG_NAME} already exists — edit it directly.")
|
|
40
|
+
return EXIT_ERROR
|
|
41
|
+
path.write_text(INIT_TEMPLATE, encoding="utf-8")
|
|
42
|
+
(root / "docs" / "specs").mkdir(parents=True, exist_ok=True)
|
|
43
|
+
print(f"Wrote {CONFIG_NAME} and docs/specs/.")
|
|
44
|
+
print("Next: set your [verify] gates, then write a Spec and run `giro implement <slug>`.")
|
|
45
|
+
return EXIT_PROOF
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def cmd_status(root: Path) -> int:
|
|
49
|
+
store = Store(root)
|
|
50
|
+
specs = store.list_specs()
|
|
51
|
+
if not specs:
|
|
52
|
+
print("No specs under docs/specs/.")
|
|
53
|
+
return EXIT_PROOF
|
|
54
|
+
for spec in specs:
|
|
55
|
+
print(f"{spec.slug} [{spec.state}] {spec.title}")
|
|
56
|
+
for issue in store.load_issues(spec.slug):
|
|
57
|
+
deps = f" blocked_by={','.join(issue.blocked_by)}" if issue.blocked_by else ""
|
|
58
|
+
print(
|
|
59
|
+
f" {issue.id} [{issue.state}] attempts={issue.attempts}{deps} {issue.title}"
|
|
60
|
+
)
|
|
61
|
+
return EXIT_PROOF
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def cmd_verify(root: Path) -> int:
|
|
65
|
+
engine = _build_engine(root)
|
|
66
|
+
ctx = build_context(
|
|
67
|
+
engine.cfg, root, "Manual `giro verify` run: judge the working tree.",
|
|
68
|
+
engine.drivers["judge"],
|
|
69
|
+
)
|
|
70
|
+
green, findings = run_gates(engine.cfg.verify_gates, ctx)
|
|
71
|
+
for finding in findings:
|
|
72
|
+
print(f"[{finding.gate}] {finding.summary}")
|
|
73
|
+
if finding.detail:
|
|
74
|
+
print(f" {finding.detail[:500]}")
|
|
75
|
+
print("all green" if green else f"{len(findings)} finding(s)")
|
|
76
|
+
return EXIT_PROOF if green else EXIT_NEEDS_HUMAN
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def cmd_implement(root: Path, target: str) -> int:
|
|
80
|
+
engine = _build_engine(root)
|
|
81
|
+
report = engine.implement(target)
|
|
82
|
+
print(f"{report.target}: {report.outcome} — {report.detail}")
|
|
83
|
+
for ref, state in sorted(report.issues.items()):
|
|
84
|
+
print(f" {ref}: {state}")
|
|
85
|
+
if report.outcome in ("done", "all-done"):
|
|
86
|
+
return EXIT_PROOF
|
|
87
|
+
return EXIT_NEEDS_HUMAN
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def main(argv: list[str] | None = None) -> int:
|
|
91
|
+
parser = argparse.ArgumentParser(
|
|
92
|
+
prog="giro",
|
|
93
|
+
description="Guarded loops: Specs become verified work — proof or escalation.",
|
|
94
|
+
)
|
|
95
|
+
parser.add_argument("--version", action="version", version=f"giro {__version__}")
|
|
96
|
+
parser.add_argument(
|
|
97
|
+
"-C", dest="root", default=".", help="project root (default: current directory)"
|
|
98
|
+
)
|
|
99
|
+
sub = parser.add_subparsers(dest="command")
|
|
100
|
+
sub.add_parser("init", help="write a starter giro.toml")
|
|
101
|
+
sub.add_parser("status", help="show spec and issue states")
|
|
102
|
+
sub.add_parser("verify", help="run the [verify] gate set once")
|
|
103
|
+
p_impl = sub.add_parser("implement", help="run the guarded loop for a spec or issue")
|
|
104
|
+
p_impl.add_argument("target", help="spec slug, issue id, or slug/issue-id")
|
|
105
|
+
|
|
106
|
+
args = parser.parse_args(argv)
|
|
107
|
+
root = Path(args.root).resolve()
|
|
108
|
+
|
|
109
|
+
try:
|
|
110
|
+
if args.command == "init":
|
|
111
|
+
return cmd_init(root)
|
|
112
|
+
if args.command == "status":
|
|
113
|
+
return cmd_status(root)
|
|
114
|
+
if args.command == "verify":
|
|
115
|
+
return cmd_verify(root)
|
|
116
|
+
if args.command == "implement":
|
|
117
|
+
return cmd_implement(root, args.target)
|
|
118
|
+
parser.print_help()
|
|
119
|
+
return EXIT_PROOF
|
|
120
|
+
except (ConfigError, StoreError, WorkspaceError, DriverError, EnvelopeError) as exc:
|
|
121
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
122
|
+
return EXIT_ERROR
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
if __name__ == "__main__":
|
|
126
|
+
sys.exit(main())
|