cohorte 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +85 -1
- package/README.md +20 -12
- package/bin/cli.js +8 -0
- package/core/agents/review.md +23 -0
- package/core/commands/audit.md +9 -1
- package/core/commands/brainstorm.md +6 -0
- package/core/commands/build.md +90 -5
- package/core/commands/doctor.md +8 -3
- package/core/commands/{loop.md → drive.md} +25 -6
- package/core/commands/fix.md +5 -0
- package/core/commands/review.md +58 -9
- package/core/commands/spec.md +20 -0
- package/core/commands/update-pipeline.md +6 -1
- package/core/templates/decisions.template.md +42 -0
- package/core/templates/spec.template.md +3 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +2 -2
- package/core/workflows/review.js +44 -2
- package/dashboard/dist/assets/{index-dkO8UUVl.css → index-BZ_LQlEj.css} +1 -1
- package/dashboard/dist/assets/{index-8owBnqyv.js → index-DYyn4p93.js} +11 -11
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +8 -1
- package/install.ps1 +8 -0
- package/install.sh +7 -0
- package/package.json +1 -1
- package/profile/SCHEMA.md +140 -1
- package/scripts/loop.sh +141 -12
- package/scripts/test-loop.mjs +227 -0
- package/scripts/test-workflows.mjs +28 -0
- package/scripts/validate-core.mjs +16 -1
|
@@ -7,8 +7,8 @@
|
|
|
7
7
|
<link rel="icon" type="image/png" sizes="16x16" href="./favicon-16.png" />
|
|
8
8
|
<link rel="apple-touch-icon" sizes="180x180" href="./apple-touch-icon-180.png" />
|
|
9
9
|
<title>cohorte · dashboard</title>
|
|
10
|
-
<script type="module" crossorigin src="./assets/index-
|
|
11
|
-
<link rel="stylesheet" crossorigin href="./assets/index-
|
|
10
|
+
<script type="module" crossorigin src="./assets/index-DYyn4p93.js"></script>
|
|
11
|
+
<link rel="stylesheet" crossorigin href="./assets/index-BZ_LQlEj.css">
|
|
12
12
|
</head>
|
|
13
13
|
<body>
|
|
14
14
|
<div id="root"></div>
|
|
@@ -19,7 +19,10 @@ const FIXED_AGENTS = new Set([
|
|
|
19
19
|
'implementer.template',
|
|
20
20
|
]);
|
|
21
21
|
|
|
22
|
-
|
|
22
|
+
// The spec lifecycle (SCHEMA.md §Spec status). `in-progress` and `blocked` are written by
|
|
23
|
+
// the /drive driver — they are what makes an interrupted autonomous loop resumable, so a
|
|
24
|
+
// dashboard that flagged them as invalid would report the pipeline's own state as a defect.
|
|
25
|
+
const VALID_STATUS = ['draft', 'frozen', 'in-progress', 'in-review', 'shipped', 'blocked'];
|
|
23
26
|
|
|
24
27
|
// Artifacts the pipeline itself writes into specs/ that are NOT feature specs and have no
|
|
25
28
|
// front-matter status. `/audit` writes specs/refactor-backlog.md by design, so scanning it
|
|
@@ -281,12 +284,16 @@ function scanSpecs(projectRoot) {
|
|
|
281
284
|
const body = fm ? fm[1] : '';
|
|
282
285
|
const get = k => { const m = body.match(new RegExp(`^${k}:\\s*(.*)$`, 'm')); return m ? m[1].trim() : null; };
|
|
283
286
|
const status = get('status');
|
|
287
|
+
// The loop driver's resume state, when a /drive is (or was) running on this spec.
|
|
288
|
+
const pass = parseInt(get('loop_pass'), 10);
|
|
289
|
+
const phase = get('loop_phase');
|
|
284
290
|
specs.push({
|
|
285
291
|
file: f,
|
|
286
292
|
id: get('feature_id') || f.replace(/\.md$/, ''),
|
|
287
293
|
title: get('title'),
|
|
288
294
|
status: status ? status.split('#')[0].trim() : null,
|
|
289
295
|
branch: get('branch'),
|
|
296
|
+
loop: pass > 0 ? { pass, phase: phase && phase !== 'done' ? phase : null } : null,
|
|
290
297
|
});
|
|
291
298
|
}
|
|
292
299
|
return specs;
|
package/install.ps1
CHANGED
|
@@ -193,6 +193,14 @@ try {
|
|
|
193
193
|
# 1.5.0 removed the /smoke phase; copy-over never deletes, so scrub the orphan agent.
|
|
194
194
|
Remove-Item -LiteralPath (Join-Path $dest 'agents\smoke.md') -Force -ErrorAction SilentlyContinue
|
|
195
195
|
Remove-Item -LiteralPath (Join-Path $dest 'commands\smoke.md') -Force -ErrorAction SilentlyContinue
|
|
196
|
+
# 1.4.0 removed /cycle and its workflow — and no installer ever scrubbed them, so every
|
|
197
|
+
# install since has kept offering a command that dispatches a workflow whose phases were
|
|
198
|
+
# later deleted. A dead command is worse than a missing one: the model can still fire it.
|
|
199
|
+
Remove-Item -LiteralPath (Join-Path $dest 'commands\cycle.md') -Force -ErrorAction SilentlyContinue
|
|
200
|
+
Remove-Item -LiteralPath (Join-Path $dest 'workflows\cycle.js') -Force -ErrorAction SilentlyContinue
|
|
201
|
+
# 1.6.0 renamed /loop → /drive: Claude Code's own built-in /loop shadowed ours, so a leftover
|
|
202
|
+
# commands\loop.md is a command the user can never reach — scrub it rather than leave a decoy.
|
|
203
|
+
Remove-Item -LiteralPath (Join-Path $dest 'commands\loop.md') -Force -ErrorAction SilentlyContinue
|
|
196
204
|
# 0.1.19 split the bi-mode questionnaire-researcher into research-agent + questionnaire-architect;
|
|
197
205
|
# copy-over never deletes, so scrub the retired agent lest a dead subagent_type linger.
|
|
198
206
|
Remove-Item -LiteralPath (Join-Path $dest 'agents\questionnaire-researcher.md') -Force -ErrorAction SilentlyContinue
|
package/install.sh
CHANGED
|
@@ -154,6 +154,13 @@ copy_fixed_agents() {
|
|
|
154
154
|
"$dest/agents/"
|
|
155
155
|
# 1.5.0 removed the /smoke phase; copy-over never deletes, so scrub the orphan agent.
|
|
156
156
|
rm -f "$dest/agents/smoke.md" "$dest/commands/smoke.md"
|
|
157
|
+
# 1.4.0 removed /cycle and its workflow — and no installer ever scrubbed them, so every
|
|
158
|
+
# install since has kept offering a command that dispatches a workflow whose phases were
|
|
159
|
+
# later deleted. A dead command is worse than a missing one: the model can still fire it.
|
|
160
|
+
rm -f "$dest/commands/cycle.md" "$dest/workflows/cycle.js"
|
|
161
|
+
# 1.6.0 renamed /loop → /drive: Claude Code's own built-in /loop shadowed ours, so a leftover
|
|
162
|
+
# commands/loop.md is a command the user can never reach — scrub it rather than leave a decoy.
|
|
163
|
+
rm -f "$dest/commands/loop.md"
|
|
157
164
|
# 0.1.19 split the bi-mode questionnaire-researcher into research-agent + questionnaire-architect;
|
|
158
165
|
# copy-over never deletes, so scrub the retired agent lest a dead subagent_type linger.
|
|
159
166
|
rm -f "$dest/agents/questionnaire-researcher.md"
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cohorte",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"description": "Portable, stack-agnostic multi-agent development pipeline for Claude Code — install the core, run /init-pipeline, and it adapts to your project's stack.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"cohorte": "bin/cli.js"
|
package/profile/SCHEMA.md
CHANGED
|
@@ -217,6 +217,129 @@ Rules for every consumer (implementers, preflight, `/audit` gates, workflow agen
|
|
|
217
217
|
`/init-pipeline` **asks** for these variants (detected defaults offered first) instead of silently
|
|
218
218
|
storing a bare `pnpm test` as the thing agents execute; `/update-pipeline` tops up older profiles.
|
|
219
219
|
|
|
220
|
+
## Spec status — the lifecycle state machine (and the loop's resume state)
|
|
221
|
+
|
|
222
|
+
A spec's front-matter `status` is not a label, it is the pipeline's **state**: every command routes on
|
|
223
|
+
it, the dashboard boards on it, the kanban backfill maps it to a column, and `/drive --resume` reads it
|
|
224
|
+
back to continue an interrupted autonomous run. Six states, and exactly one writer each:
|
|
225
|
+
|
|
226
|
+
| status | meaning | written by | who may build it |
|
|
227
|
+
| --- | --- | --- | --- |
|
|
228
|
+
| `draft` | the interview is open, nothing is frozen | `/spec` Mode A | no |
|
|
229
|
+
| `frozen` | the contract is frozen — the handoff to `/build` | `/spec` Mode A freeze | yes |
|
|
230
|
+
| `in-progress` | a `/drive` is driving this spec right now (or died doing it) | `scripts/loop.sh`, before each phase | yes |
|
|
231
|
+
| `in-review` | reviewed / awaiting the next round or `/ship` | `/spec` Mode B, `/fix`, `loop.sh` on a clean exit | yes |
|
|
232
|
+
| `blocked` | a loop gave up here (ceiling, non-convergent, no verdict, not implementable) | `loop.sh` on any non-zero exit | yes, with the reason named |
|
|
233
|
+
| `shipped` | the PR is open; the status flip is part of the release commit | `/ship` | no |
|
|
234
|
+
|
|
235
|
+
**The resume contract.** Before every phase, `loop.sh` stamps `status: in-progress` plus `loop_pass`
|
|
236
|
+
(the review pass it is on) and `loop_phase` (`build`/`review`/`fix`) into the spec — deterministically,
|
|
237
|
+
with `awk`, spending **no tokens** on state it will need later. On exit it stamps a terminal status:
|
|
238
|
+
`in-review` + `loop_phase: done` when clean, `blocked` otherwise. `/drive <id> --resume` then continues
|
|
239
|
+
at the recorded pass instead of pass 1, so a session killed at pass 3 of 5 does not re-pay passes 1–2.
|
|
240
|
+
The build is still skipped or redone by the build stamp alone (`specs/reports/<id>.built`, written only
|
|
241
|
+
after a build that finished), so an interrupted *build* correctly rebuilds.
|
|
242
|
+
|
|
243
|
+
Corollaries worth knowing:
|
|
244
|
+
|
|
245
|
+
- A spec with no front-matter makes every stamp a **silent no-op** — the state is bookkeeping, and the
|
|
246
|
+
loop must never die over a status line.
|
|
247
|
+
- Child commands write `status` too (`/fix` sets `in-review`); re-stamping before each phase is what
|
|
248
|
+
keeps `in-progress` true for the duration of the run rather than for its first phase.
|
|
249
|
+
- `blocked` is not a failure to hide: it is the resumable state. `/build` accepts it, names it, and
|
|
250
|
+
routes by the spec's `## Remediation` (open items ⇒ `/fix`).
|
|
251
|
+
|
|
252
|
+
## Dead agents — silence is not a green light
|
|
253
|
+
|
|
254
|
+
A subagent can die mid-run: a rate limit, a transport error that outlived its retries, its own context
|
|
255
|
+
exhausted on a big surface. When it does it returns **nothing** — and nothing is byte-identical to
|
|
256
|
+
"finished, nothing to report". Every phase that fans out therefore does a **roll call** before it
|
|
257
|
+
integrates anything, because the default reading of silence is the most dangerous one available:
|
|
258
|
+
|
|
259
|
+
| phase | what a dead agent looks like | what the phase must do |
|
|
260
|
+
| --- | --- | --- |
|
|
261
|
+
| `/build` | a surface with no handoff | retry it **once** alone (byte-identical prompt), then mark it `dead`, verify the tree with that surface's own quiet commands, never call the batch ok |
|
|
262
|
+
| `/review` | a reviewer with no report ⇒ **zero findings** | retry once, then list the surface in `unreviewed` and refuse to score `SHIP` |
|
|
263
|
+
| `/fix` | a re-dispatched agent with no handoff | retry once, then leave **every** one of its items `- [ ]` — a dead agent never ticks a box |
|
|
264
|
+
| workflows | `agent()` resolves to `null` | already enforced (`review.js` `unreviewedSurfaces`) — the doctrine started here |
|
|
265
|
+
|
|
266
|
+
Non-negotiables, in every phase:
|
|
267
|
+
|
|
268
|
+
- **Retry once, alone, byte-identical.** Most deaths are transient, and the other surfaces' work is
|
|
269
|
+
already on disk — so recovery costs one agent, never a rebuild. Never retry an agent that answered.
|
|
270
|
+
- **Never speak for a dead agent.** You did not see its work: report what the *tree* says (quiet
|
|
271
|
+
commands, redirected to a file, grepped), not what a handoff would have said.
|
|
272
|
+
- **Never let it reach a driver as clean.** `/build` writes `dead[]` into
|
|
273
|
+
`specs/reports/<id>.build.json`, `/review` writes `unreviewed[]` into the verdict; `scripts/loop.sh`
|
|
274
|
+
aborts on either with **exit 2** *before* it reads `blocking`, since a dead reviewer makes
|
|
275
|
+
`blocking == 0` a statement about code nobody read.
|
|
276
|
+
- **`unreviewed` is separate from `blocking` on purpose.** Faking a count in `blocking` to force a
|
|
277
|
+
driver's hand would corrupt the one field the whole contract rests on; a driver reads them as two
|
|
278
|
+
different facts — "what was found" and "what was covered".
|
|
279
|
+
- **Write the metrics line anyway** (`"<key>":"dead"`). An incomplete batch is exactly the batch worth
|
|
280
|
+
recording; holding the append back "until it's complete" deletes the evidence that anything failed.
|
|
281
|
+
|
|
282
|
+
## Readiness — the gate between a frozen spec and N implementers
|
|
283
|
+
|
|
284
|
+
`/build` §1.6 scores the frozen spec on **implementability** before authoring the contract and before
|
|
285
|
+
dispatching anything, and writes `specs/reports/<id>.readiness.json`
|
|
286
|
+
(`verdict`: `READY` · `RESERVATIONS` · `NOT-READY`, plus `gaps[]`). It costs **zero extra agents** — the
|
|
287
|
+
lead already holds the spec, the profile and the reconciled surface list — which is the whole economics
|
|
288
|
+
of the step: a spec that cannot be built does not get cheaper by being built on N surfaces in parallel.
|
|
289
|
+
|
|
290
|
+
- Five checks: contract completeness · surface coverage · dependencies exist · residual ambiguity ·
|
|
291
|
+
the design gate. Each maps to `NOT-READY` (a surface would have to invent the answer) or
|
|
292
|
+
`RESERVATIONS` (a surface can proceed on a stated assumption).
|
|
293
|
+
- **`NOT-READY` aborts the build with no agent spawned** and sends the human to `/spec`.
|
|
294
|
+
`scripts/loop.sh` reads the same file and exits **4** (`not implementable`) — the one loop outcome
|
|
295
|
+
that more passes cannot fix.
|
|
296
|
+
- **`RESERVATIONS` never blocks.** Each gap is inlined verbatim into the dispatch of the surface it
|
|
297
|
+
affects, as an assumption the implementer must apply *and* flag in its handoff. A gate that stalled a
|
|
298
|
+
sound build on a missing error case would cost more human round-trips than it saves.
|
|
299
|
+
|
|
300
|
+
## Deferred findings — real, but not this feature's problem
|
|
301
|
+
|
|
302
|
+
`/review` ends on "zero blocking findings", so everything non-blocking used to be discarded with the
|
|
303
|
+
report. A **deferred** finding is one the reviewer judges true and **out of this feature's scope**
|
|
304
|
+
(pre-existing code the staged diff never touched, adjacent debt the spec never claims to fix). The
|
|
305
|
+
review agent returns them in their own `## Deferred` section — never in `findings` — each carrying its
|
|
306
|
+
own out-of-scope reason.
|
|
307
|
+
|
|
308
|
+
- They count in **no** severity row, enter **no** verdict, and are **never** cross-checked: a deferred
|
|
309
|
+
item cannot cost a fix loop an iteration, and refuting one would spend an agent arguing about
|
|
310
|
+
something that cannot change the outcome.
|
|
311
|
+
- **Not deferrable, ever:** anything the diff touched or introduced, any spec violation, any security
|
|
312
|
+
issue on a path this feature adds, calls or modifies.
|
|
313
|
+
- `/review` §3.5 routes them, **on every verdict**, into `specs/refactor-backlog.md` under the
|
|
314
|
+
`## <domain>` heading of the owning surface, tagged `deferred:<feature_id>` — the same grouping
|
|
315
|
+
`/audit` writes, so `/refactor <domain>` picks them up with no extra plumbing. Never into the spec's
|
|
316
|
+
`## Remediation`, which is what `/fix` re-dispatches.
|
|
317
|
+
- `/audit` **carries open `deferred:` items over** when it rewrites the backlog; overwriting them away
|
|
318
|
+
is the one way they silently vanish.
|
|
319
|
+
- The verdict JSON carries `deferred: <n>` (informational, outside `blocking`), so `/drive` can name
|
|
320
|
+
them in its closing line without reading a report.
|
|
321
|
+
|
|
322
|
+
## Decisions — the transverse decision journal
|
|
323
|
+
|
|
324
|
+
`PIPELINE.md` is a **stack profile** (surfaces, commands, conventions); it says nothing about what this
|
|
325
|
+
project has *decided*. Without somewhere for those, every `/spec` re-discovers or contradicts them.
|
|
326
|
+
`specs/_decisions.md` (from `core/templates/decisions.template.md`) is that place, deliberately small:
|
|
327
|
+
|
|
328
|
+
- **Append-only, one line per decision, ≤ ~160 chars:**
|
|
329
|
+
`- <YYYY-MM-DD> · <area> · <decision> — because <reason> · <feature_id>`. Reversal never edits a line:
|
|
330
|
+
append a superseding one (`· supersedes <date> <area>`) and move the old one to `## Superseded`. When
|
|
331
|
+
`## Live` passes ~100 lines, sweep the superseded ones down.
|
|
332
|
+
- **Written by** `/spec` at freeze (the decisions that outlive the feature — typically 0–3 lines, and
|
|
333
|
+
zero is a normal outcome) and `/build` §1.5 when it adds or splits a surface.
|
|
334
|
+
- **Read by the deciding stages only** — `/brainstorm` (so the panel argues about the idea, not about
|
|
335
|
+
settled ground), `/spec` (so a new spec does not silently un-decide something), `/audit` (standing
|
|
336
|
+
decisions are part of the rulebook it audits against).
|
|
337
|
+
- **Never read by implementers or reviewers.** They work from the frozen contract, which already tells
|
|
338
|
+
them what to do; shipping them the rationale would cost `surfaces × dispatches` tokens per feature
|
|
339
|
+
for a fact they cannot act on. This is what keeps the journal cheap enough to be worth having.
|
|
340
|
+
- The `_` prefix is load-bearing: `/doctor`, the dashboard spec scanner and the kanban backfill all skip
|
|
341
|
+
`specs/_*.md`, so the journal is never mistaken for a spec (no phantom card, no bogus stage).
|
|
342
|
+
|
|
220
343
|
## Preflight — the deterministic phase gate
|
|
221
344
|
|
|
222
345
|
`/review` starts by running `pipeline/scripts/preflight.sh` — a plain shell script (no
|
|
@@ -313,6 +436,15 @@ files automatically. It works because every generated artifact is a **determinis
|
|
|
313
436
|
clobber an existing filled file; report what was seeded.
|
|
314
437
|
6. **Kanban sync.** Run the §Kanban reconcile: link/create the project's board if configured, verify
|
|
315
438
|
its columns, and backfill/sync cards from `specs/*.md`. See §Kanban.
|
|
439
|
+
7. **Spec-template top-up.** `specs/_template.md` is seeded once at install and then **never**
|
|
440
|
+
refreshed, so a repo keeps whatever front-matter the core shipped the day it was installed (a
|
|
441
|
+
pre-1.6 copy has no `loop_pass`/`loop_phase`, and its `status` comment still lists four states).
|
|
442
|
+
Top it up the same way as the profile: add the **front-matter fields** the current
|
|
443
|
+
`templates/spec.template.md` has and the repo's copy lacks, with their documented defaults, and
|
|
444
|
+
refresh the `status:` comment. Never rewrite its body — the section list is the human's to shape,
|
|
445
|
+
and some repos have deliberately trimmed it. Nothing breaks without this (the fields are written on
|
|
446
|
+
demand when a driver needs them); it just keeps a new spec's front-matter honest about the states
|
|
447
|
+
the pipeline can put it in.
|
|
316
448
|
|
|
317
449
|
Re-running `/init-pipeline` remains possible (it reconciles too) but is only *needed* when the stack
|
|
318
450
|
itself changes in ways `/build` §1.5 can't auto-grow (e.g. package manager or contract mechanism swap).
|
|
@@ -347,6 +479,9 @@ Shared design, all four scripts:
|
|
|
347
479
|
that never answered: `review.js` names them in `unreviewedSurfaces` and refuses to score
|
|
348
480
|
`SHIP`. `scripts/test-workflows.mjs`
|
|
349
481
|
pins this — it is the one invariant the structural checks in `validate-core.mjs` cannot see.
|
|
482
|
+
The conversational commands enforce the same rule by roll call (§Dead agents); it was the workflows
|
|
483
|
+
that had it first, and for three releases they had it **alone** — the same crash on the
|
|
484
|
+
conversational path went unreported.
|
|
350
485
|
- **`review.js`** — preflight gate (aborts red, zero agents), one `git diff --stat` staged per
|
|
351
486
|
touched surface, one reviewer per surface in parallel, then an **adversarial cross-check** phase
|
|
352
487
|
that tries to refute each CRITICAL/security finding before it can trigger a fix loop.
|
|
@@ -411,12 +546,16 @@ card created in the target column if missing.
|
|
|
411
546
|
| `/build` | `building` |
|
|
412
547
|
| `/review` | `review` |
|
|
413
548
|
| `/fix` | `fix` |
|
|
549
|
+
| a `/drive` is driving it (`in-progress`) | the current phase's column |
|
|
550
|
+
| a `/drive` gave up (`blocked`) | `fix` |
|
|
414
551
|
| `/ship` starts | `ship` |
|
|
415
552
|
| PR opened (`status: shipped`) | `shipped` (+ `PR #<num>` on the card) |
|
|
416
553
|
|
|
417
554
|
**Backfill / sync from specs (reconcile).** `specs/*.md` is the source of truth. For each spec, read its
|
|
418
555
|
`feature_id` (front-matter or filename) and `status`, map `status`→column — `frozen`→`ready`,
|
|
419
|
-
`in-
|
|
556
|
+
`in-progress`→the `loop_phase`'s column (`build`→`building`, `review`→`review`, `fix`→`fix`; unset ⇒
|
|
557
|
+
`building`), `in-review`→`review`, `blocked`→`fix`, `shipped`→`shipped`, anything else / a spec with no
|
|
558
|
+
status→`spec` — then **full
|
|
420
559
|
sync**: card absent ⇒ add it in that column; card present ⇒ **move it** to that column so the board
|
|
421
560
|
always reflects the specs (this repositions cards the human may have moved by hand). Report cards
|
|
422
561
|
added vs. moved vs. already-correct.
|
package/scripts/loop.sh
CHANGED
|
@@ -2,20 +2,26 @@
|
|
|
2
2
|
#
|
|
3
3
|
# loop.sh — autonomous /build → /review → /fix → /review … loop for ONE feature.
|
|
4
4
|
#
|
|
5
|
-
# loop.sh <feature-id> [--max=N] [--no-build] [--rebuild]
|
|
5
|
+
# loop.sh <feature-id> [--max=N] [--no-build] [--rebuild] [--resume]
|
|
6
6
|
#
|
|
7
7
|
# THE POINT: every phase runs as a SEPARATE `claude -p` child with its own fresh
|
|
8
|
-
# context. The session that typed /
|
|
8
|
+
# context. The session that typed /drive never sees the diff, the N review reports
|
|
9
9
|
# or the N contracts — it reads only this script's one-line-per-phase stdout and,
|
|
10
10
|
# at the end, the verdict JSON. Running the loop inside the calling session would
|
|
11
11
|
# accumulate all of it in a history that is re-sent at input price on every turn,
|
|
12
12
|
# which is the exact cost the pipeline's /clear discipline exists to avoid.
|
|
13
13
|
#
|
|
14
14
|
# Contract with the pipeline: /review writes specs/reports/<id>.verdict.json on
|
|
15
|
-
# every run
|
|
16
|
-
#
|
|
15
|
+
# every run, and /build writes <id>.readiness.json + <id>.build.json. Those three
|
|
16
|
+
# files — `blocking`, `fingerprint`, `unreviewed`, `verdict`, `dead` — are the ONLY
|
|
17
|
+
# channel between cohorte and this driver. No prose is parsed.
|
|
17
18
|
#
|
|
18
|
-
#
|
|
19
|
+
# Two of those fields exist for the same reason: a subagent that DIES returns
|
|
20
|
+
# nothing, and nothing is byte-identical to "clean". A dead implementer means a
|
|
21
|
+
# surface was never built; a dead reviewer means a surface was never audited, and
|
|
22
|
+
# `blocking == 0` would then certify code no one read. Both abort as exit 2.
|
|
23
|
+
#
|
|
24
|
+
# Exit codes (distinct diagnostics, do not collapse them):
|
|
19
25
|
# 0 clean — a review returned blocking == 0
|
|
20
26
|
# 1 ceiling — --max passes used, still blocking (the fix was progressing;
|
|
21
27
|
# re-run with a higher --max)
|
|
@@ -23,22 +29,37 @@
|
|
|
23
29
|
# preflight (typecheck/lint/tests broken; the message says which)
|
|
24
30
|
# 3 non-convergent — two consecutive reviews returned the SAME blocking
|
|
25
31
|
# fingerprint: the fix is treading water, a higher --max will not help
|
|
32
|
+
# 4 not implementable — /build's readiness gate returned NOT-READY and spawned
|
|
33
|
+
# no agent: the frozen spec cannot be built (missing contract shape, unowned
|
|
34
|
+
# area, absent dependency). Needs /spec, not more passes.
|
|
26
35
|
# 64 usage — bad flag, bad id, missing spec, no `claude` on PATH
|
|
27
36
|
#
|
|
28
37
|
# No /fix runs on the last pass: fixing without a review behind it ships
|
|
29
38
|
# unaudited code. Each fix pass is committed — that commit is the only way back
|
|
30
39
|
# after N autonomous passes.
|
|
40
|
+
#
|
|
41
|
+
# RESUME: the spec's front-matter IS the loop's state (SCHEMA.md §Spec status).
|
|
42
|
+
# Before every phase this script stamps `status: in-progress` + `loop_phase` +
|
|
43
|
+
# `loop_pass` into specs/<id>.md — deterministically, with awk, costing no tokens
|
|
44
|
+
# — and on exit stamps a terminal status (`in-review` clean, `blocked` otherwise).
|
|
45
|
+
# `--resume` reads `loop_pass` back and continues from that pass instead of 1, so
|
|
46
|
+
# a session that died at pass 3 of 5 does not re-pay passes 1 and 2. The build is
|
|
47
|
+
# skipped or redone by the same stamp logic as always (the stamp is only written
|
|
48
|
+
# on a build that finished), so an interrupted build still rebuilds.
|
|
31
49
|
|
|
32
50
|
set -uo pipefail
|
|
33
51
|
|
|
34
52
|
usage() {
|
|
35
53
|
cat >&2 <<'EOF'
|
|
36
|
-
usage: loop.sh <feature-id> [--max=N] [--no-build] [--rebuild]
|
|
54
|
+
usage: loop.sh <feature-id> [--max=N] [--no-build] [--rebuild] [--resume]
|
|
37
55
|
|
|
38
|
-
--max=N stop after N review passes (default 5)
|
|
56
|
+
--max=N stop after N review passes (default 5) — a ceiling on the TOTAL
|
|
57
|
+
pass count, so it still means "5 passes" when resuming at pass 3
|
|
39
58
|
--no-build never build — re-run the /review ⇄ /fix loop on a feature that
|
|
40
59
|
is already built (the common case; the build stamp is ignored)
|
|
41
60
|
--rebuild force a /build even if the stamp says it was already built
|
|
61
|
+
--resume continue from the pass recorded in the spec's front-matter
|
|
62
|
+
(loop_pass), instead of starting over at pass 1
|
|
42
63
|
|
|
43
64
|
env CLAUDE_FLAGS flags for every child session
|
|
44
65
|
(default: --permission-mode acceptEdits)
|
|
@@ -49,6 +70,7 @@ EOF
|
|
|
49
70
|
id=""
|
|
50
71
|
max=5
|
|
51
72
|
build_mode="auto" # auto | never | force
|
|
73
|
+
resume=0
|
|
52
74
|
|
|
53
75
|
for arg in "$@"; do
|
|
54
76
|
case "$arg" in
|
|
@@ -61,6 +83,7 @@ for arg in "$@"; do
|
|
|
61
83
|
;;
|
|
62
84
|
--no-build) build_mode="never" ;;
|
|
63
85
|
--rebuild) build_mode="force" ;;
|
|
86
|
+
--resume) resume=1 ;;
|
|
64
87
|
-h|--help) usage ;;
|
|
65
88
|
-*) echo "loop: unknown flag: $arg" >&2; usage ;;
|
|
66
89
|
*)
|
|
@@ -93,9 +116,47 @@ spec="specs/$id.md"
|
|
|
93
116
|
reports="specs/reports"
|
|
94
117
|
mkdir -p "$reports"
|
|
95
118
|
verdict="$reports/$id.verdict.json"
|
|
119
|
+
readiness="$reports/$id.readiness.json"
|
|
120
|
+
buildjson="$reports/$id.build.json"
|
|
96
121
|
stamp="$reports/$id.built"
|
|
97
122
|
log="$reports/$id.loop.log"
|
|
98
123
|
|
|
124
|
+
# --- the spec front-matter as loop state -------------------------------------
|
|
125
|
+
# Best-effort by design: a spec with no front-matter (or an unwritable one) makes
|
|
126
|
+
# every fm_* call a silent no-op. This is bookkeeping for resume + the dashboard,
|
|
127
|
+
# never a precondition — the loop must not die over a status line.
|
|
128
|
+
fm_get() { # fm_get <key> → value, or empty
|
|
129
|
+
[ -f "$spec" ] || return 0
|
|
130
|
+
awk -v k="$1" '
|
|
131
|
+
NR==1 && $0=="---" { fm=1; next }
|
|
132
|
+
fm==1 && $0=="---" { exit }
|
|
133
|
+
fm==1 && $0 ~ "^"k":" {
|
|
134
|
+
sub("^"k":[[:space:]]*", ""); sub("#.*", "")
|
|
135
|
+
gsub(/^[[:space:]]+|[[:space:]]+$/, ""); print; exit
|
|
136
|
+
}
|
|
137
|
+
' "$spec"
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
fm_set() { # fm_set <key> <value> (replace, else append)
|
|
141
|
+
[ -f "$spec" ] || return 0
|
|
142
|
+
awk -v k="$1" -v v="$2" '
|
|
143
|
+
NR==1 && $0!="---" { nofm=1 }
|
|
144
|
+
nofm { print; next }
|
|
145
|
+
NR==1 { fm=1; print; next }
|
|
146
|
+
fm==1 && $0=="---" {
|
|
147
|
+
if (!done) print k ": " v # key absent: add it before the closing ---
|
|
148
|
+
fm=2; print; next
|
|
149
|
+
}
|
|
150
|
+
fm==1 && $0 ~ "^"k":" {
|
|
151
|
+
if (done) next # a duplicate key: drop it
|
|
152
|
+
c=""; i=index($0, "#"); if (i>0) c=" " substr($0, i) # keep a trailing comment
|
|
153
|
+
print k ": " v c; done=1; next
|
|
154
|
+
}
|
|
155
|
+
{ print }
|
|
156
|
+
' "$spec" >"$spec.loop.tmp" 2>/dev/null &&
|
|
157
|
+
mv "$spec.loop.tmp" "$spec" 2>/dev/null || rm -f "$spec.loop.tmp"
|
|
158
|
+
}
|
|
159
|
+
|
|
99
160
|
: "${CLAUDE_FLAGS:=--permission-mode acceptEdits}"
|
|
100
161
|
|
|
101
162
|
: >"$log"
|
|
@@ -110,6 +171,13 @@ log="$reports/$id.loop.log"
|
|
|
110
171
|
# $CLAUDE_FLAGS is intentionally unquoted — it is a flag list, not one word.
|
|
111
172
|
run_phase() {
|
|
112
173
|
cmd="$1"
|
|
174
|
+
# Stamp the state BEFORE the phase runs: if this child dies (or the whole
|
|
175
|
+
# session does), the spec already says where the loop was — that is what
|
|
176
|
+
# --resume reads back. Child commands write `status` themselves (/fix sets
|
|
177
|
+
# in-review); re-stamping here each phase is what keeps `in-progress` true.
|
|
178
|
+
fm_set status in-progress
|
|
179
|
+
fm_set loop_pass "$pass"
|
|
180
|
+
fm_set loop_phase "$cmd"
|
|
113
181
|
printf '▶ /%-6s %-24s ' "$cmd" "$id"
|
|
114
182
|
printf '\n\n===== /%s %s =====\n' "$cmd" "$id" >>"$log"
|
|
115
183
|
# shellcheck disable=SC2086
|
|
@@ -127,7 +195,31 @@ run_phase() {
|
|
|
127
195
|
json_num() { sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\1/p' "$1" | head -n1; }
|
|
128
196
|
json_str() { sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -n1; }
|
|
129
197
|
|
|
130
|
-
|
|
198
|
+
# Terminal status goes into the spec, not just into this stdout: a clean run
|
|
199
|
+
# leaves the feature ready to /ship, any failure leaves it visibly `blocked` for
|
|
200
|
+
# the human and for the dashboard. Exit 64 never reaches here (usage dies earlier),
|
|
201
|
+
# so every code handled below is a real run outcome.
|
|
202
|
+
finish() {
|
|
203
|
+
if [ "$1" -eq 0 ]; then
|
|
204
|
+
fm_set status in-review
|
|
205
|
+
fm_set loop_pass 0
|
|
206
|
+
fm_set loop_phase done
|
|
207
|
+
else
|
|
208
|
+
fm_set status blocked
|
|
209
|
+
fi
|
|
210
|
+
echo "$2"
|
|
211
|
+
exit "$1"
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
# One short clause naming the deferred findings, appended to a closing line.
|
|
215
|
+
# They are NOT blocking (they live in the backlog, not in ## Remediation), so
|
|
216
|
+
# they never change an exit code — but a loop that silently drops them is the
|
|
217
|
+
# leak /review §3.5 exists to close, so the driver names them.
|
|
218
|
+
def_note() {
|
|
219
|
+
d="$(json_num "$verdict" deferred 2>/dev/null)"
|
|
220
|
+
case "$d" in ''|0) return 0 ;; esac
|
|
221
|
+
printf ' · %s deferred finding(s) parked in specs/refactor-backlog.md' "$d"
|
|
222
|
+
}
|
|
131
223
|
|
|
132
224
|
# --- build -------------------------------------------------------------------
|
|
133
225
|
# The stamp is the driver's own bookkeeping — /build knows nothing about it.
|
|
@@ -137,14 +229,44 @@ case "$build_mode" in
|
|
|
137
229
|
auto) [ -f "$stamp" ] && do_build=0 || do_build=1 ;;
|
|
138
230
|
esac
|
|
139
231
|
|
|
232
|
+
# --resume: continue from the pass the spec records, not from 1. A missing or
|
|
233
|
+
# junk value falls back to 1 — resuming must never be less safe than starting.
|
|
234
|
+
pass=1
|
|
235
|
+
if [ "$resume" -eq 1 ]; then
|
|
236
|
+
rp="$(fm_get loop_pass)"
|
|
237
|
+
case "$rp" in ''|*[!0-9]*|0) rp=1 ;; esac
|
|
238
|
+
[ "$rp" -le "$max" ] || {
|
|
239
|
+
echo "loop: --resume says pass $rp but --max=$max — raise --max to continue" >&2; exit 64; }
|
|
240
|
+
pass="$rp"
|
|
241
|
+
[ "$pass" -eq 1 ] || printf '↻ resuming at review pass %s (from %s)\n' "$pass" "$spec"
|
|
242
|
+
fi
|
|
243
|
+
|
|
140
244
|
if [ "$do_build" -eq 1 ]; then
|
|
141
|
-
|
|
245
|
+
# Delete first: a NOT-READY left by a previous build would abort this one on
|
|
246
|
+
# someone else's verdict (and a stale READY would hide a gate that never ran).
|
|
247
|
+
rm -f "$readiness" "$buildjson"
|
|
248
|
+
build_ok=0
|
|
249
|
+
run_phase build && build_ok=1
|
|
250
|
+
# The readiness gate is checked BEFORE the child's exit status: /build aborting
|
|
251
|
+
# on NOT-READY is a cleaner diagnosis than "/build failed", and it is the one
|
|
252
|
+
# outcome that more passes cannot fix.
|
|
253
|
+
if [ -f "$readiness" ] &&
|
|
254
|
+
grep -q '"verdict"[[:space:]]*:[[:space:]]*"NOT-READY"' "$readiness"; then
|
|
255
|
+
finish 4 "✗ spec not implementable — /build's readiness gate returned NOT-READY and spawned no agent; see $readiness, then /spec $id"
|
|
256
|
+
fi
|
|
257
|
+
# A dead implementer returns nothing, so /build can finish "successfully" having
|
|
258
|
+
# built one surface of two. Reviewing that would spend N reviewers auditing a
|
|
259
|
+
# half-built feature and report its gaps as findings to fix — the wrong diagnosis
|
|
260
|
+
# at the wrong price. `dead` is a non-empty array only when a surface died twice.
|
|
261
|
+
if [ -f "$buildjson" ] && grep -q '"dead"[[:space:]]*:[[:space:]]*\[[^]]' "$buildjson"; then
|
|
262
|
+
finish 2 "✗ an implementer died — the surface(s) in \"dead\" were never built; see $buildjson and $log"
|
|
263
|
+
fi
|
|
264
|
+
[ "$build_ok" -eq 1 ] || finish 2 "✗ /build failed — see $log"
|
|
142
265
|
date -u +%Y-%m-%dT%H:%M:%SZ >"$stamp"
|
|
143
266
|
fi
|
|
144
267
|
|
|
145
268
|
# --- review ⇄ fix ------------------------------------------------------------
|
|
146
269
|
prev_fp=""
|
|
147
|
-
pass=1
|
|
148
270
|
while [ "$pass" -le "$max" ]; do
|
|
149
271
|
# Delete first: a stale verdict from the previous pass read as this pass's
|
|
150
272
|
# answer would end the loop on someone else's numbers.
|
|
@@ -158,12 +280,19 @@ while [ "$pass" -le "$max" ]; do
|
|
|
158
280
|
finish 2 "✗ /review aborted on a red preflight — typecheck/lint/tests are broken, see $reports/$id.preflight.txt"
|
|
159
281
|
fi
|
|
160
282
|
|
|
283
|
+
# A reviewer that died twice leaves its surface unaudited, and `blocking` counts only
|
|
284
|
+
# what the SURVIVING reviewers found — so blocking == 0 here would mean "clean" about
|
|
285
|
+
# code nobody read. Checked BEFORE blocking, because it invalidates it.
|
|
286
|
+
if grep -q '"unreviewed"[[:space:]]*:[[:space:]]*\[[^]]' "$verdict"; then
|
|
287
|
+
finish 2 "✗ a reviewer died — the surface(s) in \"unreviewed\" carry no verdict (pass $pass); see $verdict"
|
|
288
|
+
fi
|
|
289
|
+
|
|
161
290
|
blocking="$(json_num "$verdict" blocking)"
|
|
162
291
|
[ -n "$blocking" ] || finish 2 \
|
|
163
292
|
"✗ verdict has no usable 'blocking' count (pass $pass) — see $verdict"
|
|
164
293
|
|
|
165
294
|
[ "$blocking" -eq 0 ] && finish 0 \
|
|
166
|
-
"✓ clean after $pass review pass(es) — no blocking findings"
|
|
295
|
+
"✓ clean after $pass review pass(es) — no blocking findings$(def_note)"
|
|
167
296
|
|
|
168
297
|
fp="$(json_str "$verdict" fingerprint)"
|
|
169
298
|
if [ -n "$fp" ] && [ "$fp" = "$prev_fp" ]; then
|
|
@@ -173,7 +302,7 @@ while [ "$pass" -le "$max" ]; do
|
|
|
173
302
|
|
|
174
303
|
# Last pass: report and stop. A /fix here would leave unreviewed code behind.
|
|
175
304
|
[ "$pass" -eq "$max" ] && finish 1 \
|
|
176
|
-
"✗ ceiling — $blocking blocking finding(s) after $max pass(es); re-run with a higher --max"
|
|
305
|
+
"✗ ceiling — $blocking blocking finding(s) after $max pass(es); re-run with a higher --max --resume$(def_note)"
|
|
177
306
|
|
|
178
307
|
run_phase fix || true
|
|
179
308
|
|