cohorte 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +151 -0
- package/README.md +43 -12
- package/bin/cli.js +13 -2
- package/core/agents/review.md +23 -0
- package/core/commands/audit.md +9 -1
- package/core/commands/brainstorm.md +6 -0
- package/core/commands/build.md +93 -8
- package/core/commands/doctor.md +13 -8
- package/core/commands/drive.md +80 -0
- package/core/commands/fix.md +10 -5
- package/core/commands/review.md +94 -12
- package/core/commands/spec.md +20 -0
- package/core/commands/update-pipeline.md +6 -1
- package/core/hooks/gate.py +4 -4
- package/core/templates/decisions.template.md +42 -0
- package/core/templates/spec.template.md +4 -2
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
- package/core/workflows/review.js +44 -2
- package/dashboard/dist/assets/{index-AFQnlfjO.css → index-BZ_LQlEj.css} +1 -1
- package/dashboard/dist/assets/index-DYyn4p93.js +43 -0
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +13 -3
- package/dashboard/server/index.js +7 -0
- package/dashboard/server/metrics.js +4 -4
- package/dashboard/server/usage.js +61 -0
- package/install.ps1 +12 -1
- package/install.sh +12 -2
- package/package.json +1 -1
- package/profile/PIPELINE.template.md +3 -3
- package/profile/SCHEMA.md +150 -12
- package/scripts/loop.sh +318 -0
- package/scripts/metrics/collect.mjs +11 -2
- package/scripts/preflight.sh +2 -2
- package/scripts/telemetry-send.sh +5 -2
- package/scripts/test-dashboard.mjs +22 -2
- package/scripts/test-gate.mjs +1 -2
- package/scripts/test-loop.mjs +227 -0
- package/scripts/test-metrics.mjs +12 -3
- package/scripts/test-workflows.mjs +28 -0
- package/scripts/validate-core.mjs +21 -6
- package/core/agents/smoke.md +0 -63
- package/core/commands/smoke.md +0 -55
- package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
|
@@ -12,11 +12,17 @@ const { versions } = require('./versions');
|
|
|
12
12
|
// Rendered surface agents live alongside these fixed (non-surface) agents; exclude them
|
|
13
13
|
// from the orphan check so they're never mistaken for a stray surface agent.
|
|
14
14
|
const FIXED_AGENTS = new Set([
|
|
15
|
-
'review', 'release', '
|
|
15
|
+
'review', 'release', 'profile-reader',
|
|
16
|
+
// retired (1.5.0) — still excluded so a stale install's leftover file isn't
|
|
17
|
+
// reported as a stray surface agent.
|
|
18
|
+
'smoke',
|
|
16
19
|
'implementer.template',
|
|
17
20
|
]);
|
|
18
21
|
|
|
19
|
-
|
|
22
|
+
// The spec lifecycle (SCHEMA.md §Spec status). `in-progress` and `blocked` are written by
|
|
23
|
+
// the /drive driver — they are what makes an interrupted autonomous loop resumable, so a
|
|
24
|
+
// dashboard that flagged them as invalid would report the pipeline's own state as a defect.
|
|
25
|
+
const VALID_STATUS = ['draft', 'frozen', 'in-progress', 'in-review', 'shipped', 'blocked'];
|
|
20
26
|
|
|
21
27
|
// Artifacts the pipeline itself writes into specs/ that are NOT feature specs and have no
|
|
22
28
|
// front-matter status. `/audit` writes specs/refactor-backlog.md by design, so scanning it
|
|
@@ -126,7 +132,7 @@ function checkGate(profile, projectRoot) {
|
|
|
126
132
|
const wantPf = gate.preflight || {};
|
|
127
133
|
const havePf = (cfg.preflight && typeof cfg.preflight === 'object') ? cfg.preflight : {};
|
|
128
134
|
if (!!wantPf.enabled !== !!havePf.enabled
|
|
129
|
-
|| !sameSet(wantPf.agents || ['review'
|
|
135
|
+
|| !sameSet(wantPf.agents || ['review'], havePf.agents || ['review'])
|
|
130
136
|
|| Number(wantPf.max_age_minutes || 30) !== Number(havePf.max_age_minutes || 30)) {
|
|
131
137
|
drifted.push('preflight');
|
|
132
138
|
}
|
|
@@ -278,12 +284,16 @@ function scanSpecs(projectRoot) {
|
|
|
278
284
|
const body = fm ? fm[1] : '';
|
|
279
285
|
const get = k => { const m = body.match(new RegExp(`^${k}:\\s*(.*)$`, 'm')); return m ? m[1].trim() : null; };
|
|
280
286
|
const status = get('status');
|
|
287
|
+
// The loop driver's resume state, when a /drive is (or was) running on this spec.
|
|
288
|
+
const pass = parseInt(get('loop_pass'), 10);
|
|
289
|
+
const phase = get('loop_phase');
|
|
281
290
|
specs.push({
|
|
282
291
|
file: f,
|
|
283
292
|
id: get('feature_id') || f.replace(/\.md$/, ''),
|
|
284
293
|
title: get('title'),
|
|
285
294
|
status: status ? status.split('#')[0].trim() : null,
|
|
286
295
|
branch: get('branch'),
|
|
296
|
+
loop: pass > 0 ? { pass, phase: phase && phase !== 'done' ? phase : null } : null,
|
|
287
297
|
});
|
|
288
298
|
}
|
|
289
299
|
return specs;
|
|
@@ -11,6 +11,7 @@ const { versions } = require('./versions');
|
|
|
11
11
|
const { state } = require('./doctor');
|
|
12
12
|
const { kanban } = require('./kanban');
|
|
13
13
|
const { metrics } = require('./metrics');
|
|
14
|
+
const { usage } = require('./usage');
|
|
14
15
|
const fleet = require('./fleet');
|
|
15
16
|
|
|
16
17
|
const MIME = {
|
|
@@ -321,6 +322,12 @@ function start({ projectRoot, globalDir, port, host, openBrowser, pkgRoot, versi
|
|
|
321
322
|
const root = q ? path.resolve(q) : projectRoot;
|
|
322
323
|
return sendJson(res, 200, metrics({ projectRoot: root }));
|
|
323
324
|
}
|
|
325
|
+
if (url === '/api/usage') {
|
|
326
|
+
const q = new URL(req.url, 'http://localhost');
|
|
327
|
+
const root = q.searchParams.get('project') ? path.resolve(q.searchParams.get('project')) : projectRoot;
|
|
328
|
+
const days = Number(q.searchParams.get('days')) || null;
|
|
329
|
+
return sendJson(res, 200, usage({ projectRoot: root, days }));
|
|
330
|
+
}
|
|
324
331
|
if (url === '/api/projects') {
|
|
325
332
|
const body = await readBody(req);
|
|
326
333
|
if (req.method === 'POST') {
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
'use strict';
|
|
2
2
|
// Read a project's `.claude/pipeline-metrics.jsonl` (one line per phase batch, appended by
|
|
3
|
-
// /build, /review
|
|
3
|
+
// /build, /review and /fix) and aggregate it per feature: wall-clock per phase, fix
|
|
4
4
|
// rounds, and per-surface results. Dependency-free; a missing file is simply "no data yet".
|
|
5
5
|
//
|
|
6
6
|
// Two line formats coexist in the file:
|
|
@@ -11,9 +11,9 @@
|
|
|
11
11
|
const fs = require('fs');
|
|
12
12
|
const path = require('path');
|
|
13
13
|
|
|
14
|
-
// `cycle`
|
|
15
|
-
// Without
|
|
16
|
-
// surface table showed rows with every cell empty.
|
|
14
|
+
// `smoke` and `cycle` are RETIRED phases, kept so metrics files written before their removal
|
|
15
|
+
// still render. Without them in this list their per-surface results parse fine but land in no
|
|
16
|
+
// column — the surface table showed rows with every cell empty.
|
|
17
17
|
const PHASES = ['build', 'review', 'fix', 'smoke', 'cycle'];
|
|
18
18
|
|
|
19
19
|
// Parse the raw JSONL into normalized batches ({ts, feature, phase, seconds, surfaces}),
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
// Serve the metrics collector's rollup (scripts/metrics/collect.mjs) to the dashboard:
|
|
3
|
+
// real cost and runtime per command, read from Claude Code's own transcripts.
|
|
4
|
+
//
|
|
5
|
+
// This is the SECOND metrics source in the cockpit, and the two answer different questions.
|
|
6
|
+
// `metrics.js` reads `.claude/pipeline-metrics.jsonl` — written by the model itself, so it
|
|
7
|
+
// carries per-surface verdicts (ok / REVISE:2 / error) that only the model knows, but it
|
|
8
|
+
// misses any run that ended early and can never report tokens. This one is derived from the
|
|
9
|
+
// transcripts, so it is complete and exact on cost and time but knows nothing about verdicts.
|
|
10
|
+
// Verdicts from one, money from the other; neither is a replacement for the other.
|
|
11
|
+
//
|
|
12
|
+
// The collector is ESM and this server is CommonJS, so it runs as a child process — the same
|
|
13
|
+
// bridge `bin/cli.js` uses. A spawn is ~1-2s on a large history, which is why the result is
|
|
14
|
+
// cached: the panel polls, and re-parsing tens of MB of transcripts on every poll would make
|
|
15
|
+
// the whole cockpit feel broken.
|
|
16
|
+
|
|
17
|
+
const path = require('path');
|
|
18
|
+
const { spawnSync } = require('child_process');
|
|
19
|
+
|
|
20
|
+
const COLLECT = path.join(__dirname, '..', '..', 'scripts', 'metrics', 'collect.mjs');
|
|
21
|
+
|
|
22
|
+
// Transcripts only grow, and nobody needs sub-minute freshness on a spend figure.
|
|
23
|
+
const CACHE_MS = 60_000;
|
|
24
|
+
const cache = new Map(); // projectRoot → { at, value }
|
|
25
|
+
|
|
26
|
+
function usage({ projectRoot, days = null, force = false }) {
|
|
27
|
+
const key = `${projectRoot}|${days || ''}`;
|
|
28
|
+
const hit = cache.get(key);
|
|
29
|
+
if (!force && hit && Date.now() - hit.at < CACHE_MS) return hit.value;
|
|
30
|
+
|
|
31
|
+
const args = [COLLECT, projectRoot, '--json'];
|
|
32
|
+
if (days) args.push(`--days=${days}`);
|
|
33
|
+
|
|
34
|
+
let value;
|
|
35
|
+
try {
|
|
36
|
+
const r = spawnSync(process.execPath, args, {
|
|
37
|
+
encoding: 'utf8',
|
|
38
|
+
// A pathological history must not wedge the cockpit's event loop forever.
|
|
39
|
+
timeout: 60_000,
|
|
40
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
41
|
+
});
|
|
42
|
+
if (r.status !== 0 || !r.stdout) {
|
|
43
|
+
value = { present: false, error: (r.stderr || '').trim().split('\n').slice(-1)[0] || 'collector failed' };
|
|
44
|
+
} else {
|
|
45
|
+
const parsed = JSON.parse(r.stdout);
|
|
46
|
+
// No transcripts for this project is a normal state (a fresh checkout, or a project
|
|
47
|
+
// driven from another machine), not an error — say so rather than rendering zeros
|
|
48
|
+
// that look like "this pipeline is free".
|
|
49
|
+
value = parsed.totals && parsed.totals.sessions
|
|
50
|
+
? { present: true, ...parsed }
|
|
51
|
+
: { present: false, error: 'no Claude Code transcripts found for this project' };
|
|
52
|
+
}
|
|
53
|
+
} catch (e) {
|
|
54
|
+
value = { present: false, error: String((e && e.message) || e) };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
cache.set(key, { at: Date.now(), value });
|
|
58
|
+
return value;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
module.exports = { usage };
|
package/install.ps1
CHANGED
|
@@ -151,6 +151,7 @@ try {
|
|
|
151
151
|
Copy-Item (Join-Path $src 'scripts\kanban-move.sh') (Join-Path $dest 'pipeline\scripts') -Force
|
|
152
152
|
Copy-Item (Join-Path $src 'scripts\telemetry-send.sh') (Join-Path $dest 'pipeline\scripts') -Force
|
|
153
153
|
Copy-Item (Join-Path $src 'scripts\preflight.sh') (Join-Path $dest 'pipeline\scripts') -Force
|
|
154
|
+
Copy-Item (Join-Path $src 'scripts\loop.sh') (Join-Path $dest 'pipeline\scripts') -Force
|
|
154
155
|
Copy-Item (Join-Path $src 'core\agents\implementer.template.md') (Join-Path $dest 'pipeline') -Force
|
|
155
156
|
if (Test-Path (Join-Path $src 'CHANGELOG.md')) { Copy-Item (Join-Path $src 'CHANGELOG.md') (Join-Path $dest 'pipeline') -Force }
|
|
156
157
|
[System.IO.File]::WriteAllText((Join-Path $dest 'pipeline\VERSION'), "$ver`n", [System.Text.UTF8Encoding]::new($false))
|
|
@@ -188,8 +189,18 @@ try {
|
|
|
188
189
|
New-Item -ItemType Directory -Force -Path (Join-Path $dest 'agents') | Out-Null
|
|
189
190
|
Copy-Item (Join-Path $src 'core\agents\review.md'),
|
|
190
191
|
(Join-Path $src 'core\agents\release.md'),
|
|
191
|
-
(Join-Path $src 'core\agents\smoke.md'),
|
|
192
192
|
(Join-Path $src 'core\agents\profile-reader.md') (Join-Path $dest 'agents') -Force
|
|
193
|
+
# 1.5.0 removed the /smoke phase; copy-over never deletes, so scrub the orphan agent.
|
|
194
|
+
Remove-Item -LiteralPath (Join-Path $dest 'agents\smoke.md') -Force -ErrorAction SilentlyContinue
|
|
195
|
+
Remove-Item -LiteralPath (Join-Path $dest 'commands\smoke.md') -Force -ErrorAction SilentlyContinue
|
|
196
|
+
# 1.4.0 removed /cycle and its workflow — and no installer ever scrubbed them, so every
|
|
197
|
+
# install since has kept offering a command that dispatches a workflow whose phases were
|
|
198
|
+
# later deleted. A dead command is worse than a missing one: the model can still fire it.
|
|
199
|
+
Remove-Item -LiteralPath (Join-Path $dest 'commands\cycle.md') -Force -ErrorAction SilentlyContinue
|
|
200
|
+
Remove-Item -LiteralPath (Join-Path $dest 'workflows\cycle.js') -Force -ErrorAction SilentlyContinue
|
|
201
|
+
# 1.6.0 renamed /loop → /drive: Claude Code's own built-in /loop shadowed ours, so a leftover
|
|
202
|
+
# commands\loop.md is a command the user can never reach — scrub it rather than leave a decoy.
|
|
203
|
+
Remove-Item -LiteralPath (Join-Path $dest 'commands\loop.md') -Force -ErrorAction SilentlyContinue
|
|
193
204
|
# 0.1.19 split the bi-mode questionnaire-researcher into research-agent + questionnaire-architect;
|
|
194
205
|
# copy-over never deletes, so scrub the retired agent lest a dead subagent_type linger.
|
|
195
206
|
Remove-Item -LiteralPath (Join-Path $dest 'agents\questionnaire-researcher.md') -Force -ErrorAction SilentlyContinue
|
package/install.sh
CHANGED
|
@@ -107,8 +107,9 @@ copy_core() {
|
|
|
107
107
|
cp "$src/scripts/kanban-move.sh" "$dest/pipeline/scripts/"
|
|
108
108
|
cp "$src/scripts/telemetry-send.sh" "$dest/pipeline/scripts/"
|
|
109
109
|
cp "$src/scripts/preflight.sh" "$dest/pipeline/scripts/"
|
|
110
|
+
cp "$src/scripts/loop.sh" "$dest/pipeline/scripts/"
|
|
110
111
|
chmod +x "$dest/pipeline/scripts/kanban-move.sh" "$dest/pipeline/scripts/telemetry-send.sh" \
|
|
111
|
-
"$dest/pipeline/scripts/preflight.sh" 2>/dev/null || true
|
|
112
|
+
"$dest/pipeline/scripts/preflight.sh" "$dest/pipeline/scripts/loop.sh" 2>/dev/null || true
|
|
112
113
|
cp "$src/core/agents/implementer.template.md" "$dest/pipeline/"
|
|
113
114
|
[ -f "$src/CHANGELOG.md" ] && cp "$src/CHANGELOG.md" "$dest/pipeline/"
|
|
114
115
|
printf '%s\n' "$ver" > "$dest/pipeline/VERSION"
|
|
@@ -149,8 +150,17 @@ PY
|
|
|
149
150
|
copy_fixed_agents() {
|
|
150
151
|
mkdir -p "$dest/agents"
|
|
151
152
|
cp "$src/core/agents/review.md" "$src/core/agents/release.md" \
|
|
152
|
-
"$src/core/agents/
|
|
153
|
+
"$src/core/agents/profile-reader.md" \
|
|
153
154
|
"$dest/agents/"
|
|
155
|
+
# 1.5.0 removed the /smoke phase; copy-over never deletes, so scrub the orphan agent.
|
|
156
|
+
rm -f "$dest/agents/smoke.md" "$dest/commands/smoke.md"
|
|
157
|
+
# 1.4.0 removed /cycle and its workflow — and no installer ever scrubbed them, so every
|
|
158
|
+
# install since has kept offering a command that dispatches a workflow whose phases were
|
|
159
|
+
# later deleted. A dead command is worse than a missing one: the model can still fire it.
|
|
160
|
+
rm -f "$dest/commands/cycle.md" "$dest/workflows/cycle.js"
|
|
161
|
+
# 1.6.0 renamed /loop → /drive: Claude Code's own built-in /loop shadowed ours, so a leftover
|
|
162
|
+
# commands/loop.md is a command the user can never reach — scrub it rather than leave a decoy.
|
|
163
|
+
rm -f "$dest/commands/loop.md"
|
|
154
164
|
# 0.1.19 split the bi-mode questionnaire-researcher into research-agent + questionnaire-architect;
|
|
155
165
|
# copy-over never deletes, so scrub the retired agent lest a dead subagent_type linger.
|
|
156
166
|
rm -f "$dest/agents/questionnaire-researcher.md"
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "cohorte",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.6.0",
|
|
4
4
|
"description": "Portable, stack-agnostic multi-agent development pipeline for Claude Code — install the core, run /init-pipeline, and it adapts to your project's stack.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"cohorte": "bin/cli.js"
|
|
@@ -106,7 +106,7 @@ commands:
|
|
|
106
106
|
format: pnpm format
|
|
107
107
|
typecheck: pnpm check-types
|
|
108
108
|
test: pnpm test
|
|
109
|
-
test_quiet: pnpm test --reporter=dot # bridled variant — what /review
|
|
109
|
+
test_quiet: pnpm test --reporter=dot # bridled variant — what the /review preflight runs
|
|
110
110
|
# migration commands — omit / leave "" if the project has no DB migrations
|
|
111
111
|
migrate: "cd apps/api && node ace migration:run"
|
|
112
112
|
make_migration: "cd apps/api && node ace make:migration"
|
|
@@ -164,12 +164,12 @@ gate:
|
|
|
164
164
|
- "git rebase"
|
|
165
165
|
- "git reset"
|
|
166
166
|
- "docker compose"
|
|
167
|
-
# Phase gate: review
|
|
167
|
+
# Phase gate: review dispatches require a fresh `.claude/preflight.ok` stamp,
|
|
168
168
|
# written by pipeline/scripts/preflight.sh when typecheck+lint+tests are green —
|
|
169
169
|
# gate.py "ask"s the dispatch when the stamp is missing, stale, or HEAD moved.
|
|
170
170
|
preflight:
|
|
171
171
|
enabled: true
|
|
172
|
-
agents: [review
|
|
172
|
+
agents: [review] # subagent_types the stamp gates
|
|
173
173
|
max_age_minutes: 30
|
|
174
174
|
|
|
175
175
|
```
|
package/profile/SCHEMA.md
CHANGED
|
@@ -34,7 +34,7 @@ generic pipeline uses it, so a stateless agent can read/regenerate the profile c
|
|
|
34
34
|
| `contract.path` `.ext` `.index` | string | build | Where `<feature_id>` contract is authored + barrel. |
|
|
35
35
|
| `contract.authored_by` | const `lead` | build | Implementers import it read-only, never edit. |
|
|
36
36
|
| `commands.*` | string | all | Repo-wide install/dev/lint/format/typecheck/test + migrate. |
|
|
37
|
-
| `commands.test_quiet` `.lint_quiet` | string | review,
|
|
37
|
+
| `commands.test_quiet` `.lint_quiet` | string | review, audit, workflows | Repo-wide bridled variants — what the `/review` pre-flight runs. Same fallback as the per-surface ones. |
|
|
38
38
|
| `rbac.enabled` | bool | brainstorm, review | Toggle RBAC personas + authz audit. |
|
|
39
39
|
| `rbac.hierarchy` | list | review | Highest→lowest role list. |
|
|
40
40
|
| `design.enabled` | bool | build, frontend, align-ds | `false` ⇒ design steps are no-ops. |
|
|
@@ -52,8 +52,8 @@ generic pipeline uses it, so a stateless agent can read/regenerate the profile c
|
|
|
52
52
|
| `gate.ask[]` | list | hooks/gate.py, settings | Command substrings that require confirm, on any branch. |
|
|
53
53
|
| `gate.ask_on_default_branch[]` | list | hooks/gate.py | Confirm ONLY on `default_branch`; free on feature branches. |
|
|
54
54
|
| `gate.default_branch` | string | hooks/gate.py | Protected branch (default `main`); gate resolves via git. |
|
|
55
|
-
| `gate.preflight.enabled` | bool | hooks/gate.py, review
|
|
56
|
-
| `gate.preflight.agents[]` | list | hooks/gate.py | `subagent_type`s the stamp gates (default `[review
|
|
55
|
+
| `gate.preflight.enabled` | bool | hooks/gate.py, review | Phase gate: review dispatches need a fresh preflight stamp. See §Preflight. |
|
|
56
|
+
| `gate.preflight.agents[]` | list | hooks/gate.py | `subagent_type`s the stamp gates (default `[review]`). |
|
|
57
57
|
| `gate.preflight.max_age_minutes` | number | hooks/gate.py | Stamp freshness window (default 30). |
|
|
58
58
|
|
|
59
59
|
## Prose sections
|
|
@@ -161,8 +161,8 @@ the frozen contract as the only cross-surface channel**. So specialization means
|
|
|
161
161
|
Coarse first, specialize on evidence: start with one `frontend` / `backend` surface each; split only a
|
|
162
162
|
surface that's proven slow and cleanly separable. The evidence lives in
|
|
163
163
|
the **main checkout's** `.claude/pipeline-metrics.jsonl` (gitignored) — one JSONL line per phase batch
|
|
164
|
-
(`ts`/`feature`/`phase`/`seconds`/`surfaces:{key: result}`), appended by `/build`, `/review
|
|
165
|
-
and `/
|
|
164
|
+
(`ts`/`feature`/`phase`/`seconds`/`surfaces:{key: result}`), appended by `/build`, `/review`
|
|
165
|
+
and `/fix`.
|
|
166
166
|
**`surfaces` keys are surface keys, nothing else** — run-level facts go in their own top-level
|
|
167
167
|
fields. Anything put inside `surfaces` is read
|
|
168
168
|
as a surface: the dashboard renders it as a row in the per-surface table and scores a non-`ok`
|
|
@@ -188,7 +188,7 @@ to log it. For what's EXPENSIVE, use Claude Code's own accounting:
|
|
|
188
188
|
per-subagent attribution needs traces (`CLAUDE_CODE_ENHANCED_TELEMETRY_BETA=1`, beta).
|
|
189
189
|
|
|
190
190
|
**Lead context discipline — the silent bill.** The lead session's conversation history is re-sent as
|
|
191
|
-
input on EVERY turn; a session that spans spec→build→
|
|
191
|
+
input on EVERY turn; a session that spans spec→build→review→fix without clearing re-pays the
|
|
192
192
|
accumulated spec walk-through, handoffs, and reports on each turn. The pipeline is built so this is
|
|
193
193
|
never necessary: every phase handoff (spec, contract, diff, staged reports) lives on disk, so `/clear`
|
|
194
194
|
at each phase boundary is always safe — each command's closing line recommends it. Corollaries the
|
|
@@ -217,9 +217,132 @@ Rules for every consumer (implementers, preflight, `/audit` gates, workflow agen
|
|
|
217
217
|
`/init-pipeline` **asks** for these variants (detected defaults offered first) instead of silently
|
|
218
218
|
storing a bare `pnpm test` as the thing agents execute; `/update-pipeline` tops up older profiles.
|
|
219
219
|
|
|
220
|
+
## Spec status — the lifecycle state machine (and the loop's resume state)
|
|
221
|
+
|
|
222
|
+
A spec's front-matter `status` is not a label, it is the pipeline's **state**: every command routes on
|
|
223
|
+
it, the dashboard boards on it, the kanban backfill maps it to a column, and `/drive --resume` reads it
|
|
224
|
+
back to continue an interrupted autonomous run. Six states, and exactly one writer each:
|
|
225
|
+
|
|
226
|
+
| status | meaning | written by | who may build it |
|
|
227
|
+
| --- | --- | --- | --- |
|
|
228
|
+
| `draft` | the interview is open, nothing is frozen | `/spec` Mode A | no |
|
|
229
|
+
| `frozen` | the contract is frozen — the handoff to `/build` | `/spec` Mode A freeze | yes |
|
|
230
|
+
| `in-progress` | a `/drive` is driving this spec right now (or died doing it) | `scripts/loop.sh`, before each phase | yes |
|
|
231
|
+
| `in-review` | reviewed / awaiting the next round or `/ship` | `/spec` Mode B, `/fix`, `loop.sh` on a clean exit | yes |
|
|
232
|
+
| `blocked` | a loop gave up here (ceiling, non-convergent, no verdict, not implementable) | `loop.sh` on any non-zero exit | yes, with the reason named |
|
|
233
|
+
| `shipped` | the PR is open; the status flip is part of the release commit | `/ship` | no |
|
|
234
|
+
|
|
235
|
+
**The resume contract.** Before every phase, `loop.sh` stamps `status: in-progress` plus `loop_pass`
|
|
236
|
+
(the review pass it is on) and `loop_phase` (`build`/`review`/`fix`) into the spec — deterministically,
|
|
237
|
+
with `awk`, spending **no tokens** on state it will need later. On exit it stamps a terminal status:
|
|
238
|
+
`in-review` + `loop_phase: done` when clean, `blocked` otherwise. `/drive <id> --resume` then continues
|
|
239
|
+
at the recorded pass instead of pass 1, so a session killed at pass 3 of 5 does not re-pay passes 1–2.
|
|
240
|
+
The build is still skipped or redone by the build stamp alone (`specs/reports/<id>.built`, written only
|
|
241
|
+
after a build that finished), so an interrupted *build* correctly rebuilds.
|
|
242
|
+
|
|
243
|
+
Corollaries worth knowing:
|
|
244
|
+
|
|
245
|
+
- A spec with no front-matter makes every stamp a **silent no-op** — the state is bookkeeping, and the
|
|
246
|
+
loop must never die over a status line.
|
|
247
|
+
- Child commands write `status` too (`/fix` sets `in-review`); re-stamping before each phase is what
|
|
248
|
+
keeps `in-progress` true for the duration of the run rather than for its first phase.
|
|
249
|
+
- `blocked` is not a failure to hide: it is the resumable state. `/build` accepts it, names it, and
|
|
250
|
+
routes by the spec's `## Remediation` (open items ⇒ `/fix`).
|
|
251
|
+
|
|
252
|
+
## Dead agents — silence is not a green light
|
|
253
|
+
|
|
254
|
+
A subagent can die mid-run: a rate limit, a transport error that outlived its retries, its own context
|
|
255
|
+
exhausted on a big surface. When it does it returns **nothing** — and nothing is byte-identical to
|
|
256
|
+
"finished, nothing to report". Every phase that fans out therefore does a **roll call** before it
|
|
257
|
+
integrates anything, because the default reading of silence is the most dangerous one available:
|
|
258
|
+
|
|
259
|
+
| phase | what a dead agent looks like | what the phase must do |
|
|
260
|
+
| --- | --- | --- |
|
|
261
|
+
| `/build` | a surface with no handoff | retry it **once** alone (byte-identical prompt), then mark it `dead`, verify the tree with that surface's own quiet commands, never call the batch ok |
|
|
262
|
+
| `/review` | a reviewer with no report ⇒ **zero findings** | retry once, then list the surface in `unreviewed` and refuse to score `SHIP` |
|
|
263
|
+
| `/fix` | a re-dispatched agent with no handoff | retry once, then leave **every** one of its items `- [ ]` — a dead agent never ticks a box |
|
|
264
|
+
| workflows | `agent()` resolves to `null` | already enforced (`review.js` `unreviewedSurfaces`) — the doctrine started here |
|
|
265
|
+
|
|
266
|
+
Non-negotiables, in every phase:
|
|
267
|
+
|
|
268
|
+
- **Retry once, alone, byte-identical.** Most deaths are transient, and the other surfaces' work is
|
|
269
|
+
already on disk — so recovery costs one agent, never a rebuild. Never retry an agent that answered.
|
|
270
|
+
- **Never speak for a dead agent.** You did not see its work: report what the *tree* says (quiet
|
|
271
|
+
commands, redirected to a file, grepped), not what a handoff would have said.
|
|
272
|
+
- **Never let it reach a driver as clean.** `/build` writes `dead[]` into
|
|
273
|
+
`specs/reports/<id>.build.json`, `/review` writes `unreviewed[]` into the verdict; `scripts/loop.sh`
|
|
274
|
+
aborts on either with **exit 2** *before* it reads `blocking`, since a dead reviewer makes
|
|
275
|
+
`blocking == 0` a statement about code nobody read.
|
|
276
|
+
- **`unreviewed` is separate from `blocking` on purpose.** Faking a count in `blocking` to force a
|
|
277
|
+
driver's hand would corrupt the one field the whole contract rests on; a driver reads them as two
|
|
278
|
+
different facts — "what was found" and "what was covered".
|
|
279
|
+
- **Write the metrics line anyway** (`"<key>":"dead"`). An incomplete batch is exactly the batch worth
|
|
280
|
+
recording; holding the append back "until it's complete" deletes the evidence that anything failed.
|
|
281
|
+
|
|
282
|
+
## Readiness — the gate between a frozen spec and N implementers
|
|
283
|
+
|
|
284
|
+
`/build` §1.6 scores the frozen spec on **implementability** before authoring the contract and before
|
|
285
|
+
dispatching anything, and writes `specs/reports/<id>.readiness.json`
|
|
286
|
+
(`verdict`: `READY` · `RESERVATIONS` · `NOT-READY`, plus `gaps[]`). It costs **zero extra agents** — the
|
|
287
|
+
lead already holds the spec, the profile and the reconciled surface list — which is the whole economics
|
|
288
|
+
of the step: a spec that cannot be built does not get cheaper by being built on N surfaces in parallel.
|
|
289
|
+
|
|
290
|
+
- Five checks: contract completeness · surface coverage · dependencies exist · residual ambiguity ·
|
|
291
|
+
the design gate. Each maps to `NOT-READY` (a surface would have to invent the answer) or
|
|
292
|
+
`RESERVATIONS` (a surface can proceed on a stated assumption).
|
|
293
|
+
- **`NOT-READY` aborts the build with no agent spawned** and sends the human to `/spec`.
|
|
294
|
+
`scripts/loop.sh` reads the same file and exits **4** (`not implementable`) — the one loop outcome
|
|
295
|
+
that more passes cannot fix.
|
|
296
|
+
- **`RESERVATIONS` never blocks.** Each gap is inlined verbatim into the dispatch of the surface it
|
|
297
|
+
affects, as an assumption the implementer must apply *and* flag in its handoff. A gate that stalled a
|
|
298
|
+
sound build on a missing error case would cost more human round-trips than it saves.
|
|
299
|
+
|
|
300
|
+
## Deferred findings — real, but not this feature's problem
|
|
301
|
+
|
|
302
|
+
`/review` ends on "zero blocking findings", so everything non-blocking used to be discarded with the
|
|
303
|
+
report. A **deferred** finding is one the reviewer judges true and **out of this feature's scope**
|
|
304
|
+
(pre-existing code the staged diff never touched, adjacent debt the spec never claims to fix). The
|
|
305
|
+
review agent returns them in their own `## Deferred` section — never in `findings` — each carrying its
|
|
306
|
+
own out-of-scope reason.
|
|
307
|
+
|
|
308
|
+
- They count in **no** severity row, enter **no** verdict, and are **never** cross-checked: a deferred
|
|
309
|
+
item cannot cost a fix loop an iteration, and refuting one would spend an agent arguing about
|
|
310
|
+
something that cannot change the outcome.
|
|
311
|
+
- **Not deferrable, ever:** anything the diff touched or introduced, any spec violation, any security
|
|
312
|
+
issue on a path this feature adds, calls or modifies.
|
|
313
|
+
- `/review` §3.5 routes them, **on every verdict**, into `specs/refactor-backlog.md` under the
|
|
314
|
+
`## <domain>` heading of the owning surface, tagged `deferred:<feature_id>` — the same grouping
|
|
315
|
+
`/audit` writes, so `/refactor <domain>` picks them up with no extra plumbing. Never into the spec's
|
|
316
|
+
`## Remediation`, which is what `/fix` re-dispatches.
|
|
317
|
+
- `/audit` **carries open `deferred:` items over** when it rewrites the backlog; overwriting them away
|
|
318
|
+
is the one way they silently vanish.
|
|
319
|
+
- The verdict JSON carries `deferred: <n>` (informational, outside `blocking`), so `/drive` can name
|
|
320
|
+
them in its closing line without reading a report.
|
|
321
|
+
|
|
322
|
+
## Decisions — the transverse decision journal
|
|
323
|
+
|
|
324
|
+
`PIPELINE.md` is a **stack profile** (surfaces, commands, conventions); it says nothing about what this
|
|
325
|
+
project has *decided*. Without somewhere for those, every `/spec` re-discovers or contradicts them.
|
|
326
|
+
`specs/_decisions.md` (from `core/templates/decisions.template.md`) is that place, deliberately small:
|
|
327
|
+
|
|
328
|
+
- **Append-only, one line per decision, ≤ ~160 chars:**
|
|
329
|
+
`- <YYYY-MM-DD> · <area> · <decision> — because <reason> · <feature_id>`. Reversal never edits a line:
|
|
330
|
+
append a superseding one (`· supersedes <date> <area>`) and move the old one to `## Superseded`. When
|
|
331
|
+
`## Live` passes ~100 lines, sweep the superseded ones down.
|
|
332
|
+
- **Written by** `/spec` at freeze (the decisions that outlive the feature — typically 0–3 lines, and
|
|
333
|
+
zero is a normal outcome) and `/build` §1.5 when it adds or splits a surface.
|
|
334
|
+
- **Read by the deciding stages only** — `/brainstorm` (so the panel argues about the idea, not about
|
|
335
|
+
settled ground), `/spec` (so a new spec does not silently un-decide something), `/audit` (standing
|
|
336
|
+
decisions are part of the rulebook it audits against).
|
|
337
|
+
- **Never read by implementers or reviewers.** They work from the frozen contract, which already tells
|
|
338
|
+
them what to do; shipping them the rationale would cost `surfaces × dispatches` tokens per feature
|
|
339
|
+
for a fact they cannot act on. This is what keeps the journal cheap enough to be worth having.
|
|
340
|
+
- The `_` prefix is load-bearing: `/doctor`, the dashboard spec scanner and the kanban backfill all skip
|
|
341
|
+
`specs/_*.md`, so the journal is never mistaken for a spec (no phantom card, no bogus stage).
|
|
342
|
+
|
|
220
343
|
## Preflight — the deterministic phase gate
|
|
221
344
|
|
|
222
|
-
`/review`
|
|
345
|
+
`/review` starts by running `pipeline/scripts/preflight.sh` — a plain shell script (no
|
|
223
346
|
agent) that executes the profile's mechanical checks in order (typecheck → lint → tests, quiet
|
|
224
347
|
variants) with all output redirected to `specs/reports/<id>.preflight.txt`:
|
|
225
348
|
|
|
@@ -231,7 +354,7 @@ variants) with all output redirected to `specs/reports/<id>.preflight.txt`:
|
|
|
231
354
|
|
|
232
355
|
`hooks/gate.py` enforces the stamp as a **phase gate** (the `preflight` block of `gate-config.json`,
|
|
233
356
|
generated from `gate.preflight`): a Task dispatch of a listed `subagent_type` (default
|
|
234
|
-
`review
|
|
357
|
+
`review`) with a missing/stale stamp — older than `max_age_minutes`, or HEAD moved — gets an
|
|
235
358
|
"ask", so a lead can't accidentally skip the gate but a human can consciously override it. The gate
|
|
236
359
|
hook fires for **every** agent in the session, including subagents spawned by the Workflow runtime
|
|
237
360
|
(they run in `acceptEdits` whatever the session mode — Write/Edit auto-approved — but Bash and Task
|
|
@@ -313,6 +436,15 @@ files automatically. It works because every generated artifact is a **determinis
|
|
|
313
436
|
clobber an existing filled file; report what was seeded.
|
|
314
437
|
6. **Kanban sync.** Run the §Kanban reconcile: link/create the project's board if configured, verify
|
|
315
438
|
its columns, and backfill/sync cards from `specs/*.md`. See §Kanban.
|
|
439
|
+
7. **Spec-template top-up.** `specs/_template.md` is seeded once at install and then **never**
|
|
440
|
+
refreshed, so a repo keeps whatever front-matter the core shipped the day it was installed (a
|
|
441
|
+
pre-1.6 copy has no `loop_pass`/`loop_phase`, and its `status` comment still lists four states).
|
|
442
|
+
Top it up the same way as the profile: add the **front-matter fields** the current
|
|
443
|
+
`templates/spec.template.md` has and the repo's copy lacks, with their documented defaults, and
|
|
444
|
+
refresh the `status:` comment. Never rewrite its body — the section list is the human's to shape,
|
|
445
|
+
and some repos have deliberately trimmed it. Nothing breaks without this (the fields are written on
|
|
446
|
+
demand when a driver needs them); it just keeps a new spec's front-matter honest about the states
|
|
447
|
+
the pipeline can put it in.
|
|
316
448
|
|
|
317
449
|
Re-running `/init-pipeline` remains possible (it reconciles too) but is only *needed* when the stack
|
|
318
450
|
itself changes in ways `/build` §1.5 can't auto-grow (e.g. package manager or contract mechanism swap).
|
|
@@ -347,6 +479,9 @@ Shared design, all four scripts:
|
|
|
347
479
|
that never answered: `review.js` names them in `unreviewedSurfaces` and refuses to score
|
|
348
480
|
`SHIP`. `scripts/test-workflows.mjs`
|
|
349
481
|
pins this — it is the one invariant the structural checks in `validate-core.mjs` cannot see.
|
|
482
|
+
The conversational commands enforce the same rule by roll call (§Dead agents); it was the workflows
|
|
483
|
+
that had it first, and for three releases they had it **alone** — the same crash on the
|
|
484
|
+
conversational path went unreported.
|
|
350
485
|
- **`review.js`** — preflight gate (aborts red, zero agents), one `git diff --stat` staged per
|
|
351
486
|
touched surface, one reviewer per surface in parallel, then an **adversarial cross-check** phase
|
|
352
487
|
that tries to refute each CRITICAL/security finding before it can trigger a fix loop.
|
|
@@ -409,14 +544,18 @@ card created in the target column if missing.
|
|
|
409
544
|
| `/spec` opens (draft) | `spec` |
|
|
410
545
|
| `/spec` freezes (`status: frozen`) | `ready` |
|
|
411
546
|
| `/build` | `building` |
|
|
412
|
-
| `/review`
|
|
547
|
+
| `/review` | `review` |
|
|
413
548
|
| `/fix` | `fix` |
|
|
549
|
+
| a `/drive` is driving it (`in-progress`) | the current phase's column |
|
|
550
|
+
| a `/drive` gave up (`blocked`) | `fix` |
|
|
414
551
|
| `/ship` starts | `ship` |
|
|
415
552
|
| PR opened (`status: shipped`) | `shipped` (+ `PR #<num>` on the card) |
|
|
416
553
|
|
|
417
554
|
**Backfill / sync from specs (reconcile).** `specs/*.md` is the source of truth. For each spec, read its
|
|
418
555
|
`feature_id` (front-matter or filename) and `status`, map `status`→column — `frozen`→`ready`,
|
|
419
|
-
`in-
|
|
556
|
+
`in-progress`→the `loop_phase`'s column (`build`→`building`, `review`→`review`, `fix`→`fix`; unset ⇒
|
|
557
|
+
`building`), `in-review`→`review`, `blocked`→`fix`, `shipped`→`shipped`, anything else / a spec with no
|
|
558
|
+
status→`spec` — then **full
|
|
420
559
|
sync**: card absent ⇒ add it in that column; card present ⇒ **move it** to that column so the board
|
|
421
560
|
always reflects the specs (this repositions cards the human may have moved by hand). Report cards
|
|
422
561
|
added vs. moved vs. already-correct.
|
|
@@ -437,7 +576,7 @@ pre-telemetry installs) ask ONE question, once per machine, default **No**, and
|
|
|
437
576
|
`|| true`, so a **missing** script is equally silent: `/doctor` check 1 verifies `pipeline/scripts/`
|
|
438
577
|
is fully populated.
|
|
439
578
|
|
|
440
|
-
**Which commands ping** — the
|
|
579
|
+
**Which commands ping** — the six that make up the feature funnel, and only those. The point is to
|
|
441
580
|
see where features stall, so every stage of `idea → PR` reports and nothing else does:
|
|
442
581
|
|
|
443
582
|
| phase | fired when | `seconds` | `results` |
|
|
@@ -445,7 +584,6 @@ see where features stall, so every stage of `idea → PR` reports and nothing el
|
|
|
445
584
|
| `brainstorm` | the return is staged | `0` | — |
|
|
446
585
|
| `spec` | a freeze lands (Mode A only) | `0` | `frozen` |
|
|
447
586
|
| `build` | after the batch metrics line | wall-clock | `ok,ok` / `error` |
|
|
448
|
-
| `smoke` | after the verdict | wall-clock | `PASS` / `FAIL:<n>` |
|
|
449
587
|
| `review` | after the merged verdict | wall-clock | `<verdict>:<count>` |
|
|
450
588
|
| `fix` | after the batch metrics line | wall-clock | `<fixed>/<found>` |
|
|
451
589
|
| `ship` | the release agent succeeded | `0` | `pr` / `compare` |
|