cohorte 1.3.4 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +110 -0
- package/README.md +36 -16
- package/bin/cli.js +34 -4
- package/core/agents/profile-reader.md +22 -0
- package/core/commands/build.md +4 -4
- package/core/commands/doctor.md +10 -8
- package/core/commands/fix.md +5 -5
- package/core/commands/loop.md +61 -0
- package/core/commands/review.md +38 -5
- package/core/hooks/gate.py +4 -4
- package/core/templates/spec.template.md +1 -1
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
- package/core/workflows/audit.js +68 -4
- package/core/workflows/refactor.js +69 -4
- package/core/workflows/review.js +73 -4
- package/dashboard/dist/assets/index-8owBnqyv.js +43 -0
- package/dashboard/dist/assets/{index-AFQnlfjO.css → index-dkO8UUVl.css} +1 -1
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +15 -5
- package/dashboard/server/index.js +7 -0
- package/dashboard/server/metrics.js +6 -5
- package/dashboard/server/usage.js +61 -0
- package/install.ps1 +4 -1
- package/install.sh +5 -2
- package/package.json +1 -1
- package/profile/PIPELINE.template.md +3 -3
- package/profile/SCHEMA.md +21 -44
- package/scripts/loop.sh +189 -0
- package/scripts/metrics/collect.mjs +504 -0
- package/scripts/metrics/prices.json +39 -0
- package/scripts/preflight.sh +2 -2
- package/scripts/telemetry-send.sh +5 -2
- package/scripts/test-dashboard.mjs +29 -3
- package/scripts/test-gate.mjs +1 -2
- package/scripts/test-metrics.mjs +144 -0
- package/scripts/test-workflows.mjs +56 -178
- package/scripts/validate-core.mjs +10 -9
- package/core/agents/smoke.md +0 -63
- package/core/commands/cycle.md +0 -61
- package/core/commands/smoke.md +0 -55
- package/core/workflows/cycle.js +0 -513
- package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
package/scripts/test-gate.mjs
CHANGED
|
@@ -181,7 +181,7 @@ console.log("gate.py — config robustness");
|
|
|
181
181
|
// ── the preflight phase gate (Task dispatches) ───────────────────────────────
|
|
182
182
|
console.log("gate.py — preflight phase gate");
|
|
183
183
|
{
|
|
184
|
-
const pf = { enabled: true, agents: ["review"
|
|
184
|
+
const pf = { enabled: true, agents: ["review"], max_age_minutes: 30 };
|
|
185
185
|
const d = scratch(); writeConfig(d, { ...GATE_CFG, preflight: pf });
|
|
186
186
|
const head = gitRepo(d, "main");
|
|
187
187
|
const stamp = (epoch, sha) =>
|
|
@@ -195,7 +195,6 @@ console.log("gate.py — preflight phase gate");
|
|
|
195
195
|
|
|
196
196
|
stamp(now(), head);
|
|
197
197
|
check("fresh stamp at the current HEAD ⇒ passes", run(task("review"), at).decision === null);
|
|
198
|
-
check("smoke is gated too", run(task("smoke"), at).decision === null);
|
|
199
198
|
|
|
200
199
|
stamp(now() - 60 * 60, head);
|
|
201
200
|
check("stamp older than max_age_minutes ⇒ ask", run(task("review"), at).decision === "ask");
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// test-metrics.mjs — end-to-end checks for scripts/metrics/collect.mjs.
|
|
3
|
+
//
|
|
4
|
+
// Builds a throwaway repo plus a synthetic ~/.claude/projects transcript, runs the real
|
|
5
|
+
// collector against it via --json, and asserts the numbers. The cases are the ones that
|
|
6
|
+
// silently produce plausible-but-wrong output rather than crashing:
|
|
7
|
+
//
|
|
8
|
+
// 1. one API response written as several transcript lines, each repeating `usage`
|
|
9
|
+
// 2. a <task-notification> arriving mid-command (must not split the run)
|
|
10
|
+
// 3. subagent spend, which lives in a separate file tree
|
|
11
|
+
// 4. <synthetic> harness messages, which carry usage but cost nothing
|
|
12
|
+
// 5. the cache-tier pricing arithmetic itself
|
|
13
|
+
//
|
|
14
|
+
// Run: node scripts/test-metrics.mjs
|
|
15
|
+
|
|
16
|
+
import fs from 'node:fs';
|
|
17
|
+
import os from 'node:os';
|
|
18
|
+
import path from 'node:path';
|
|
19
|
+
import { execFileSync, spawnSync } from 'node:child_process';
|
|
20
|
+
import { fileURLToPath } from 'node:url';
|
|
21
|
+
|
|
22
|
+
const HERE = path.dirname(fileURLToPath(import.meta.url));
|
|
23
|
+
const COLLECT = path.join(HERE, 'metrics', 'collect.mjs');
|
|
24
|
+
|
|
25
|
+
let failures = 0;
|
|
26
|
+
const ok = (label) => console.log(` ✓ ${label}`);
|
|
27
|
+
function check(label, actual, expected) {
|
|
28
|
+
const a = JSON.stringify(actual), e = JSON.stringify(expected);
|
|
29
|
+
if (a === e) return ok(label);
|
|
30
|
+
failures += 1;
|
|
31
|
+
console.log(` ✗ ${label}\n expected ${e}\n actual ${a}`);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
const tmp = fs.mkdtempSync(path.join(os.tmpdir(), 'cohorte-metrics-'));
|
|
35
|
+
const repo = path.join(tmp, 'repo');
|
|
36
|
+
const cfg = path.join(tmp, 'claude');
|
|
37
|
+
const SESSION = 'sess-test-0001';
|
|
38
|
+
const projectDir = path.join(cfg, 'projects', 'test-slug');
|
|
39
|
+
fs.mkdirSync(path.join(projectDir, SESSION, 'subagents'), { recursive: true });
|
|
40
|
+
fs.mkdirSync(repo, { recursive: true });
|
|
41
|
+
execFileSync('git', ['init', '-q'], { cwd: repo, stdio: 'ignore' });
|
|
42
|
+
|
|
43
|
+
const T0 = Date.parse('2026-07-30T10:00:00.000Z');
|
|
44
|
+
const at = (s) => new Date(T0 + s * 1000).toISOString();
|
|
45
|
+
|
|
46
|
+
const assistant = (id, tsS, model, usage, content = [{ type: 'text', text: 'x' }]) => ({
|
|
47
|
+
type: 'assistant', timestamp: at(tsS), cwd: repo, sessionId: SESSION,
|
|
48
|
+
message: { id, model, content, usage },
|
|
49
|
+
});
|
|
50
|
+
const user = (tsS, text) => ({
|
|
51
|
+
type: 'user', timestamp: at(tsS), cwd: repo, sessionId: SESSION,
|
|
52
|
+
message: { role: 'user', content: text },
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
const usageOpus = {
|
|
56
|
+
input_tokens: 100, output_tokens: 1000,
|
|
57
|
+
cache_creation_input_tokens: 1000, cache_read_input_tokens: 10000,
|
|
58
|
+
cache_creation: { ephemeral_5m_input_tokens: 1000, ephemeral_1h_input_tokens: 0 },
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
const lines = [
|
|
62
|
+
user(0, '<command-message>build</command-message>\n<command-name>/build</command-name>'),
|
|
63
|
+
// Case 1: one response, three lines, identical usage on each. Only one should be billed.
|
|
64
|
+
assistant('m1', 5, 'claude-opus-5', usageOpus, [{ type: 'thinking', thinking: '...' }]),
|
|
65
|
+
assistant('m1', 5, 'claude-opus-5', usageOpus, [{ type: 'text', text: 'hello' }]),
|
|
66
|
+
assistant('m1', 6, 'claude-opus-5', usageOpus, [{ type: 'tool_use', id: 'toolu_A', name: 'Task', input: {} }]),
|
|
67
|
+
// Case 2: a background agent finished mid-run. This is not the human starting anything.
|
|
68
|
+
user(10, '<task-notification>\n<task-id>a1</task-id>\n</task-notification>'),
|
|
69
|
+
// Case 4: harness-authored message, has usage, costs nothing.
|
|
70
|
+
assistant('m2', 12, '<synthetic>', { input_tokens: 0, output_tokens: 999999 }),
|
|
71
|
+
assistant('m3', 20, 'claude-opus-5', { input_tokens: 0, output_tokens: 500 }),
|
|
72
|
+
// A second, genuinely separate run — long enough not to read as a steering turn.
|
|
73
|
+
user(600, 'unrelated question about the repository layout and its conventions'),
|
|
74
|
+
assistant('m4', 605, 'claude-opus-5', { input_tokens: 0, output_tokens: 40 }),
|
|
75
|
+
// Case 6: a command named inside ordinary prose. The harness emits no <command-name>
|
|
76
|
+
// for this, but it is the way commands actually get invoked in practice.
|
|
77
|
+
user(1200, 'move on branding-ramp and /review'),
|
|
78
|
+
assistant('m5', 1205, 'claude-opus-5', { input_tokens: 0, output_tokens: 60 }),
|
|
79
|
+
// Case 7: a short steer continues the /review rather than opening an anonymous run.
|
|
80
|
+
user(1260, 'continue'),
|
|
81
|
+
assistant('m6', 1265, 'claude-opus-5', { input_tokens: 0, output_tokens: 70 }),
|
|
82
|
+
// Case 8: a slash token that is not a command must not invent one.
|
|
83
|
+
user(1800, 'look at the /usr/local/share directory and report what you find there'),
|
|
84
|
+
assistant('m7', 1805, 'claude-opus-5', { input_tokens: 0, output_tokens: 10 }),
|
|
85
|
+
// Case 9: a long prompt that merely DISCUSSES a command is not an invocation of it.
|
|
86
|
+
// Without the length gate, writing about /review bills the conversation to /review —
|
|
87
|
+
// which is what happened in cohorte's own repo while the pipeline was being designed.
|
|
88
|
+
user(2400, 'I want to talk through how /review behaves when a surface has no findings at '
|
|
89
|
+
+ 'all, because the verdict logic there is what produced the false green we saw last week '
|
|
90
|
+
+ 'and I am not convinced the fix covers the case where every reviewer dies at once.'),
|
|
91
|
+
assistant('m8', 2405, 'claude-opus-5', { input_tokens: 0, output_tokens: 20 }),
|
|
92
|
+
];
|
|
93
|
+
fs.writeFileSync(path.join(projectDir, `${SESSION}.jsonl`),
|
|
94
|
+
lines.map((l) => JSON.stringify(l)).join('\n') + '\n');
|
|
95
|
+
|
|
96
|
+
// Case 3: subagent spend, linked back to /build by the Task tool_use id.
|
|
97
|
+
const agentDir = path.join(projectDir, SESSION, 'subagents');
|
|
98
|
+
fs.writeFileSync(path.join(agentDir, 'agent-a1.meta.json'),
|
|
99
|
+
JSON.stringify({ agentType: 'core', description: 'Build core surface', toolUseId: 'toolu_A', spawnDepth: 1 }));
|
|
100
|
+
fs.writeFileSync(path.join(agentDir, 'agent-a1.jsonl'),
|
|
101
|
+
JSON.stringify(assistant('s1', 8, 'claude-sonnet-5', { input_tokens: 0, output_tokens: 2000 })) + '\n');
|
|
102
|
+
|
|
103
|
+
const run = spawnSync(process.execPath, [COLLECT, repo, '--json', '--runs'], {
|
|
104
|
+
encoding: 'utf8',
|
|
105
|
+
env: { ...process.env, CLAUDE_CONFIG_DIR: cfg },
|
|
106
|
+
});
|
|
107
|
+
if (run.status !== 0) {
|
|
108
|
+
console.error('collector failed:\n' + (run.stderr || run.stdout));
|
|
109
|
+
process.exit(1);
|
|
110
|
+
}
|
|
111
|
+
const out = JSON.parse(run.stdout);
|
|
112
|
+
const build = out.commands.find((c) => c.command === '/build');
|
|
113
|
+
const chat = out.commands.find((c) => c.command === '(chat)');
|
|
114
|
+
const review = out.commands.find((c) => c.command === '/review');
|
|
115
|
+
|
|
116
|
+
console.log('test-metrics');
|
|
117
|
+
check('the mid-command task-notification did not split the run', out.totals.runs, 5);
|
|
118
|
+
check('/build is one run, not three', build.runs, 1);
|
|
119
|
+
check('duplicate lines of one response are billed once', build.tokens.output, 1000 + 500 + 2000);
|
|
120
|
+
check('the <synthetic> message contributed no tokens', build.tokens.output < 999999, true);
|
|
121
|
+
check('cache-write tokens are kept on their own tier', build.tokens.cacheWrite5m, 1000);
|
|
122
|
+
check('cache-read tokens are kept on their own tier', build.tokens.cacheRead, 10000);
|
|
123
|
+
check('the subagent was attributed to the command that spawned it', build.agents.total, 1);
|
|
124
|
+
check('the second prompt is a separate (chat) run', chat.runs, 3);
|
|
125
|
+
check('a command named inside prose is attributed to that command', review && review.runs, 1);
|
|
126
|
+
check('a short steer continues the run instead of opening a new one', review.continuations, 1);
|
|
127
|
+
check('the continued turn counts toward the command it continued', review.tokens.output, 60 + 70);
|
|
128
|
+
check('a non-command slash token does not invent a command', chat.tokens.output, 40 + 10 + 20);
|
|
129
|
+
check('a long prompt that discusses a command is not counted as running it',
|
|
130
|
+
review.runs, 1);
|
|
131
|
+
|
|
132
|
+
// opus-5 $5 in / $25 out per MTok; 5m cache write 1.25x input, cache read 0.1x input.
|
|
133
|
+
// m1 100*5 + 1000*25 + 1000*6.25 + 10000*0.5 = 36750
|
|
134
|
+
// m3 500*25 = 12500
|
|
135
|
+
// s1 sonnet-5 2000*15 = 30000 (subagent)
|
|
136
|
+
check('cost sums the cache tiers at their own rates', Number(build.cost.total.toFixed(6)), 0.07925);
|
|
137
|
+
check('the unpriced list stays empty for known models', build.unpriced, []);
|
|
138
|
+
|
|
139
|
+
const detail = out.runs.find((r) => r.command === '/build');
|
|
140
|
+
check('per-run detail carries the subagent', detail.agents.map((a) => a.type), ['core']);
|
|
141
|
+
|
|
142
|
+
fs.rmSync(tmp, { recursive: true, force: true });
|
|
143
|
+
console.log(failures ? `\ntest-metrics: ${failures} FAILED` : '\ntest-metrics: OK');
|
|
144
|
+
process.exit(failures ? 1 : 0);
|
|
@@ -8,9 +8,8 @@
|
|
|
8
8
|
//
|
|
9
9
|
// This exists because of one specific failure mode: agent() resolves to `null`
|
|
10
10
|
// when a subagent dies, and a dead reviewer produces zero findings — which is
|
|
11
|
-
// byte-identical to a clean surface.
|
|
12
|
-
//
|
|
13
|
-
// code no reviewer had read. A unit test is the only thing that catches it: the
|
|
11
|
+
// byte-identical to a clean surface. review.js scored that as SHIP over code no
|
|
12
|
+
// reviewer had read. A unit test is the only thing that catches it: the
|
|
14
13
|
// structural checks in validate-core.mjs cannot see verdict logic.
|
|
15
14
|
//
|
|
16
15
|
// node scripts/test-workflows.mjs
|
|
@@ -99,17 +98,6 @@ const BASE_REVIEW = [
|
|
|
99
98
|
["stage-report", "done"],
|
|
100
99
|
];
|
|
101
100
|
|
|
102
|
-
const BASE_CYCLE = [
|
|
103
|
-
["profile", PROFILE],
|
|
104
|
-
["ready", { frozen: true, gaps: [], designLinks: "none" }],
|
|
105
|
-
["preflight", { pass: true }],
|
|
106
|
-
["stage-diff", { surfaces: TOUCHED }],
|
|
107
|
-
["build:", "handoff ok"],
|
|
108
|
-
["fix:", "handoff ok"],
|
|
109
|
-
["close", "done"],
|
|
110
|
-
];
|
|
111
|
-
|
|
112
|
-
// ── review.js ────────────────────────────────────────────────────────────────
|
|
113
101
|
console.log("review.js");
|
|
114
102
|
{
|
|
115
103
|
const { result } = await run("review.js", replier([
|
|
@@ -172,165 +160,74 @@ console.log("review.js");
|
|
|
172
160
|
!calls.some(c => c.startsWith("review:")), calls.join(","));
|
|
173
161
|
}
|
|
174
162
|
|
|
175
|
-
// ──
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
]));
|
|
181
|
-
check("clean run, smoke off ⇒ SHIP-READY", result.outcome === "SHIP-READY", `got ${result.outcome}`);
|
|
182
|
-
check("smoke off ⇒ smoke: SKIPPED", result.smoke === "SKIPPED", result.smoke);
|
|
183
|
-
check("smoke off ⇒ next warns nobody ran the code",
|
|
184
|
-
/\/smoke/.test(result.next), result.next);
|
|
185
|
-
check("clean run ⇒ no questions", (result.questions || []).length === 0, JSON.stringify(result.questions));
|
|
186
|
-
}
|
|
187
|
-
{
|
|
188
|
-
const { result } = await run("cycle.js", replier([
|
|
189
|
-
["smoke", { pass: true, failures: [] }],
|
|
190
|
-
["review:", { verdict: "SHIP", findings: [] }], ...BASE_CYCLE,
|
|
191
|
-
]), { feature: "feat-x", smoke: true });
|
|
192
|
-
check("clean run, smoke on ⇒ SHIP-READY", result.outcome === "SHIP-READY", `got ${result.outcome}`);
|
|
193
|
-
check("smoke on ⇒ smoke: PASS", result.smoke === "PASS", result.smoke);
|
|
194
|
-
check("smoke on + clean ⇒ next is a straight /ship",
|
|
195
|
-
/straight shot/.test(result.next), result.next);
|
|
196
|
-
}
|
|
163
|
+
// ── args normalisation ───────────────────────────────────────────────────────
|
|
164
|
+
// The runtime passes `args` through verbatim, so a caller that JSON-encodes it
|
|
165
|
+
// hands the script a string. That string used to become the feature id itself —
|
|
166
|
+
// which is how a report was written to `specs/reports/{"feature": "x"}.md`.
|
|
167
|
+
console.log("args");
|
|
197
168
|
{
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
check("all reviewers dead ⇒ not SHIP-READY", result.outcome !== "SHIP-READY", `got ${result.outcome}`);
|
|
204
|
-
check("all reviewers dead ⇒ verdict not SHIP", result.verdict !== "SHIP", result.verdict);
|
|
205
|
-
check("all reviewers dead ⇒ surfaces reported",
|
|
206
|
-
(result.unreviewedSurfaces || []).length === 2, JSON.stringify(result.unreviewedSurfaces));
|
|
207
|
-
check("all reviewers dead ⇒ a question names them",
|
|
208
|
-
(result.questions || []).some(q => /not reviewed/i.test(q)), JSON.stringify(result.questions));
|
|
209
|
-
}
|
|
210
|
-
{
|
|
211
|
-
// …and it must retry the review round rather than dispatching an empty fix round.
|
|
212
|
-
const { calls } = await run("cycle.js", replier([
|
|
213
|
-
["review:", null], ...BASE_CYCLE,
|
|
214
|
-
]), { feature: "feat-x", maxRounds: 3 });
|
|
215
|
-
check("dead reviewers ⇒ review retried across rounds",
|
|
216
|
-
calls.filter(c => c.startsWith("review:")).length > 2,
|
|
217
|
-
`review calls: ${calls.filter(c => c.startsWith("review:")).length}`);
|
|
218
|
-
check("dead reviewers ⇒ no empty fix round dispatched",
|
|
219
|
-
!calls.some(c => c.startsWith("fix:")), calls.join(","));
|
|
220
|
-
}
|
|
221
|
-
{
|
|
222
|
-
const { result } = await run("cycle.js", replier([
|
|
223
|
-
["ready", { frozen: false, gaps: ["status is draft"], designLinks: "none" }], ...BASE_CYCLE,
|
|
224
|
-
]));
|
|
225
|
-
check("unfrozen spec ⇒ NOT-READY", result.outcome === "NOT-READY", `got ${result.outcome}`);
|
|
226
|
-
check("unfrozen spec ⇒ the gap is in questions",
|
|
227
|
-
(result.questions || []).some(q => /draft/.test(q)), JSON.stringify(result.questions));
|
|
228
|
-
}
|
|
229
|
-
{
|
|
230
|
-
const { result } = await run("cycle.js", replier([
|
|
231
|
-
["smoke", { pass: false, failures: ["❌ POST /x · expected 201 got 500 · apps/api/a.ts"] }],
|
|
232
|
-
["review:", { verdict: "SHIP", findings: [] }], ...BASE_CYCLE,
|
|
233
|
-
]), { feature: "feat-x", smoke: true, maxRounds: 1 });
|
|
234
|
-
check("smoke on + FAIL ⇒ not SHIP-READY", result.outcome !== "SHIP-READY", `got ${result.outcome}`);
|
|
235
|
-
check("smoke on + FAIL ⇒ smoke: FAIL", result.smoke === "FAIL", result.smoke);
|
|
236
|
-
}
|
|
237
|
-
{
|
|
238
|
-
// A finding in round 1 that the fix clears must let round 2 exit clean.
|
|
239
|
-
let round = 0;
|
|
240
|
-
const { result } = await run("cycle.js", (prompt, opts) => {
|
|
241
|
-
const l = opts.label || "";
|
|
242
|
-
if (l.startsWith("review:")) {
|
|
243
|
-
round++;
|
|
244
|
-
return round <= 2 ? { verdict: "REVISE", findings: [finding({ severity: "CRITICAL" })] }
|
|
245
|
-
: { verdict: "SHIP", findings: [] };
|
|
246
|
-
}
|
|
247
|
-
if (l.startsWith("verify:")) return { refuted: false, reason: "holds" };
|
|
248
|
-
return replier(BASE_CYCLE)(prompt, opts);
|
|
249
|
-
}, { feature: "feat-x", maxRounds: 4 });
|
|
250
|
-
check("findings then clean ⇒ SHIP-READY", result.outcome === "SHIP-READY", `got ${result.outcome}`);
|
|
251
|
-
check("findings then clean ⇒ took >1 round", result.rounds > 1, `rounds ${result.rounds}`);
|
|
169
|
+
const { result } = await run("review.js", replier([
|
|
170
|
+
["review:", { verdict: "SHIP", findings: [] }], ...BASE_REVIEW,
|
|
171
|
+
]), JSON.stringify({ feature: "feat-x" }));
|
|
172
|
+
check("review: a JSON-encoded args string is parsed, not used as the id",
|
|
173
|
+
result.verdict === "SHIP", `got ${result.verdict}`);
|
|
252
174
|
}
|
|
253
175
|
{
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
}, { feature: "feat-x", maxRounds: 2 });
|
|
261
|
-
check("cross-check refutes the only CRITICAL ⇒ SHIP-READY",
|
|
262
|
-
result.outcome === "SHIP-READY", `got ${result.outcome}`);
|
|
263
|
-
check("refuted finding ⇒ no fix round", !calls.some(c => c.startsWith("fix:")), calls.join(","));
|
|
176
|
+
let threw = "";
|
|
177
|
+
try {
|
|
178
|
+
await run("review.js", replier([...BASE_REVIEW]), { feature: '{"feature": "feat-x"}' });
|
|
179
|
+
} catch (e) { threw = e.message; }
|
|
180
|
+
check("review: a non-slug feature id throws before anything is written",
|
|
181
|
+
/not a slug/.test(threw), threw || "(did not throw)");
|
|
264
182
|
}
|
|
265
183
|
{
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
184
|
+
let threw = "";
|
|
185
|
+
try {
|
|
186
|
+
await run("review.js", replier([...BASE_REVIEW]), { feature: "../../etc/passwd" });
|
|
187
|
+
} catch (e) { threw = e.message; }
|
|
188
|
+
check("review: a path-shaped feature id is rejected",
|
|
189
|
+
/not a slug/.test(threw), threw || "(did not throw)");
|
|
271
190
|
}
|
|
191
|
+
|
|
192
|
+
// ── Phase 0 profile handling ─────────────────────────────────────────────────
|
|
193
|
+
// A haiku profile-reader intermittently returns the profile as a JSON *string*
|
|
194
|
+
// under a wrapper field instead of at the top level. The old schema accepted that
|
|
195
|
+
// wrapper, so `surfaces` read as undefined ⇒ [] ⇒ parallel([]) ⇒ zero agents
|
|
196
|
+
// dispatched — and because every later guard compares against `surfaces`, an
|
|
197
|
+
// empty list made them all vacuously pass: a run reported a verdict having done
|
|
198
|
+
// nothing, indistinguishable from a clean run with an empty diff. Two properties
|
|
199
|
+
// are pinned per workflow: a wrapped return is recovered, an empty one aborts.
|
|
200
|
+
console.log("profile phase");
|
|
201
|
+
const WRAPPED = { output: JSON.stringify(PROFILE) };
|
|
202
|
+
const EMPTY_PROFILE = { ...PROFILE, surfaces: [] };
|
|
272
203
|
{
|
|
273
|
-
const { result } = await run("
|
|
274
|
-
["profile",
|
|
204
|
+
const { result } = await run("review.js", replier([
|
|
205
|
+
["profile", WRAPPED],
|
|
206
|
+
["review:", { verdict: "SHIP", findings: [] }], ...BASE_REVIEW,
|
|
275
207
|
]));
|
|
276
|
-
check("
|
|
208
|
+
check("review: a string-wrapped profile is unwrapped, not silently empty",
|
|
209
|
+
result.verdict === "SHIP", `got ${result.verdict}`);
|
|
277
210
|
}
|
|
278
211
|
{
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
contract: { enabled: true, path: "packages/shared/src", ext: "ts", mechanism: "shared-types-zod", index: "" },
|
|
285
|
-
};
|
|
286
|
-
const contractFinding = finding({ severity: "CRITICAL", file: "packages/shared/src/feat-x.ts" });
|
|
287
|
-
const { result, calls } = await run("cycle.js", (prompt, opts) => {
|
|
288
|
-
const l = opts.label || "";
|
|
289
|
-
if (l === "profile") return CONTRACT_PROFILE;
|
|
290
|
-
if (l === "contract-fix") return null; // the agent dies
|
|
291
|
-
if (l.startsWith("review:")) return { verdict: "REVISE", findings: [contractFinding] };
|
|
292
|
-
if (l.startsWith("verify:")) return { refuted: false, reason: "holds" };
|
|
293
|
-
return replier(BASE_CYCLE)(prompt, opts);
|
|
294
|
-
// maxRounds ≥ 2: the loop breaks at the cap BEFORE the fix block, so a
|
|
295
|
-
// 1-round run never reaches the contract path at all (a vacuous test).
|
|
296
|
-
}, { feature: "feat-x", maxRounds: 2 });
|
|
297
|
-
check("dead contract agent ⇒ no fabricated contractChanges entry",
|
|
298
|
-
(result.contractChanges || []).length === 0, JSON.stringify(result.contractChanges));
|
|
299
|
-
check("dead contract agent ⇒ a question says the contract is UNCHANGED",
|
|
300
|
-
(result.questions || []).some(q => /contract agent died/.test(q)), JSON.stringify(result.questions));
|
|
301
|
-
check("dead contract agent ⇒ no surface told to realign against it",
|
|
302
|
-
!calls.some(c => c.startsWith("fix:")), calls.join(","));
|
|
212
|
+
const { result, calls } = await run("review.js", replier([["profile", EMPTY_PROFILE], ...BASE_REVIEW]));
|
|
213
|
+
check("review: no surfaces ⇒ ABORTED, not a verdict",
|
|
214
|
+
result.verdict === "ABORTED", `got ${result.verdict}`);
|
|
215
|
+
check("review: no surfaces ⇒ zero reviewers spawned",
|
|
216
|
+
!calls.some(c => c.startsWith("review:")), calls.join(","));
|
|
303
217
|
}
|
|
304
218
|
{
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
310
|
-
...BASE_CYCLE,
|
|
311
|
-
]), { feature: "feat-x", maxRounds: 5 });
|
|
312
|
-
check("red preflight with no owning surface ⇒ stops instead of spinning",
|
|
313
|
-
result.rounds === 1, `burned ${result.rounds} round(s)`);
|
|
314
|
-
check("…and dispatches no fix agent", !calls.some(c => c.startsWith("fix:")), calls.join(","));
|
|
315
|
-
check("…and the question carries the failure tail",
|
|
316
|
-
(result.questions || []).some(q => /no surface owns the failure/.test(q)),
|
|
317
|
-
JSON.stringify(result.questions));
|
|
219
|
+
const { result } = await run("audit.js", replier([
|
|
220
|
+
["profile", EMPTY_PROFILE], ["gates", { failures: [] }], ["write-backlog", "done"],
|
|
221
|
+
]), {});
|
|
222
|
+
check("audit: no surfaces ⇒ error, not an empty backlog",
|
|
223
|
+
/no surfaces/.test(result.error || ""), JSON.stringify(result));
|
|
318
224
|
}
|
|
319
225
|
{
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
const l = opts.label || "";
|
|
326
|
-
if (l === "stage-diff") return { surfaces: [{ key: "tools", diff: "d", files: ["tools/thing.sh"] }] };
|
|
327
|
-
if (l.startsWith("review:")) return { verdict: "REVISE", findings: [orphan] };
|
|
328
|
-
if (l.startsWith("verify:")) return { refuted: false, reason: "holds" };
|
|
329
|
-
return replier(BASE_CYCLE)(prompt, opts);
|
|
330
|
-
}, { feature: "feat-x", maxRounds: 3 });
|
|
331
|
-
check("a finding owned by no surface names the file, not just 'run /fix manually'",
|
|
332
|
-
(result.questions || []).some(q => /never dispatched/.test(q) && /tools\/thing\.sh:9/.test(q)),
|
|
333
|
-
JSON.stringify(result.questions));
|
|
226
|
+
const { result } = await run("refactor.js", replier([
|
|
227
|
+
["profile", EMPTY_PROFILE], ["read-backlog", { domains: [] }],
|
|
228
|
+
]), { domains: "all" });
|
|
229
|
+
check("refactor: no surfaces ⇒ error, not a no-op success",
|
|
230
|
+
/no surfaces/.test(result.error || ""), JSON.stringify(result));
|
|
334
231
|
}
|
|
335
232
|
|
|
336
233
|
// ── the dead-agent family, swept across every terminal/staging agent ─────────
|
|
@@ -356,25 +253,6 @@ console.log("dead-agent sweep");
|
|
|
356
253
|
check("review: dead report-stager ⇒ next says nothing was written",
|
|
357
254
|
/NEVER written/.test(result.next), result.next);
|
|
358
255
|
}
|
|
359
|
-
{
|
|
360
|
-
const { result } = await run("cycle.js", replier([
|
|
361
|
-
["stage-diff", null], ["review:", { verdict: "SHIP", findings: [] }], ...BASE_CYCLE,
|
|
362
|
-
]));
|
|
363
|
-
check("cycle: dead diff-stager ⇒ diagnosed as such, not 'wrong branch'",
|
|
364
|
-
(result.questions || []).some(q => /diff-staging agent died/.test(q)),
|
|
365
|
-
JSON.stringify(result.questions));
|
|
366
|
-
}
|
|
367
|
-
{
|
|
368
|
-
const { result } = await run("cycle.js", replier([
|
|
369
|
-
["close", null], ["review:", { verdict: "SHIP", findings: [] }], ...BASE_CYCLE,
|
|
370
|
-
]));
|
|
371
|
-
check("cycle: dead close agent ⇒ NOT SHIP-READY",
|
|
372
|
-
result.outcome !== "SHIP-READY", `got ${result.outcome}`);
|
|
373
|
-
check("cycle: dead close agent ⇒ a question says nothing was written",
|
|
374
|
-
(result.questions || []).some(q => /NEVER written/.test(q)), JSON.stringify(result.questions));
|
|
375
|
-
check("cycle: dead close agent ⇒ report path not claimed",
|
|
376
|
-
!/^specs\//.test(String(result.report)), result.report);
|
|
377
|
-
}
|
|
378
256
|
{
|
|
379
257
|
const { result } = await run("audit.js", replier([
|
|
380
258
|
["profile", PROFILE], ["gates", { failures: [] }],
|
|
@@ -22,8 +22,8 @@ const frontmatter = (text) => {
|
|
|
22
22
|
// Mechanical commands must pin model: sonnet (otherwise the lead's
|
|
23
23
|
// orchestration turn silently bills at the session model — Opus/Fable).
|
|
24
24
|
// Interactive commands must stay unpinned (they inherit on purpose).
|
|
25
|
-
const PINNED = ["build", "review", "fix", "
|
|
26
|
-
"refactor", "doctor", "align-ds", "update-pipeline"
|
|
25
|
+
const PINNED = ["build", "review", "fix", "ship", "audit",
|
|
26
|
+
"refactor", "doctor", "align-ds", "update-pipeline"];
|
|
27
27
|
const UNPINNED = ["brainstorm", "spec", "init-pipeline"];
|
|
28
28
|
|
|
29
29
|
for (const f of readdirSync(join(root, "core/commands"))) {
|
|
@@ -43,7 +43,7 @@ for (const f of readdirSync(join(root, "core/commands"))) {
|
|
|
43
43
|
// Every non-template agent needs name/tools/model, and must be shipped by
|
|
44
44
|
// both installers (a new agent that install.sh doesn't copy never reaches
|
|
45
45
|
// a global install — the exact bug that motivated this check).
|
|
46
|
-
const AGENT_MODEL = { review: "sonnet", release: "haiku",
|
|
46
|
+
const AGENT_MODEL = { review: "sonnet", release: "haiku",
|
|
47
47
|
"profile-reader": "haiku" };
|
|
48
48
|
const installSh = read("install.sh");
|
|
49
49
|
const installPs1 = read("install.ps1");
|
|
@@ -90,7 +90,7 @@ for (const path of allDocs) {
|
|
|
90
90
|
}
|
|
91
91
|
for (const m of text.matchAll(/subagent_type:\s*(?:`|)([a-z-]+)(?:`|)/g)) {
|
|
92
92
|
const t = m[1];
|
|
93
|
-
if (["review", "release", "
|
|
93
|
+
if (["review", "release", "profile-reader"].includes(t)) continue;
|
|
94
94
|
if (t.startsWith("<")) continue; // <surface.agent> placeholder
|
|
95
95
|
if (!existsSync(join(root, "core/agents", `${t}.md`)))
|
|
96
96
|
fail(path, `dispatches subagent_type ${t} with no core/agents/${t}.md`);
|
|
@@ -115,9 +115,9 @@ if (!existsSync(steps) || readdirSync(steps).length === 0)
|
|
|
115
115
|
|
|
116
116
|
// ── telemetry coverage ──────────────────────────────────────────────────────
|
|
117
117
|
// The funnel is only readable if every one of its stages pings — a single missing
|
|
118
|
-
// one silently truncates it (that is how /
|
|
118
|
+
// one silently truncates it (that is how /review and /fix went unreported
|
|
119
119
|
// until 1.2.3). The phase list here must match SCHEMA.md §Telemetry's table.
|
|
120
|
-
const FUNNEL = ["brainstorm", "spec", "build", "
|
|
120
|
+
const FUNNEL = ["brainstorm", "spec", "build", "review", "fix", "ship"];
|
|
121
121
|
for (const c of FUNNEL)
|
|
122
122
|
if (!/usage ping/i.test(read(`core/commands/${c}.md`)))
|
|
123
123
|
fail(`core/commands/${c}.md`, "funnel command with no usage ping — breaks the telemetry funnel");
|
|
@@ -188,8 +188,9 @@ if (!installPs1.includes("core\\workflows"))
|
|
|
188
188
|
fail("install.ps1", "does not copy core\\workflows (Copy-Core)");
|
|
189
189
|
|
|
190
190
|
// A new workflow script must also be KNOWN to the things that check for it, or it
|
|
191
|
-
// ships and nothing notices when an installer stops copying it.
|
|
192
|
-
//
|
|
191
|
+
// ships and nothing notices when an installer stops copying it. This check exists
|
|
192
|
+
// because a workflow once shipped while three call sites still named only the
|
|
193
|
+
// three that preceded it.
|
|
193
194
|
const workflowNames = existsSync(workflowsDir)
|
|
194
195
|
? readdirSync(workflowsDir).filter((f) => f.endsWith(".js"))
|
|
195
196
|
: [];
|
|
@@ -204,7 +205,7 @@ for (const f of workflowNames) {
|
|
|
204
205
|
|
|
205
206
|
// ── dashboard: the metrics phase list is duplicated server/client ────────────
|
|
206
207
|
// A phase present in one and not the other parses fine and renders in no column —
|
|
207
|
-
// silently invisible data, which is how
|
|
208
|
+
// silently invisible data, which is how a phase batch once went unnoticed.
|
|
208
209
|
const phaseList = (text, file) => {
|
|
209
210
|
const m = text.match(/const PHASES = \[([^\]]*)\]/);
|
|
210
211
|
if (!m) { fail(file, "no `const PHASES = [...]` found"); return null; }
|
package/core/agents/smoke.md
DELETED
|
@@ -1,63 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
name: smoke
|
|
3
|
-
description: Executes the end-to-end smoke run for one feature in its worktree — infra up, migrations, contract endpoints, key UI flows, visual check vs design — then stages the SMOKE REPORT. Dispatched by /smoke. Observes honestly, never fixes anything.
|
|
4
|
-
tools: Read, Write, Grep, Glob, Bash, DesignSync
|
|
5
|
-
model: sonnet
|
|
6
|
-
---
|
|
7
|
-
|
|
8
|
-
You are the **smoke** agent for one feature. You actually run the built feature — `/review` audits
|
|
9
|
-
code read-only; nobody has executed it yet. You verify it *works*; you never fix it (failures go
|
|
10
|
-
through `/fix`). Observe honestly: report what happened, not what should have happened.
|
|
11
|
-
|
|
12
|
-
> **First action, always:** read `PIPELINE.md` §`pipeline-profile`: `commands` (migrate/dev),
|
|
13
|
-
> `isolation` (worktree, slot ports, db), `contract`, `design`, `surfaces` — then the spec
|
|
14
|
-
> `specs/<id>.md` (§5 contract, §8 flows, §9 acceptance).
|
|
15
|
-
|
|
16
|
-
## Your inputs (supplied at dispatch — you have no memory)
|
|
17
|
-
|
|
18
|
-
1. The feature id and spec path `specs/<id>.md`.
|
|
19
|
-
2. The contract path `<contract.path>/<id>.<ext>`.
|
|
20
|
-
3. The checkout to work in: the worktree path + slot ports/db, or the main checkout on the feature branch.
|
|
21
|
-
|
|
22
|
-
## Keep your own context lean
|
|
23
|
-
|
|
24
|
-
Redirect every bulky output to a file and inspect it with `grep`/`jq` — never print full curl bodies,
|
|
25
|
-
server logs, or poll loops into your transcript. `curl -s … -o /tmp/resp.json -w '%{http_code}'` then
|
|
26
|
-
assert on the pieces you need.
|
|
27
|
-
|
|
28
|
-
## 1. Bring the feature up
|
|
29
|
-
|
|
30
|
-
- Work in the checkout your dispatch names. Infra as needed: the compose stack if one is declared
|
|
31
|
-
(the gate will ask — that's expected), then `commands.migrate`, then `commands.dev` **in the
|
|
32
|
-
background**. Wait for ready (poll the ports), don't assume.
|
|
33
|
-
|
|
34
|
-
## 2. Exercise the contract (the real server, not the tests)
|
|
35
|
-
|
|
36
|
-
- Hit a representative set of spec §5 endpoints with `curl`: every route domain, every auth level,
|
|
37
|
-
at least one error case per class (validation `422`, unauthenticated `401`, wrong-role `403`,
|
|
38
|
-
conflict `409`). Compare status + response envelope against the contract.
|
|
39
|
-
- If `rbac.enabled`: verify at least one denial per role boundary the spec declares.
|
|
40
|
-
- A mismatch is a FAIL entry with the exact command, expected, and actual — precise enough for a
|
|
41
|
-
stateless `/fix` agent.
|
|
42
|
-
|
|
43
|
-
## 3. Exercise the UI (only if a touched surface has `uses_design`)
|
|
44
|
-
|
|
45
|
-
- Drive the spec §8 flows against the running app, **mobile viewport first** (375px), then desktop.
|
|
46
|
-
- If a browser/screenshot tool is available (a project driver, playwright, an agent browser), capture
|
|
47
|
-
each §8 screen and compare against the feature's design pages: each `design_files` entry is a full
|
|
48
|
-
`https://claude.ai/design/p/<projectId>?file=<file>` link — extract its `<projectId>` (the `/p/…`
|
|
49
|
-
segment) + `<file>` (the `?file=` query) and fetch read-only via `DesignSync get_file(<projectId>,
|
|
50
|
-
<file>)`. Compare layout, states (empty/loading/error/suppressed…), copy language. Note deviations.
|
|
51
|
-
- No browser tooling available ⇒ **say so and skip the visual diff** — never claim a visual check
|
|
52
|
-
you didn't perform.
|
|
53
|
-
|
|
54
|
-
## 4. Stage the SMOKE REPORT, tear down, return
|
|
55
|
-
|
|
56
|
-
- One line per check: ✅/❌ · what was exercised · (on ❌) command → expected vs actual.
|
|
57
|
-
- **Write the full report to `specs/reports/<id>.md`** (overwrite) — the same gitignored buffer
|
|
58
|
-
`/review` uses, so a `/fix` after a `/clear` still has the failures.
|
|
59
|
-
- Tear down what you started (kill the dev server); leave shared infra as you found it.
|
|
60
|
-
- **Your return to the lead is ONLY:** the verdict line (`PASS` / `FAIL:<n>`), **at most 10 ❌
|
|
61
|
-
lines** — one line each (`❌ <flow/endpoint> · expected <x> got <y>`), no command output, no code
|
|
62
|
-
or body excerpts; more than 10 ⇒ keep the 10 most severe and add `+<n> more — see the report` —
|
|
63
|
-
and `Full report: specs/reports/<id>.md`. No logs, no bodies, no screenshots.
|
package/core/commands/cycle.md
DELETED
|
@@ -1,61 +0,0 @@
|
|
|
1
|
-
---
|
|
2
|
-
model: sonnet
|
|
3
|
-
description: Launch the full dev-cycle workflow (contract → build → review → fix, until zero findings; add `smoke` to also run the app each round) for a frozen spec; relay its verdict + deferred questions.
|
|
4
|
-
argument-hint: <feature_id> [max_rounds] [smoke]
|
|
5
|
-
---
|
|
6
|
-
|
|
7
|
-
You are the **lead**. Launch the full dev-cycle **workflow** for feature **$ARGUMENTS** — the
|
|
8
|
-
deterministic script does the orchestration (SCHEMA.md §Workflows, `cycle.js`); your job is only to
|
|
9
|
-
start it and relay its result. Do NOT run the phases yourself here — that's the conversational path
|
|
10
|
-
(`/build` → `/smoke` → `/review` → `/fix`), which remains the fallback below.
|
|
11
|
-
|
|
12
|
-
> **Kanban** (SCHEMA.md §Kanban): move card `#<feature_id>` → **Building** at launch. No-op silently
|
|
13
|
-
> if no board.
|
|
14
|
-
|
|
15
|
-
## 1. Resolve & check (fail fast, before spending anything)
|
|
16
|
-
|
|
17
|
-
- Parse `$ARGUMENTS`: the first token is `<feature_id>`, an optional numeric token is
|
|
18
|
-
`<max_rounds>` (the workflow defaults to 5), and an optional literal token `smoke` opts the
|
|
19
|
-
smoke agent INTO the verify rounds (default: review-only — smoke is expensive, it boots the
|
|
20
|
-
app every round; skipping it is the human's explicit risk, `/smoke <feature_id>` stays
|
|
21
|
-
available standalone before `/ship`).
|
|
22
|
-
- Resolve the script: `.claude/workflows/cycle.js` if it exists, else `~/.claude/workflows/cycle.js`
|
|
23
|
-
(`test -f`). **Missing both** ⇒ the core predates 1.3.0 or is half-copied: tell the human to run
|
|
24
|
-
`/update-pipeline`, and stop.
|
|
25
|
-
- **Workflow runtime available?** If the `Workflow` tool is not in your toolset (Claude Code
|
|
26
|
-
< 2.1.154 or workflows disabled), say so and hand over the conversational path instead:
|
|
27
|
-
`/build <feature_id>` → `/smoke` → `/review` → `/fix` — same phases, interactive. Stop.
|
|
28
|
-
- Quick spec sanity (the workflow re-checks properly — this just saves a doomed launch):
|
|
29
|
-
`grep '^status:' specs/<feature_id>.md` must say `frozen` or `in-review`; otherwise tell the human
|
|
30
|
-
to run `/spec` first, and stop.
|
|
31
|
-
|
|
32
|
-
## 2. Launch
|
|
33
|
-
|
|
34
|
-
Call the `Workflow` tool: `scriptPath: <resolved cycle.js path>`,
|
|
35
|
-
`args: {"feature": "<feature_id>", "maxRounds": <max_rounds, omit if not given>, "smoke": <true
|
|
36
|
-
only if the smoke token was given — omit otherwise>}`.
|
|
37
|
-
It runs in the background — tell the human it's off and what it will do (build, then review → fix
|
|
38
|
-
rounds until zero findings, with smoke alongside each round if opted in; no questions mid-run), and
|
|
39
|
-
that `/workflows` shows live progress. Then END YOUR TURN — never poll, never sleep; the completion
|
|
40
|
-
notification re-wakes you.
|
|
41
|
-
|
|
42
|
-
## 3. Relay the result (when the task notification arrives)
|
|
43
|
-
|
|
44
|
-
The workflow returns only a verdict object — the bulk is already on disk
|
|
45
|
-
(`specs/reports/<feature_id>.md`, spec `## Remediation`). Print, without re-reading any of it into
|
|
46
|
-
context:
|
|
47
|
-
|
|
48
|
-
- `outcome` · rounds used · review verdict · smoke result.
|
|
49
|
-
- `contractChanges` if any — flag them explicitly: the loop re-authored the frozen contract
|
|
50
|
-
lead-style; the human should eyeball those hunks in the diff.
|
|
51
|
-
- `unreviewedSurfaces` if non-empty — **say it plainly**: a reviewer died, so those surfaces
|
|
52
|
-
carry NO verdict. The run's verdict covers the others only; the code there was never read.
|
|
53
|
-
- **The `questions` array, verbatim** — this is the human's inbox from the run (empty when the spec
|
|
54
|
-
pre-answered everything). Each one is a decision to make, usually by sharpening the spec.
|
|
55
|
-
- The `next` line: **SHIP-READY** ⇒ `/ship <feature_id>` (DoD ticked + freshness stamped — ship is a
|
|
56
|
-
straight shot, its human confirmation stays). **STOPPED** ⇒ answer the questions, then rerun
|
|
57
|
-
`/cycle <feature_id>` (it picks up from the spec's Remediation) or finish conversationally with
|
|
58
|
-
`/fix <feature_id>` + `/review <feature_id>`.
|
|
59
|
-
- **Kanban:** outcome SHIP-READY ⇒ move card → **Review** (the cycle's last verdict is a review);
|
|
60
|
-
otherwise → **Fix**. No-op silently if no board.
|
|
61
|
-
- **Recommend a `/clear`** — everything the next command needs is on disk.
|