cohorte 1.4.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +151 -0
- package/README.md +43 -12
- package/bin/cli.js +13 -2
- package/core/agents/review.md +23 -0
- package/core/commands/audit.md +9 -1
- package/core/commands/brainstorm.md +6 -0
- package/core/commands/build.md +93 -8
- package/core/commands/doctor.md +13 -8
- package/core/commands/drive.md +80 -0
- package/core/commands/fix.md +10 -5
- package/core/commands/review.md +94 -12
- package/core/commands/spec.md +20 -0
- package/core/commands/update-pipeline.md +6 -1
- package/core/hooks/gate.py +4 -4
- package/core/templates/decisions.template.md +42 -0
- package/core/templates/spec.template.md +4 -2
- package/core/templates/steps/init-pipeline/02-interview-gaps.md +1 -1
- package/core/templates/steps/init-pipeline/04-write-render.md +8 -4
- package/core/workflows/review.js +44 -2
- package/dashboard/dist/assets/{index-AFQnlfjO.css → index-BZ_LQlEj.css} +1 -1
- package/dashboard/dist/assets/index-DYyn4p93.js +43 -0
- package/dashboard/dist/index.html +2 -2
- package/dashboard/server/doctor.js +13 -3
- package/dashboard/server/index.js +7 -0
- package/dashboard/server/metrics.js +4 -4
- package/dashboard/server/usage.js +61 -0
- package/install.ps1 +12 -1
- package/install.sh +12 -2
- package/package.json +1 -1
- package/profile/PIPELINE.template.md +3 -3
- package/profile/SCHEMA.md +150 -12
- package/scripts/loop.sh +318 -0
- package/scripts/metrics/collect.mjs +11 -2
- package/scripts/preflight.sh +2 -2
- package/scripts/telemetry-send.sh +5 -2
- package/scripts/test-dashboard.mjs +22 -2
- package/scripts/test-gate.mjs +1 -2
- package/scripts/test-loop.mjs +227 -0
- package/scripts/test-metrics.mjs +12 -3
- package/scripts/test-workflows.mjs +28 -0
- package/scripts/validate-core.mjs +21 -6
- package/core/agents/smoke.md +0 -63
- package/core/commands/smoke.md +0 -55
- package/dashboard/dist/assets/index-DLBzciIC.js +0 -43
package/scripts/loop.sh
ADDED
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
#
|
|
3
|
+
# loop.sh — autonomous /build → /review → /fix → /review … loop for ONE feature.
|
|
4
|
+
#
|
|
5
|
+
# loop.sh <feature-id> [--max=N] [--no-build] [--rebuild] [--resume]
|
|
6
|
+
#
|
|
7
|
+
# THE POINT: every phase runs as a SEPARATE `claude -p` child with its own fresh
|
|
8
|
+
# context. The session that typed /drive never sees the diff, the N review reports
|
|
9
|
+
# or the N contracts — it reads only this script's one-line-per-phase stdout and,
|
|
10
|
+
# at the end, the verdict JSON. Running the loop inside the calling session would
|
|
11
|
+
# accumulate all of it in a history that is re-sent at input price on every turn,
|
|
12
|
+
# which is the exact cost the pipeline's /clear discipline exists to avoid.
|
|
13
|
+
#
|
|
14
|
+
# Contract with the pipeline: /review writes specs/reports/<id>.verdict.json on
|
|
15
|
+
# every run, and /build writes <id>.readiness.json + <id>.build.json. Those three
|
|
16
|
+
# files — `blocking`, `fingerprint`, `unreviewed`, `verdict`, `dead` — are the ONLY
|
|
17
|
+
# channel between cohorte and this driver. No prose is parsed.
|
|
18
|
+
#
|
|
19
|
+
# Two of those fields exist for the same reason: a subagent that DIES returns
|
|
20
|
+
# nothing, and nothing is byte-identical to "clean". A dead implementer means a
|
|
21
|
+
# surface was never built; a dead reviewer means a surface was never audited, and
|
|
22
|
+
# `blocking == 0` would then certify code no one read. Both abort as exit 2.
|
|
23
|
+
#
|
|
24
|
+
# Exit codes (distinct diagnostics, do not collapse them):
|
|
25
|
+
# 0 clean — a review returned blocking == 0
|
|
26
|
+
# 1 ceiling — --max passes used, still blocking (the fix was progressing;
|
|
27
|
+
# re-run with a higher --max)
|
|
28
|
+
# 2 no usable verdict — /review produced nothing, or aborted on a red
|
|
29
|
+
# preflight (typecheck/lint/tests broken; the message says which)
|
|
30
|
+
# 3 non-convergent — two consecutive reviews returned the SAME blocking
|
|
31
|
+
# fingerprint: the fix is treading water, a higher --max will not help
|
|
32
|
+
# 4 not implementable — /build's readiness gate returned NOT-READY and spawned
|
|
33
|
+
# no agent: the frozen spec cannot be built (missing contract shape, unowned
|
|
34
|
+
# area, absent dependency). Needs /spec, not more passes.
|
|
35
|
+
# 64 usage — bad flag, bad id, missing spec, no `claude` on PATH
|
|
36
|
+
#
|
|
37
|
+
# No /fix runs on the last pass: fixing without a review behind it ships
|
|
38
|
+
# unaudited code. Each fix pass is committed — that commit is the only way back
|
|
39
|
+
# after N autonomous passes.
|
|
40
|
+
#
|
|
41
|
+
# RESUME: the spec's front-matter IS the loop's state (SCHEMA.md §Spec status).
|
|
42
|
+
# Before every phase this script stamps `status: in-progress` + `loop_phase` +
|
|
43
|
+
# `loop_pass` into specs/<id>.md — deterministically, with awk, costing no tokens
|
|
44
|
+
# — and on exit stamps a terminal status (`in-review` clean, `blocked` otherwise).
|
|
45
|
+
# `--resume` reads `loop_pass` back and continues from that pass instead of 1, so
|
|
46
|
+
# a session that died at pass 3 of 5 does not re-pay passes 1 and 2. The build is
|
|
47
|
+
# skipped or redone by the same stamp logic as always (the stamp is only written
|
|
48
|
+
# on a build that finished), so an interrupted build still rebuilds.
|
|
49
|
+
|
|
50
|
+
set -uo pipefail
|
|
51
|
+
|
|
52
|
+
usage() {
|
|
53
|
+
cat >&2 <<'EOF'
|
|
54
|
+
usage: loop.sh <feature-id> [--max=N] [--no-build] [--rebuild] [--resume]
|
|
55
|
+
|
|
56
|
+
--max=N stop after N review passes (default 5) — a ceiling on the TOTAL
|
|
57
|
+
pass count, so it still means "5 passes" when resuming at pass 3
|
|
58
|
+
--no-build never build — re-run the /review ⇄ /fix loop on a feature that
|
|
59
|
+
is already built (the common case; the build stamp is ignored)
|
|
60
|
+
--rebuild force a /build even if the stamp says it was already built
|
|
61
|
+
--resume continue from the pass recorded in the spec's front-matter
|
|
62
|
+
(loop_pass), instead of starting over at pass 1
|
|
63
|
+
|
|
64
|
+
env CLAUDE_FLAGS flags for every child session
|
|
65
|
+
(default: --permission-mode acceptEdits)
|
|
66
|
+
EOF
|
|
67
|
+
exit 64
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
id=""
|
|
71
|
+
max=5
|
|
72
|
+
build_mode="auto" # auto | never | force
|
|
73
|
+
resume=0
|
|
74
|
+
|
|
75
|
+
for arg in "$@"; do
|
|
76
|
+
case "$arg" in
|
|
77
|
+
--max=*)
|
|
78
|
+
max="${arg#--max=}"
|
|
79
|
+
case "$max" in
|
|
80
|
+
''|*[!0-9]*) echo "loop: --max must be a positive integer (got '${arg#--max=}')" >&2; exit 64 ;;
|
|
81
|
+
esac
|
|
82
|
+
[ "$max" -ge 1 ] || { echo "loop: --max must be >= 1" >&2; exit 64; }
|
|
83
|
+
;;
|
|
84
|
+
--no-build) build_mode="never" ;;
|
|
85
|
+
--rebuild) build_mode="force" ;;
|
|
86
|
+
--resume) resume=1 ;;
|
|
87
|
+
-h|--help) usage ;;
|
|
88
|
+
-*) echo "loop: unknown flag: $arg" >&2; usage ;;
|
|
89
|
+
*)
|
|
90
|
+
[ -z "$id" ] || { echo "loop: unexpected argument: $arg" >&2; usage; }
|
|
91
|
+
id="$arg"
|
|
92
|
+
;;
|
|
93
|
+
esac
|
|
94
|
+
done
|
|
95
|
+
|
|
96
|
+
[ -n "$id" ] || usage
|
|
97
|
+
# --no-build --rebuild together is a contradiction, not a precedence puzzle.
|
|
98
|
+
case " $* " in
|
|
99
|
+
*" --no-build "*) case " $* " in *" --rebuild "*)
|
|
100
|
+
echo "loop: --no-build and --rebuild are mutually exclusive" >&2; exit 64 ;; esac ;;
|
|
101
|
+
esac
|
|
102
|
+
|
|
103
|
+
command -v claude >/dev/null 2>&1 || {
|
|
104
|
+
echo "loop: no 'claude' on PATH — the loop drives child claude -p sessions" >&2
|
|
105
|
+
exit 64
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
root="$(git rev-parse --show-toplevel 2>/dev/null)" || {
|
|
109
|
+
echo "loop: not inside a git checkout" >&2; exit 64; }
|
|
110
|
+
cd "$root" || exit 64
|
|
111
|
+
|
|
112
|
+
spec="specs/$id.md"
|
|
113
|
+
[ -f "$spec" ] || {
|
|
114
|
+
echo "loop: no spec at $spec — run /spec $id first" >&2; exit 64; }
|
|
115
|
+
|
|
116
|
+
reports="specs/reports"
|
|
117
|
+
mkdir -p "$reports"
|
|
118
|
+
verdict="$reports/$id.verdict.json"
|
|
119
|
+
readiness="$reports/$id.readiness.json"
|
|
120
|
+
buildjson="$reports/$id.build.json"
|
|
121
|
+
stamp="$reports/$id.built"
|
|
122
|
+
log="$reports/$id.loop.log"
|
|
123
|
+
|
|
124
|
+
# --- the spec front-matter as loop state -------------------------------------
|
|
125
|
+
# Best-effort by design: a spec with no front-matter (or an unwritable one) makes
|
|
126
|
+
# every fm_* call a silent no-op. This is bookkeeping for resume + the dashboard,
|
|
127
|
+
# never a precondition — the loop must not die over a status line.
|
|
128
|
+
fm_get() { # fm_get <key> → value, or empty
|
|
129
|
+
[ -f "$spec" ] || return 0
|
|
130
|
+
awk -v k="$1" '
|
|
131
|
+
NR==1 && $0=="---" { fm=1; next }
|
|
132
|
+
fm==1 && $0=="---" { exit }
|
|
133
|
+
fm==1 && $0 ~ "^"k":" {
|
|
134
|
+
sub("^"k":[[:space:]]*", ""); sub("#.*", "")
|
|
135
|
+
gsub(/^[[:space:]]+|[[:space:]]+$/, ""); print; exit
|
|
136
|
+
}
|
|
137
|
+
' "$spec"
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
fm_set() { # fm_set <key> <value> (replace, else append)
|
|
141
|
+
[ -f "$spec" ] || return 0
|
|
142
|
+
awk -v k="$1" -v v="$2" '
|
|
143
|
+
NR==1 && $0!="---" { nofm=1 }
|
|
144
|
+
nofm { print; next }
|
|
145
|
+
NR==1 { fm=1; print; next }
|
|
146
|
+
fm==1 && $0=="---" {
|
|
147
|
+
if (!done) print k ": " v # key absent: add it before the closing ---
|
|
148
|
+
fm=2; print; next
|
|
149
|
+
}
|
|
150
|
+
fm==1 && $0 ~ "^"k":" {
|
|
151
|
+
if (done) next # a duplicate key: drop it
|
|
152
|
+
c=""; i=index($0, "#"); if (i>0) c=" " substr($0, i) # keep a trailing comment
|
|
153
|
+
print k ": " v c; done=1; next
|
|
154
|
+
}
|
|
155
|
+
{ print }
|
|
156
|
+
' "$spec" >"$spec.loop.tmp" 2>/dev/null &&
|
|
157
|
+
mv "$spec.loop.tmp" "$spec" 2>/dev/null || rm -f "$spec.loop.tmp"
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
: "${CLAUDE_FLAGS:=--permission-mode acceptEdits}"
|
|
161
|
+
|
|
162
|
+
: >"$log"
|
|
163
|
+
{
|
|
164
|
+
printf '# loop %s — max=%s build=%s\n' "$id" "$max" "$build_mode"
|
|
165
|
+
printf '# flags: %s\n' "$CLAUDE_FLAGS"
|
|
166
|
+
} >>"$log"
|
|
167
|
+
|
|
168
|
+
# --- one phase = one throwaway child session ---------------------------------
|
|
169
|
+
# ALL child output is redirected into $log and never surfaces here: if the
|
|
170
|
+
# parent re-imports the children's transcripts, the whole point is lost.
|
|
171
|
+
# $CLAUDE_FLAGS is intentionally unquoted — it is a flag list, not one word.
|
|
172
|
+
run_phase() {
|
|
173
|
+
cmd="$1"
|
|
174
|
+
# Stamp the state BEFORE the phase runs: if this child dies (or the whole
|
|
175
|
+
# session does), the spec already says where the loop was — that is what
|
|
176
|
+
# --resume reads back. Child commands write `status` themselves (/fix sets
|
|
177
|
+
# in-review); re-stamping here each phase is what keeps `in-progress` true.
|
|
178
|
+
fm_set status in-progress
|
|
179
|
+
fm_set loop_pass "$pass"
|
|
180
|
+
fm_set loop_phase "$cmd"
|
|
181
|
+
printf '▶ /%-6s %-24s ' "$cmd" "$id"
|
|
182
|
+
printf '\n\n===== /%s %s =====\n' "$cmd" "$id" >>"$log"
|
|
183
|
+
# shellcheck disable=SC2086
|
|
184
|
+
if claude -p "/$cmd $id" $CLAUDE_FLAGS >>"$log" 2>&1; then
|
|
185
|
+
echo "ok"
|
|
186
|
+
return 0
|
|
187
|
+
fi
|
|
188
|
+
echo "fail"
|
|
189
|
+
return 1
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
# Scalar reads on a flat JSON object — no jq dependency (the pipeline ships no
|
|
193
|
+
# runtime deps). Only `blocking` and `fingerprint` are ever read; both are
|
|
194
|
+
# top-level scalars by construction of the verdict contract.
|
|
195
|
+
json_num() { sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*\([0-9][0-9]*\).*/\1/p' "$1" | head -n1; }
|
|
196
|
+
json_str() { sed -n 's/.*"'"$2"'"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p' "$1" | head -n1; }
|
|
197
|
+
|
|
198
|
+
# Terminal status goes into the spec, not just into this stdout: a clean run
|
|
199
|
+
# leaves the feature ready to /ship, any failure leaves it visibly `blocked` for
|
|
200
|
+
# the human and for the dashboard. Exit 64 never reaches here (usage dies earlier),
|
|
201
|
+
# so every code handled below is a real run outcome.
|
|
202
|
+
finish() {
|
|
203
|
+
if [ "$1" -eq 0 ]; then
|
|
204
|
+
fm_set status in-review
|
|
205
|
+
fm_set loop_pass 0
|
|
206
|
+
fm_set loop_phase done
|
|
207
|
+
else
|
|
208
|
+
fm_set status blocked
|
|
209
|
+
fi
|
|
210
|
+
echo "$2"
|
|
211
|
+
exit "$1"
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
# One short clause naming the deferred findings, appended to a closing line.
|
|
215
|
+
# They are NOT blocking (they live in the backlog, not in ## Remediation), so
|
|
216
|
+
# they never change an exit code — but a loop that silently drops them is the
|
|
217
|
+
# leak /review §3.5 exists to close, so the driver names them.
|
|
218
|
+
def_note() {
|
|
219
|
+
d="$(json_num "$verdict" deferred 2>/dev/null)"
|
|
220
|
+
case "$d" in ''|0) return 0 ;; esac
|
|
221
|
+
printf ' · %s deferred finding(s) parked in specs/refactor-backlog.md' "$d"
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
# --- build -------------------------------------------------------------------
|
|
225
|
+
# The stamp is the driver's own bookkeeping — /build knows nothing about it.
|
|
226
|
+
case "$build_mode" in
|
|
227
|
+
force) do_build=1 ;;
|
|
228
|
+
never) do_build=0 ;;
|
|
229
|
+
auto) [ -f "$stamp" ] && do_build=0 || do_build=1 ;;
|
|
230
|
+
esac
|
|
231
|
+
|
|
232
|
+
# --resume: continue from the pass the spec records, not from 1. A missing or
|
|
233
|
+
# junk value falls back to 1 — resuming must never be less safe than starting.
|
|
234
|
+
pass=1
|
|
235
|
+
if [ "$resume" -eq 1 ]; then
|
|
236
|
+
rp="$(fm_get loop_pass)"
|
|
237
|
+
case "$rp" in ''|*[!0-9]*|0) rp=1 ;; esac
|
|
238
|
+
[ "$rp" -le "$max" ] || {
|
|
239
|
+
echo "loop: --resume says pass $rp but --max=$max — raise --max to continue" >&2; exit 64; }
|
|
240
|
+
pass="$rp"
|
|
241
|
+
[ "$pass" -eq 1 ] || printf '↻ resuming at review pass %s (from %s)\n' "$pass" "$spec"
|
|
242
|
+
fi
|
|
243
|
+
|
|
244
|
+
if [ "$do_build" -eq 1 ]; then
|
|
245
|
+
# Delete first: a NOT-READY left by a previous build would abort this one on
|
|
246
|
+
# someone else's verdict (and a stale READY would hide a gate that never ran).
|
|
247
|
+
rm -f "$readiness" "$buildjson"
|
|
248
|
+
build_ok=0
|
|
249
|
+
run_phase build && build_ok=1
|
|
250
|
+
# The readiness gate is checked BEFORE the child's exit status: /build aborting
|
|
251
|
+
# on NOT-READY is a cleaner diagnosis than "/build failed", and it is the one
|
|
252
|
+
# outcome that more passes cannot fix.
|
|
253
|
+
if [ -f "$readiness" ] &&
|
|
254
|
+
grep -q '"verdict"[[:space:]]*:[[:space:]]*"NOT-READY"' "$readiness"; then
|
|
255
|
+
finish 4 "✗ spec not implementable — /build's readiness gate returned NOT-READY and spawned no agent; see $readiness, then /spec $id"
|
|
256
|
+
fi
|
|
257
|
+
# A dead implementer returns nothing, so /build can finish "successfully" having
|
|
258
|
+
# built one surface of two. Reviewing that would spend N reviewers auditing a
|
|
259
|
+
# half-built feature and report its gaps as findings to fix — the wrong diagnosis
|
|
260
|
+
# at the wrong price. `dead` is a non-empty array only when a surface died twice.
|
|
261
|
+
if [ -f "$buildjson" ] && grep -q '"dead"[[:space:]]*:[[:space:]]*\[[^]]' "$buildjson"; then
|
|
262
|
+
finish 2 "✗ an implementer died — the surface(s) in \"dead\" were never built; see $buildjson and $log"
|
|
263
|
+
fi
|
|
264
|
+
[ "$build_ok" -eq 1 ] || finish 2 "✗ /build failed — see $log"
|
|
265
|
+
date -u +%Y-%m-%dT%H:%M:%SZ >"$stamp"
|
|
266
|
+
fi
|
|
267
|
+
|
|
268
|
+
# --- review ⇄ fix ------------------------------------------------------------
|
|
269
|
+
prev_fp=""
|
|
270
|
+
while [ "$pass" -le "$max" ]; do
|
|
271
|
+
# Delete first: a stale verdict from the previous pass read as this pass's
|
|
272
|
+
# answer would end the loop on someone else's numbers.
|
|
273
|
+
rm -f "$verdict"
|
|
274
|
+
run_phase review || true # exit status of the child is not the verdict
|
|
275
|
+
|
|
276
|
+
[ -f "$verdict" ] || finish 2 \
|
|
277
|
+
"✗ /review wrote no verdict (pass $pass) — see $log"
|
|
278
|
+
|
|
279
|
+
if grep -q '"aborted"' "$verdict"; then
|
|
280
|
+
finish 2 "✗ /review aborted on a red preflight — typecheck/lint/tests are broken, see $reports/$id.preflight.txt"
|
|
281
|
+
fi
|
|
282
|
+
|
|
283
|
+
# A reviewer that died twice leaves its surface unaudited, and `blocking` counts only
|
|
284
|
+
# what the SURVIVING reviewers found — so blocking == 0 here would mean "clean" about
|
|
285
|
+
# code nobody read. Checked BEFORE blocking, because it invalidates it.
|
|
286
|
+
if grep -q '"unreviewed"[[:space:]]*:[[:space:]]*\[[^]]' "$verdict"; then
|
|
287
|
+
finish 2 "✗ a reviewer died — the surface(s) in \"unreviewed\" carry no verdict (pass $pass); see $verdict"
|
|
288
|
+
fi
|
|
289
|
+
|
|
290
|
+
blocking="$(json_num "$verdict" blocking)"
|
|
291
|
+
[ -n "$blocking" ] || finish 2 \
|
|
292
|
+
"✗ verdict has no usable 'blocking' count (pass $pass) — see $verdict"
|
|
293
|
+
|
|
294
|
+
[ "$blocking" -eq 0 ] && finish 0 \
|
|
295
|
+
"✓ clean after $pass review pass(es) — no blocking findings$(def_note)"
|
|
296
|
+
|
|
297
|
+
fp="$(json_str "$verdict" fingerprint)"
|
|
298
|
+
if [ -n "$fp" ] && [ "$fp" = "$prev_fp" ]; then
|
|
299
|
+
finish 3 "✗ non-convergent — the same $blocking blocking finding(s) survived a fix pass; see $verdict"
|
|
300
|
+
fi
|
|
301
|
+
prev_fp="$fp"
|
|
302
|
+
|
|
303
|
+
# Last pass: report and stop. A /fix here would leave unreviewed code behind.
|
|
304
|
+
[ "$pass" -eq "$max" ] && finish 1 \
|
|
305
|
+
"✗ ceiling — $blocking blocking finding(s) after $max pass(es); re-run with a higher --max --resume$(def_note)"
|
|
306
|
+
|
|
307
|
+
run_phase fix || true
|
|
308
|
+
|
|
309
|
+
# Non-fatal by design: nothing to commit is a legitimate outcome (an agent
|
|
310
|
+
# that decided a finding needed no code change). The commit itself is the
|
|
311
|
+
# rollback point for the pass that just ran.
|
|
312
|
+
git add -A >>"$log" 2>&1
|
|
313
|
+
git commit -m "loop($id): fix pass $pass" >>"$log" 2>&1 || true
|
|
314
|
+
|
|
315
|
+
pass=$((pass + 1))
|
|
316
|
+
done
|
|
317
|
+
|
|
318
|
+
finish 1 "✗ ceiling — $max pass(es) exhausted"
|
|
@@ -60,21 +60,30 @@ function knownCommands() {
|
|
|
60
60
|
}
|
|
61
61
|
// Fallback for a collector run outside the package (e.g. copied into a repo on its own).
|
|
62
62
|
if (!names.size) {
|
|
63
|
-
for (const n of ['brainstorm', 'spec', 'build', '
|
|
63
|
+
for (const n of ['brainstorm', 'spec', 'build', 'review', 'fix', 'ship',
|
|
64
64
|
'audit', 'refactor', 'align-ds', 'doctor', 'init-pipeline', 'update-pipeline']) names.add(n);
|
|
65
65
|
}
|
|
66
66
|
// Retired commands. The list above is read from the shipped core, so a command that is
|
|
67
67
|
// removed stops being recognised — and every run of it already in the transcripts silently
|
|
68
68
|
// reclassifies as (chat), rewriting history and inflating the catch-all bucket. Keep the
|
|
69
69
|
// names here so past runs stay attributed to what actually ran.
|
|
70
|
-
for (const n of ['cycle']) names.add(n);
|
|
70
|
+
for (const n of ['cycle', 'smoke']) names.add(n);
|
|
71
71
|
return names;
|
|
72
72
|
}
|
|
73
73
|
const COMMANDS = knownCommands();
|
|
74
74
|
|
|
75
|
+
// An invocation is a short instruction that is mostly the command ("move on branding-ramp
|
|
76
|
+
// and /review"). A long prompt that happens to name one is someone TALKING ABOUT the
|
|
77
|
+
// command — a bug report, a design discussion, a pasted transcript. Counting those as runs
|
|
78
|
+
// inflates a command's run count and cost with conversation that never invoked it, which is
|
|
79
|
+
// exactly what happened in cohorte's own repo while this pipeline was being discussed.
|
|
80
|
+
// Only inline mentions are length-gated; an explicit <command-name> is always an invocation.
|
|
81
|
+
const MENTION_MAX_CHARS = 120;
|
|
82
|
+
|
|
75
83
|
function commandIn(text) {
|
|
76
84
|
const explicit = /<command-name>\s*(\/?[\w:-]+)\s*<\/command-name>/.exec(text);
|
|
77
85
|
if (explicit) return explicit[1].replace(/^\//, '');
|
|
86
|
+
if (text.trim().length > MENTION_MAX_CHARS) return null;
|
|
78
87
|
// Last mention wins: "finish /build then /review" ends on the one being asked for.
|
|
79
88
|
let found = null;
|
|
80
89
|
for (const m of text.matchAll(/(?:^|\s)\/([a-z][a-z0-9-]{2,})\b/g)) {
|
package/scripts/preflight.sh
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/bin/sh
|
|
2
2
|
#
|
|
3
|
-
# preflight.sh — deterministic phase gate for /review
|
|
3
|
+
# preflight.sh — deterministic phase gate for /review.
|
|
4
4
|
#
|
|
5
5
|
# Runs the profile's mechanical checks (typecheck, lint, tests — whatever the caller
|
|
6
6
|
# passes) BEFORE any agent is spawned. A red gate means the caller aborts and relays
|
|
@@ -15,7 +15,7 @@
|
|
|
15
15
|
# The caller must stop there — no agents.
|
|
16
16
|
# - All green: writes `<project>/.claude/preflight.ok` ("<epoch> <HEAD sha>") — the
|
|
17
17
|
# stamp `hooks/gate.py` checks (gate-config.json `preflight` block) before letting
|
|
18
|
-
# review
|
|
18
|
+
# review agents dispatch.
|
|
19
19
|
|
|
20
20
|
set -u
|
|
21
21
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
# telemetry-send.sh — fire-and-forget anonymous usage ping (SCHEMA.md §Telemetry).
|
|
3
3
|
#
|
|
4
4
|
# telemetry-send.sh <phase> <feature_id> <seconds> [results]
|
|
5
|
-
# phase brainstorm|spec|build|
|
|
5
|
+
# phase brainstorm|spec|build|review|fix|ship — the feature funnel, and only it.
|
|
6
6
|
# Setup/maintenance commands never ping (SCHEMA.md §Telemetry).
|
|
7
7
|
# feature the feature id — NEVER sent raw; SHA-256-hashed to 12 hex chars
|
|
8
8
|
# seconds batch wall-clock
|
|
@@ -39,7 +39,10 @@ phase="${1:-}"; feature="${2:-}"; seconds="${3:-0}"; results="${4:-}"
|
|
|
39
39
|
# Allowlist the phase here — the collector accepts any string, so a typo in a command
|
|
40
40
|
# file would silently pollute the dataset with a phantom phase nobody notices.
|
|
41
41
|
case "$phase" in
|
|
42
|
-
brainstorm|spec|build|
|
|
42
|
+
brainstorm|spec|build|review|fix|ship) ;;
|
|
43
|
+
# `smoke` is a RETIRED phase (removed in 1.5.0) — still accepted so a stale install
|
|
44
|
+
# pinging it lands in its own bucket instead of being silently dropped.
|
|
45
|
+
smoke) ;;
|
|
43
46
|
*) exit 0 ;;
|
|
44
47
|
esac
|
|
45
48
|
|
|
@@ -20,6 +20,7 @@ const require = createRequire(import.meta.url);
|
|
|
20
20
|
const root = fileURLToPath(new URL("..", import.meta.url));
|
|
21
21
|
const { parse, parseProfileBlock } = require(join(root, "dashboard/server/yaml.js"));
|
|
22
22
|
const { metrics } = require(join(root, "dashboard/server/metrics.js"));
|
|
23
|
+
const { usage } = require(join(root, "dashboard/server/usage.js"));
|
|
23
24
|
const { state, scanSpecs } = require(join(root, "dashboard/server/doctor.js"));
|
|
24
25
|
const { kanban } = require(join(root, "dashboard/server/kanban.js"));
|
|
25
26
|
const fleet = require(join(root, "dashboard/server/fleet.js"));
|
|
@@ -106,6 +107,25 @@ console.log("metrics.js — the funnel aggregate");
|
|
|
106
107
|
check("no metrics file ⇒ present:false", metrics({ projectRoot: scratch() }).present === false);
|
|
107
108
|
}
|
|
108
109
|
|
|
110
|
+
// ── usage.js ─────────────────────────────────────────────────────────────────
|
|
111
|
+
// Wraps the ESM metrics collector for the CJS server. The failure that matters is
|
|
112
|
+
// not a crash: a project with no transcripts must say so, because rendering zeros
|
|
113
|
+
// reads as "this pipeline costs nothing" rather than "nothing was measured".
|
|
114
|
+
console.log("usage.js — the collector bridge");
|
|
115
|
+
{
|
|
116
|
+
const empty = usage({ projectRoot: scratch() });
|
|
117
|
+
check("a project with no transcripts reports present:false", empty.present === false);
|
|
118
|
+
check("…and says why rather than returning silent zeros",
|
|
119
|
+
typeof empty.error === "string" && empty.error.length > 0, JSON.stringify(empty));
|
|
120
|
+
|
|
121
|
+
// Same project twice: the second call must come from cache, or the panel's polling
|
|
122
|
+
// would re-parse tens of MB of transcripts on every refresh.
|
|
123
|
+
const d = scratch();
|
|
124
|
+
const t0 = Date.now(); usage({ projectRoot: d });
|
|
125
|
+
const t1 = Date.now(); usage({ projectRoot: d }); const cached = Date.now() - t1;
|
|
126
|
+
check("a repeated read is served from cache", cached <= Math.max(50, (t1 - t0)), `${cached}ms`);
|
|
127
|
+
}
|
|
128
|
+
|
|
109
129
|
// ── doctor.js ────────────────────────────────────────────────────────────────
|
|
110
130
|
console.log("doctor.js — the /doctor port");
|
|
111
131
|
{
|
|
@@ -135,7 +155,7 @@ console.log("doctor.js — the /doctor port");
|
|
|
135
155
|
const d = scratch();
|
|
136
156
|
const gate = {
|
|
137
157
|
deny: ["x"], ask: ["y"], ask_on_default_branch: ["git push"], default_branch: "main",
|
|
138
|
-
preflight: { enabled: true, agents: ["review"
|
|
158
|
+
preflight: { enabled: true, agents: ["review"], max_age_minutes: 30 },
|
|
139
159
|
};
|
|
140
160
|
writeFileSync(join(d, "PIPELINE.md"), [
|
|
141
161
|
"```yaml pipeline-profile",
|
|
@@ -153,7 +173,7 @@ console.log("doctor.js — the /doctor port");
|
|
|
153
173
|
' ask_on_default_branch: ["git push"]',
|
|
154
174
|
" preflight:",
|
|
155
175
|
" enabled: true",
|
|
156
|
-
" agents: [review
|
|
176
|
+
" agents: [review]",
|
|
157
177
|
" max_age_minutes: 30",
|
|
158
178
|
"```",
|
|
159
179
|
].join("\n"));
|
package/scripts/test-gate.mjs
CHANGED
|
@@ -181,7 +181,7 @@ console.log("gate.py — config robustness");
|
|
|
181
181
|
// ── the preflight phase gate (Task dispatches) ───────────────────────────────
|
|
182
182
|
console.log("gate.py — preflight phase gate");
|
|
183
183
|
{
|
|
184
|
-
const pf = { enabled: true, agents: ["review"
|
|
184
|
+
const pf = { enabled: true, agents: ["review"], max_age_minutes: 30 };
|
|
185
185
|
const d = scratch(); writeConfig(d, { ...GATE_CFG, preflight: pf });
|
|
186
186
|
const head = gitRepo(d, "main");
|
|
187
187
|
const stamp = (epoch, sha) =>
|
|
@@ -195,7 +195,6 @@ console.log("gate.py — preflight phase gate");
|
|
|
195
195
|
|
|
196
196
|
stamp(now(), head);
|
|
197
197
|
check("fresh stamp at the current HEAD ⇒ passes", run(task("review"), at).decision === null);
|
|
198
|
-
check("smoke is gated too", run(task("smoke"), at).decision === null);
|
|
199
198
|
|
|
200
199
|
stamp(now() - 60 * 60, head);
|
|
201
200
|
check("stamp older than max_age_minutes ⇒ ask", run(task("review"), at).decision === "ask");
|