hstack 0.7.1 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +271 -0
- package/README.md +39 -13
- package/VERSION +1 -1
- package/dist/commands/doctor.js +51 -1
- package/dist/commands/doctor.js.map +1 -1
- package/dist/commands/update.js +8 -2
- package/dist/commands/update.js.map +1 -1
- package/dist/lib/descriptions.js +167 -0
- package/dist/lib/descriptions.js.map +1 -0
- package/dist/lib/diff.js +1 -1
- package/dist/lib/git.js +16 -0
- package/dist/lib/git.js.map +1 -1
- package/dist/lib/wire.js +108 -4
- package/dist/lib/wire.js.map +1 -1
- package/dist/manifest.js +17 -2
- package/dist/manifest.js.map +1 -1
- package/package.json +3 -1
- package/template/.claude/agents/adversarial-reviewer.md +16 -64
- package/template/.claude/agents/app-architect.md +12 -49
- package/template/.claude/agents/data-architect.md +13 -51
- package/template/.claude/agents/data-specialist.md +5 -50
- package/template/.claude/agents/implementer.md +8 -65
- package/template/.claude/agents/kernel-fit-analyst.md +7 -68
- package/template/.claude/agents/planner.md +7 -42
- package/template/.claude/agents/product-discovery.md +12 -48
- package/template/.claude/agents/product-manager.md +8 -43
- package/template/.claude/agents/researcher.md +5 -41
- package/template/.claude/agents/security-reviewer.md +19 -54
- package/template/.claude/agents/spec-author.md +18 -52
- package/template/.claude/agents/stack-architect.md +14 -43
- package/template/.claude/agents/test-strategist.md +16 -57
- package/template/.claude/agents/ui-ux-briefer.md +6 -36
- package/template/.claude/agents/verifier.md +13 -45
- package/template/.claude/skills/hstack-adr-new/SKILL.md +6 -33
- package/template/.claude/skills/hstack-adversarial-review/SKILL.md +31 -52
- package/template/.claude/skills/hstack-adversarial-review/references/finding-categories.md +157 -0
- package/template/.claude/skills/hstack-app-architecture/SKILL.md +2 -29
- package/template/.claude/skills/hstack-branch/SKILL.md +4 -31
- package/template/.claude/skills/hstack-brownfield-init/SKILL.md +10 -37
- package/template/.claude/skills/hstack-change-new/SKILL.md +4 -31
- package/template/.claude/skills/hstack-change-plan/SKILL.md +21 -32
- package/template/.claude/skills/hstack-commit/SKILL.md +7 -35
- package/template/.claude/skills/hstack-configure/SKILL.md +7 -34
- package/template/.claude/skills/hstack-coord/SKILL.md +3 -39
- package/template/.claude/skills/hstack-data-architecture/SKILL.md +4 -30
- package/template/.claude/skills/hstack-data-review/SKILL.md +3 -42
- package/template/.claude/skills/hstack-finalize/SKILL.md +30 -49
- package/template/.claude/skills/hstack-flag/SKILL.md +9 -48
- package/template/.claude/skills/hstack-greenfield-init/SKILL.md +9 -36
- package/template/.claude/skills/hstack-help/SKILL.md +11 -37
- package/template/.claude/skills/hstack-implement/SKILL.md +28 -58
- package/template/.claude/skills/hstack-kernel-fit-promote/SKILL.md +7 -46
- package/template/.claude/skills/hstack-kernel-fit-scan/SKILL.md +5 -60
- package/template/.claude/skills/hstack-kernel-fit-scan/references/slack-setup.md +42 -0
- package/template/.claude/skills/hstack-kernel-fit-triage/SKILL.md +12 -50
- package/template/.claude/skills/hstack-module-spec/SKILL.md +5 -32
- package/template/.claude/skills/hstack-product-discovery/SKILL.md +5 -31
- package/template/.claude/skills/hstack-research/SKILL.md +3 -33
- package/template/.claude/skills/hstack-scaffold/SKILL.md +2 -29
- package/template/.claude/skills/hstack-security-review/SKILL.md +5 -43
- package/template/.claude/skills/hstack-ship/SKILL.md +43 -53
- package/template/.claude/skills/hstack-stack-decide/SKILL.md +3 -30
- package/template/.claude/skills/hstack-story-draft/SKILL.md +6 -33
- package/template/.claude/skills/hstack-tech-debt-new/SKILL.md +4 -31
- package/template/.claude/skills/hstack-tech-debt-resolve/SKILL.md +9 -44
- package/template/.claude/skills/hstack-tech-debt-stale/SKILL.md +10 -37
- package/template/.claude/skills/hstack-tech-debt-wontfix/SKILL.md +8 -35
- package/template/.claude/skills/hstack-telemetry/SKILL.md +5 -30
- package/template/.claude/skills/hstack-test-plan/SKILL.md +23 -46
- package/template/.claude/skills/hstack-ui-brief/SKILL.md +3 -30
- package/template/.claude/skills/hstack-verify/SKILL.md +26 -48
- package/template/KERNEL.md +410 -0
- package/template/scripts/compute-merge-readiness.mjs +780 -0
- package/template/scripts/run-gates.sh +388 -0
- package/template/scripts/telemetry/insights/kernel_fit.py +1 -1
- package/template/scripts/telemetry/insights/token_economics.py +181 -8
- package/template/scripts/telemetry/parsers/sidecars.py +61 -0
- package/template/scripts/telemetry/parsers/transcripts.py +135 -22
- package/template/scripts/telemetry/render.py +68 -3
- package/template/scripts/telemetry/report.py +16 -4
- package/template/scripts/telemetry/run_kernel_fit.py +6 -2
- package/template/scripts/telemetry/session_id.py +139 -0
- package/template/scripts/validate-spec.mjs +3303 -0
- package/template/templates/adr.md +7 -0
- package/template/templates/adversarial-review.md +5 -5
- package/template/templates/ci-cd.md +14 -0
- package/template/templates/coord-message.md +3 -2
- package/template/templates/data-architecture.md +3 -6
- package/template/templates/kernel-fit-finding.md +2 -2
- package/template/templates/kernel-fit-flag.md +2 -2
- package/template/templates/plan.md +4 -0
- package/template/templates/product-brief.md +2 -2
- package/template/templates/roadmap.md +41 -0
- package/template/templates/security-review.md +1 -1
- package/template/templates/telemetry-sidecar.md +56 -13
- package/template/templates/test-plan.md +1 -1
- package/template/CLAUDE.md +0 -443
- package/template/templates/mvp-scope.md +0 -34
|
@@ -0,0 +1,388 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
#
|
|
3
|
+
# hstack gate runner — the verifier's machine hands, and the thing
|
|
4
|
+
# {{TODO-SCRIPT: hstack/scripts/run-gates.sh}} stood in for.
|
|
5
|
+
#
|
|
6
|
+
# hstack/scripts/run-gates.sh --change <change-id>
|
|
7
|
+
# hstack/scripts/run-gates.sh --change <id> --suite unit --suite lint
|
|
8
|
+
# hstack/scripts/run-gates.sh --list
|
|
9
|
+
# hstack/scripts/run-gates.sh --change <id> --json
|
|
10
|
+
#
|
|
11
|
+
# It reads the canonical commands declared in hstack/context/ci-cd.md, runs
|
|
12
|
+
# every one of them, captures combined stdout/stderr to the pointer file the
|
|
13
|
+
# verification artifact references, and emits an observed-test-count PER SUITE
|
|
14
|
+
# so V-05 ("a suite that executed zero tests cannot be recorded as pass") is a
|
|
15
|
+
# measurement rather than a paragraph of parsing instructions in a prompt.
|
|
16
|
+
#
|
|
17
|
+
# Exit codes:
|
|
18
|
+
# 0 every suite ran, exited 0, and every test suite observed > 0 tests
|
|
19
|
+
# 1 a suite failed, or a test suite observed zero tests (V-05)
|
|
20
|
+
# 2 usage / environment error (no ci-cd.md, no canonical-commands block)
|
|
21
|
+
#
|
|
22
|
+
# Dependency-free by construction: POSIX tools only, no jq, no node. The
|
|
23
|
+
# consuming repo has no node_modules for hstack — same constraint that made
|
|
24
|
+
# validate-spec.mjs plain ESM (ADR-0001).
|
|
25
|
+
|
|
26
|
+
set -uo pipefail
|
|
27
|
+
|
|
28
|
+
# ---------------------------------------------------------------------------
|
|
29
|
+
# 1. Argument parsing
|
|
30
|
+
# ---------------------------------------------------------------------------
|
|
31
|
+
|
|
32
|
+
CHANGE_ID=""
|
|
33
|
+
OUT=""
|
|
34
|
+
ROOT=""
|
|
35
|
+
JSON=0
|
|
36
|
+
LIST=0
|
|
37
|
+
SUITES_REQUESTED=""
|
|
38
|
+
|
|
39
|
+
usage() {
|
|
40
|
+
cat <<'EOF'
|
|
41
|
+
hstack run-gates — run the canonical test / lint / typecheck suites
|
|
42
|
+
|
|
43
|
+
hstack/scripts/run-gates.sh [options]
|
|
44
|
+
|
|
45
|
+
Options
|
|
46
|
+
--change ID change-spec id; the pointer file defaults to
|
|
47
|
+
hstack/specs/changes/<ID>/test-output.txt
|
|
48
|
+
--out PATH pointer file path (overrides --change)
|
|
49
|
+
--suite NAME run only this suite; repeatable
|
|
50
|
+
--list print the parsed canonical commands and exit
|
|
51
|
+
--json emit the per-suite summary as JSON on stdout
|
|
52
|
+
--root DIR repo root to resolve hstack/ from (default: search upward)
|
|
53
|
+
-h, --help this text
|
|
54
|
+
|
|
55
|
+
Exit codes: 0 all green with a non-zero test count per test suite,
|
|
56
|
+
1 a suite failed or observed zero tests (V-05), 2 usage / environment error.
|
|
57
|
+
EOF
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
while [ $# -gt 0 ]; do
|
|
61
|
+
case "$1" in
|
|
62
|
+
--change) CHANGE_ID="${2:-}"; shift 2 ;;
|
|
63
|
+
--out) OUT="${2:-}"; shift 2 ;;
|
|
64
|
+
--suite) SUITES_REQUESTED="$SUITES_REQUESTED ${2:-}"; shift 2 ;;
|
|
65
|
+
--root) ROOT="${2:-}"; shift 2 ;;
|
|
66
|
+
--json) JSON=1; shift ;;
|
|
67
|
+
--list) LIST=1; shift ;;
|
|
68
|
+
-h|--help) usage; exit 0 ;;
|
|
69
|
+
*) echo "run-gates: unknown option $1" >&2; usage >&2; exit 2 ;;
|
|
70
|
+
esac
|
|
71
|
+
done
|
|
72
|
+
|
|
73
|
+
# ---------------------------------------------------------------------------
|
|
74
|
+
# 2. Locate the hstack tree
|
|
75
|
+
# ---------------------------------------------------------------------------
|
|
76
|
+
|
|
77
|
+
find_hstack_root() {
|
|
78
|
+
dir="${1:-$PWD}"
|
|
79
|
+
dir=$(cd "$dir" 2>/dev/null && pwd) || return 1
|
|
80
|
+
while :; do
|
|
81
|
+
if [ -f "$dir/hstack/KERNEL.md" ] || [ -f "$dir/hstack/CLAUDE.md" ] || [ -f "$dir/hstack/config.yaml" ]; then
|
|
82
|
+
printf '%s\n' "$dir"
|
|
83
|
+
return 0
|
|
84
|
+
fi
|
|
85
|
+
parent=$(dirname "$dir")
|
|
86
|
+
[ "$parent" = "$dir" ] && return 1
|
|
87
|
+
dir="$parent"
|
|
88
|
+
done
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
REPO_ROOT=$(find_hstack_root "${ROOT:-$PWD}") || {
|
|
92
|
+
echo "run-gates: no hstack/ tree found (looked for hstack/KERNEL.md upward from ${ROOT:-$PWD}). Pass --root <repo>." >&2
|
|
93
|
+
exit 2
|
|
94
|
+
}
|
|
95
|
+
HSTACK="$REPO_ROOT/hstack"
|
|
96
|
+
CI_CD="$HSTACK/context/ci-cd.md"
|
|
97
|
+
|
|
98
|
+
[ -f "$CI_CD" ] || {
|
|
99
|
+
echo "run-gates: $CI_CD not found. The canonical commands live there; run \`/hstack:configure --interview ci-cd\` first." >&2
|
|
100
|
+
exit 2
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
# ---------------------------------------------------------------------------
|
|
104
|
+
# 3. Parse the canonical commands
|
|
105
|
+
# ---------------------------------------------------------------------------
|
|
106
|
+
#
|
|
107
|
+
# ci-cd.md declares them in a fenced block with the info string `hstack-gates`,
|
|
108
|
+
# one `suite: command` pair per line. The fence is the contract: everything
|
|
109
|
+
# else in ci-cd.md is prose written for humans, and a runner that guessed at
|
|
110
|
+
# prose would produce a confident wrong answer about what the repo's tests are.
|
|
111
|
+
|
|
112
|
+
CANONICAL=$(awk '
|
|
113
|
+
/^```[[:space:]]*hstack-gates[[:space:]]*$/ { inblock=1; next }
|
|
114
|
+
inblock && /^```/ { inblock=0; next }
|
|
115
|
+
inblock {
|
|
116
|
+
line=$0
|
|
117
|
+
sub(/#.*$/, "", line) # trailing comment
|
|
118
|
+
if (line ~ /^[[:space:]]*$/) next
|
|
119
|
+
idx = index(line, ":")
|
|
120
|
+
if (idx == 0) next
|
|
121
|
+
key = substr(line, 1, idx-1)
|
|
122
|
+
val = substr(line, idx+1)
|
|
123
|
+
gsub(/^[[:space:]]+|[[:space:]]+$/, "", key)
|
|
124
|
+
gsub(/^[[:space:]]+|[[:space:]]+$/, "", val)
|
|
125
|
+
if (val == "" || val == "none" || val == "null") next # declared absent
|
|
126
|
+
printf "%s\t%s\n", key, val
|
|
127
|
+
}
|
|
128
|
+
' "$CI_CD")
|
|
129
|
+
|
|
130
|
+
if [ -z "$CANONICAL" ]; then
|
|
131
|
+
echo "run-gates: no \`hstack-gates\` fenced block in hstack/context/ci-cd.md." >&2
|
|
132
|
+
echo " Declare the canonical commands there — see hstack/templates/ci-cd.md § Canonical Commands." >&2
|
|
133
|
+
exit 2
|
|
134
|
+
fi
|
|
135
|
+
|
|
136
|
+
# Suites that are evidence of behaviour, and therefore subject to V-05. Lint
|
|
137
|
+
# and typecheck are exempt: both produce a diagnostic count whose floor is
|
|
138
|
+
# naturally zero on a clean repo, so zero is not a signal of a skipped run.
|
|
139
|
+
is_test_suite() {
|
|
140
|
+
case "$1" in
|
|
141
|
+
unit|integration|e2e) return 0 ;;
|
|
142
|
+
*) return 1 ;;
|
|
143
|
+
esac
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
wanted() {
|
|
147
|
+
[ -z "$SUITES_REQUESTED" ] && return 0
|
|
148
|
+
for s in $SUITES_REQUESTED; do [ "$s" = "$1" ] && return 0; done
|
|
149
|
+
return 1
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
if [ "$LIST" -eq 1 ]; then
|
|
153
|
+
echo "hstack run-gates — canonical commands from hstack/context/ci-cd.md"
|
|
154
|
+
echo ""
|
|
155
|
+
printf '%s\n' "$CANONICAL" | while IFS="$(printf '\t')" read -r suite cmd; do
|
|
156
|
+
printf ' %-12s %s\n' "$suite" "$cmd"
|
|
157
|
+
done
|
|
158
|
+
exit 0
|
|
159
|
+
fi
|
|
160
|
+
|
|
161
|
+
# ---------------------------------------------------------------------------
|
|
162
|
+
# 4. Pointer file
|
|
163
|
+
# ---------------------------------------------------------------------------
|
|
164
|
+
|
|
165
|
+
if [ -z "$OUT" ]; then
|
|
166
|
+
if [ -n "$CHANGE_ID" ]; then
|
|
167
|
+
OUT="$HSTACK/specs/changes/$CHANGE_ID/test-output.txt"
|
|
168
|
+
[ -d "$HSTACK/specs/changes/$CHANGE_ID" ] || {
|
|
169
|
+
echo "run-gates: no change folder at hstack/specs/changes/$CHANGE_ID/" >&2
|
|
170
|
+
exit 2
|
|
171
|
+
}
|
|
172
|
+
else
|
|
173
|
+
OUT="$REPO_ROOT/hstack-gates-output.txt"
|
|
174
|
+
fi
|
|
175
|
+
fi
|
|
176
|
+
mkdir -p "$(dirname "$OUT")" || exit 2
|
|
177
|
+
: > "$OUT" || { echo "run-gates: cannot write $OUT" >&2; exit 2; }
|
|
178
|
+
|
|
179
|
+
{
|
|
180
|
+
echo "hstack run-gates"
|
|
181
|
+
echo "repo: $REPO_ROOT"
|
|
182
|
+
echo "source: hstack/context/ci-cd.md"
|
|
183
|
+
echo "====================================================================="
|
|
184
|
+
} >> "$OUT"
|
|
185
|
+
|
|
186
|
+
# ---------------------------------------------------------------------------
|
|
187
|
+
# 5. Observed-test-count extraction
|
|
188
|
+
# ---------------------------------------------------------------------------
|
|
189
|
+
#
|
|
190
|
+
# One awk pass per suite over that suite's captured output. The runners hstack
|
|
191
|
+
# meets in practice all print a summary line; the point is not to understand
|
|
192
|
+
# every runner, it is to answer one question honestly: did this suite execute
|
|
193
|
+
# anything? When no known pattern matches, the answer is "unknown" — and
|
|
194
|
+
# unknown is treated as zero, because a count nobody could read is not evidence.
|
|
195
|
+
#
|
|
196
|
+
# "Executed" is passed + failed, NOT total. A run that collected fifteen tests
|
|
197
|
+
# and skipped all fifteen executed nothing, and V-05 exists precisely for that
|
|
198
|
+
# case: `Tests: 15 skipped, 15 total` is a non-zero total with zero assertions.
|
|
199
|
+
#
|
|
200
|
+
# Jest / Vitest Tests: 12 passed, 3 skipped, 15 total | No tests found
|
|
201
|
+
# Mocha 12 passing / 3 pending / 1 failing
|
|
202
|
+
# Playwright 12 passed (4.2s) / 1 failed / 3 skipped
|
|
203
|
+
# pytest collected 15 items | 12 passed, 3 skipped | no tests ran
|
|
204
|
+
# go test ok pkg 0.4s | testing: warning: no tests to run
|
|
205
|
+
|
|
206
|
+
count_tests() {
|
|
207
|
+
# $1 = file holding this suite's output
|
|
208
|
+
awk '
|
|
209
|
+
function num(s) { return s + 0 }
|
|
210
|
+
# --- explicit zero-collection statements, strongest signal ----------------
|
|
211
|
+
/[Nn]o tests found/ { zero=1 }
|
|
212
|
+
/collected 0 items/ { zero=1; seen=1 }
|
|
213
|
+
/no tests ran/ { zero=1; seen=1 }
|
|
214
|
+
/no tests to run/ { zero=1 }
|
|
215
|
+
/^[[:space:]]*Test Files[[:space:]]+no tests/ { zero=1 }
|
|
216
|
+
|
|
217
|
+
# --- Jest / Vitest summary ----------------------------------------------
|
|
218
|
+
/^[[:space:]]*Tests:?[[:space:]]/ {
|
|
219
|
+
seen=1
|
|
220
|
+
line=$0
|
|
221
|
+
if (match(line, /[0-9]+ passed/)) { s=substr(line, RSTART, RLENGTH); passed=num(s) }
|
|
222
|
+
if (match(line, /[0-9]+ failed/)) { s=substr(line, RSTART, RLENGTH); failed=num(s) }
|
|
223
|
+
if (match(line, /[0-9]+ skipped/)) { s=substr(line, RSTART, RLENGTH); skipped=num(s) }
|
|
224
|
+
if (match(line, /[0-9]+ todo/)) { s=substr(line, RSTART, RLENGTH); skipped+=num(s) }
|
|
225
|
+
if (match(line, /[0-9]+ total/)) { s=substr(line, RSTART, RLENGTH); total=num(s) }
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
# --- pytest short summary ------------------------------------------------
|
|
229
|
+
/=+ .*(passed|failed|error|skipped).* =+/ {
|
|
230
|
+
seen=1
|
|
231
|
+
line=$0
|
|
232
|
+
if (match(line, /[0-9]+ passed/)) { s=substr(line, RSTART, RLENGTH); passed=num(s) }
|
|
233
|
+
if (match(line, /[0-9]+ failed/)) { s=substr(line, RSTART, RLENGTH); failed=num(s) }
|
|
234
|
+
if (match(line, /[0-9]+ error/)) { s=substr(line, RSTART, RLENGTH); failed+=num(s) }
|
|
235
|
+
if (match(line, /[0-9]+ skipped/)) { s=substr(line, RSTART, RLENGTH); skipped=num(s) }
|
|
236
|
+
}
|
|
237
|
+
/collected [0-9]+ item/ {
|
|
238
|
+
seen=1
|
|
239
|
+
if (match($0, /collected [0-9]+/)) { s=substr($0, RSTART+10, RLENGTH-10); collected=num(s) }
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
# --- Mocha ---------------------------------------------------------------
|
|
243
|
+
/^[[:space:]]*[0-9]+ passing/ { seen=1; if (match($0, /[0-9]+/)) { passed=num(substr($0, RSTART, RLENGTH)) } }
|
|
244
|
+
/^[[:space:]]*[0-9]+ pending/ { seen=1; if (match($0, /[0-9]+/)) { skipped=num(substr($0, RSTART, RLENGTH)) } }
|
|
245
|
+
/^[[:space:]]*[0-9]+ failing/ { seen=1; if (match($0, /[0-9]+/)) { failed=num(substr($0, RSTART, RLENGTH)) } }
|
|
246
|
+
|
|
247
|
+
# --- Playwright ("12 passed (4.2s)") and bare runner tallies -------------
|
|
248
|
+
# No \b here: POSIX awk reads it as a backspace, not a word boundary.
|
|
249
|
+
/^[[:space:]]*[0-9]+ (passed|failed|skipped|flaky)([^a-z]|$)/ {
|
|
250
|
+
seen=1
|
|
251
|
+
line=$0
|
|
252
|
+
if (match(line, /[0-9]+ passed/)) { s=substr(line, RSTART, RLENGTH); passed=num(s) }
|
|
253
|
+
if (match(line, /[0-9]+ failed/)) { s=substr(line, RSTART, RLENGTH); failed=num(s) }
|
|
254
|
+
if (match(line, /[0-9]+ skipped/)) { s=substr(line, RSTART, RLENGTH); skipped=num(s) }
|
|
255
|
+
if (match(line, /[0-9]+ flaky/)) { s=substr(line, RSTART, RLENGTH); passed+=num(s) }
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
END {
|
|
259
|
+
if (total == 0) total = passed + failed + skipped
|
|
260
|
+
if (total == 0 && collected > 0) { total = collected }
|
|
261
|
+
if (zero) { total = 0; passed = 0; failed = 0 }
|
|
262
|
+
executed = passed + failed
|
|
263
|
+
# `known` says whether any pattern matched at all. An unreadable summary
|
|
264
|
+
# is reported as unknown and treated as zero downstream — a count nobody
|
|
265
|
+
# could read is not evidence that tests ran.
|
|
266
|
+
known = (seen || zero) ? 1 : 0
|
|
267
|
+
printf "%d %d %d %d %d %d\n", passed+0, failed+0, skipped+0, total+0, executed+0, known
|
|
268
|
+
}
|
|
269
|
+
' "$1"
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
# ---------------------------------------------------------------------------
|
|
273
|
+
# 6. Run
|
|
274
|
+
# ---------------------------------------------------------------------------
|
|
275
|
+
|
|
276
|
+
TMPDIR_RUN=$(mktemp -d "${TMPDIR:-/tmp}/hstack-run-gates.XXXXXX") || exit 2
|
|
277
|
+
trap 'rm -rf "$TMPDIR_RUN"' EXIT
|
|
278
|
+
|
|
279
|
+
SUMMARY="$TMPDIR_RUN/summary"
|
|
280
|
+
: > "$SUMMARY"
|
|
281
|
+
OVERALL=0
|
|
282
|
+
RAN_ANY=0
|
|
283
|
+
|
|
284
|
+
while IFS="$(printf '\t')" read -r suite cmd; do
|
|
285
|
+
[ -n "$suite" ] || continue
|
|
286
|
+
wanted "$suite" || continue
|
|
287
|
+
RAN_ANY=1
|
|
288
|
+
|
|
289
|
+
suite_out="$TMPDIR_RUN/$suite.out"
|
|
290
|
+
{
|
|
291
|
+
echo ""
|
|
292
|
+
echo "--- suite: $suite ------------------------------------------------"
|
|
293
|
+
echo "\$ $cmd"
|
|
294
|
+
} >> "$OUT"
|
|
295
|
+
|
|
296
|
+
# stdin from /dev/null, not inherited: the loop below is fed by a heredoc of
|
|
297
|
+
# the canonical commands, and a suite that reads stdin (an interactive watch
|
|
298
|
+
# mode, a prompt) would otherwise eat the remaining suites.
|
|
299
|
+
( cd "$REPO_ROOT" && eval "$cmd" ) > "$suite_out" 2>&1 < /dev/null
|
|
300
|
+
code=$?
|
|
301
|
+
cat "$suite_out" >> "$OUT"
|
|
302
|
+
|
|
303
|
+
if is_test_suite "$suite"; then
|
|
304
|
+
read -r passed failed skipped total executed known <<EOF
|
|
305
|
+
$(count_tests "$suite_out")
|
|
306
|
+
EOF
|
|
307
|
+
else
|
|
308
|
+
passed=0; failed=0; skipped=0; total=0; executed=0; known=1
|
|
309
|
+
fi
|
|
310
|
+
|
|
311
|
+
# V-05: zero executed tests is `not-run`, never `pass`. "Zero failures" is
|
|
312
|
+
# not evidence of correctness when there were zero assertions to fail.
|
|
313
|
+
if [ "$code" -ne 0 ]; then
|
|
314
|
+
verdict="fail"
|
|
315
|
+
reason="command exited $code"
|
|
316
|
+
OVERALL=1
|
|
317
|
+
elif is_test_suite "$suite" && [ "$executed" -eq 0 ]; then
|
|
318
|
+
verdict="not-run"
|
|
319
|
+
if [ "$known" -eq 0 ]; then
|
|
320
|
+
reason="no test count could be read from the runner's output (unrecognised summary format)"
|
|
321
|
+
elif [ "$skipped" -gt 0 ]; then
|
|
322
|
+
reason="the runner reported zero executed tests — $skipped skipped of $total collected (all-skipped, or a filter that collapsed the set)"
|
|
323
|
+
else
|
|
324
|
+
reason="the runner reported zero executed tests (env-gated, empty-collection, or filter-collapse)"
|
|
325
|
+
fi
|
|
326
|
+
OVERALL=1
|
|
327
|
+
else
|
|
328
|
+
verdict="pass"
|
|
329
|
+
reason=""
|
|
330
|
+
fi
|
|
331
|
+
|
|
332
|
+
printf '%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n' \
|
|
333
|
+
"$suite" "$cmd" "$code" "$verdict" "$passed" "$failed" "$skipped" "$total" "$executed" "$reason" >> "$SUMMARY"
|
|
334
|
+
|
|
335
|
+
{
|
|
336
|
+
echo "--- suite: $suite → $verdict (exit $code, $executed of $total test(s) executed)"
|
|
337
|
+
} >> "$OUT"
|
|
338
|
+
done <<EOF
|
|
339
|
+
$CANONICAL
|
|
340
|
+
EOF
|
|
341
|
+
|
|
342
|
+
if [ "$RAN_ANY" -eq 0 ]; then
|
|
343
|
+
echo "run-gates: no suite matched --suite${SUITES_REQUESTED}" >&2
|
|
344
|
+
exit 2
|
|
345
|
+
fi
|
|
346
|
+
|
|
347
|
+
# ---------------------------------------------------------------------------
|
|
348
|
+
# 7. Report
|
|
349
|
+
# ---------------------------------------------------------------------------
|
|
350
|
+
|
|
351
|
+
REL_OUT="${OUT#"$REPO_ROOT"/}"
|
|
352
|
+
|
|
353
|
+
if [ "$JSON" -eq 1 ]; then
|
|
354
|
+
printf '{\n'
|
|
355
|
+
printf ' "ok": %s,\n' "$([ "$OVERALL" -eq 0 ] && echo true || echo false)"
|
|
356
|
+
printf ' "test-output": "%s",\n' "$REL_OUT"
|
|
357
|
+
printf ' "suites": [\n'
|
|
358
|
+
first=1
|
|
359
|
+
while IFS="$(printf '\t')" read -r suite cmd code verdict passed failed skipped total executed reason; do
|
|
360
|
+
[ "$first" -eq 1 ] || printf ',\n'
|
|
361
|
+
first=0
|
|
362
|
+
esc_cmd=$(printf '%s' "$cmd" | sed 's/\\/\\\\/g; s/"/\\"/g')
|
|
363
|
+
esc_reason=$(printf '%s' "$reason" | sed 's/\\/\\\\/g; s/"/\\"/g')
|
|
364
|
+
printf ' {"suite": "%s", "command": "%s", "exit": %s, "verdict": "%s", "observed": {"passed": %s, "failed": %s, "skipped": %s, "total": %s, "executed": %s}, "reason": "%s"}' \
|
|
365
|
+
"$suite" "$esc_cmd" "$code" "$verdict" "$passed" "$failed" "$skipped" "$total" "$executed" "$esc_reason"
|
|
366
|
+
done < "$SUMMARY"
|
|
367
|
+
printf '\n ]\n}\n'
|
|
368
|
+
else
|
|
369
|
+
echo ""
|
|
370
|
+
echo "hstack run-gates — $REPO_ROOT"
|
|
371
|
+
echo ""
|
|
372
|
+
printf ' %-12s %-9s %-6s %s\n' "suite" "verdict" "exit" "observed (passed/failed/skipped/total)"
|
|
373
|
+
while IFS="$(printf '\t')" read -r suite cmd code verdict passed failed skipped total executed reason; do
|
|
374
|
+
printf ' %-12s %-9s %-6s %s/%s/%s/%s\n' "$suite" "$verdict" "$code" "$passed" "$failed" "$skipped" "$total"
|
|
375
|
+
[ -n "$reason" ] && printf ' %s\n' "$reason"
|
|
376
|
+
done < "$SUMMARY"
|
|
377
|
+
echo ""
|
|
378
|
+
echo " captured output: $REL_OUT"
|
|
379
|
+
echo " → verification.artifacts.test-output: $REL_OUT"
|
|
380
|
+
echo ""
|
|
381
|
+
if [ "$OVERALL" -eq 0 ]; then
|
|
382
|
+
echo "run-gates: all suites green with a non-zero executed-test count."
|
|
383
|
+
else
|
|
384
|
+
echo "run-gates: at least one suite failed or executed zero tests (V-05). status: passed is blocked."
|
|
385
|
+
fi
|
|
386
|
+
fi
|
|
387
|
+
|
|
388
|
+
exit "$OVERALL"
|
|
@@ -4,7 +4,7 @@ This module is the detection layer of the kernel-fit closed-loop system. It
|
|
|
4
4
|
pattern-matches across shipped artifacts and emits evidence rows; an LLM
|
|
5
5
|
subagent (`kernel-fit-analyst`) then synthesizes findings from these rows.
|
|
6
6
|
|
|
7
|
-
See ADR-0004 for the full design rationale and `template/
|
|
7
|
+
See ADR-0004 for the full design rationale and `template/KERNEL.md` § How
|
|
8
8
|
hstack improves itself for the loop contract.
|
|
9
9
|
|
|
10
10
|
Three starter patterns:
|
|
@@ -1,24 +1,49 @@
|
|
|
1
|
-
"""Token-economics insights: TE-1 cost per
|
|
2
|
-
subagent, TE-3 subagent entry-tax amortization
|
|
1
|
+
"""Token-economics insights: TE-1 cost per Skill, TE-2 cache-hit ratio per
|
|
2
|
+
subagent, TE-3 subagent entry-tax amortization, TE-4 cost per phase, TE-5 cost
|
|
3
|
+
per change (ADR-0009).
|
|
4
|
+
|
|
5
|
+
TE-1/TE-2/TE-3 are *session-scoped*: they attribute a whole session to the first
|
|
6
|
+
Skill it invoked, because a Skill has a start marker and no end marker. TE-4/TE-5
|
|
7
|
+
are *phase-scoped*: they read the sidecar's `[phase_opened_at, phase_closed_at]`
|
|
8
|
+
window and sum only the turns inside it. Where a sidecar exists, TE-4/TE-5
|
|
9
|
+
supersede TE-1.
|
|
10
|
+
"""
|
|
3
11
|
|
|
4
12
|
from __future__ import annotations
|
|
5
13
|
|
|
6
14
|
from collections import defaultdict
|
|
7
15
|
|
|
16
|
+
from telemetry.parsers.transcripts import phase_usage
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
#: The five Skills that emit sidecars (ADR-0001 § v1 emission list). Every other
|
|
20
|
+
#: Skill is invisible to TE-4/TE-5 — which is what the coverage fraction says.
|
|
21
|
+
EMITTING_SKILLS = (
|
|
22
|
+
"hstack-test-plan", "hstack-implement", "hstack-verify",
|
|
23
|
+
"hstack-adversarial-review", "hstack-finalize",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
UNATTRIBUTED = "(unattributed)"
|
|
27
|
+
|
|
8
28
|
|
|
9
|
-
def compute(session_rows: list[dict], changes: dict) -> dict:
|
|
10
|
-
"""Compute the
|
|
29
|
+
def compute(session_rows: list[dict], changes: dict, sidecars: list[dict] | None = None) -> dict:
|
|
30
|
+
"""Compute the five TE metrics.
|
|
11
31
|
|
|
12
32
|
Args:
|
|
13
33
|
session_rows: from transcripts.collect_session_rows
|
|
14
34
|
changes: from frontmatter.load_change_artifacts
|
|
35
|
+
sidecars: from sidecars.load_sidecars (empty/None → TE-4/TE-5 report no
|
|
36
|
+
coverage rather than silently vanishing)
|
|
15
37
|
|
|
16
38
|
Returns a dict ready for rendering.
|
|
17
39
|
"""
|
|
40
|
+
phases = _te_4(sidecars or [])
|
|
18
41
|
return {
|
|
19
42
|
"te_1_cost_per_change": _te_1(session_rows, changes),
|
|
20
43
|
"te_2_cache_hit_per_subagent": _te_2(session_rows),
|
|
21
44
|
"te_3_subagent_entry_tax": _te_3(session_rows),
|
|
45
|
+
"te_4_cost_per_phase": phases,
|
|
46
|
+
"te_5_cost_per_change": _te_5(phases["rows"]),
|
|
22
47
|
}
|
|
23
48
|
|
|
24
49
|
|
|
@@ -44,7 +69,10 @@ def _te_1(session_rows: list[dict], changes: dict) -> dict:
|
|
|
44
69
|
}
|
|
45
70
|
cost_total = defaultdict(int)
|
|
46
71
|
session_counts = defaultdict(int)
|
|
72
|
+
unattributed = 0
|
|
47
73
|
for s in session_rows:
|
|
74
|
+
if s["skill"] is None:
|
|
75
|
+
unattributed += 1
|
|
48
76
|
if s["skill"] in per_change_skills:
|
|
49
77
|
# Heuristic: we don't have a structured change-id-per-session yet,
|
|
50
78
|
# so accumulate by skill until sidecars exist. The (skill, total)
|
|
@@ -62,9 +90,16 @@ def _te_1(session_rows: list[dict], changes: dict) -> dict:
|
|
|
62
90
|
})
|
|
63
91
|
return {
|
|
64
92
|
"rows": rows,
|
|
93
|
+
"unattributed_sessions": unattributed,
|
|
65
94
|
"note": (
|
|
66
|
-
"
|
|
67
|
-
"
|
|
95
|
+
"Session-scoped, not phase-scoped: a Skill has a start marker and no "
|
|
96
|
+
"end marker, so everything a session spends after the invocation "
|
|
97
|
+
"lands in the first bucket — including later phases and unrelated "
|
|
98
|
+
"work. Superseded by TE-4/TE-5 for any change that carries sidecars. "
|
|
99
|
+
"Attribution reads structured invocation markers only (<command-name> "
|
|
100
|
+
"tags, Skill tool_use blocks); sessions with no marker are "
|
|
101
|
+
f"unattributed ({unattributed} of {len(session_rows)} in window) "
|
|
102
|
+
"rather than credited to whichever Skill their prompt mentioned."
|
|
68
103
|
),
|
|
69
104
|
}
|
|
70
105
|
|
|
@@ -76,7 +111,7 @@ def _te_2(session_rows: list[dict]) -> dict:
|
|
|
76
111
|
"""
|
|
77
112
|
per_skill = defaultdict(lambda: {"cache_read": 0, "cache_creation": 0, "turns": 0})
|
|
78
113
|
for s in session_rows:
|
|
79
|
-
key = s["skill"] or
|
|
114
|
+
key = s["skill"] or UNATTRIBUTED
|
|
80
115
|
t = s["totals"]
|
|
81
116
|
per_skill[key]["cache_read"] += t.get("cache_read_input_tokens", 0)
|
|
82
117
|
per_skill[key]["cache_creation"] += t.get("cache_creation_input_tokens", 0)
|
|
@@ -92,7 +127,16 @@ def _te_2(session_rows: list[dict]) -> dict:
|
|
|
92
127
|
"cache_creation": agg["cache_creation"],
|
|
93
128
|
"ratio": ratio,
|
|
94
129
|
})
|
|
95
|
-
return {
|
|
130
|
+
return {
|
|
131
|
+
"rows": rows,
|
|
132
|
+
"note": (
|
|
133
|
+
"Session-scoped, same caveat as TE-1 — the whole session's cache "
|
|
134
|
+
f"behaviour is credited to its first Skill. `{UNATTRIBUTED}` holds "
|
|
135
|
+
"every session with no structured hstack invocation marker, "
|
|
136
|
+
"including plain non-hstack work. Superseded by TE-4/TE-5 wherever "
|
|
137
|
+
"sidecars exist."
|
|
138
|
+
),
|
|
139
|
+
}
|
|
96
140
|
|
|
97
141
|
|
|
98
142
|
def _te_3(session_rows: list[dict]) -> dict:
|
|
@@ -127,3 +171,132 @@ def _te_3(session_rows: list[dict]) -> dict:
|
|
|
127
171
|
"timestamps will sharpen this."
|
|
128
172
|
),
|
|
129
173
|
}
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def _te_4(sidecars: list[dict]) -> dict:
|
|
177
|
+
"""TE-4: cost per phase — the sidecar's window, summed from the transcript.
|
|
178
|
+
|
|
179
|
+
One row per sidecar. A row is *measured* when the sidecar carries a phase
|
|
180
|
+
window (schema_version ≥ 2) whose session transcript is still on disk;
|
|
181
|
+
otherwise `tokens` is `None` and the row is unmeasured. Never zero: a phase
|
|
182
|
+
whose transcript was swept spent tokens we can no longer count, and printing
|
|
183
|
+
0 would fold it into the average as if it were free.
|
|
184
|
+
"""
|
|
185
|
+
rows = []
|
|
186
|
+
for sc in sidecars:
|
|
187
|
+
usage = phase_usage(sc.get("data") or {})
|
|
188
|
+
data = sc.get("data") or {}
|
|
189
|
+
rows.append({
|
|
190
|
+
"skill": sc.get("skill"),
|
|
191
|
+
"change": sc.get("change_id"),
|
|
192
|
+
"phase_id": sc.get("phase_id"),
|
|
193
|
+
"sidecar": sc.get("file"),
|
|
194
|
+
"schema_version": sc.get("schema_version"),
|
|
195
|
+
"session_id": data.get("session_id"),
|
|
196
|
+
"opened_at": data.get("phase_opened_at"),
|
|
197
|
+
"closed_at": data.get("phase_closed_at"),
|
|
198
|
+
"measured": usage is not None,
|
|
199
|
+
"unmeasured_reason": None if usage is not None else _unmeasured_reason(sc),
|
|
200
|
+
"tokens": usage["total_tokens"] if usage else None,
|
|
201
|
+
"cost_score": usage["cost_score"] if usage else None,
|
|
202
|
+
"turns": usage["turns"] if usage else None,
|
|
203
|
+
"wall_clock_h": round(usage["wall_clock_s"] / 3600, 2) if usage else None,
|
|
204
|
+
})
|
|
205
|
+
rows.sort(key=lambda r: (-(r["tokens"] or 0), r["change"] or "", r["sidecar"] or ""))
|
|
206
|
+
measured = [r for r in rows if r["measured"]]
|
|
207
|
+
return {
|
|
208
|
+
"rows": rows,
|
|
209
|
+
"phases_emitted": len(rows),
|
|
210
|
+
"phases_measured": len(measured),
|
|
211
|
+
"coverage_fraction": (len(measured) / len(rows)) if rows else None,
|
|
212
|
+
"note": (
|
|
213
|
+
"Phase-scoped: tokens are summed over assistant turns whose "
|
|
214
|
+
"timestamp falls inside the sidecar's [phase_opened_at, "
|
|
215
|
+
"phase_closed_at] window. Only the five sidecar-emitting Skills "
|
|
216
|
+
f"({', '.join(EMITTING_SKILLS)}) appear here at all — every other "
|
|
217
|
+
"Skill is invisible, and subagent spend lands in its host's window "
|
|
218
|
+
"(isSidechain=False). Unmeasured rows are phases whose window or "
|
|
219
|
+
"transcript could not be read; they are never counted as zero. "
|
|
220
|
+
"Sidecars are not window-filtered — every change folder on disk is "
|
|
221
|
+
"read, unlike the session and git tables above. Read "
|
|
222
|
+
"this table next to QO-4 (observed vs promised) and WS-2 (gate "
|
|
223
|
+
"findings density): cost without an outcome beside it can only "
|
|
224
|
+
"argue for spending less, never for spending well."
|
|
225
|
+
),
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _unmeasured_reason(sidecar: dict) -> str:
|
|
230
|
+
data = sidecar.get("data") or {}
|
|
231
|
+
if not data.get("phase_opened_at") or not data.get("phase_closed_at"):
|
|
232
|
+
version = sidecar.get("schema_version")
|
|
233
|
+
return ("pre-ADR-0009 sidecar (schema_version "
|
|
234
|
+
f"{version if version is not None else '?'}) — no phase window")
|
|
235
|
+
if not data.get("session_id"):
|
|
236
|
+
return "session id unresolved at write time"
|
|
237
|
+
return "transcript not found (retention sweep, or written on another machine)"
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _te_5(phase_rows: list[dict]) -> dict:
|
|
241
|
+
"""TE-5: cost per change — the sum of that change's measured phases.
|
|
242
|
+
|
|
243
|
+
The coverage fraction is not decoration. Five of the 27 Skills emit
|
|
244
|
+
sidecars, so `tokens` is a sum over a subset by construction: the spec, the
|
|
245
|
+
plan, the security- and data-reviews, the ship gate and the whole configure
|
|
246
|
+
family are absent, and so is any phase whose transcript has aged out. A
|
|
247
|
+
reader who takes this column for a change's total cost will read it low.
|
|
248
|
+
"""
|
|
249
|
+
per_change: dict[str, dict] = {}
|
|
250
|
+
for r in phase_rows:
|
|
251
|
+
change = r.get("change") or "(unknown)"
|
|
252
|
+
agg = per_change.setdefault(change, {
|
|
253
|
+
"change": change,
|
|
254
|
+
"phases_emitted": 0,
|
|
255
|
+
"phases_measured": 0,
|
|
256
|
+
"skills": set(),
|
|
257
|
+
"tokens": 0,
|
|
258
|
+
"cost_score": 0,
|
|
259
|
+
"turns": 0,
|
|
260
|
+
"wall_clock_h": 0.0,
|
|
261
|
+
})
|
|
262
|
+
agg["phases_emitted"] += 1
|
|
263
|
+
if r.get("skill"):
|
|
264
|
+
agg["skills"].add(r["skill"])
|
|
265
|
+
if not r["measured"]:
|
|
266
|
+
continue
|
|
267
|
+
agg["phases_measured"] += 1
|
|
268
|
+
agg["tokens"] += r["tokens"] or 0
|
|
269
|
+
agg["cost_score"] += r["cost_score"] or 0
|
|
270
|
+
agg["turns"] += r["turns"] or 0
|
|
271
|
+
agg["wall_clock_h"] += r["wall_clock_h"] or 0.0
|
|
272
|
+
rows = []
|
|
273
|
+
for agg in per_change.values():
|
|
274
|
+
measured = agg["phases_measured"]
|
|
275
|
+
rows.append({
|
|
276
|
+
"change": agg["change"],
|
|
277
|
+
"phases_measured": measured,
|
|
278
|
+
"phases_emitted": agg["phases_emitted"],
|
|
279
|
+
"coverage_fraction": (measured / agg["phases_emitted"]) if agg["phases_emitted"] else None,
|
|
280
|
+
"skills_measured": sorted(agg["skills"]),
|
|
281
|
+
"tokens": agg["tokens"] if measured else None,
|
|
282
|
+
"cost_score": agg["cost_score"] if measured else None,
|
|
283
|
+
"turns": agg["turns"] if measured else None,
|
|
284
|
+
"wall_clock_h": round(agg["wall_clock_h"], 2) if measured else None,
|
|
285
|
+
})
|
|
286
|
+
rows.sort(key=lambda r: (-(r["tokens"] or 0), r["change"]))
|
|
287
|
+
total_emitted = sum(r["phases_emitted"] for r in rows)
|
|
288
|
+
total_measured = sum(r["phases_measured"] for r in rows)
|
|
289
|
+
return {
|
|
290
|
+
"rows": rows,
|
|
291
|
+
"phases_emitted": total_emitted,
|
|
292
|
+
"phases_measured": total_measured,
|
|
293
|
+
"coverage_fraction": (total_measured / total_emitted) if total_emitted else None,
|
|
294
|
+
"note": (
|
|
295
|
+
"A subset, not a total. Coverage fraction = measured phases / "
|
|
296
|
+
"emitted sidecars, and sidecars are emitted by five Skills only — "
|
|
297
|
+
"change-new, change-plan, security-review, data-review, ship and the "
|
|
298
|
+
"configure family contribute nothing to these sums. Pair with QO-4 "
|
|
299
|
+
"and the adversarial-review findings density before concluding that "
|
|
300
|
+
"an expensive change was a wasteful one."
|
|
301
|
+
),
|
|
302
|
+
}
|