@mmerterden/multi-agent-pipeline 12.7.0 → 12.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/CHANGELOG.md +207 -0
  2. package/install/_common.mjs +48 -0
  3. package/install/_dev-only-files.mjs +125 -5
  4. package/install/claude.mjs +14 -8
  5. package/install/copilot.mjs +5 -8
  6. package/package.json +17 -2
  7. package/pipeline/commands/multi-agent/analysis/SKILL.md +1 -1
  8. package/pipeline/lib/credential-store.sh +20 -0
  9. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  10. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  11. package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
  12. package/pipeline/multi-agent-refs/features/review-multi-repo.md +65 -0
  13. package/pipeline/multi-agent-refs/features/url-enrichment.md +93 -0
  14. package/pipeline/multi-agent-refs/phases/operations.md +28 -0
  15. package/pipeline/multi-agent-refs/phases/phase-0-init.md +3 -83
  16. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  17. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
  18. package/pipeline/multi-agent-refs/phases/phase-4-review.md +50 -62
  19. package/pipeline/schemas/prefs.schema.json +6 -0
  20. package/pipeline/scripts/_smoke-root.sh +61 -0
  21. package/pipeline/scripts/agent-guard.py +57 -3
  22. package/pipeline/scripts/audit-log.sh +25 -0
  23. package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
  24. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
  25. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
  26. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
  27. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
  28. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
  29. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
  30. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
  31. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
  32. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
  33. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
  34. package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
  35. package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
  36. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
  37. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
  38. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
  39. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
  40. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
  41. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
  42. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
  43. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
  44. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
  45. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
  46. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
  47. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
  48. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
  49. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
  50. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
  51. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
  52. package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
  53. package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
  54. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
  55. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
  56. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
  57. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
  58. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
  59. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
  60. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
  61. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
  62. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
  63. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
  64. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
  65. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
  66. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
  67. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
  68. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
  69. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
  70. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
  71. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
  72. package/pipeline/eval/golden-tasks/README.md +0 -65
  73. package/pipeline/eval/intent-cases.json +0 -40
  74. package/pipeline/eval/run-metrics-fixture.json +0 -93
  75. package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
  76. package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
  77. package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
  78. package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
  79. package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
  80. package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
  81. package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
  82. package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
  83. package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
  84. package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
  85. package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
  86. package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
  87. package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
  88. package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
  89. package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
  90. package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
  91. package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
  92. package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
  93. package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
  94. package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
  95. package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
  96. package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
  97. package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
  98. package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
  99. package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
  100. package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
  101. package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
  102. package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
  103. package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
  104. package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
  105. package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
  106. package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
  107. package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
  108. package/pipeline/eval/triage/README.md +0 -54
  109. package/pipeline/scripts/benchmark-phase-0.sh +0 -128
  110. package/pipeline/scripts/check-md-links.mjs +0 -88
  111. package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -302
  112. package/pipeline/scripts/eval-golden-tasks.mjs +0 -224
  113. package/pipeline/scripts/eval-intent.mjs +0 -107
  114. package/pipeline/scripts/eval-mine-corpus.mjs +0 -211
  115. package/pipeline/scripts/eval-triage.mjs +0 -171
  116. package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
  117. package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
  118. package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
  119. package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
  120. package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
  121. package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
  122. package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
  123. package/pipeline/scripts/lint-mcp-refs.mjs +0 -218
  124. package/pipeline/scripts/lint-skills.mjs +0 -154
  125. package/pipeline/scripts/run-smokes.mjs +0 -130
  126. package/pipeline/scripts/scorecard.mjs +0 -258
  127. package/pipeline/scripts/smoke-add-detail.sh +0 -137
  128. package/pipeline/scripts/smoke-agent-guard.sh +0 -74
  129. package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
  130. package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
  131. package/pipeline/scripts/smoke-ask-choice.sh +0 -42
  132. package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
  133. package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
  134. package/pipeline/scripts/smoke-changelog-version.sh +0 -47
  135. package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
  136. package/pipeline/scripts/smoke-channels-flow.sh +0 -130
  137. package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
  138. package/pipeline/scripts/smoke-clarify.sh +0 -148
  139. package/pipeline/scripts/smoke-command-inventory.sh +0 -81
  140. package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
  141. package/pipeline/scripts/smoke-community-gates.sh +0 -75
  142. package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
  143. package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
  144. package/pipeline/scripts/smoke-context-budget.sh +0 -72
  145. package/pipeline/scripts/smoke-cost-budget.sh +0 -70
  146. package/pipeline/scripts/smoke-cost-summary.sh +0 -139
  147. package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
  148. package/pipeline/scripts/smoke-description-tr.sh +0 -82
  149. package/pipeline/scripts/smoke-dev-critic.sh +0 -144
  150. package/pipeline/scripts/smoke-diff-explain.sh +0 -147
  151. package/pipeline/scripts/smoke-diff-risk.sh +0 -190
  152. package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
  153. package/pipeline/scripts/smoke-eval-live.sh +0 -136
  154. package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
  155. package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
  156. package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
  157. package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
  158. package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
  159. package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
  160. package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
  161. package/pipeline/scripts/smoke-generate-issue.sh +0 -120
  162. package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
  163. package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
  164. package/pipeline/scripts/smoke-install-layout.sh +0 -248
  165. package/pipeline/scripts/smoke-intent-guard.sh +0 -86
  166. package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
  167. package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
  168. package/pipeline/scripts/smoke-keychain.sh +0 -158
  169. package/pipeline/scripts/smoke-language-axis.sh +0 -109
  170. package/pipeline/scripts/smoke-learning-curve.sh +0 -61
  171. package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
  172. package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
  173. package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
  174. package/pipeline/scripts/smoke-md-links.sh +0 -8
  175. package/pipeline/scripts/smoke-md2confluence.sh +0 -126
  176. package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
  177. package/pipeline/scripts/smoke-migrate-state.sh +0 -102
  178. package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
  179. package/pipeline/scripts/smoke-model-fallback.sh +0 -89
  180. package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
  181. package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
  182. package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -194
  183. package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
  184. package/pipeline/scripts/smoke-own-punctuation.sh +0 -103
  185. package/pipeline/scripts/smoke-pack-contents.sh +0 -140
  186. package/pipeline/scripts/smoke-pat-audit.sh +0 -128
  187. package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
  188. package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
  189. package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
  190. package/pipeline/scripts/smoke-phase-banner.sh +0 -101
  191. package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
  192. package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
  193. package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
  194. package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
  195. package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
  196. package/pipeline/scripts/smoke-plan-safety.sh +0 -139
  197. package/pipeline/scripts/smoke-plan-todos.sh +0 -196
  198. package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
  199. package/pipeline/scripts/smoke-pre-commit.sh +0 -170
  200. package/pipeline/scripts/smoke-pref-migration.sh +0 -226
  201. package/pipeline/scripts/smoke-prefs-language.sh +0 -134
  202. package/pipeline/scripts/smoke-progress-contract.sh +0 -127
  203. package/pipeline/scripts/smoke-prune-logs.sh +0 -137
  204. package/pipeline/scripts/smoke-purge.sh +0 -138
  205. package/pipeline/scripts/smoke-push-retry.sh +0 -75
  206. package/pipeline/scripts/smoke-repo-map.sh +0 -300
  207. package/pipeline/scripts/smoke-review-readiness.sh +0 -92
  208. package/pipeline/scripts/smoke-review-watch.sh +0 -146
  209. package/pipeline/scripts/smoke-routines.sh +0 -84
  210. package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
  211. package/pipeline/scripts/smoke-run-metrics.sh +0 -50
  212. package/pipeline/scripts/smoke-search.sh +0 -187
  213. package/pipeline/scripts/smoke-shadow-git.sh +0 -224
  214. package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
  215. package/pipeline/scripts/smoke-skill-language.sh +0 -83
  216. package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
  217. package/pipeline/scripts/smoke-skill-scan.sh +0 -198
  218. package/pipeline/scripts/smoke-source-parity.sh +0 -85
  219. package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
  220. package/pipeline/scripts/smoke-sync-parity.sh +0 -92
  221. package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
  222. package/pipeline/scripts/smoke-telemetry.sh +0 -147
  223. package/pipeline/scripts/smoke-test-gap.sh +0 -183
  224. package/pipeline/scripts/smoke-token-budget.sh +0 -67
  225. package/pipeline/scripts/smoke-token-preflight.sh +0 -82
  226. package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
  227. package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
  228. package/pipeline/scripts/smoke-triage-memory.sh +0 -174
  229. package/pipeline/scripts/smoke-update-check.sh +0 -135
  230. package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
  231. package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
  232. package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
  233. package/pipeline/scripts/smoke-validator-gates.sh +0 -164
  234. package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
  235. package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
  236. package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
  237. package/pipeline/scripts/smoke-work-summary.sh +0 -163
  238. package/pipeline/scripts/smoke-workflow-audit.sh +0 -101
  239. package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
  240. package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
  241. package/pipeline/scripts/smoke-write-state.sh +0 -159
  242. package/pipeline/scripts/sync-parity-check.sh +0 -135
  243. package/pipeline/scripts/test-gap-rules/android.json +0 -25
  244. package/pipeline/scripts/test-gap-rules/ios.json +0 -34
  245. package/pipeline/scripts/test-gap-rules/node.json +0 -29
  246. package/pipeline/scripts/test-gap-rules/python.json +0 -25
  247. package/pipeline/scripts/validate-schemas.mjs +0 -88
@@ -1,147 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-diff-explain.sh - v7.8.0 Paket B: Phase 4 triage to diff bridge.
3
- #
4
- # Validates:
5
- # 1. Slash command + Copilot peer present (parity smoke also catches this)
6
- # 2. Empty triage produces a "no findings" stub without crashing
7
- # 3. Mixed-classification fixture renders all bucket icons (✅ ⏭️ ❌)
8
- # 4. Severity icons (🚫 ⚠️ 💡) render correctly
9
- # 5. Hunk binding: synthetic diff + matching line → diff fence in output
10
- # 6. Out-of-diff line → fallback message (no crash)
11
- # 7. --help works without inputs
12
- # 8. Missing inputs exit 1 with a clear error
13
- # 9. Read-only contract: script does NOT call git apply / git checkout / write APIs
14
-
15
- set -uo pipefail
16
-
17
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
18
- SCRIPT="$REPO_ROOT/pipeline/scripts/diff-explain.mjs"
19
-
20
- PASS=0
21
- FAIL=0
22
- pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
23
- fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
24
-
25
- cleanup() { [ -n "${TMP:-}" ] && rm -rf "$TMP"; }
26
- trap cleanup EXIT
27
- TMP=$(mktemp -d)
28
-
29
- # ──────────────────────────────────────────────────────────────────────────
30
- echo "→ 1. Command + skill files exist"
31
- [ -f "$REPO_ROOT/pipeline/commands/multi-agent/diff-explain/SKILL.md" ] && pass "diff-explain.md present" || fail "diff-explain.md missing"
32
- [ -f "$REPO_ROOT/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md" ] && pass "multi-agent-diff-explain SKILL.md present" || fail "Copilot peer missing"
33
- [ -x "$SCRIPT" ] && pass "diff-explain.mjs is executable" || fail "diff-explain.mjs not executable"
34
-
35
- # ──────────────────────────────────────────────────────────────────────────
36
- echo "→ 2. Empty findings stub"
37
- cat > "$TMP/empty.json" <<'EOF'
38
- {"accepted": [], "deferred": [], "rejected": [], "approved": true}
39
- EOF
40
- OUT=$(node "$SCRIPT" --triage "$TMP/empty.json" --diff /dev/null 2>&1)
41
- if echo "$OUT" | grep -q "no findings to map"; then
42
- pass "empty triage → 'no findings' stub"
43
- else
44
- fail "empty triage did not produce stub message"
45
- fi
46
-
47
- # ──────────────────────────────────────────────────────────────────────────
48
- echo "→ 3. Bucket icons render correctly"
49
- OUT=$(node "$SCRIPT" --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" --diff /dev/null 2>&1)
50
- echo "$OUT" | grep -q "✅" && pass "accepted bucket icon (✅) rendered" || fail "accepted icon missing"
51
- echo "$OUT" | grep -q "⏭️" && pass "deferred bucket icon (⏭️) rendered" || fail "deferred icon missing"
52
-
53
- # ──────────────────────────────────────────────────────────────────────────
54
- echo "→ 4. Severity icons render"
55
- echo "$OUT" | grep -q "🚫" && pass "blocking severity icon (🚫) rendered" || fail "blocking icon missing"
56
- echo "$OUT" | grep -q "⚠️" && pass "important severity icon (⚠️) rendered" || fail "important icon missing"
57
- echo "$OUT" | grep -q "💡" && pass "suggestion severity icon (💡) rendered" || fail "suggestion icon missing"
58
-
59
- # ──────────────────────────────────────────────────────────────────────────
60
- echo "→ 5. Hunk binding - synthetic diff matches finding line"
61
- cat > "$TMP/syn.diff" <<'EOF'
62
- diff --git a/src/api/checkout.ts b/src/api/checkout.ts
63
- index a..b 100644
64
- --- a/src/api/checkout.ts
65
- +++ b/src/api/checkout.ts
66
- @@ -98,5 +98,7 @@ ...
67
- const total = sum(items);
68
- + const total2 = sum(items);
69
- + // duplicate
70
- await chargeCard(total);
71
- EOF
72
- OUT=$(node "$SCRIPT" --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" --diff "$TMP/syn.diff" 2>&1)
73
- if echo "$OUT" | grep -q '```diff'; then
74
- pass "hunk fenced diff block rendered"
75
- else
76
- fail "no diff code fence in output"
77
- fi
78
- if echo "$OUT" | grep -q "@@ -98,5 +98,7 @@"; then
79
- pass "hunk header preserved in output"
80
- else
81
- fail "hunk header missing in output"
82
- fi
83
-
84
- # ──────────────────────────────────────────────────────────────────────────
85
- echo "→ 6. Out-of-diff line fallback"
86
- # checkout.ts is in the diff but its lines 145, 200 are NOT covered by the
87
- # single hunk above. Should produce fallback messages, not crash.
88
- fallback_count=$(echo "$OUT" | grep -c "line not in diff" || true)
89
- if [ "$fallback_count" -ge 1 ]; then
90
- pass "out-of-diff lines produce fallback message ($fallback_count occurrences)"
91
- else
92
- # Acceptable alternative: hunk renderer outputs nothing visible. Both options OK.
93
- pass "out-of-diff lines handled (no crash)"
94
- fi
95
-
96
- # ──────────────────────────────────────────────────────────────────────────
97
- echo "→ 7. --help works without inputs"
98
- HELP=$(node "$SCRIPT" --help 2>&1)
99
- if echo "$HELP" | grep -q "Usage:"; then
100
- pass "--help prints usage"
101
- else
102
- fail "--help missing Usage block"
103
- fi
104
-
105
- # ──────────────────────────────────────────────────────────────────────────
106
- echo "→ 8. Missing inputs exit non-zero"
107
- set +e
108
- node "$SCRIPT" 2>"$TMP/err.txt"
109
- EC=$?
110
- set -e
111
- if [ "$EC" -ne 0 ] && grep -q "required" "$TMP/err.txt"; then
112
- pass "no-input exit $EC with 'required' error"
113
- else
114
- fail "no-input did not error correctly (exit=$EC)"
115
- fi
116
-
117
- # ──────────────────────────────────────────────────────────────────────────
118
- echo "→ 9. Read-only contract - no destructive git APIs"
119
- if grep -qE "git apply|git checkout|git reset|git rebase|git push|git commit|writeFileSync" "$SCRIPT"; then
120
- fail "diff-explain.mjs references destructive git API or writeFileSync (read-only contract violated)"
121
- else
122
- pass "no destructive git/write API references in diff-explain.mjs"
123
- fi
124
-
125
- # ──────────────────────────────────────────────────────────────────────────
126
- echo "→ 10. Injection-safe - shell metacharacters in --base are not executed"
127
- SENTINEL="$TMP/PWNED"
128
- set +e
129
- node "$SCRIPT" \
130
- --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" \
131
- --base "main; touch $SENTINEL" --branch HEAD >/dev/null 2>&1
132
- set -e
133
- if [ -f "$SENTINEL" ]; then
134
- fail "command injection: --base executed arbitrary shell (sentinel created)"
135
- else
136
- pass "shell metacharacters in --base not executed (execFileSync)"
137
- fi
138
- if grep -q "execSync" "$SCRIPT"; then
139
- fail "diff-explain.mjs still uses execSync (shell interpolation risk)"
140
- else
141
- pass "diff-explain.mjs uses execFileSync only"
142
- fi
143
-
144
- # ──────────────────────────────────────────────────────────────────────────
145
- echo ""
146
- echo "══ diff-explain smoke: $PASS passed, $FAIL failed ══"
147
- [ "$FAIL" -eq 0 ]
@@ -1,190 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-diff-risk.sh - v8.3.0
3
- #
4
- # Verifies the Phase 4 advisory diff-risk pipeline:
5
- # 1. diff-risk-score.mjs runs against the iOS fixture and produces valid JSON
6
- # 2. KeychainStore (security path) ranks first
7
- # 3. Tests file ranks last (no security/public_api/migration signal)
8
- # 4. validate-diff-risk.mjs accepts the output, rejects malformed
9
- # 5. Android fixture: migration outranks AuthRepository outranks HomeScreen
10
- # 6. --top truncation honored
11
- # 7. Empty diff returns exit 2
12
- # 8. phase-4-review.md ref doc declares Step 1.75 + diff-risk-score.mjs
13
- # 9. code-reviewer.md agent template carries the priority-files placeholder
14
- # 10. prefs.schema.json exposes diffRisk advisory toggle
15
- # 11. test-removal fixture fires the test_lines_removed signal (v1.1.0)
16
- #
17
- # Exit 0 = all pass, 1 = any failure.
18
-
19
- set -euo pipefail
20
-
21
- ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
22
- SCORE="$ROOT/pipeline/scripts/diff-risk-score.mjs"
23
- VALIDATE="$ROOT/pipeline/scripts/validate-diff-risk.mjs"
24
- SCHEMA="$ROOT/pipeline/schemas/diff-risk.schema.json"
25
- PHASE4="$ROOT/pipeline/multi-agent-refs/phases/phase-4-review.md"
26
- REVIEWER="$ROOT/pipeline/agents/code-reviewer.md"
27
- PREFS="$ROOT/pipeline/schemas/prefs.schema.json"
28
- FIX_IOS="$ROOT/pipeline/scripts/fixtures/diff-risk-ios.diff"
29
- FIX_AND="$ROOT/pipeline/scripts/fixtures/diff-risk-android.diff"
30
- FIX_TESTRM="$ROOT/pipeline/scripts/fixtures/diff-risk-test-removal.diff"
31
-
32
- pass=0
33
- fail=0
34
- failures=()
35
- record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
36
- record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
37
-
38
- printf '→ smoke-diff-risk (v8.3.0): pre-review risk scoring contract\n'
39
-
40
- [ -f "$SCHEMA" ] || { record_fail "schema missing: $SCHEMA"; exit 1; }
41
- [ -f "$FIX_IOS" ] || { record_fail "fixture missing: $FIX_IOS"; exit 1; }
42
- [ -f "$FIX_AND" ] || { record_fail "fixture missing: $FIX_AND"; exit 1; }
43
- [ -f "$FIX_TESTRM" ] || { record_fail "fixture missing: $FIX_TESTRM"; exit 1; }
44
-
45
- # --- 1: iOS fixture produces JSON ---
46
- out_ios=$(node "$SCORE" --diff "$FIX_IOS" 2>/dev/null)
47
- if jq -e '.schemaVersion == "1.1.0"' <<< "$out_ios" >/dev/null 2>&1; then
48
- record_pass "iOS fixture renders schema-versioned JSON"
49
- else
50
- record_fail "iOS fixture JSON malformed or missing schemaVersion"
51
- fi
52
-
53
- # --- 2: KeychainStore ranks first ---
54
- top1=$(jq -r '.files[0].path' <<< "$out_ios")
55
- if [ "$top1" = "MyApp/Sources/Auth/KeychainStore.swift" ]; then
56
- record_pass "iOS top-1 = KeychainStore (security_path + public_api + no_test)"
57
- else
58
- record_fail "iOS top-1 should be KeychainStore, got: $top1"
59
- fi
60
-
61
- # --- 3: Test file ranks last ---
62
- last=$(jq -r '.files[-1].path' <<< "$out_ios")
63
- if [ "$last" = "MyAppTests/SettingsViewTests.swift" ]; then
64
- record_pass "iOS last = test file (no risk signals beyond loc_changed)"
65
- else
66
- record_fail "iOS last should be test file, got: $last"
67
- fi
68
-
69
- # --- 4: Validator accepts valid output, rejects malformed ---
70
- set +e
71
- echo "$out_ios" | node "$VALIDATE" - >/dev/null 2>&1
72
- rc_ok=$?
73
- echo '{"schemaVersion":"0.0.0","files":[]}' | node "$VALIDATE" - >/dev/null 2>&1
74
- rc_bad=$?
75
- set -e
76
- if [ "$rc_ok" -eq 0 ]; then
77
- record_pass "validator accepts valid output"
78
- else
79
- record_fail "validator rejected valid output (rc=$rc_ok)"
80
- fi
81
- if [ "$rc_bad" -ne 0 ]; then
82
- record_pass "validator rejects malformed input"
83
- else
84
- record_fail "validator should reject malformed (got rc=$rc_bad)"
85
- fi
86
-
87
- # --- 5: Android ranking ---
88
- out_and=$(node "$SCORE" --diff "$FIX_AND" 2>/dev/null)
89
- top1=$(jq -r '.files[0].path' <<< "$out_and")
90
- top2=$(jq -r '.files[1].path' <<< "$out_and")
91
- top3=$(jq -r '.files[2].path' <<< "$out_and")
92
- if [[ "$top1" == *"Migration_3_to_4.kt"* ]]; then
93
- record_pass "Android top-1 = migration file"
94
- else
95
- record_fail "Android top-1 should be migration, got: $top1"
96
- fi
97
- if [[ "$top2" == *"AuthRepository.kt"* ]]; then
98
- record_pass "Android top-2 = AuthRepository (security_path)"
99
- else
100
- record_fail "Android top-2 should be AuthRepository, got: $top2"
101
- fi
102
- if [[ "$top3" == *"HomeScreen.kt"* ]]; then
103
- record_pass "Android top-3 = HomeScreen (ui_critical)"
104
- else
105
- record_fail "Android top-3 should be HomeScreen, got: $top3"
106
- fi
107
-
108
- # --- 6: --top N truncates ---
109
- out_top2=$(node "$SCORE" --diff "$FIX_IOS" --top 2 2>/dev/null)
110
- n=$(jq '.files | length' <<< "$out_top2")
111
- if [ "$n" = "2" ]; then
112
- record_pass "--top 2 truncates files array"
113
- else
114
- record_fail "--top 2 should keep 2 rows, got $n"
115
- fi
116
-
117
- # --- 7: Empty diff exits 2 ---
118
- empty=$(mktemp)
119
- set +e
120
- node "$SCORE" --diff "$empty" >/dev/null 2>&1
121
- rc=$?
122
- set -e
123
- rm -f "$empty"
124
- if [ "$rc" -eq 2 ]; then
125
- record_pass "empty diff exits 2"
126
- else
127
- record_fail "empty diff should exit 2, got $rc"
128
- fi
129
-
130
- # --- 8: phase-4-review.md declares the advisory step ---
131
- if grep -qE '^####? .*Diff Risk' "$PHASE4"; then
132
- record_pass "phase-4-review.md declares Diff Risk step"
133
- else
134
- record_fail "phase-4-review.md missing Diff Risk step heading"
135
- fi
136
- if grep -q 'diff-risk-score.mjs' "$PHASE4"; then
137
- record_pass "phase-4-review.md references diff-risk-score.mjs"
138
- else
139
- record_fail "phase-4-review.md missing script reference"
140
- fi
141
-
142
- # --- 9: code-reviewer.md template carries priority placeholder ---
143
- if grep -q 'Priority Files' "$REVIEWER"; then
144
- record_pass "code-reviewer.md declares Priority Files section"
145
- else
146
- record_fail "code-reviewer.md missing Priority Files section"
147
- fi
148
-
149
- # --- 10: prefs schema exposes diffRisk advisory toggle ---
150
- if jq -e '.properties.global.properties.diffRiskAdvisory' "$PREFS" >/dev/null 2>&1; then
151
- record_pass "prefs schema exposes diffRiskAdvisory toggle"
152
- else
153
- record_fail "prefs.schema.json missing global.diffRiskAdvisory"
154
- fi
155
-
156
- # --- 11: test_lines_removed signal fires on the test-removal fixture ---
157
- out_testrm=$(node "$SCORE" --diff "$FIX_TESTRM" 2>/dev/null)
158
- sig_value=$(jq -r '.files[] | select(.path == "MyAppTests/LoginViewModelTests.swift")
159
- | .signals[] | select(.name == "test_lines_removed") | .value' <<< "$out_testrm")
160
- if [ "$sig_value" = "16" ]; then
161
- record_pass "test_lines_removed fires with value=16 (18 removed - 2 added)"
162
- else
163
- record_fail "test_lines_removed should fire with value=16, got: ${sig_value:-missing}"
164
- fi
165
- sig_on_source=$(jq -r '[.files[] | select(.path == "MyApp/Sources/Auth/LoginViewModel.swift")
166
- | .signals[] | select(.name == "test_lines_removed")] | length' <<< "$out_testrm")
167
- if [ "$sig_on_source" = "0" ]; then
168
- record_pass "test_lines_removed does not fire on source files"
169
- else
170
- record_fail "test_lines_removed must only fire on test-classified paths"
171
- fi
172
- set +e
173
- echo "$out_testrm" | node "$VALIDATE" - >/dev/null 2>&1
174
- rc_testrm=$?
175
- set -e
176
- if [ "$rc_testrm" -eq 0 ]; then
177
- record_pass "validator accepts output carrying test_lines_removed"
178
- else
179
- record_fail "validator rejected test_lines_removed output (rc=$rc_testrm)"
180
- fi
181
-
182
- # --- Summary ---
183
- total=$((pass + fail))
184
- printf '\n→ smoke-diff-risk: %d/%d passed\n' "$pass" "$total"
185
- if [ "$fail" -ne 0 ]; then
186
- printf '\nFailures:\n'
187
- for f in "${failures[@]}"; do printf ' - %s\n' "$f"; done
188
- exit 1
189
- fi
190
- exit 0
@@ -1,160 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-dynamic-skill-loading.sh - v7.0.I
3
- #
4
- # Verifies the index + matcher contract:
5
- # 1. build-skills-index.mjs produces valid JSON with schemaVersion + entries
6
- # 2. index skill count matches the raw filesystem count
7
- # 3. match-skills.mjs ranks iOS skills higher on a SwiftUI task
8
- # 4. match-skills.mjs --json returns sorted matches[]
9
- # 5. matcher platform boost applies when --stack matches frontmatter
10
- # 6. matcher returns empty matches[] for a task with no relevant skills
11
- # 7. matcher --limit honored
12
- # 8. prefs schema exposes dynamicSkillLoading default false
13
- #
14
- # Exit 0 all pass, 1 any failure.
15
-
16
- set -euo pipefail
17
-
18
- ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
19
- BUILD="$ROOT/pipeline/scripts/build-skills-index.mjs"
20
- MATCH="$ROOT/pipeline/scripts/match-skills.mjs"
21
- SCHEMA="$ROOT/pipeline/schemas/prefs.schema.json"
22
-
23
- pass=0; fail=0; failures=()
24
- record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
25
- record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
26
-
27
- printf '→ smoke-dynamic-skill-loading (v7.0.I): index + matcher contract\n'
28
-
29
- # Build into a sandbox so we don't clobber the shipping index
30
- sandbox=$(mktemp -d)
31
- skills="$sandbox/skills"
32
- mkdir -p "$skills/core/multi-agent" "$skills/core/testing-backend" "$skills/external/swiftui-pro" "$skills/external/compose-components"
33
-
34
- cat > "$skills/core/multi-agent/SKILL.md" <<'S'
35
- ---
36
- name: multi-agent
37
- description: Task orchestrator for the multi-agent pipeline
38
- platform: generic
39
- ---
40
- body
41
- S
42
-
43
- cat > "$skills/core/testing-backend/SKILL.md" <<'S'
44
- ---
45
- name: testing-backend
46
- description: Backend testing patterns for pytest / Jest
47
- platform: backend
48
- trigger-keywords: pytest, jest, integration test
49
- ---
50
- body
51
- S
52
-
53
- cat > "$skills/external/swiftui-pro/SKILL.md" <<'S'
54
- ---
55
- name: swiftui-pro
56
- description: SwiftUI review for modern APIs, state management, performance
57
- platform: ios
58
- trigger-keywords: swiftui, swift, ios, swiftui performance
59
- trigger-paths: *.swift, **/Sources/**/*.swift
60
- ---
61
- body
62
- S
63
-
64
- cat > "$skills/external/compose-components/SKILL.md" <<'S'
65
- ---
66
- name: compose-components
67
- description: Material 3 Jetpack Compose components
68
- platform: android
69
- trigger-keywords: compose, jetpack, material 3, android
70
- trigger-paths: *.kt, **/src/main/**/*.kt
71
- ---
72
- body
73
- S
74
-
75
- # --- 1: build index ---
76
- node "$BUILD" --root "$skills" >/dev/null
77
- INDEX="$skills/.skills-index.json"
78
- if [ -f "$INDEX" ] && jq -e '.schemaVersion and .entries and (.skillCount == 4)' "$INDEX" >/dev/null; then
79
- record_pass "build-skills-index emits valid JSON with 4 entries"
80
- else
81
- record_fail "build-skills-index output malformed"
82
- fi
83
-
84
- # --- 2: count matches filesystem ---
85
- raw_count=$(find "$skills" -name SKILL.md | wc -l | tr -d ' ')
86
- index_count=$(jq -r '.skillCount' "$INDEX")
87
- if [ "$raw_count" = "$index_count" ]; then
88
- record_pass "index count matches filesystem ($index_count)"
89
- else
90
- record_fail "index count $index_count != filesystem $raw_count"
91
- fi
92
-
93
- # --- 3: iOS task ranks SwiftUI skills higher ---
94
- out=$(node "$MATCH" "SwiftUI dark mode fix on LoginView" \
95
- --touched-files "Sources/Auth/LoginView.swift" \
96
- --stack ios --limit 4 --index "$INDEX" --json)
97
- top=$(jq -r '.matches[0].name' <<< "$out")
98
- if [ "$top" = "swiftui-pro" ]; then
99
- record_pass "iOS task ranks swiftui-pro top ($top)"
100
- else
101
- record_fail "iOS task should rank swiftui-pro first, got $top"
102
- fi
103
-
104
- # --- 4: JSON output is sorted ---
105
- scores=$(jq -r '.matches[].score' <<< "$out")
106
- prev=9999; sorted=1
107
- for s in $scores; do
108
- if [ "$s" -gt "$prev" ]; then sorted=0; break; fi
109
- prev=$s
110
- done
111
- if [ "$sorted" = "1" ]; then
112
- record_pass "JSON matches[] sorted by score desc"
113
- else
114
- record_fail "matches[] not sorted"
115
- fi
116
-
117
- # --- 5: platform boost ---
118
- # A keyword-free task with only --stack android should still boost Android skill
119
- out=$(node "$MATCH" "Build the feature end to end" --stack android --limit 4 --index "$INDEX" --json)
120
- rules=$(jq -r '.matches[] | select(.name=="compose-components") | .reasons[].rule' <<< "$out" 2>/dev/null || echo "")
121
- if echo "$rules" | grep -q 'platform'; then
122
- record_pass "platform boost reason present for matched stack"
123
- else
124
- record_fail "platform boost missing for --stack android"
125
- fi
126
-
127
- # --- 6: empty matches ---
128
- out=$(node "$MATCH" "Totally unrelated lmn xyz qrs foobar baz" --limit 4 --index "$INDEX" --json)
129
- n=$(jq '.matches | length' <<< "$out")
130
- if [ "$n" = "0" ]; then
131
- record_pass "empty matches[] for unrelated task"
132
- else
133
- record_fail "expected 0 matches for unrelated task, got $n"
134
- fi
135
-
136
- # --- 7: --limit honored ---
137
- out=$(node "$MATCH" "swiftui compose test pipeline" --limit 2 --index "$INDEX" --json)
138
- n=$(jq '.matches | length' <<< "$out")
139
- if [ "$n" -le "2" ]; then
140
- record_pass "--limit 2 returns at most 2 matches ($n)"
141
- else
142
- record_fail "--limit 2 returned $n"
143
- fi
144
-
145
- # --- 8: schema pref ---
146
- if jq -e '.properties.global.properties.dynamicSkillLoading.default == false' "$SCHEMA" >/dev/null 2>&1; then
147
- record_pass "prefs schema exposes dynamicSkillLoading default false"
148
- else
149
- record_fail "prefs schema missing dynamicSkillLoading or wrong default"
150
- fi
151
-
152
- rm -rf "$sandbox"
153
-
154
- printf '\n══ dynamic-skill-loading smoke: %d passed, %d failed ══\n' "$pass" "$fail"
155
- if [ "$fail" -gt 0 ]; then
156
- printf '\nFailures:\n'
157
- for m in "${failures[@]}"; do printf ' - %s\n' "$m"; done
158
- exit 1
159
- fi
160
- exit 0
@@ -1,136 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-eval-live.sh - v7.8.0 Paket A: opt-in live golden-task runner.
3
- #
4
- # Validates the cost-guarding contract WITHOUT making any model calls. Live
5
- # mode is gated behind two independent signals (--live flag AND
6
- # MULTI_AGENT_LIVE_EVAL=1 env) so accidental cost is impossible during CI
7
- # or local `npm test` runs.
8
- #
9
- # All assertions exercise the dry-run code path or expect early-exit before
10
- # any model invocation.
11
-
12
- set -uo pipefail
13
-
14
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
15
- SCRIPT="$REPO_ROOT/pipeline/scripts/eval-golden-tasks-live.mjs"
16
-
17
- PASS=0
18
- FAIL=0
19
- pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
20
- fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
21
-
22
- cleanup() { [ -n "${TMP:-}" ] && rm -rf "$TMP"; }
23
- trap cleanup EXIT
24
- TMP=$(mktemp -d)
25
-
26
- # Strip env that would unlock live mode; smoke MUST never spend money.
27
- unset MULTI_AGENT_LIVE_EVAL
28
-
29
- # ──────────────────────────────────────────────────────────────────────────
30
- echo "→ 1. Script presence + syntax"
31
- [ -x "$SCRIPT" ] && pass "eval-golden-tasks-live.mjs is executable" || fail "script not executable"
32
- node --check "$SCRIPT" 2>/dev/null && pass "syntax check" || fail "syntax error"
33
-
34
- # ──────────────────────────────────────────────────────────────────────────
35
- echo "→ 2. Default mode is dry-run (no --live, no env)"
36
- OUT=$(node "$SCRIPT" --json 2>/dev/null)
37
- if echo "$OUT" | grep -q '"mode": "dry-run"'; then
38
- pass "default mode = dry-run"
39
- else
40
- fail "default mode not dry-run: $(echo "$OUT" | head -3)"
41
- fi
42
-
43
- # ──────────────────────────────────────────────────────────────────────────
44
- echo "→ 3. Dry-run does not invoke model"
45
- # All cases must end with status "skipped-dry-run" - i.e. no live call
46
- SKIPPED_COUNT=$(echo "$OUT" | grep -c '"pass": "skipped-dry-run"' || true)
47
- if [ "$SKIPPED_COUNT" -ge 1 ]; then
48
- pass "dry-run skipped all cases ($SKIPPED_COUNT) without invoking model"
49
- else
50
- fail "dry-run did not skip cases properly"
51
- fi
52
-
53
- # ──────────────────────────────────────────────────────────────────────────
54
- echo "→ 4. --live without env exits 2 (cost gate)"
55
- node "$SCRIPT" --live 2>"$TMP/err.txt"
56
- EC=$?
57
- if [ "$EC" -eq 2 ] && grep -q "MULTI_AGENT_LIVE_EVAL" "$TMP/err.txt"; then
58
- pass "--live without env: exit $EC + cost gate message"
59
- else
60
- fail "--live cost gate weak (exit=$EC, err='$(cat "$TMP/err.txt")')"
61
- fi
62
-
63
- # ──────────────────────────────────────────────────────────────────────────
64
- echo "→ 5. Static check - no fall-through past cost gate"
65
- # The cost gate must be the first thing checked after flag parsing. If --live
66
- # is set without env, no model call should be reachable. Verify by static
67
- # grep: spawnSync("claude", ...) must come AFTER the env-check block.
68
- GATE_LINE=$(grep -n 'MULTI_AGENT_LIVE_EVAL' "$SCRIPT" | head -1 | cut -d: -f1)
69
- CALL_LINE=$(grep -n 'spawnSync("claude"' "$SCRIPT" | head -1 | cut -d: -f1)
70
- if [ -n "$GATE_LINE" ] && [ -n "$CALL_LINE" ] && [ "$GATE_LINE" -lt "$CALL_LINE" ]; then
71
- pass "cost gate at line $GATE_LINE precedes claude call at line $CALL_LINE"
72
- else
73
- fail "cost gate ordering wrong (gate=$GATE_LINE, call=$CALL_LINE)"
74
- fi
75
-
76
- # ──────────────────────────────────────────────────────────────────────────
77
- echo "→ 6. Budget cap honored - total-budget < single case budget skips all"
78
- OUT=$(node "$SCRIPT" --max-cases=5 --budget=0.5 --total-budget=0.1 --json 2>/dev/null)
79
- RUN_COUNT=$(echo "$OUT" | grep -oE '"cases_run":[[:space:]]*[0-9]+' | grep -oE '[0-9]+')
80
- SKIPPED=$(echo "$OUT" | grep -c '"skipped-budget-exhausted"' || true)
81
- if [ "$SKIPPED" -ge 1 ]; then
82
- pass "budget exhaustion produces 'skipped-budget-exhausted' result ($SKIPPED times)"
83
- else
84
- pass "budget very tight: cases_run=$RUN_COUNT (acceptable in dry-run mode)"
85
- fi
86
-
87
- # ──────────────────────────────────────────────────────────────────────────
88
- echo "→ 7. --max-cases caps the run"
89
- OUT=$(node "$SCRIPT" --max-cases=1 --json 2>/dev/null)
90
- RUN_COUNT=$(echo "$OUT" | grep -oE '"cases_run":[[:space:]]*[0-9]+' | grep -oE '[0-9]+')
91
- if [ "$RUN_COUNT" = "1" ]; then
92
- pass "--max-cases=1 limits cases_run to 1"
93
- else
94
- fail "--max-cases not honored: cases_run=$RUN_COUNT"
95
- fi
96
-
97
- # ──────────────────────────────────────────────────────────────────────────
98
- echo "→ 8. --case selector"
99
- OUT=$(node "$SCRIPT" --case 01-ios-bugfix-darkmode --json 2>/dev/null)
100
- if echo "$OUT" | grep -q '"case": "01-ios-bugfix-darkmode"'; then
101
- pass "--case selector picks the right fixture"
102
- else
103
- fail "--case selector did not target fixture"
104
- fi
105
-
106
- node "$SCRIPT" --case nonexistent-case 2>"$TMP/err.txt"
107
- EC=$?
108
- if [ "$EC" -eq 2 ] && grep -q "not found" "$TMP/err.txt"; then
109
- pass "unknown --case exits 2 with 'not found'"
110
- else
111
- fail "unknown --case error weak (exit=$EC)"
112
- fi
113
-
114
- # ──────────────────────────────────────────────────────────────────────────
115
- echo "→ 9. Help text mentions cost gate prominently"
116
- HELP=$(node "$SCRIPT" --help 2>&1)
117
- if echo "$HELP" | grep -q "MULTI_AGENT_LIVE_EVAL"; then
118
- pass "--help documents the cost gate env var"
119
- else
120
- fail "--help missing cost gate documentation"
121
- fi
122
-
123
- # ──────────────────────────────────────────────────────────────────────────
124
- echo "→ 10. Bad --budget exits 2"
125
- node "$SCRIPT" --budget=invalid 2>/dev/null
126
- EC=$?
127
- if [ "$EC" -eq 2 ]; then
128
- pass "invalid --budget exits 2"
129
- else
130
- fail "invalid --budget did not error (exit=$EC)"
131
- fi
132
-
133
- # ──────────────────────────────────────────────────────────────────────────
134
- echo ""
135
- echo "══ eval-live smoke: $PASS passed, $FAIL failed ══"
136
- [ "$FAIL" -eq 0 ]