@mmerterden/multi-agent-pipeline 12.7.0 → 12.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (244) hide show
  1. package/CHANGELOG.md +126 -0
  2. package/install/_common.mjs +48 -0
  3. package/install/_dev-only-files.mjs +125 -5
  4. package/install/claude.mjs +14 -8
  5. package/install/copilot.mjs +5 -8
  6. package/package.json +17 -2
  7. package/pipeline/commands/multi-agent/analysis/SKILL.md +1 -1
  8. package/pipeline/lib/credential-store.sh +20 -0
  9. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  10. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  11. package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
  12. package/pipeline/multi-agent-refs/phases/operations.md +28 -0
  13. package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -0
  14. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  15. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
  16. package/pipeline/multi-agent-refs/phases/phase-4-review.md +49 -4
  17. package/pipeline/schemas/prefs.schema.json +6 -0
  18. package/pipeline/scripts/_smoke-root.sh +61 -0
  19. package/pipeline/scripts/audit-log.sh +25 -0
  20. package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
  21. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
  22. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
  23. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
  24. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
  25. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
  26. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
  27. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
  28. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
  29. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
  30. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
  31. package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
  32. package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
  33. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
  34. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
  35. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
  36. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
  37. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
  38. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
  39. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
  40. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
  41. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
  42. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
  43. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
  44. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
  45. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
  46. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
  47. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
  48. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
  49. package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
  50. package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
  51. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
  52. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
  53. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
  54. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
  55. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
  56. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
  57. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
  58. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
  59. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
  60. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
  61. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
  62. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
  63. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
  64. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
  65. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
  66. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
  67. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
  68. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
  69. package/pipeline/eval/golden-tasks/README.md +0 -65
  70. package/pipeline/eval/intent-cases.json +0 -40
  71. package/pipeline/eval/run-metrics-fixture.json +0 -93
  72. package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
  73. package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
  74. package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
  75. package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
  76. package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
  77. package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
  78. package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
  79. package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
  80. package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
  81. package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
  82. package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
  83. package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
  84. package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
  85. package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
  86. package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
  87. package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
  88. package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
  89. package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
  90. package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
  91. package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
  92. package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
  93. package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
  94. package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
  95. package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
  96. package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
  97. package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
  98. package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
  99. package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
  100. package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
  101. package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
  102. package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
  103. package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
  104. package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
  105. package/pipeline/eval/triage/README.md +0 -54
  106. package/pipeline/scripts/benchmark-phase-0.sh +0 -128
  107. package/pipeline/scripts/check-md-links.mjs +0 -88
  108. package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -302
  109. package/pipeline/scripts/eval-golden-tasks.mjs +0 -224
  110. package/pipeline/scripts/eval-intent.mjs +0 -107
  111. package/pipeline/scripts/eval-mine-corpus.mjs +0 -211
  112. package/pipeline/scripts/eval-triage.mjs +0 -171
  113. package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
  114. package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
  115. package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
  116. package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
  117. package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
  118. package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
  119. package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
  120. package/pipeline/scripts/lint-mcp-refs.mjs +0 -218
  121. package/pipeline/scripts/lint-skills.mjs +0 -154
  122. package/pipeline/scripts/run-smokes.mjs +0 -130
  123. package/pipeline/scripts/scorecard.mjs +0 -258
  124. package/pipeline/scripts/smoke-add-detail.sh +0 -137
  125. package/pipeline/scripts/smoke-agent-guard.sh +0 -74
  126. package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
  127. package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
  128. package/pipeline/scripts/smoke-ask-choice.sh +0 -42
  129. package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
  130. package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
  131. package/pipeline/scripts/smoke-changelog-version.sh +0 -47
  132. package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
  133. package/pipeline/scripts/smoke-channels-flow.sh +0 -130
  134. package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
  135. package/pipeline/scripts/smoke-clarify.sh +0 -148
  136. package/pipeline/scripts/smoke-command-inventory.sh +0 -81
  137. package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
  138. package/pipeline/scripts/smoke-community-gates.sh +0 -75
  139. package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
  140. package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
  141. package/pipeline/scripts/smoke-context-budget.sh +0 -72
  142. package/pipeline/scripts/smoke-cost-budget.sh +0 -70
  143. package/pipeline/scripts/smoke-cost-summary.sh +0 -139
  144. package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
  145. package/pipeline/scripts/smoke-description-tr.sh +0 -82
  146. package/pipeline/scripts/smoke-dev-critic.sh +0 -144
  147. package/pipeline/scripts/smoke-diff-explain.sh +0 -147
  148. package/pipeline/scripts/smoke-diff-risk.sh +0 -190
  149. package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
  150. package/pipeline/scripts/smoke-eval-live.sh +0 -136
  151. package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
  152. package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
  153. package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
  154. package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
  155. package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
  156. package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
  157. package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
  158. package/pipeline/scripts/smoke-generate-issue.sh +0 -120
  159. package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
  160. package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
  161. package/pipeline/scripts/smoke-install-layout.sh +0 -248
  162. package/pipeline/scripts/smoke-intent-guard.sh +0 -86
  163. package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
  164. package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
  165. package/pipeline/scripts/smoke-keychain.sh +0 -158
  166. package/pipeline/scripts/smoke-language-axis.sh +0 -109
  167. package/pipeline/scripts/smoke-learning-curve.sh +0 -61
  168. package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
  169. package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
  170. package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
  171. package/pipeline/scripts/smoke-md-links.sh +0 -8
  172. package/pipeline/scripts/smoke-md2confluence.sh +0 -126
  173. package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
  174. package/pipeline/scripts/smoke-migrate-state.sh +0 -102
  175. package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
  176. package/pipeline/scripts/smoke-model-fallback.sh +0 -89
  177. package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
  178. package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
  179. package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -194
  180. package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
  181. package/pipeline/scripts/smoke-own-punctuation.sh +0 -103
  182. package/pipeline/scripts/smoke-pack-contents.sh +0 -140
  183. package/pipeline/scripts/smoke-pat-audit.sh +0 -128
  184. package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
  185. package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
  186. package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
  187. package/pipeline/scripts/smoke-phase-banner.sh +0 -101
  188. package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
  189. package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
  190. package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
  191. package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
  192. package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
  193. package/pipeline/scripts/smoke-plan-safety.sh +0 -139
  194. package/pipeline/scripts/smoke-plan-todos.sh +0 -196
  195. package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
  196. package/pipeline/scripts/smoke-pre-commit.sh +0 -170
  197. package/pipeline/scripts/smoke-pref-migration.sh +0 -226
  198. package/pipeline/scripts/smoke-prefs-language.sh +0 -134
  199. package/pipeline/scripts/smoke-progress-contract.sh +0 -127
  200. package/pipeline/scripts/smoke-prune-logs.sh +0 -137
  201. package/pipeline/scripts/smoke-purge.sh +0 -138
  202. package/pipeline/scripts/smoke-push-retry.sh +0 -75
  203. package/pipeline/scripts/smoke-repo-map.sh +0 -300
  204. package/pipeline/scripts/smoke-review-readiness.sh +0 -92
  205. package/pipeline/scripts/smoke-review-watch.sh +0 -146
  206. package/pipeline/scripts/smoke-routines.sh +0 -84
  207. package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
  208. package/pipeline/scripts/smoke-run-metrics.sh +0 -50
  209. package/pipeline/scripts/smoke-search.sh +0 -187
  210. package/pipeline/scripts/smoke-shadow-git.sh +0 -224
  211. package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
  212. package/pipeline/scripts/smoke-skill-language.sh +0 -83
  213. package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
  214. package/pipeline/scripts/smoke-skill-scan.sh +0 -198
  215. package/pipeline/scripts/smoke-source-parity.sh +0 -85
  216. package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
  217. package/pipeline/scripts/smoke-sync-parity.sh +0 -92
  218. package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
  219. package/pipeline/scripts/smoke-telemetry.sh +0 -147
  220. package/pipeline/scripts/smoke-test-gap.sh +0 -183
  221. package/pipeline/scripts/smoke-token-budget.sh +0 -67
  222. package/pipeline/scripts/smoke-token-preflight.sh +0 -82
  223. package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
  224. package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
  225. package/pipeline/scripts/smoke-triage-memory.sh +0 -174
  226. package/pipeline/scripts/smoke-update-check.sh +0 -135
  227. package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
  228. package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
  229. package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
  230. package/pipeline/scripts/smoke-validator-gates.sh +0 -164
  231. package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
  232. package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
  233. package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
  234. package/pipeline/scripts/smoke-work-summary.sh +0 -163
  235. package/pipeline/scripts/smoke-workflow-audit.sh +0 -101
  236. package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
  237. package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
  238. package/pipeline/scripts/smoke-write-state.sh +0 -159
  239. package/pipeline/scripts/sync-parity-check.sh +0 -135
  240. package/pipeline/scripts/test-gap-rules/android.json +0 -25
  241. package/pipeline/scripts/test-gap-rules/ios.json +0 -34
  242. package/pipeline/scripts/test-gap-rules/node.json +0 -29
  243. package/pipeline/scripts/test-gap-rules/python.json +0 -25
  244. package/pipeline/scripts/validate-schemas.mjs +0 -88
@@ -1,147 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-diff-explain.sh - v7.8.0 Paket B: Phase 4 triage to diff bridge.
3
- #
4
- # Validates:
5
- # 1. Slash command + Copilot peer present (parity smoke also catches this)
6
- # 2. Empty triage produces a "no findings" stub without crashing
7
- # 3. Mixed-classification fixture renders all bucket icons (✅ ⏭️ ❌)
8
- # 4. Severity icons (🚫 ⚠️ 💡) render correctly
9
- # 5. Hunk binding: synthetic diff + matching line → diff fence in output
10
- # 6. Out-of-diff line → fallback message (no crash)
11
- # 7. --help works without inputs
12
- # 8. Missing inputs exit 1 with a clear error
13
- # 9. Read-only contract: script does NOT call git apply / git checkout / write APIs
14
-
15
- set -uo pipefail
16
-
17
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
18
- SCRIPT="$REPO_ROOT/pipeline/scripts/diff-explain.mjs"
19
-
20
- PASS=0
21
- FAIL=0
22
- pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
23
- fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
24
-
25
- cleanup() { [ -n "${TMP:-}" ] && rm -rf "$TMP"; }
26
- trap cleanup EXIT
27
- TMP=$(mktemp -d)
28
-
29
- # ──────────────────────────────────────────────────────────────────────────
30
- echo "→ 1. Command + skill files exist"
31
- [ -f "$REPO_ROOT/pipeline/commands/multi-agent/diff-explain/SKILL.md" ] && pass "diff-explain.md present" || fail "diff-explain.md missing"
32
- [ -f "$REPO_ROOT/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md" ] && pass "multi-agent-diff-explain SKILL.md present" || fail "Copilot peer missing"
33
- [ -x "$SCRIPT" ] && pass "diff-explain.mjs is executable" || fail "diff-explain.mjs not executable"
34
-
35
- # ──────────────────────────────────────────────────────────────────────────
36
- echo "→ 2. Empty findings stub"
37
- cat > "$TMP/empty.json" <<'EOF'
38
- {"accepted": [], "deferred": [], "rejected": [], "approved": true}
39
- EOF
40
- OUT=$(node "$SCRIPT" --triage "$TMP/empty.json" --diff /dev/null 2>&1)
41
- if echo "$OUT" | grep -q "no findings to map"; then
42
- pass "empty triage → 'no findings' stub"
43
- else
44
- fail "empty triage did not produce stub message"
45
- fi
46
-
47
- # ──────────────────────────────────────────────────────────────────────────
48
- echo "→ 3. Bucket icons render correctly"
49
- OUT=$(node "$SCRIPT" --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" --diff /dev/null 2>&1)
50
- echo "$OUT" | grep -q "✅" && pass "accepted bucket icon (✅) rendered" || fail "accepted icon missing"
51
- echo "$OUT" | grep -q "⏭️" && pass "deferred bucket icon (⏭️) rendered" || fail "deferred icon missing"
52
-
53
- # ──────────────────────────────────────────────────────────────────────────
54
- echo "→ 4. Severity icons render"
55
- echo "$OUT" | grep -q "🚫" && pass "blocking severity icon (🚫) rendered" || fail "blocking icon missing"
56
- echo "$OUT" | grep -q "⚠️" && pass "important severity icon (⚠️) rendered" || fail "important icon missing"
57
- echo "$OUT" | grep -q "💡" && pass "suggestion severity icon (💡) rendered" || fail "suggestion icon missing"
58
-
59
- # ──────────────────────────────────────────────────────────────────────────
60
- echo "→ 5. Hunk binding - synthetic diff matches finding line"
61
- cat > "$TMP/syn.diff" <<'EOF'
62
- diff --git a/src/api/checkout.ts b/src/api/checkout.ts
63
- index a..b 100644
64
- --- a/src/api/checkout.ts
65
- +++ b/src/api/checkout.ts
66
- @@ -98,5 +98,7 @@ ...
67
- const total = sum(items);
68
- + const total2 = sum(items);
69
- + // duplicate
70
- await chargeCard(total);
71
- EOF
72
- OUT=$(node "$SCRIPT" --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" --diff "$TMP/syn.diff" 2>&1)
73
- if echo "$OUT" | grep -q '```diff'; then
74
- pass "hunk fenced diff block rendered"
75
- else
76
- fail "no diff code fence in output"
77
- fi
78
- if echo "$OUT" | grep -q "@@ -98,5 +98,7 @@"; then
79
- pass "hunk header preserved in output"
80
- else
81
- fail "hunk header missing in output"
82
- fi
83
-
84
- # ──────────────────────────────────────────────────────────────────────────
85
- echo "→ 6. Out-of-diff line fallback"
86
- # checkout.ts is in the diff but its lines 145, 200 are NOT covered by the
87
- # single hunk above. Should produce fallback messages, not crash.
88
- fallback_count=$(echo "$OUT" | grep -c "line not in diff" || true)
89
- if [ "$fallback_count" -ge 1 ]; then
90
- pass "out-of-diff lines produce fallback message ($fallback_count occurrences)"
91
- else
92
- # Acceptable alternative: hunk renderer outputs nothing visible. Both options OK.
93
- pass "out-of-diff lines handled (no crash)"
94
- fi
95
-
96
- # ──────────────────────────────────────────────────────────────────────────
97
- echo "→ 7. --help works without inputs"
98
- HELP=$(node "$SCRIPT" --help 2>&1)
99
- if echo "$HELP" | grep -q "Usage:"; then
100
- pass "--help prints usage"
101
- else
102
- fail "--help missing Usage block"
103
- fi
104
-
105
- # ──────────────────────────────────────────────────────────────────────────
106
- echo "→ 8. Missing inputs exit non-zero"
107
- set +e
108
- node "$SCRIPT" 2>"$TMP/err.txt"
109
- EC=$?
110
- set -e
111
- if [ "$EC" -ne 0 ] && grep -q "required" "$TMP/err.txt"; then
112
- pass "no-input exit $EC with 'required' error"
113
- else
114
- fail "no-input did not error correctly (exit=$EC)"
115
- fi
116
-
117
- # ──────────────────────────────────────────────────────────────────────────
118
- echo "→ 9. Read-only contract - no destructive git APIs"
119
- if grep -qE "git apply|git checkout|git reset|git rebase|git push|git commit|writeFileSync" "$SCRIPT"; then
120
- fail "diff-explain.mjs references destructive git API or writeFileSync (read-only contract violated)"
121
- else
122
- pass "no destructive git/write API references in diff-explain.mjs"
123
- fi
124
-
125
- # ──────────────────────────────────────────────────────────────────────────
126
- echo "→ 10. Injection-safe - shell metacharacters in --base are not executed"
127
- SENTINEL="$TMP/PWNED"
128
- set +e
129
- node "$SCRIPT" \
130
- --triage "$REPO_ROOT/pipeline/eval/triage/05-mixed-classification/expected.json" \
131
- --base "main; touch $SENTINEL" --branch HEAD >/dev/null 2>&1
132
- set -e
133
- if [ -f "$SENTINEL" ]; then
134
- fail "command injection: --base executed arbitrary shell (sentinel created)"
135
- else
136
- pass "shell metacharacters in --base not executed (execFileSync)"
137
- fi
138
- if grep -q "execSync" "$SCRIPT"; then
139
- fail "diff-explain.mjs still uses execSync (shell interpolation risk)"
140
- else
141
- pass "diff-explain.mjs uses execFileSync only"
142
- fi
143
-
144
- # ──────────────────────────────────────────────────────────────────────────
145
- echo ""
146
- echo "══ diff-explain smoke: $PASS passed, $FAIL failed ══"
147
- [ "$FAIL" -eq 0 ]
@@ -1,190 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-diff-risk.sh - v8.3.0
3
- #
4
- # Verifies the Phase 4 advisory diff-risk pipeline:
5
- # 1. diff-risk-score.mjs runs against the iOS fixture and produces valid JSON
6
- # 2. KeychainStore (security path) ranks first
7
- # 3. Tests file ranks last (no security/public_api/migration signal)
8
- # 4. validate-diff-risk.mjs accepts the output, rejects malformed
9
- # 5. Android fixture: migration outranks AuthRepository outranks HomeScreen
10
- # 6. --top truncation honored
11
- # 7. Empty diff returns exit 2
12
- # 8. phase-4-review.md ref doc declares Step 1.75 + diff-risk-score.mjs
13
- # 9. code-reviewer.md agent template carries the priority-files placeholder
14
- # 10. prefs.schema.json exposes diffRisk advisory toggle
15
- # 11. test-removal fixture fires the test_lines_removed signal (v1.1.0)
16
- #
17
- # Exit 0 = all pass, 1 = any failure.
18
-
19
- set -euo pipefail
20
-
21
- ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
22
- SCORE="$ROOT/pipeline/scripts/diff-risk-score.mjs"
23
- VALIDATE="$ROOT/pipeline/scripts/validate-diff-risk.mjs"
24
- SCHEMA="$ROOT/pipeline/schemas/diff-risk.schema.json"
25
- PHASE4="$ROOT/pipeline/multi-agent-refs/phases/phase-4-review.md"
26
- REVIEWER="$ROOT/pipeline/agents/code-reviewer.md"
27
- PREFS="$ROOT/pipeline/schemas/prefs.schema.json"
28
- FIX_IOS="$ROOT/pipeline/scripts/fixtures/diff-risk-ios.diff"
29
- FIX_AND="$ROOT/pipeline/scripts/fixtures/diff-risk-android.diff"
30
- FIX_TESTRM="$ROOT/pipeline/scripts/fixtures/diff-risk-test-removal.diff"
31
-
32
- pass=0
33
- fail=0
34
- failures=()
35
- record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
36
- record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
37
-
38
- printf '→ smoke-diff-risk (v8.3.0): pre-review risk scoring contract\n'
39
-
40
- [ -f "$SCHEMA" ] || { record_fail "schema missing: $SCHEMA"; exit 1; }
41
- [ -f "$FIX_IOS" ] || { record_fail "fixture missing: $FIX_IOS"; exit 1; }
42
- [ -f "$FIX_AND" ] || { record_fail "fixture missing: $FIX_AND"; exit 1; }
43
- [ -f "$FIX_TESTRM" ] || { record_fail "fixture missing: $FIX_TESTRM"; exit 1; }
44
-
45
- # --- 1: iOS fixture produces JSON ---
46
- out_ios=$(node "$SCORE" --diff "$FIX_IOS" 2>/dev/null)
47
- if jq -e '.schemaVersion == "1.1.0"' <<< "$out_ios" >/dev/null 2>&1; then
48
- record_pass "iOS fixture renders schema-versioned JSON"
49
- else
50
- record_fail "iOS fixture JSON malformed or missing schemaVersion"
51
- fi
52
-
53
- # --- 2: KeychainStore ranks first ---
54
- top1=$(jq -r '.files[0].path' <<< "$out_ios")
55
- if [ "$top1" = "MyApp/Sources/Auth/KeychainStore.swift" ]; then
56
- record_pass "iOS top-1 = KeychainStore (security_path + public_api + no_test)"
57
- else
58
- record_fail "iOS top-1 should be KeychainStore, got: $top1"
59
- fi
60
-
61
- # --- 3: Test file ranks last ---
62
- last=$(jq -r '.files[-1].path' <<< "$out_ios")
63
- if [ "$last" = "MyAppTests/SettingsViewTests.swift" ]; then
64
- record_pass "iOS last = test file (no risk signals beyond loc_changed)"
65
- else
66
- record_fail "iOS last should be test file, got: $last"
67
- fi
68
-
69
- # --- 4: Validator accepts valid output, rejects malformed ---
70
- set +e
71
- echo "$out_ios" | node "$VALIDATE" - >/dev/null 2>&1
72
- rc_ok=$?
73
- echo '{"schemaVersion":"0.0.0","files":[]}' | node "$VALIDATE" - >/dev/null 2>&1
74
- rc_bad=$?
75
- set -e
76
- if [ "$rc_ok" -eq 0 ]; then
77
- record_pass "validator accepts valid output"
78
- else
79
- record_fail "validator rejected valid output (rc=$rc_ok)"
80
- fi
81
- if [ "$rc_bad" -ne 0 ]; then
82
- record_pass "validator rejects malformed input"
83
- else
84
- record_fail "validator should reject malformed (got rc=$rc_bad)"
85
- fi
86
-
87
- # --- 5: Android ranking ---
88
- out_and=$(node "$SCORE" --diff "$FIX_AND" 2>/dev/null)
89
- top1=$(jq -r '.files[0].path' <<< "$out_and")
90
- top2=$(jq -r '.files[1].path' <<< "$out_and")
91
- top3=$(jq -r '.files[2].path' <<< "$out_and")
92
- if [[ "$top1" == *"Migration_3_to_4.kt"* ]]; then
93
- record_pass "Android top-1 = migration file"
94
- else
95
- record_fail "Android top-1 should be migration, got: $top1"
96
- fi
97
- if [[ "$top2" == *"AuthRepository.kt"* ]]; then
98
- record_pass "Android top-2 = AuthRepository (security_path)"
99
- else
100
- record_fail "Android top-2 should be AuthRepository, got: $top2"
101
- fi
102
- if [[ "$top3" == *"HomeScreen.kt"* ]]; then
103
- record_pass "Android top-3 = HomeScreen (ui_critical)"
104
- else
105
- record_fail "Android top-3 should be HomeScreen, got: $top3"
106
- fi
107
-
108
- # --- 6: --top N truncates ---
109
- out_top2=$(node "$SCORE" --diff "$FIX_IOS" --top 2 2>/dev/null)
110
- n=$(jq '.files | length' <<< "$out_top2")
111
- if [ "$n" = "2" ]; then
112
- record_pass "--top 2 truncates files array"
113
- else
114
- record_fail "--top 2 should keep 2 rows, got $n"
115
- fi
116
-
117
- # --- 7: Empty diff exits 2 ---
118
- empty=$(mktemp)
119
- set +e
120
- node "$SCORE" --diff "$empty" >/dev/null 2>&1
121
- rc=$?
122
- set -e
123
- rm -f "$empty"
124
- if [ "$rc" -eq 2 ]; then
125
- record_pass "empty diff exits 2"
126
- else
127
- record_fail "empty diff should exit 2, got $rc"
128
- fi
129
-
130
- # --- 8: phase-4-review.md declares the advisory step ---
131
- if grep -qE '^####? .*Diff Risk' "$PHASE4"; then
132
- record_pass "phase-4-review.md declares Diff Risk step"
133
- else
134
- record_fail "phase-4-review.md missing Diff Risk step heading"
135
- fi
136
- if grep -q 'diff-risk-score.mjs' "$PHASE4"; then
137
- record_pass "phase-4-review.md references diff-risk-score.mjs"
138
- else
139
- record_fail "phase-4-review.md missing script reference"
140
- fi
141
-
142
- # --- 9: code-reviewer.md template carries priority placeholder ---
143
- if grep -q 'Priority Files' "$REVIEWER"; then
144
- record_pass "code-reviewer.md declares Priority Files section"
145
- else
146
- record_fail "code-reviewer.md missing Priority Files section"
147
- fi
148
-
149
- # --- 10: prefs schema exposes diffRisk advisory toggle ---
150
- if jq -e '.properties.global.properties.diffRiskAdvisory' "$PREFS" >/dev/null 2>&1; then
151
- record_pass "prefs schema exposes diffRiskAdvisory toggle"
152
- else
153
- record_fail "prefs.schema.json missing global.diffRiskAdvisory"
154
- fi
155
-
156
- # --- 11: test_lines_removed signal fires on the test-removal fixture ---
157
- out_testrm=$(node "$SCORE" --diff "$FIX_TESTRM" 2>/dev/null)
158
- sig_value=$(jq -r '.files[] | select(.path == "MyAppTests/LoginViewModelTests.swift")
159
- | .signals[] | select(.name == "test_lines_removed") | .value' <<< "$out_testrm")
160
- if [ "$sig_value" = "16" ]; then
161
- record_pass "test_lines_removed fires with value=16 (18 removed - 2 added)"
162
- else
163
- record_fail "test_lines_removed should fire with value=16, got: ${sig_value:-missing}"
164
- fi
165
- sig_on_source=$(jq -r '[.files[] | select(.path == "MyApp/Sources/Auth/LoginViewModel.swift")
166
- | .signals[] | select(.name == "test_lines_removed")] | length' <<< "$out_testrm")
167
- if [ "$sig_on_source" = "0" ]; then
168
- record_pass "test_lines_removed does not fire on source files"
169
- else
170
- record_fail "test_lines_removed must only fire on test-classified paths"
171
- fi
172
- set +e
173
- echo "$out_testrm" | node "$VALIDATE" - >/dev/null 2>&1
174
- rc_testrm=$?
175
- set -e
176
- if [ "$rc_testrm" -eq 0 ]; then
177
- record_pass "validator accepts output carrying test_lines_removed"
178
- else
179
- record_fail "validator rejected test_lines_removed output (rc=$rc_testrm)"
180
- fi
181
-
182
- # --- Summary ---
183
- total=$((pass + fail))
184
- printf '\n→ smoke-diff-risk: %d/%d passed\n' "$pass" "$total"
185
- if [ "$fail" -ne 0 ]; then
186
- printf '\nFailures:\n'
187
- for f in "${failures[@]}"; do printf ' - %s\n' "$f"; done
188
- exit 1
189
- fi
190
- exit 0
@@ -1,160 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-dynamic-skill-loading.sh - v7.0.I
3
- #
4
- # Verifies the index + matcher contract:
5
- # 1. build-skills-index.mjs produces valid JSON with schemaVersion + entries
6
- # 2. index skill count matches the raw filesystem count
7
- # 3. match-skills.mjs ranks iOS skills higher on a SwiftUI task
8
- # 4. match-skills.mjs --json returns sorted matches[]
9
- # 5. matcher platform boost applies when --stack matches frontmatter
10
- # 6. matcher returns empty matches[] for a task with no relevant skills
11
- # 7. matcher --limit honored
12
- # 8. prefs schema exposes dynamicSkillLoading default false
13
- #
14
- # Exit 0 all pass, 1 any failure.
15
-
16
- set -euo pipefail
17
-
18
- ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
19
- BUILD="$ROOT/pipeline/scripts/build-skills-index.mjs"
20
- MATCH="$ROOT/pipeline/scripts/match-skills.mjs"
21
- SCHEMA="$ROOT/pipeline/schemas/prefs.schema.json"
22
-
23
- pass=0; fail=0; failures=()
24
- record_pass() { pass=$((pass + 1)); printf ' \033[0;32mPASS\033[0m %s\n' "$1"; }
25
- record_fail() { fail=$((fail + 1)); failures+=("$1"); printf ' \033[0;31mFAIL\033[0m %s\n' "$1"; }
26
-
27
- printf '→ smoke-dynamic-skill-loading (v7.0.I): index + matcher contract\n'
28
-
29
- # Build into a sandbox so we don't clobber the shipping index
30
- sandbox=$(mktemp -d)
31
- skills="$sandbox/skills"
32
- mkdir -p "$skills/core/multi-agent" "$skills/core/testing-backend" "$skills/external/swiftui-pro" "$skills/external/compose-components"
33
-
34
- cat > "$skills/core/multi-agent/SKILL.md" <<'S'
35
- ---
36
- name: multi-agent
37
- description: Task orchestrator for the multi-agent pipeline
38
- platform: generic
39
- ---
40
- body
41
- S
42
-
43
- cat > "$skills/core/testing-backend/SKILL.md" <<'S'
44
- ---
45
- name: testing-backend
46
- description: Backend testing patterns for pytest / Jest
47
- platform: backend
48
- trigger-keywords: pytest, jest, integration test
49
- ---
50
- body
51
- S
52
-
53
- cat > "$skills/external/swiftui-pro/SKILL.md" <<'S'
54
- ---
55
- name: swiftui-pro
56
- description: SwiftUI review for modern APIs, state management, performance
57
- platform: ios
58
- trigger-keywords: swiftui, swift, ios, swiftui performance
59
- trigger-paths: *.swift, **/Sources/**/*.swift
60
- ---
61
- body
62
- S
63
-
64
- cat > "$skills/external/compose-components/SKILL.md" <<'S'
65
- ---
66
- name: compose-components
67
- description: Material 3 Jetpack Compose components
68
- platform: android
69
- trigger-keywords: compose, jetpack, material 3, android
70
- trigger-paths: *.kt, **/src/main/**/*.kt
71
- ---
72
- body
73
- S
74
-
75
- # --- 1: build index ---
76
- node "$BUILD" --root "$skills" >/dev/null
77
- INDEX="$skills/.skills-index.json"
78
- if [ -f "$INDEX" ] && jq -e '.schemaVersion and .entries and (.skillCount == 4)' "$INDEX" >/dev/null; then
79
- record_pass "build-skills-index emits valid JSON with 4 entries"
80
- else
81
- record_fail "build-skills-index output malformed"
82
- fi
83
-
84
- # --- 2: count matches filesystem ---
85
- raw_count=$(find "$skills" -name SKILL.md | wc -l | tr -d ' ')
86
- index_count=$(jq -r '.skillCount' "$INDEX")
87
- if [ "$raw_count" = "$index_count" ]; then
88
- record_pass "index count matches filesystem ($index_count)"
89
- else
90
- record_fail "index count $index_count != filesystem $raw_count"
91
- fi
92
-
93
- # --- 3: iOS task ranks SwiftUI skills higher ---
94
- out=$(node "$MATCH" "SwiftUI dark mode fix on LoginView" \
95
- --touched-files "Sources/Auth/LoginView.swift" \
96
- --stack ios --limit 4 --index "$INDEX" --json)
97
- top=$(jq -r '.matches[0].name' <<< "$out")
98
- if [ "$top" = "swiftui-pro" ]; then
99
- record_pass "iOS task ranks swiftui-pro top ($top)"
100
- else
101
- record_fail "iOS task should rank swiftui-pro first, got $top"
102
- fi
103
-
104
- # --- 4: JSON output is sorted ---
105
- scores=$(jq -r '.matches[].score' <<< "$out")
106
- prev=9999; sorted=1
107
- for s in $scores; do
108
- if [ "$s" -gt "$prev" ]; then sorted=0; break; fi
109
- prev=$s
110
- done
111
- if [ "$sorted" = "1" ]; then
112
- record_pass "JSON matches[] sorted by score desc"
113
- else
114
- record_fail "matches[] not sorted"
115
- fi
116
-
117
- # --- 5: platform boost ---
118
- # A keyword-free task with only --stack android should still boost Android skill
119
- out=$(node "$MATCH" "Build the feature end to end" --stack android --limit 4 --index "$INDEX" --json)
120
- rules=$(jq -r '.matches[] | select(.name=="compose-components") | .reasons[].rule' <<< "$out" 2>/dev/null || echo "")
121
- if echo "$rules" | grep -q 'platform'; then
122
- record_pass "platform boost reason present for matched stack"
123
- else
124
- record_fail "platform boost missing for --stack android"
125
- fi
126
-
127
- # --- 6: empty matches ---
128
- out=$(node "$MATCH" "Totally unrelated lmn xyz qrs foobar baz" --limit 4 --index "$INDEX" --json)
129
- n=$(jq '.matches | length' <<< "$out")
130
- if [ "$n" = "0" ]; then
131
- record_pass "empty matches[] for unrelated task"
132
- else
133
- record_fail "expected 0 matches for unrelated task, got $n"
134
- fi
135
-
136
- # --- 7: --limit honored ---
137
- out=$(node "$MATCH" "swiftui compose test pipeline" --limit 2 --index "$INDEX" --json)
138
- n=$(jq '.matches | length' <<< "$out")
139
- if [ "$n" -le "2" ]; then
140
- record_pass "--limit 2 returns at most 2 matches ($n)"
141
- else
142
- record_fail "--limit 2 returned $n"
143
- fi
144
-
145
- # --- 8: schema pref ---
146
- if jq -e '.properties.global.properties.dynamicSkillLoading.default == false' "$SCHEMA" >/dev/null 2>&1; then
147
- record_pass "prefs schema exposes dynamicSkillLoading default false"
148
- else
149
- record_fail "prefs schema missing dynamicSkillLoading or wrong default"
150
- fi
151
-
152
- rm -rf "$sandbox"
153
-
154
- printf '\n══ dynamic-skill-loading smoke: %d passed, %d failed ══\n' "$pass" "$fail"
155
- if [ "$fail" -gt 0 ]; then
156
- printf '\nFailures:\n'
157
- for m in "${failures[@]}"; do printf ' - %s\n' "$m"; done
158
- exit 1
159
- fi
160
- exit 0
@@ -1,136 +0,0 @@
1
- #!/usr/bin/env bash
2
- # smoke-eval-live.sh - v7.8.0 Paket A: opt-in live golden-task runner.
3
- #
4
- # Validates the cost-guarding contract WITHOUT making any model calls. Live
5
- # mode is gated behind two independent signals (--live flag AND
6
- # MULTI_AGENT_LIVE_EVAL=1 env) so accidental cost is impossible during CI
7
- # or local `npm test` runs.
8
- #
9
- # All assertions exercise the dry-run code path or expect early-exit before
10
- # any model invocation.
11
-
12
- set -uo pipefail
13
-
14
- REPO_ROOT="$(cd "$(dirname "$0")/../.." && pwd)"
15
- SCRIPT="$REPO_ROOT/pipeline/scripts/eval-golden-tasks-live.mjs"
16
-
17
- PASS=0
18
- FAIL=0
19
- pass() { PASS=$((PASS + 1)); echo " ✓ $1"; }
20
- fail() { FAIL=$((FAIL + 1)); echo " ✗ $1"; }
21
-
22
- cleanup() { [ -n "${TMP:-}" ] && rm -rf "$TMP"; }
23
- trap cleanup EXIT
24
- TMP=$(mktemp -d)
25
-
26
- # Strip env that would unlock live mode; smoke MUST never spend money.
27
- unset MULTI_AGENT_LIVE_EVAL
28
-
29
- # ──────────────────────────────────────────────────────────────────────────
30
- echo "→ 1. Script presence + syntax"
31
- [ -x "$SCRIPT" ] && pass "eval-golden-tasks-live.mjs is executable" || fail "script not executable"
32
- node --check "$SCRIPT" 2>/dev/null && pass "syntax check" || fail "syntax error"
33
-
34
- # ──────────────────────────────────────────────────────────────────────────
35
- echo "→ 2. Default mode is dry-run (no --live, no env)"
36
- OUT=$(node "$SCRIPT" --json 2>/dev/null)
37
- if echo "$OUT" | grep -q '"mode": "dry-run"'; then
38
- pass "default mode = dry-run"
39
- else
40
- fail "default mode not dry-run: $(echo "$OUT" | head -3)"
41
- fi
42
-
43
- # ──────────────────────────────────────────────────────────────────────────
44
- echo "→ 3. Dry-run does not invoke model"
45
- # All cases must end with status "skipped-dry-run" - i.e. no live call
46
- SKIPPED_COUNT=$(echo "$OUT" | grep -c '"pass": "skipped-dry-run"' || true)
47
- if [ "$SKIPPED_COUNT" -ge 1 ]; then
48
- pass "dry-run skipped all cases ($SKIPPED_COUNT) without invoking model"
49
- else
50
- fail "dry-run did not skip cases properly"
51
- fi
52
-
53
- # ──────────────────────────────────────────────────────────────────────────
54
- echo "→ 4. --live without env exits 2 (cost gate)"
55
- node "$SCRIPT" --live 2>"$TMP/err.txt"
56
- EC=$?
57
- if [ "$EC" -eq 2 ] && grep -q "MULTI_AGENT_LIVE_EVAL" "$TMP/err.txt"; then
58
- pass "--live without env: exit $EC + cost gate message"
59
- else
60
- fail "--live cost gate weak (exit=$EC, err='$(cat "$TMP/err.txt")')"
61
- fi
62
-
63
- # ──────────────────────────────────────────────────────────────────────────
64
- echo "→ 5. Static check - no fall-through past cost gate"
65
- # The cost gate must be the first thing checked after flag parsing. If --live
66
- # is set without env, no model call should be reachable. Verify by static
67
- # grep: spawnSync("claude", ...) must come AFTER the env-check block.
68
- GATE_LINE=$(grep -n 'MULTI_AGENT_LIVE_EVAL' "$SCRIPT" | head -1 | cut -d: -f1)
69
- CALL_LINE=$(grep -n 'spawnSync("claude"' "$SCRIPT" | head -1 | cut -d: -f1)
70
- if [ -n "$GATE_LINE" ] && [ -n "$CALL_LINE" ] && [ "$GATE_LINE" -lt "$CALL_LINE" ]; then
71
- pass "cost gate at line $GATE_LINE precedes claude call at line $CALL_LINE"
72
- else
73
- fail "cost gate ordering wrong (gate=$GATE_LINE, call=$CALL_LINE)"
74
- fi
75
-
76
- # ──────────────────────────────────────────────────────────────────────────
77
- echo "→ 6. Budget cap honored - total-budget < single case budget skips all"
78
- OUT=$(node "$SCRIPT" --max-cases=5 --budget=0.5 --total-budget=0.1 --json 2>/dev/null)
79
- RUN_COUNT=$(echo "$OUT" | grep -oE '"cases_run":[[:space:]]*[0-9]+' | grep -oE '[0-9]+')
80
- SKIPPED=$(echo "$OUT" | grep -c '"skipped-budget-exhausted"' || true)
81
- if [ "$SKIPPED" -ge 1 ]; then
82
- pass "budget exhaustion produces 'skipped-budget-exhausted' result ($SKIPPED times)"
83
- else
84
- pass "budget very tight: cases_run=$RUN_COUNT (acceptable in dry-run mode)"
85
- fi
86
-
87
- # ──────────────────────────────────────────────────────────────────────────
88
- echo "→ 7. --max-cases caps the run"
89
- OUT=$(node "$SCRIPT" --max-cases=1 --json 2>/dev/null)
90
- RUN_COUNT=$(echo "$OUT" | grep -oE '"cases_run":[[:space:]]*[0-9]+' | grep -oE '[0-9]+')
91
- if [ "$RUN_COUNT" = "1" ]; then
92
- pass "--max-cases=1 limits cases_run to 1"
93
- else
94
- fail "--max-cases not honored: cases_run=$RUN_COUNT"
95
- fi
96
-
97
- # ──────────────────────────────────────────────────────────────────────────
98
- echo "→ 8. --case selector"
99
- OUT=$(node "$SCRIPT" --case 01-ios-bugfix-darkmode --json 2>/dev/null)
100
- if echo "$OUT" | grep -q '"case": "01-ios-bugfix-darkmode"'; then
101
- pass "--case selector picks the right fixture"
102
- else
103
- fail "--case selector did not target fixture"
104
- fi
105
-
106
- node "$SCRIPT" --case nonexistent-case 2>"$TMP/err.txt"
107
- EC=$?
108
- if [ "$EC" -eq 2 ] && grep -q "not found" "$TMP/err.txt"; then
109
- pass "unknown --case exits 2 with 'not found'"
110
- else
111
- fail "unknown --case error weak (exit=$EC)"
112
- fi
113
-
114
- # ──────────────────────────────────────────────────────────────────────────
115
- echo "→ 9. Help text mentions cost gate prominently"
116
- HELP=$(node "$SCRIPT" --help 2>&1)
117
- if echo "$HELP" | grep -q "MULTI_AGENT_LIVE_EVAL"; then
118
- pass "--help documents the cost gate env var"
119
- else
120
- fail "--help missing cost gate documentation"
121
- fi
122
-
123
- # ──────────────────────────────────────────────────────────────────────────
124
- echo "→ 10. Bad --budget exits 2"
125
- node "$SCRIPT" --budget=invalid 2>/dev/null
126
- EC=$?
127
- if [ "$EC" -eq 2 ]; then
128
- pass "invalid --budget exits 2"
129
- else
130
- fail "invalid --budget did not error (exit=$EC)"
131
- fi
132
-
133
- # ──────────────────────────────────────────────────────────────────────────
134
- echo ""
135
- echo "══ eval-live smoke: $PASS passed, $FAIL failed ══"
136
- [ "$FAIL" -eq 0 ]