@mmerterden/multi-agent-pipeline 12.7.0 → 12.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/CHANGELOG.md +207 -0
  2. package/install/_common.mjs +48 -0
  3. package/install/_dev-only-files.mjs +125 -5
  4. package/install/claude.mjs +14 -8
  5. package/install/copilot.mjs +5 -8
  6. package/package.json +17 -2
  7. package/pipeline/commands/multi-agent/analysis/SKILL.md +1 -1
  8. package/pipeline/lib/credential-store.sh +20 -0
  9. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  10. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  11. package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
  12. package/pipeline/multi-agent-refs/features/review-multi-repo.md +65 -0
  13. package/pipeline/multi-agent-refs/features/url-enrichment.md +93 -0
  14. package/pipeline/multi-agent-refs/phases/operations.md +28 -0
  15. package/pipeline/multi-agent-refs/phases/phase-0-init.md +3 -83
  16. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  17. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
  18. package/pipeline/multi-agent-refs/phases/phase-4-review.md +50 -62
  19. package/pipeline/schemas/prefs.schema.json +6 -0
  20. package/pipeline/scripts/_smoke-root.sh +61 -0
  21. package/pipeline/scripts/agent-guard.py +57 -3
  22. package/pipeline/scripts/audit-log.sh +25 -0
  23. package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
  24. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
  25. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
  26. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
  27. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
  28. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
  29. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
  30. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
  31. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
  32. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
  33. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
  34. package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
  35. package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
  36. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
  37. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
  38. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
  39. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
  40. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
  41. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
  42. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
  43. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
  44. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
  45. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
  46. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
  47. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
  48. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
  49. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
  50. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
  51. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
  52. package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
  53. package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
  54. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
  55. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
  56. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
  57. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
  58. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
  59. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
  60. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
  61. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
  62. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
  63. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
  64. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
  65. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
  66. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
  67. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
  68. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
  69. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
  70. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
  71. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
  72. package/pipeline/eval/golden-tasks/README.md +0 -65
  73. package/pipeline/eval/intent-cases.json +0 -40
  74. package/pipeline/eval/run-metrics-fixture.json +0 -93
  75. package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
  76. package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
  77. package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
  78. package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
  79. package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
  80. package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
  81. package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
  82. package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
  83. package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
  84. package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
  85. package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
  86. package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
  87. package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
  88. package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
  89. package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
  90. package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
  91. package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
  92. package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
  93. package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
  94. package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
  95. package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
  96. package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
  97. package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
  98. package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
  99. package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
  100. package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
  101. package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
  102. package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
  103. package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
  104. package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
  105. package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
  106. package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
  107. package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
  108. package/pipeline/eval/triage/README.md +0 -54
  109. package/pipeline/scripts/benchmark-phase-0.sh +0 -128
  110. package/pipeline/scripts/check-md-links.mjs +0 -88
  111. package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -302
  112. package/pipeline/scripts/eval-golden-tasks.mjs +0 -224
  113. package/pipeline/scripts/eval-intent.mjs +0 -107
  114. package/pipeline/scripts/eval-mine-corpus.mjs +0 -211
  115. package/pipeline/scripts/eval-triage.mjs +0 -171
  116. package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
  117. package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
  118. package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
  119. package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
  120. package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
  121. package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
  122. package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
  123. package/pipeline/scripts/lint-mcp-refs.mjs +0 -218
  124. package/pipeline/scripts/lint-skills.mjs +0 -154
  125. package/pipeline/scripts/run-smokes.mjs +0 -130
  126. package/pipeline/scripts/scorecard.mjs +0 -258
  127. package/pipeline/scripts/smoke-add-detail.sh +0 -137
  128. package/pipeline/scripts/smoke-agent-guard.sh +0 -74
  129. package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
  130. package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
  131. package/pipeline/scripts/smoke-ask-choice.sh +0 -42
  132. package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
  133. package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
  134. package/pipeline/scripts/smoke-changelog-version.sh +0 -47
  135. package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
  136. package/pipeline/scripts/smoke-channels-flow.sh +0 -130
  137. package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
  138. package/pipeline/scripts/smoke-clarify.sh +0 -148
  139. package/pipeline/scripts/smoke-command-inventory.sh +0 -81
  140. package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
  141. package/pipeline/scripts/smoke-community-gates.sh +0 -75
  142. package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
  143. package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
  144. package/pipeline/scripts/smoke-context-budget.sh +0 -72
  145. package/pipeline/scripts/smoke-cost-budget.sh +0 -70
  146. package/pipeline/scripts/smoke-cost-summary.sh +0 -139
  147. package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
  148. package/pipeline/scripts/smoke-description-tr.sh +0 -82
  149. package/pipeline/scripts/smoke-dev-critic.sh +0 -144
  150. package/pipeline/scripts/smoke-diff-explain.sh +0 -147
  151. package/pipeline/scripts/smoke-diff-risk.sh +0 -190
  152. package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
  153. package/pipeline/scripts/smoke-eval-live.sh +0 -136
  154. package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
  155. package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
  156. package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
  157. package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
  158. package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
  159. package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
  160. package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
  161. package/pipeline/scripts/smoke-generate-issue.sh +0 -120
  162. package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
  163. package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
  164. package/pipeline/scripts/smoke-install-layout.sh +0 -248
  165. package/pipeline/scripts/smoke-intent-guard.sh +0 -86
  166. package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
  167. package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
  168. package/pipeline/scripts/smoke-keychain.sh +0 -158
  169. package/pipeline/scripts/smoke-language-axis.sh +0 -109
  170. package/pipeline/scripts/smoke-learning-curve.sh +0 -61
  171. package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
  172. package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
  173. package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
  174. package/pipeline/scripts/smoke-md-links.sh +0 -8
  175. package/pipeline/scripts/smoke-md2confluence.sh +0 -126
  176. package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
  177. package/pipeline/scripts/smoke-migrate-state.sh +0 -102
  178. package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
  179. package/pipeline/scripts/smoke-model-fallback.sh +0 -89
  180. package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
  181. package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
  182. package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -194
  183. package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
  184. package/pipeline/scripts/smoke-own-punctuation.sh +0 -103
  185. package/pipeline/scripts/smoke-pack-contents.sh +0 -140
  186. package/pipeline/scripts/smoke-pat-audit.sh +0 -128
  187. package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
  188. package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
  189. package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
  190. package/pipeline/scripts/smoke-phase-banner.sh +0 -101
  191. package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
  192. package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
  193. package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
  194. package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
  195. package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
  196. package/pipeline/scripts/smoke-plan-safety.sh +0 -139
  197. package/pipeline/scripts/smoke-plan-todos.sh +0 -196
  198. package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
  199. package/pipeline/scripts/smoke-pre-commit.sh +0 -170
  200. package/pipeline/scripts/smoke-pref-migration.sh +0 -226
  201. package/pipeline/scripts/smoke-prefs-language.sh +0 -134
  202. package/pipeline/scripts/smoke-progress-contract.sh +0 -127
  203. package/pipeline/scripts/smoke-prune-logs.sh +0 -137
  204. package/pipeline/scripts/smoke-purge.sh +0 -138
  205. package/pipeline/scripts/smoke-push-retry.sh +0 -75
  206. package/pipeline/scripts/smoke-repo-map.sh +0 -300
  207. package/pipeline/scripts/smoke-review-readiness.sh +0 -92
  208. package/pipeline/scripts/smoke-review-watch.sh +0 -146
  209. package/pipeline/scripts/smoke-routines.sh +0 -84
  210. package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
  211. package/pipeline/scripts/smoke-run-metrics.sh +0 -50
  212. package/pipeline/scripts/smoke-search.sh +0 -187
  213. package/pipeline/scripts/smoke-shadow-git.sh +0 -224
  214. package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
  215. package/pipeline/scripts/smoke-skill-language.sh +0 -83
  216. package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
  217. package/pipeline/scripts/smoke-skill-scan.sh +0 -198
  218. package/pipeline/scripts/smoke-source-parity.sh +0 -85
  219. package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
  220. package/pipeline/scripts/smoke-sync-parity.sh +0 -92
  221. package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
  222. package/pipeline/scripts/smoke-telemetry.sh +0 -147
  223. package/pipeline/scripts/smoke-test-gap.sh +0 -183
  224. package/pipeline/scripts/smoke-token-budget.sh +0 -67
  225. package/pipeline/scripts/smoke-token-preflight.sh +0 -82
  226. package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
  227. package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
  228. package/pipeline/scripts/smoke-triage-memory.sh +0 -174
  229. package/pipeline/scripts/smoke-update-check.sh +0 -135
  230. package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
  231. package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
  232. package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
  233. package/pipeline/scripts/smoke-validator-gates.sh +0 -164
  234. package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
  235. package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
  236. package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
  237. package/pipeline/scripts/smoke-work-summary.sh +0 -163
  238. package/pipeline/scripts/smoke-workflow-audit.sh +0 -101
  239. package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
  240. package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
  241. package/pipeline/scripts/smoke-write-state.sh +0 -159
  242. package/pipeline/scripts/sync-parity-check.sh +0 -135
  243. package/pipeline/scripts/test-gap-rules/android.json +0 -25
  244. package/pipeline/scripts/test-gap-rules/ios.json +0 -34
  245. package/pipeline/scripts/test-gap-rules/node.json +0 -29
  246. package/pipeline/scripts/test-gap-rules/python.json +0 -25
  247. package/pipeline/scripts/validate-schemas.mjs +0 -88
@@ -1,302 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- /**
4
- * @file eval-golden-tasks-live.mjs - v7.8.0 Paket A
5
- *
6
- * Opt-in live evaluation harness for golden-task fixtures. Unlike the
7
- * always-on contract harness (`eval-golden-tasks.mjs`) which only validates
8
- * schema shape, this harness optionally invokes a real model via the local
9
- * `claude` CLI and compares the actual output against the fixture's expected
10
- * scope (stack, blocker count, deferral count).
11
- *
12
- * **Default behavior is dry-run** - no model is invoked unless `--live` is
13
- * passed AND `MULTI_AGENT_LIVE_EVAL=1` is set in the environment. This
14
- * double-gate prevents accidental cost.
15
- *
16
- * Cost guard:
17
- * - Per-case budget cap (default $1 USD) - case skipped if exceeded
18
- * - Run-wide max-cases cap (default 1) - even with --live, only N cases run
19
- * - Run-wide hard ceiling: budget * max_cases (default $1 total)
20
- *
21
- * Zero-dep: shells out to the user's installed `claude` CLI rather than the
22
- * Anthropic SDK. Honors ADR-4 and uses the user's existing CLI auth.
23
- *
24
- * Usage:
25
- * node eval-golden-tasks-live.mjs # dry-run (default)
26
- * node eval-golden-tasks-live.mjs --live # blocked unless env set
27
- * MULTI_AGENT_LIVE_EVAL=1 node eval-golden-tasks-live.mjs --live --case 01-...
28
- * node eval-golden-tasks-live.mjs --budget=0.50 --max-cases=2
29
- * node eval-golden-tasks-live.mjs --json # machine-readable summary
30
- *
31
- * Exit codes:
32
- * 0 all run cases produced output matching expected scope
33
- * 1 one or more cases mismatched expected
34
- * 2 setup/usage error
35
- * 3 budget exhausted before run completed
36
- *
37
- * @module pipeline/scripts/eval-golden-tasks-live
38
- */
39
-
40
- import { readdirSync, readFileSync, existsSync } from "node:fs";
41
- import { join, dirname, resolve } from "node:path";
42
- import { fileURLToPath } from "node:url";
43
- import { spawnSync } from "node:child_process";
44
-
45
- const here = dirname(fileURLToPath(import.meta.url));
46
- const root = resolve(here, "..", "..");
47
- const evalDir = join(root, "pipeline", "eval", "golden-tasks");
48
-
49
- const argv = process.argv.slice(2);
50
- const flags = {};
51
- for (let i = 0; i < argv.length; i++) {
52
- const a = argv[i];
53
- if (a.startsWith("--")) {
54
- const [key, ...rest] = a.slice(2).split("=");
55
- if (rest.length) flags[key] = rest.join("=");
56
- else if (argv[i + 1] && !argv[i + 1].startsWith("--")) {
57
- flags[key] = argv[i + 1];
58
- i++;
59
- } else flags[key] = true;
60
- }
61
- }
62
-
63
- if (flags.help || flags.h) {
64
- console.log(`Usage: eval-golden-tasks-live.mjs [options]
65
-
66
- Modes:
67
- (default) dry-run: list what would run, no model invocation
68
- --live live: call \`claude -p\` for real (requires
69
- MULTI_AGENT_LIVE_EVAL=1 in env)
70
-
71
- Selection:
72
- --case <slug> run only this case (e.g. 01-ios-bugfix-darkmode)
73
- --max-cases <N> cap total cases to run (default: 1)
74
-
75
- Cost guard:
76
- --budget <usd> per-case budget cap (default: 1.00 USD)
77
- --total-budget <usd> run-wide ceiling (default: budget * max-cases)
78
-
79
- Output:
80
- --json machine-readable JSON summary
81
- `);
82
- process.exit(0);
83
- }
84
-
85
- const isLive = !!flags.live;
86
- const liveEnvOK = process.env.MULTI_AGENT_LIVE_EVAL === "1";
87
- const budget = parseFloat(flags.budget || "1.00");
88
- const maxCases = parseInt(flags["max-cases"] || "1", 10);
89
- const totalBudget = parseFloat(flags["total-budget"] || String(budget * maxCases));
90
- const asJson = !!flags.json;
91
-
92
- if (isLive && !liveEnvOK) {
93
- process.stderr.write(
94
- "eval-golden-tasks-live: --live requires MULTI_AGENT_LIVE_EVAL=1 in env (cost gate).\n",
95
- );
96
- process.exit(2);
97
- }
98
-
99
- if (!Number.isFinite(budget) || budget <= 0) {
100
- process.stderr.write(`eval-golden-tasks-live: invalid --budget ${flags.budget}\n`);
101
- process.exit(2);
102
- }
103
-
104
- function listCases() {
105
- if (!existsSync(evalDir)) return [];
106
- return readdirSync(evalDir, { withFileTypes: true })
107
- .filter((e) => e.isDirectory() && /^\d{2}-/.test(e.name))
108
- .map((e) => e.name)
109
- .sort();
110
- }
111
-
112
- function loadCase(slug) {
113
- const caseDir = join(evalDir, slug);
114
- const taskFile = join(caseDir, "task.json");
115
- if (!existsSync(taskFile)) return null;
116
- let task;
117
- try {
118
- task = JSON.parse(readFileSync(taskFile, "utf-8"));
119
- } catch (e) {
120
- process.stderr.write(`[eval-live] malformed task.json for "${slug}": ${e.message}\n`);
121
- return null;
122
- }
123
- return { slug, dir: caseDir, task };
124
- }
125
-
126
- function hasClaudeCLI() {
127
- const r = spawnSync("which", ["claude"], { encoding: "utf-8" });
128
- return r.status === 0 && (r.stdout || "").trim().length > 0;
129
- }
130
-
131
- /**
132
- * Live invocation - wraps the task as a prompt for `claude -p`. Returns the
133
- * raw stdout. The current iteration captures only the analysis-phase output
134
- * by asking for a JSON-only response per the analysis schema. Future work
135
- * extends this to full Phase 1/2/4 multi-call evaluation.
136
- */
137
- function invokeLive(taskJson) {
138
- const prompt = [
139
- "Perform Phase 1 (Analysis) of the multi-agent-pipeline for the task below.",
140
- "Return ONLY the analysis JSON (matching pipeline/schemas/analysis-output.schema.json).",
141
- "Do not include explanatory prose, markdown fences, or commentary.",
142
- "",
143
- "Task:",
144
- JSON.stringify(taskJson, null, 2),
145
- ].join("\n");
146
-
147
- const r = spawnSync("claude", ["-p", prompt], {
148
- encoding: "utf-8",
149
- timeout: 5 * 60 * 1000,
150
- });
151
- if (r.status !== 0) {
152
- return { ok: false, error: `claude exit ${r.status}: ${r.stderr || ""}` };
153
- }
154
- return { ok: true, raw: r.stdout || "" };
155
- }
156
-
157
- function tryParseJson(s) {
158
- try {
159
- return { ok: true, value: JSON.parse(s) };
160
- } catch (e) {
161
- // Try to extract first {...} block
162
- const m = /\{[\s\S]*\}/.exec(s);
163
- if (m) {
164
- try {
165
- return { ok: true, value: JSON.parse(m[0]) };
166
- } catch {
167
- /* fallthrough */
168
- }
169
- }
170
- return { ok: false, error: e.message };
171
- }
172
- }
173
-
174
- function evaluateOne(c) {
175
- const result = {
176
- case: c.slug,
177
- mode: isLive ? "live" : "dry-run",
178
- expected: {
179
- stack: c.task.expectedStack,
180
- blockers: c.task.expectedBlockers,
181
- deferrals: c.task.expectedDeferrals,
182
- },
183
- actual: null,
184
- pass: null,
185
- notes: [],
186
- };
187
-
188
- if (!isLive) {
189
- result.pass = "skipped-dry-run";
190
- result.notes.push("would invoke claude -p with task prompt");
191
- return result;
192
- }
193
-
194
- if (!hasClaudeCLI()) {
195
- result.pass = "skipped-no-cli";
196
- result.notes.push("claude CLI not on PATH");
197
- return result;
198
- }
199
-
200
- const live = invokeLive(c.task);
201
- if (!live.ok) {
202
- result.pass = false;
203
- result.notes.push(`live invocation failed: ${live.error}`);
204
- return result;
205
- }
206
-
207
- const parsed = tryParseJson(live.raw);
208
- if (!parsed.ok) {
209
- result.pass = false;
210
- result.notes.push(`could not parse model output as JSON: ${parsed.error}`);
211
- return result;
212
- }
213
-
214
- const actual = parsed.value;
215
- result.actual = {
216
- stack: actual?.stack?.primary ?? actual?.stack ?? null,
217
- };
218
- if (actual?.stack?.primary === c.task.expectedStack) {
219
- result.pass = true;
220
- result.notes.push(`stack match: ${actual.stack.primary}`);
221
- } else {
222
- result.pass = false;
223
- result.notes.push(
224
- `stack mismatch: expected ${c.task.expectedStack}, got ${result.actual.stack}`,
225
- );
226
- }
227
- return result;
228
- }
229
-
230
- // ─────────────────────────────────────────────────────────────────────────
231
- const allCases = listCases();
232
- let selected = allCases;
233
- if (flags.case) {
234
- selected = selected.filter((s) => s === flags.case);
235
- if (selected.length === 0) {
236
- process.stderr.write(`eval-golden-tasks-live: case not found: ${flags.case}\n`);
237
- process.exit(2);
238
- }
239
- }
240
- selected = selected.slice(0, maxCases);
241
-
242
- const results = [];
243
- let projectedSpend = 0;
244
-
245
- for (const slug of selected) {
246
- const c = loadCase(slug);
247
- if (!c) {
248
- results.push({ case: slug, pass: false, notes: ["fixture missing"] });
249
- continue;
250
- }
251
- if (projectedSpend + budget > totalBudget) {
252
- results.push({
253
- case: slug,
254
- pass: "skipped-budget-exhausted",
255
- notes: [`would exceed total-budget ${totalBudget} (already projected ${projectedSpend})`],
256
- });
257
- continue;
258
- }
259
- const r = evaluateOne(c);
260
- if (r.pass !== "skipped-dry-run" && r.pass !== "skipped-no-cli") {
261
- projectedSpend += budget;
262
- }
263
- results.push(r);
264
- }
265
-
266
- const summary = {
267
- mode: isLive ? "live" : "dry-run",
268
- cases_total: allCases.length,
269
- cases_run: results.length,
270
- cases_passed: results.filter((r) => r.pass === true).length,
271
- cases_failed: results.filter((r) => r.pass === false).length,
272
- cases_skipped: results.filter((r) => typeof r.pass === "string" && r.pass.startsWith("skipped"))
273
- .length,
274
- budget_per_case: budget,
275
- total_budget: totalBudget,
276
- projected_spend: projectedSpend,
277
- };
278
-
279
- if (asJson) {
280
- console.log(JSON.stringify({ summary, results }, null, 2));
281
- } else {
282
- console.log("");
283
- console.log(`eval-golden-tasks-live (${summary.mode})`);
284
- console.log(
285
- ` cases: ${summary.cases_run}/${summary.cases_total} run, ${summary.cases_passed} passed, ${summary.cases_failed} failed, ${summary.cases_skipped} skipped`,
286
- );
287
- console.log(
288
- ` budget: $${budget.toFixed(2)}/case · total cap $${totalBudget.toFixed(2)} · projected spend $${projectedSpend.toFixed(2)}`,
289
- );
290
- console.log("");
291
- for (const r of results) {
292
- const icon = r.pass === true ? "✓" : r.pass === false ? "✗" : "·";
293
- console.log(` ${icon} ${r.case} [${r.pass}]`);
294
- for (const n of r.notes) console.log(` ${n}`);
295
- }
296
- console.log("");
297
- }
298
-
299
- const hasFailure = results.some((r) => r.pass === false);
300
- const budgetExhausted = results.some((r) => r.pass === "skipped-budget-exhausted");
301
-
302
- process.exit(hasFailure ? 1 : budgetExhausted ? 3 : 0);
@@ -1,224 +0,0 @@
1
- #!/usr/bin/env node
2
- // eval-golden-tasks.mjs - contract regression check for whole-pipeline fixtures.
3
- //
4
- // Walks pipeline/eval/golden-tasks/<NN>-<name>/ subdirectories. For each case:
5
- // 1. Reads task.json (input) + expected/phase-{1,2,4-review,4-triage}.json.
6
- // 2. Validates each expected/ file against its paired validator script
7
- // (validate-analysis.mjs / validate-planning.mjs / validate-reviewer.mjs /
8
- // validate-triage.mjs).
9
- // 3. Cross-file consistency: Phase 2 tasks[].files[] ⊆ Phase 1
10
- // touchedAreas[].path. Phase 4 review findings[].file referenced in
11
- // Phase 2 planned files (or flagged as deferred in triage).
12
- // 4. Stack match: phase-1.stack.primary === task.expectedStack.
13
- // 5. Blocker count match: triage.accepted[].filter(severity=blocking).length
14
- // === task.expectedBlockers. Same for deferred bucket.
15
- //
16
- // What this catches:
17
- // - Schema drift (any of the 4 phase schemas changes in a breaking way).
18
- // - Fixture-vs-spec divergence (someone updates a schema without re-syncing
19
- // the golden fixtures).
20
- // - Coverage gap (a triage that loses findings vs. reviewer input).
21
- //
22
- // What this does NOT catch:
23
- // - Real model output quality. This is a CONTRACT test, not a model eval.
24
- // A separate multi-model harness (with API keys, non-deterministic) is
25
- // future work. See pipeline/eval/golden-tasks/README.md.
26
- //
27
- // Usage:
28
- // node pipeline/scripts/eval-golden-tasks.mjs # all cases
29
- // node pipeline/scripts/eval-golden-tasks.mjs --case 01-... # one case
30
- // node pipeline/scripts/eval-golden-tasks.mjs --json # CI-friendly
31
- //
32
- // Exit codes: 0 all pass, 1 one or more fail, 2 usage/setup error.
33
-
34
- import { readdirSync, readFileSync, statSync, existsSync } from "node:fs";
35
- import { dirname, join, resolve } from "node:path";
36
- import { fileURLToPath } from "node:url";
37
- import { spawnSync } from "node:child_process";
38
-
39
- const here = dirname(fileURLToPath(import.meta.url));
40
- const root = resolve(here, "..", "..");
41
- const evalDir = join(root, "pipeline", "eval", "golden-tasks");
42
-
43
- const validators = {
44
- analysis: join(root, "pipeline", "scripts", "validate-analysis.mjs"),
45
- planning: join(root, "pipeline", "scripts", "validate-planning.mjs"),
46
- reviewer: join(root, "pipeline", "scripts", "validate-reviewer.mjs"),
47
- triage: join(root, "pipeline", "scripts", "validate-triage.mjs"),
48
- };
49
-
50
- const args = process.argv.slice(2);
51
- const opts = { json: false, case: null };
52
- for (let i = 0; i < args.length; i++) {
53
- if (args[i] === "--json") opts.json = true;
54
- else if (args[i] === "--case") opts.case = args[++i];
55
- else if (args[i] === "-h" || args[i] === "--help") {
56
- console.log("Usage: eval-golden-tasks.mjs [--case <name>] [--json]");
57
- process.exit(0);
58
- }
59
- }
60
-
61
- function listCases() {
62
- if (!existsSync(evalDir)) return [];
63
- return readdirSync(evalDir)
64
- .filter((f) => f !== "README.md" && statSync(join(evalDir, f)).isDirectory())
65
- .sort();
66
- }
67
-
68
- function runValidator(scriptPath, payloadPath) {
69
- // validate-reviewer.mjs reads stdin with `-`; others take a path.
70
- // For simplicity, all validators accept either stdin or file path. Pass path.
71
- const r = spawnSync("node", [scriptPath, payloadPath], { encoding: "utf-8" });
72
- return { code: r.status, stdout: r.stdout || "", stderr: r.stderr || "" };
73
- }
74
-
75
- function runCase(name) {
76
- const dir = join(evalDir, name);
77
- const errors = [];
78
-
79
- const taskPath = join(dir, "task.json");
80
- const analysisPath = join(dir, "expected", "phase-1-analysis.json");
81
- const planPath = join(dir, "expected", "phase-2-plan.json");
82
- const reviewPath = join(dir, "expected", "phase-4-review.json");
83
- const triagePath = join(dir, "expected", "phase-4-triage.json");
84
-
85
- for (const [label, p] of [
86
- ["task.json", taskPath],
87
- ["phase-1-analysis.json", analysisPath],
88
- ["phase-2-plan.json", planPath],
89
- ["phase-4-review.json", reviewPath],
90
- ["phase-4-triage.json", triagePath],
91
- ]) {
92
- if (!existsSync(p)) {
93
- errors.push(`missing ${label}`);
94
- }
95
- }
96
- if (errors.length) return { name, ok: false, errors };
97
-
98
- let task, analysis, plan, review, triage;
99
- try {
100
- task = JSON.parse(readFileSync(taskPath, "utf-8"));
101
- analysis = JSON.parse(readFileSync(analysisPath, "utf-8"));
102
- plan = JSON.parse(readFileSync(planPath, "utf-8"));
103
- review = JSON.parse(readFileSync(reviewPath, "utf-8"));
104
- triage = JSON.parse(readFileSync(triagePath, "utf-8"));
105
- } catch (e) {
106
- return { name, ok: false, errors: [`JSON parse: ${e.message}`] };
107
- }
108
-
109
- // --- schema checks ---
110
- const a = runValidator(validators.analysis, analysisPath);
111
- if (a.code !== 0)
112
- errors.push(`phase-1-analysis fails validate-analysis (exit ${a.code}): ${a.stderr.trim()}`);
113
-
114
- const p = runValidator(validators.planning, planPath);
115
- if (p.code !== 0)
116
- errors.push(`phase-2-plan fails validate-planning (exit ${p.code}): ${p.stderr.trim()}`);
117
-
118
- // phase-4-review is an array (one entry per reviewer) - validate each
119
- if (!Array.isArray(review)) {
120
- errors.push("phase-4-review.json must be a JSON array (one entry per reviewer)");
121
- } else {
122
- review.forEach((entry, idx) => {
123
- // write temp file for each reviewer entry
124
- const tmpPayload = JSON.stringify(entry);
125
- const r = spawnSync("node", [validators.reviewer, "-"], {
126
- input: tmpPayload,
127
- encoding: "utf-8",
128
- });
129
- if (r.status !== 0)
130
- errors.push(
131
- `phase-4-review[${idx}] fails validate-reviewer (exit ${r.status}): ${(r.stderr || "").trim()}`,
132
- );
133
- });
134
- }
135
-
136
- const t = runValidator(validators.triage, triagePath);
137
- // validate-triage exit codes: 0 valid+clean, 2 contradiction, 3 correction.
138
- // For fixtures we accept 0 only.
139
- if (t.code !== 0)
140
- errors.push(`phase-4-triage fails validate-triage (exit ${t.code}): ${t.stderr.trim()}`);
141
-
142
- // --- cross-file consistency ---
143
-
144
- // 1. Stack match
145
- if (analysis.stack?.primary !== task.expectedStack) {
146
- errors.push(
147
- `stack mismatch: task.expectedStack=${task.expectedStack}, phase-1.stack.primary=${analysis.stack?.primary}`,
148
- );
149
- }
150
-
151
- // 2. plan files ⊆ analysis touchedAreas paths
152
- const touched = new Set((analysis.touchedAreas || []).map((a) => a.path));
153
- for (const task of plan.tasks || []) {
154
- for (const f of task.files || []) {
155
- if (!touched.has(f)) {
156
- errors.push(`plan task ${task.id} references file not in Phase 1 touchedAreas: ${f}`);
157
- }
158
- }
159
- }
160
-
161
- // 3. every reviewer finding traces to a planned file OR appears in triage
162
- // (as accepted / deferred / rejected).
163
- const plannedFiles = new Set();
164
- for (const task of plan.tasks || []) for (const f of task.files || []) plannedFiles.add(f);
165
-
166
- const triageAll = [
167
- ...(triage.accepted || []).map((x) => x),
168
- ...(triage.deferred || []).map((x) => x.finding),
169
- ...(triage.rejected || []).map((x) => x.finding),
170
- ].filter((f) => f && typeof f === "object");
171
- const triageKeys = new Set(triageAll.map((f) => `${f.file}::${f.line}::${f.issue}`));
172
-
173
- for (const rev of Array.isArray(review) ? review : []) {
174
- for (const f of rev.findings || []) {
175
- const key = `${f.file}::${f.line}::${f.issue}`;
176
- const inPlan = plannedFiles.has(f.file);
177
- const inTriage = triageKeys.has(key);
178
- if (!inPlan && !inTriage) {
179
- errors.push(`review finding not in plan+triage: ${key}`);
180
- }
181
- }
182
- }
183
-
184
- // 4. blocker + deferral counts match task.expected*
185
- const acceptedBlockers = (triage.accepted || []).filter((x) => x.severity === "blocking").length;
186
- if (typeof task.expectedBlockers === "number" && acceptedBlockers !== task.expectedBlockers) {
187
- errors.push(
188
- `expectedBlockers=${task.expectedBlockers}, triage.accepted blocking count=${acceptedBlockers}`,
189
- );
190
- }
191
- const deferredCount = (triage.deferred || []).length;
192
- if (typeof task.expectedDeferrals === "number" && deferredCount !== task.expectedDeferrals) {
193
- errors.push(
194
- `expectedDeferrals=${task.expectedDeferrals}, triage.deferred count=${deferredCount}`,
195
- );
196
- }
197
-
198
- return { name, ok: errors.length === 0, errors };
199
- }
200
-
201
- // --- main ---
202
-
203
- const cases = opts.case ? [opts.case] : listCases();
204
- if (cases.length === 0) {
205
- console.error("No golden-task fixtures found under pipeline/eval/golden-tasks/");
206
- process.exit(2);
207
- }
208
-
209
- const results = cases.map(runCase);
210
- const passed = results.filter((r) => r.ok).length;
211
- const failed = results.length - passed;
212
-
213
- if (opts.json) {
214
- console.log(JSON.stringify({ passed, failed, results }, null, 2));
215
- } else {
216
- for (const r of results) {
217
- const mark = r.ok ? "\x1b[32m✓\x1b[0m" : "\x1b[31m✗\x1b[0m";
218
- console.log(`${mark} ${r.name.padEnd(38)} ${r.ok ? "" : "FAIL"}`);
219
- if (!r.ok) for (const e of r.errors) console.log(` - ${e}`);
220
- }
221
- console.log(`\n══ eval-golden-tasks: ${passed}/${results.length} passed ══`);
222
- }
223
-
224
- process.exit(failed > 0 ? 1 : 0);
@@ -1,107 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- /**
4
- * @file eval-intent.mjs - measured accuracy for the Phase 0 intent guard.
5
- *
6
- * Runs every labeled case in pipeline/eval/intent-cases.json through
7
- * pipeline/lib/classify-intent.sh and reports accuracy. This turns the
8
- * classifier from "a heuristic we hope works" into a number with a regression
9
- * gate: a dangerous miss is a clear task read as `question` (work skipped) or a
10
- * clear question read as `task` (a worktree spun up for nothing). `ambiguous`
11
- * is the safe middle and is only correct when the case labels it so.
12
- *
13
- * Exit 0 if accuracy >= threshold (default 0.95), else 1. Override with
14
- * --min <0..1>. Prints a JSON summary + the mismatches.
15
- *
16
- * @module pipeline/scripts/eval-intent
17
- */
18
-
19
- import { readFileSync } from "node:fs";
20
- import { execFileSync } from "node:child_process";
21
- import { join, dirname } from "node:path";
22
- import { fileURLToPath } from "node:url";
23
-
24
- const HERE = dirname(fileURLToPath(import.meta.url));
25
- const ROOT = join(HERE, "..", "..");
26
- const CLASSIFY = join(ROOT, "pipeline", "lib", "classify-intent.sh");
27
- const CASES = join(ROOT, "pipeline", "eval", "intent-cases.json");
28
-
29
- const argv = process.argv.slice(2);
30
- let minAccuracy = 0.95;
31
- for (let i = 0; i < argv.length; i++) {
32
- if (argv[i] === "--min") {
33
- minAccuracy = Number(argv[i + 1]);
34
- // A typo'd `--min` with no/NaN value would set the threshold to NaN, and
35
- // `accuracy < NaN` is always false - silently turning the gate into a
36
- // no-op. Fail loudly instead.
37
- if (!Number.isFinite(minAccuracy) || minAccuracy < 0 || minAccuracy > 1) {
38
- console.error(`eval-intent: --min needs a number in [0,1], got: ${argv[i + 1]}`);
39
- process.exit(2);
40
- }
41
- }
42
- }
43
-
44
- function classify(input) {
45
- try {
46
- return execFileSync("bash", [CLASSIFY, input], { encoding: "utf8" }).trim();
47
- } catch {
48
- return "ERROR";
49
- }
50
- }
51
-
52
- const { cases } = JSON.parse(readFileSync(CASES, "utf8"));
53
- if (!Array.isArray(cases) || cases.length === 0) {
54
- console.error("eval-intent: no cases found");
55
- process.exit(2);
56
- }
57
-
58
- // Operationally-safe match: only TWO outcomes are dangerous, and they are what
59
- // the gate measures -
60
- // - a `task` read as `question` -> real work silently skipped
61
- // - a `question` read as `task` -> a worktree spun up for nothing
62
- // `ambiguous` proceeds as a task in Phase 0, so it is SAFE for a task/ambiguous
63
- // case but UNSAFE for a question case (ambiguous still avoids the worktree, so
64
- // it is acceptable for a question too - only `task` is the dangerous read).
65
- function isSafe(expected, got) {
66
- if (got === expected) return true;
67
- if (expected === "task") return got === "ambiguous"; // proceeds as task anyway
68
- if (expected === "ambiguous") return got === "task"; // both proceed to dev
69
- if (expected === "question") return got === "ambiguous"; // still no worktree spun up
70
- return false;
71
- }
72
-
73
- const mismatches = []; // operationally unsafe (gate-failing)
74
- const exactMisses = []; // not exact, but operationally safe (informational)
75
- let safe = 0;
76
- let exact = 0;
77
- for (const c of cases) {
78
- const got = classify(c.input);
79
- if (got === c.expected) exact++;
80
- else exactMisses.push({ input: c.input, expected: c.expected, got });
81
- if (isSafe(c.expected, got)) safe++;
82
- else mismatches.push({ input: c.input, expected: c.expected, got, danger: true });
83
- }
84
-
85
- const accuracy = safe / cases.length;
86
- const summary = {
87
- total: cases.length,
88
- safeMatches: safe,
89
- safeAccuracy: Math.round(accuracy * 1000) / 1000,
90
- exactMatches: exact,
91
- exactAccuracy: Math.round((exact / cases.length) * 1000) / 1000,
92
- minAccuracy,
93
- dangerousMisclassifications: mismatches,
94
- safeButInexact: exactMisses,
95
- };
96
- console.log(JSON.stringify(summary, null, 2));
97
-
98
- if (accuracy < minAccuracy) {
99
- console.error(
100
- `\neval-intent: safe accuracy ${(accuracy * 100).toFixed(1)}% < required ${(minAccuracy * 100).toFixed(0)}% (${mismatches.length} dangerous misclassification(s))`,
101
- );
102
- process.exit(1);
103
- }
104
- console.error(
105
- `\neval-intent: safe ${(accuracy * 100).toFixed(1)}% (${safe}/${cases.length}), exact ${((exact / cases.length) * 100).toFixed(1)}% - pass`,
106
- );
107
- process.exit(0);