@mmerterden/multi-agent-pipeline 12.6.0 → 12.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (481) hide show
  1. package/CHANGELOG.md +209 -0
  2. package/README.md +18 -18
  3. package/docs/FIGMA_PIPELINE.md +34 -34
  4. package/docs/adr/0001-three-model-triage.md +12 -12
  5. package/docs/adr/0002-instruction-driven-flag.md +5 -5
  6. package/docs/adr/0003-unified-shared-skills.md +5 -5
  7. package/docs/adr/0004-zero-dependency-philosophy.md +5 -5
  8. package/docs/adr/0005-lazy-phase-docs.md +2 -2
  9. package/docs/adr/0006-skills-core-external-split.md +6 -6
  10. package/docs/adr/0007-multi-tool-adapter-framework.md +19 -19
  11. package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +19 -19
  12. package/docs/adr/README.md +1 -1
  13. package/docs/best-practices.md +3 -3
  14. package/docs/features.md +28 -28
  15. package/docs/performance.md +16 -16
  16. package/docs/recovery-guide.md +39 -39
  17. package/index.js +4 -4
  18. package/install/_common.mjs +53 -11
  19. package/install/_copilot-instructions.mjs +2 -2
  20. package/install/_dev-only-files.mjs +126 -6
  21. package/install/_platform-filter.mjs +1 -1
  22. package/install/_telemetry.mjs +1 -1
  23. package/install/claude.mjs +20 -13
  24. package/install/copilot.mjs +11 -23
  25. package/install/index.mjs +7 -15
  26. package/install/templates/copilot-instructions.md +54 -54
  27. package/install.js +1 -1
  28. package/package.json +29 -11
  29. package/pipeline/commands/multi-agent/SKILL.md +1 -1
  30. package/pipeline/commands/multi-agent/analysis/SKILL.md +2 -2
  31. package/pipeline/commands/multi-agent/analysis-resolve/SKILL.md +1 -1
  32. package/pipeline/commands/multi-agent/autopilot/SKILL.md +1 -1
  33. package/pipeline/commands/multi-agent/build-optimize/SKILL.md +1 -1
  34. package/pipeline/commands/multi-agent/create-jira/SKILL.md +1 -1
  35. package/pipeline/commands/multi-agent/design-check/SKILL.md +1 -1
  36. package/pipeline/commands/multi-agent/dev/SKILL.md +1 -1
  37. package/pipeline/commands/multi-agent/dev-autopilot/SKILL.md +1 -1
  38. package/pipeline/commands/multi-agent/dev-local/SKILL.md +1 -1
  39. package/pipeline/commands/multi-agent/dev-local-autopilot/SKILL.md +1 -1
  40. package/pipeline/commands/multi-agent/diff-explain/SKILL.md +1 -1
  41. package/pipeline/commands/multi-agent/finish/SKILL.md +6 -6
  42. package/pipeline/commands/multi-agent/forget/SKILL.md +1 -1
  43. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  44. package/pipeline/commands/multi-agent/help/SKILL.md +3 -3
  45. package/pipeline/commands/multi-agent/issue/SKILL.md +1 -1
  46. package/pipeline/commands/multi-agent/jira/SKILL.md +1 -1
  47. package/pipeline/commands/multi-agent/kill/SKILL.md +1 -1
  48. package/pipeline/commands/multi-agent/language/SKILL.md +1 -1
  49. package/pipeline/commands/multi-agent/local/SKILL.md +1 -1
  50. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +1 -1
  51. package/pipeline/commands/multi-agent/log/SKILL.md +1 -1
  52. package/pipeline/commands/multi-agent/manual-test/SKILL.md +1 -1
  53. package/pipeline/commands/multi-agent/prune-logs/SKILL.md +1 -1
  54. package/pipeline/commands/multi-agent/purge/SKILL.md +1 -1
  55. package/pipeline/commands/multi-agent/refactor/SKILL.md +16 -8
  56. package/pipeline/commands/multi-agent/resume/SKILL.md +2 -2
  57. package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
  58. package/pipeline/commands/multi-agent/review-issue/SKILL.md +1 -1
  59. package/pipeline/commands/multi-agent/review-jira/SKILL.md +1 -1
  60. package/pipeline/commands/multi-agent/routines/SKILL.md +1 -1
  61. package/pipeline/commands/multi-agent/save/SKILL.md +1 -1
  62. package/pipeline/commands/multi-agent/scan/SKILL.md +1 -1
  63. package/pipeline/commands/multi-agent/search/SKILL.md +1 -1
  64. package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
  65. package/pipeline/commands/multi-agent/stack/SKILL.md +3 -3
  66. package/pipeline/commands/multi-agent/status/SKILL.md +1 -1
  67. package/pipeline/commands/multi-agent/sync/SKILL.md +5 -5
  68. package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
  69. package/pipeline/commands/multi-agent/uninstall/SKILL.md +1 -1
  70. package/pipeline/commands/multi-agent/update/SKILL.md +3 -3
  71. package/pipeline/lib/account-resolver.sh +1 -1
  72. package/pipeline/lib/channels-multi-repo.sh +1 -1
  73. package/pipeline/lib/context-link-extractor.sh +1 -1
  74. package/pipeline/lib/credential-store.sh +21 -1
  75. package/pipeline/lib/fetch-confluence.sh +1 -1
  76. package/pipeline/lib/fetch-crashlytics.sh +1 -1
  77. package/pipeline/lib/fetch-fortify.sh +1 -1
  78. package/pipeline/lib/fetch-graylog.sh +1 -1
  79. package/pipeline/lib/fetch-swagger.sh +1 -1
  80. package/pipeline/lib/issue-fetcher.sh +1 -1
  81. package/pipeline/lib/multi-repo-pipeline.sh +1 -1
  82. package/pipeline/lib/repo-cache.sh +1 -1
  83. package/pipeline/lib/submodule-detector.sh +1 -1
  84. package/pipeline/multi-agent-refs/_account-picker.md +1 -1
  85. package/pipeline/multi-agent-refs/_dev-context.md +1 -1
  86. package/pipeline/multi-agent-refs/_repo-picker.md +1 -1
  87. package/pipeline/multi-agent-refs/component-dispatch.md +1 -1
  88. package/pipeline/multi-agent-refs/component-generation.md +121 -0
  89. package/pipeline/multi-agent-refs/cross-cli-contract.md +1 -1
  90. package/pipeline/multi-agent-refs/features/model-fallback.md +2 -2
  91. package/pipeline/multi-agent-refs/phases/operations.md +28 -0
  92. package/pipeline/multi-agent-refs/phases/phase-0-init.md +1 -0
  93. package/pipeline/multi-agent-refs/phases/phase-2-planning.md +1 -1
  94. package/pipeline/multi-agent-refs/phases/phase-3-dev.md +1 -2
  95. package/pipeline/multi-agent-refs/phases/phase-4-review.md +50 -5
  96. package/pipeline/preferences-template.json +5 -11
  97. package/pipeline/schemas/agent-state.schema.json +39 -9
  98. package/pipeline/schemas/analysis-output.schema.json +18 -4
  99. package/pipeline/schemas/analysis-spec.schema.json +120 -32
  100. package/pipeline/schemas/clarify-output.schema.json +15 -5
  101. package/pipeline/schemas/design-check-config.schema.json +32 -11
  102. package/pipeline/schemas/dev-critic-output.schema.json +20 -5
  103. package/pipeline/schemas/figma-project-config.schema.json +42 -10
  104. package/pipeline/schemas/learnings-ledger.schema.json +10 -2
  105. package/pipeline/schemas/migrations/figma-config-1.0.0-to-2.0.0.mjs +1 -4
  106. package/pipeline/schemas/migrations/prefs-2.0.0-to-2.1.0.mjs +24 -7
  107. package/pipeline/schemas/plan-todos.schema.json +6 -3
  108. package/pipeline/schemas/planning-output.schema.json +5 -1
  109. package/pipeline/schemas/prefs.schema.json +97 -229
  110. package/pipeline/schemas/test-gap.schema.json +5 -5
  111. package/pipeline/schemas/token-budget.json +8 -8
  112. package/pipeline/schemas/triage-corpus.schema.json +1 -1
  113. package/pipeline/scripts/_smoke-root.sh +61 -0
  114. package/pipeline/scripts/aggregate-metrics.mjs +18 -6
  115. package/pipeline/scripts/audit-log.sh +25 -0
  116. package/pipeline/scripts/build-skills-index.mjs +6 -2
  117. package/pipeline/scripts/build-stack-plugins.mjs +142 -39
  118. package/pipeline/scripts/check-derived-drift.mjs +196 -0
  119. package/pipeline/scripts/classify-plan-safety.mjs +20 -7
  120. package/pipeline/scripts/cost-budget-check.mjs +2 -1
  121. package/pipeline/scripts/cost-table.json +1 -1
  122. package/pipeline/scripts/diff-explain.mjs +7 -3
  123. package/pipeline/scripts/diff-risk-score.mjs +13 -3
  124. package/pipeline/scripts/evidence-gate.mjs +7 -2
  125. package/pipeline/scripts/gen-mode-dispatch.mjs +38 -21
  126. package/pipeline/scripts/gen-skills-index.mjs +18 -3
  127. package/pipeline/scripts/learning-curve.mjs +13 -3
  128. package/pipeline/scripts/learnings-ledger.mjs +103 -36
  129. package/pipeline/scripts/localize-commands.mjs +6 -1
  130. package/pipeline/scripts/match-skills.mjs +15 -4
  131. package/pipeline/scripts/migrate-prefs.mjs +33 -16
  132. package/pipeline/scripts/phase-tracker.sh +3 -1
  133. package/pipeline/scripts/repo-map.mjs +110 -64
  134. package/pipeline/scripts/review-scope.mjs +7 -1
  135. package/pipeline/scripts/routine-registry.mjs +4 -9
  136. package/pipeline/scripts/run-aggregator.mjs +11 -5
  137. package/pipeline/scripts/run-metrics.mjs +13 -8
  138. package/pipeline/scripts/smoke-cross-cli-behavior.sh +21 -7
  139. package/pipeline/scripts/test-gap-scan.mjs +44 -12
  140. package/pipeline/scripts/test-integrity-gate.mjs +5 -1
  141. package/pipeline/scripts/token-budget-report.mjs +44 -21
  142. package/pipeline/scripts/triage-memory.mjs +126 -34
  143. package/pipeline/scripts/uninstall.mjs +74 -30
  144. package/pipeline/scripts/validate-analysis-doc.mjs +15 -5
  145. package/pipeline/scripts/validate-diff-risk.mjs +32 -18
  146. package/pipeline/scripts/validate-test-gap.mjs +17 -7
  147. package/pipeline/scripts/validate-triage.mjs +17 -5
  148. package/pipeline/scripts/write-state.mjs +32 -9
  149. package/pipeline/skills/.skills-index.json +91 -91
  150. package/pipeline/skills/shared/README.md +57 -57
  151. package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +1 -1
  152. package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +1 -1
  153. package/pipeline/skills/shared/core/multi-agent/SKILL.md +26 -279
  154. package/pipeline/skills/shared/core/multi-agent-analysis/SKILL.md +1 -1
  155. package/pipeline/skills/shared/core/multi-agent-analysis-resolve/SKILL.md +1 -1
  156. package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +1 -1
  157. package/pipeline/skills/shared/core/multi-agent-build-optimize/SKILL.md +1 -1
  158. package/pipeline/skills/shared/core/multi-agent-create-jira/SKILL.md +1 -1
  159. package/pipeline/skills/shared/core/multi-agent-design-check/SKILL.md +1 -1
  160. package/pipeline/skills/shared/core/multi-agent-dev/SKILL.md +1 -1
  161. package/pipeline/skills/shared/core/multi-agent-dev-autopilot/SKILL.md +1 -1
  162. package/pipeline/skills/shared/core/multi-agent-dev-local/SKILL.md +1 -1
  163. package/pipeline/skills/shared/core/multi-agent-dev-local-autopilot/SKILL.md +1 -1
  164. package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +1 -1
  165. package/pipeline/skills/shared/core/multi-agent-finish/SKILL.md +1 -1
  166. package/pipeline/skills/shared/core/multi-agent-forget/SKILL.md +1 -1
  167. package/pipeline/skills/shared/core/multi-agent-garbage-collect/SKILL.md +1 -1
  168. package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +1 -1
  169. package/pipeline/skills/shared/core/multi-agent-issue/SKILL.md +1 -1
  170. package/pipeline/skills/shared/core/multi-agent-jira/SKILL.md +1 -1
  171. package/pipeline/skills/shared/core/multi-agent-kill/SKILL.md +1 -1
  172. package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +1 -1
  173. package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +1 -1
  174. package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +1 -1
  175. package/pipeline/skills/shared/core/multi-agent-log/SKILL.md +1 -1
  176. package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +1 -1
  177. package/pipeline/skills/shared/core/multi-agent-prune-logs/SKILL.md +1 -1
  178. package/pipeline/skills/shared/core/multi-agent-purge/SKILL.md +1 -1
  179. package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +16 -8
  180. package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
  181. package/pipeline/skills/shared/core/multi-agent-review/SKILL.md +1 -1
  182. package/pipeline/skills/shared/core/multi-agent-review-issue/SKILL.md +1 -1
  183. package/pipeline/skills/shared/core/multi-agent-review-jira/SKILL.md +1 -1
  184. package/pipeline/skills/shared/core/multi-agent-routines/SKILL.md +1 -1
  185. package/pipeline/skills/shared/core/multi-agent-save/SKILL.md +1 -1
  186. package/pipeline/skills/shared/core/multi-agent-scan/SKILL.md +1 -1
  187. package/pipeline/skills/shared/core/multi-agent-search/SKILL.md +1 -1
  188. package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
  189. package/pipeline/skills/shared/core/multi-agent-stack/SKILL.md +3 -3
  190. package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +1 -1
  191. package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +1 -1
  192. package/pipeline/skills/shared/core/multi-agent-test/SKILL.md +1 -1
  193. package/pipeline/skills/shared/core/multi-agent-uninstall/SKILL.md +1 -1
  194. package/pipeline/skills/shared/core/multi-agent-update/SKILL.md +1 -1
  195. package/pipeline/skills/shared/external/accessibility-compliance-accessibility-audit/SKILL.md +1 -1
  196. package/pipeline/skills/shared/external/agent-introspection-debugging/SKILL.md +4 -4
  197. package/pipeline/skills/shared/external/agentflow/SKILL.md +1 -1
  198. package/pipeline/skills/shared/external/android-jetpack-compose-expert/SKILL.md +1 -1
  199. package/pipeline/skills/shared/external/android_ui_verification/SKILL.md +1 -1
  200. package/pipeline/skills/shared/external/api-patterns/SKILL.md +1 -1
  201. package/pipeline/skills/shared/external/api-security-best-practices/SKILL.md +1 -1
  202. package/pipeline/skills/shared/external/app-store-changelog/SKILL.md +1 -1
  203. package/pipeline/skills/shared/external/backlog/BACKLOG.md +1 -1
  204. package/pipeline/skills/shared/external/backlog/SKILL.md +12 -12
  205. package/pipeline/skills/shared/external/ci-cd-pipelines/SKILL.md +1 -1
  206. package/pipeline/skills/shared/external/context-compression/SKILL.md +1 -1
  207. package/pipeline/skills/shared/external/council/SKILL.md +3 -3
  208. package/pipeline/skills/shared/external/css-modern/SKILL.md +1 -1
  209. package/pipeline/skills/shared/external/database-patterns/SKILL.md +1 -1
  210. package/pipeline/skills/shared/external/debugging-strategies/SKILL.md +1 -1
  211. package/pipeline/skills/shared/external/docker-expert/SKILL.md +1 -1
  212. package/pipeline/skills/shared/external/fastapi-pro/SKILL.md +1 -1
  213. package/pipeline/skills/shared/external/firebase/SKILL.md +1 -1
  214. package/pipeline/skills/shared/external/github-actions-templates/SKILL.md +1 -1
  215. package/pipeline/skills/shared/external/help-skills/SKILL.md +1 -1
  216. package/pipeline/skills/shared/external/hig-components-content/SKILL.md +1 -1
  217. package/pipeline/skills/shared/external/hig-components-layout/SKILL.md +1 -1
  218. package/pipeline/skills/shared/external/hig-components-status/SKILL.md +1 -1
  219. package/pipeline/skills/shared/external/hig-components-system/SKILL.md +1 -1
  220. package/pipeline/skills/shared/external/hig-foundations/SKILL.md +1 -1
  221. package/pipeline/skills/shared/external/hig-inputs/SKILL.md +1 -1
  222. package/pipeline/skills/shared/external/hig-patterns/SKILL.md +1 -1
  223. package/pipeline/skills/shared/external/hig-platforms/SKILL.md +1 -1
  224. package/pipeline/skills/shared/external/hig-technologies/SKILL.md +1 -1
  225. package/pipeline/skills/shared/external/html-semantic/SKILL.md +1 -1
  226. package/pipeline/skills/shared/external/humanizer/SKILL.md +1 -1
  227. package/pipeline/skills/shared/external/ios-debugger-agent/SKILL.md +1 -1
  228. package/pipeline/skills/shared/external/ios-developer/SKILL.md +1 -1
  229. package/pipeline/skills/shared/external/kotlin-coroutines-expert/SKILL.md +1 -1
  230. package/pipeline/skills/shared/external/macos-menubar-tuist-app/SKILL.md +1 -1
  231. package/pipeline/skills/shared/external/macos-spm-app-packaging/SKILL.md +1 -1
  232. package/pipeline/skills/shared/external/monorepo-architect/SKILL.md +1 -1
  233. package/pipeline/skills/shared/external/nextjs-app-router/SKILL.md +1 -1
  234. package/pipeline/skills/shared/external/nodejs-backend-patterns/SKILL.md +1 -1
  235. package/pipeline/skills/shared/external/observability-engineer/SKILL.md +1 -1
  236. package/pipeline/skills/shared/external/python-patterns/SKILL.md +1 -1
  237. package/pipeline/skills/shared/external/react-best-practices/SKILL.md +1 -1
  238. package/pipeline/skills/shared/external/rest-api-design/SKILL.md +1 -1
  239. package/pipeline/skills/shared/external/search-first/SKILL.md +2 -2
  240. package/pipeline/skills/shared/external/skill-creator/SKILL.md +12 -12
  241. package/pipeline/skills/shared/external/skill-creator/audit.md +21 -21
  242. package/pipeline/skills/shared/external/skill-creator/checklist.md +3 -3
  243. package/pipeline/skills/shared/external/skill-creator/examples.md +10 -10
  244. package/pipeline/skills/shared/external/skill-creator/label-check.md +17 -17
  245. package/pipeline/skills/shared/external/skill-creator/scripts/audit-panel.js +86 -50
  246. package/pipeline/skills/shared/external/skill-creator/template.md +9 -9
  247. package/pipeline/skills/shared/external/swift-concurrency-expert/SKILL.md +1 -1
  248. package/pipeline/skills/shared/external/swiftui-performance-audit/SKILL.md +1 -1
  249. package/pipeline/skills/shared/external/swiftui-ui-patterns/SKILL.md +1 -1
  250. package/pipeline/skills/shared/external/swiftui-view-refactor/SKILL.md +1 -1
  251. package/pipeline/skills/shared/external/tailwind-css/SKILL.md +1 -1
  252. package/pipeline/skills/shared/external/testing-backend/SKILL.md +1 -1
  253. package/pipeline/skills/shared/external/typescript-patterns/SKILL.md +1 -1
  254. package/pipeline/skills/shared/external/vue-composition/SKILL.md +1 -1
  255. package/pipeline/skills/shared/external/web-accessibility/SKILL.md +1 -1
  256. package/pipeline/skills/shared/external/web-performance/SKILL.md +1 -1
  257. package/pipeline/skills/shared/external/web-testing/SKILL.md +1 -1
  258. package/pipeline/skills/shared/external/xcode-build-benchmark/schemas/build-benchmark.schema.json +9 -49
  259. package/pipeline/skills/skills-index.md +57 -57
  260. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-1-analysis.json +0 -25
  261. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-2-plan.json +0 -30
  262. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-review.json +0 -20
  263. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/expected/phase-4-triage.json +0 -15
  264. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/metadata.json +0 -14
  265. package/pipeline/eval/golden-tasks/01-ios-bugfix-darkmode/task.json +0 -12
  266. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-1-analysis.json +0 -29
  267. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-2-plan.json +0 -43
  268. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-review.json +0 -35
  269. package/pipeline/eval/golden-tasks/02-android-feature-compose/expected/phase-4-triage.json +0 -35
  270. package/pipeline/eval/golden-tasks/02-android-feature-compose/metadata.json +0 -14
  271. package/pipeline/eval/golden-tasks/02-android-feature-compose/task.json +0 -12
  272. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-1-analysis.json +0 -29
  273. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-2-plan.json +0 -42
  274. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-review.json +0 -20
  275. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/expected/phase-4-triage.json +0 -15
  276. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/metadata.json +0 -14
  277. package/pipeline/eval/golden-tasks/03-backend-python-ratelimit/task.json +0 -12
  278. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-1-analysis.json +0 -29
  279. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-2-plan.json +0 -40
  280. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-review.json +0 -20
  281. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/expected/phase-4-triage.json +0 -15
  282. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/metadata.json +0 -14
  283. package/pipeline/eval/golden-tasks/04-frontend-next-hydration/task.json +0 -12
  284. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-1-analysis.json +0 -29
  285. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-2-plan.json +0 -42
  286. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-review.json +0 -28
  287. package/pipeline/eval/golden-tasks/05-ios-security-keychain/expected/phase-4-triage.json +0 -27
  288. package/pipeline/eval/golden-tasks/05-ios-security-keychain/metadata.json +0 -14
  289. package/pipeline/eval/golden-tasks/05-ios-security-keychain/task.json +0 -12
  290. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-1-analysis.json +0 -29
  291. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-2-plan.json +0 -41
  292. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-review.json +0 -12
  293. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/expected/phase-4-triage.json +0 -6
  294. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/metadata.json +0 -14
  295. package/pipeline/eval/golden-tasks/06-android-refactor-usecase/task.json +0 -12
  296. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-1-analysis.json +0 -29
  297. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-2-plan.json +0 -42
  298. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-review.json +0 -28
  299. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/expected/phase-4-triage.json +0 -27
  300. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/metadata.json +0 -14
  301. package/pipeline/eval/golden-tasks/07-backend-node-idempotency/task.json +0 -12
  302. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-1-analysis.json +0 -25
  303. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-2-plan.json +0 -31
  304. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-review.json +0 -12
  305. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/expected/phase-4-triage.json +0 -18
  306. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/metadata.json +0 -14
  307. package/pipeline/eval/golden-tasks/08-ios-auth-consensus-unverified/task.json +0 -12
  308. package/pipeline/eval/golden-tasks/README.md +0 -65
  309. package/pipeline/eval/intent-cases.json +0 -40
  310. package/pipeline/eval/run-metrics-fixture.json +0 -46
  311. package/pipeline/eval/triage/01-empty-findings/expected.json +0 -6
  312. package/pipeline/eval/triage/01-empty-findings/input.json +0 -5
  313. package/pipeline/eval/triage/01-empty-findings/notes.md +0 -7
  314. package/pipeline/eval/triage/02-real-blocker/expected.json +0 -15
  315. package/pipeline/eval/triage/02-real-blocker/input.json +0 -14
  316. package/pipeline/eval/triage/02-real-blocker/notes.md +0 -7
  317. package/pipeline/eval/triage/03-out-of-scope-defer/expected.json +0 -18
  318. package/pipeline/eval/triage/03-out-of-scope-defer/input.json +0 -14
  319. package/pipeline/eval/triage/03-out-of-scope-defer/notes.md +0 -10
  320. package/pipeline/eval/triage/04-false-positive-reject/expected.json +0 -18
  321. package/pipeline/eval/triage/04-false-positive-reject/input.json +0 -14
  322. package/pipeline/eval/triage/04-false-positive-reject/notes.md +0 -10
  323. package/pipeline/eval/triage/05-mixed-classification/expected.json +0 -43
  324. package/pipeline/eval/triage/05-mixed-classification/input.json +0 -38
  325. package/pipeline/eval/triage/05-mixed-classification/notes.md +0 -17
  326. package/pipeline/eval/triage/06-severity-mismatch/expected.json +0 -15
  327. package/pipeline/eval/triage/06-severity-mismatch/input.json +0 -14
  328. package/pipeline/eval/triage/06-severity-mismatch/notes.md +0 -9
  329. package/pipeline/eval/triage/07-duplicate-reviewers/expected.json +0 -27
  330. package/pipeline/eval/triage/07-duplicate-reviewers/input.json +0 -22
  331. package/pipeline/eval/triage/07-duplicate-reviewers/notes.md +0 -9
  332. package/pipeline/eval/triage/08-style-misclassified/expected.json +0 -18
  333. package/pipeline/eval/triage/08-style-misclassified/input.json +0 -14
  334. package/pipeline/eval/triage/08-style-misclassified/notes.md +0 -9
  335. package/pipeline/eval/triage/09-cascading-finding/expected.json +0 -23
  336. package/pipeline/eval/triage/09-cascading-finding/input.json +0 -22
  337. package/pipeline/eval/triage/09-cascading-finding/notes.md +0 -9
  338. package/pipeline/eval/triage/10-deferred-crossref/expected.json +0 -18
  339. package/pipeline/eval/triage/10-deferred-crossref/input.json +0 -14
  340. package/pipeline/eval/triage/10-deferred-crossref/notes.md +0 -9
  341. package/pipeline/eval/triage/11-vercel-token-leak-blocker/expected.json +0 -27
  342. package/pipeline/eval/triage/11-vercel-token-leak-blocker/input.json +0 -22
  343. package/pipeline/eval/triage/11-vercel-token-leak-blocker/notes.md +0 -14
  344. package/pipeline/eval/triage/README.md +0 -54
  345. package/pipeline/scripts/benchmark-phase-0.sh +0 -128
  346. package/pipeline/scripts/check-md-links.mjs +0 -84
  347. package/pipeline/scripts/eval-golden-tasks-live.mjs +0 -297
  348. package/pipeline/scripts/eval-golden-tasks.mjs +0 -212
  349. package/pipeline/scripts/eval-intent.mjs +0 -103
  350. package/pipeline/scripts/eval-mine-corpus.mjs +0 -201
  351. package/pipeline/scripts/eval-triage.mjs +0 -171
  352. package/pipeline/scripts/fixtures/diff-risk-android.diff +0 -40
  353. package/pipeline/scripts/fixtures/diff-risk-ios.diff +0 -48
  354. package/pipeline/scripts/fixtures/diff-risk-test-removal.diff +0 -40
  355. package/pipeline/scripts/fixtures/install-layout.tsv +0 -19
  356. package/pipeline/scripts/fixtures/pack-expected-count.txt +0 -1
  357. package/pipeline/scripts/fixtures/test-gap-node.diff +0 -30
  358. package/pipeline/scripts/fixtures/test-gap-python.diff +0 -32
  359. package/pipeline/scripts/lint-mcp-refs.mjs +0 -207
  360. package/pipeline/scripts/lint-skills.mjs +0 -143
  361. package/pipeline/scripts/run-smokes.mjs +0 -76
  362. package/pipeline/scripts/smoke-add-detail.sh +0 -137
  363. package/pipeline/scripts/smoke-agent-guard.sh +0 -74
  364. package/pipeline/scripts/smoke-agent-log-cost.sh +0 -262
  365. package/pipeline/scripts/smoke-agent-model-routing.sh +0 -87
  366. package/pipeline/scripts/smoke-ask-choice.sh +0 -42
  367. package/pipeline/scripts/smoke-autopilot-circuit-breaker.sh +0 -36
  368. package/pipeline/scripts/smoke-bitbucket-contract.sh +0 -255
  369. package/pipeline/scripts/smoke-changelog-version.sh +0 -47
  370. package/pipeline/scripts/smoke-channels-approval-gate.sh +0 -60
  371. package/pipeline/scripts/smoke-channels-flow.sh +0 -130
  372. package/pipeline/scripts/smoke-ci-workflows.sh +0 -88
  373. package/pipeline/scripts/smoke-clarify.sh +0 -148
  374. package/pipeline/scripts/smoke-command-inventory.sh +0 -81
  375. package/pipeline/scripts/smoke-commands-skills-parity.sh +0 -87
  376. package/pipeline/scripts/smoke-community-gates.sh +0 -75
  377. package/pipeline/scripts/smoke-compliance-skills.sh +0 -119
  378. package/pipeline/scripts/smoke-config-hygiene.sh +0 -58
  379. package/pipeline/scripts/smoke-cost-budget.sh +0 -70
  380. package/pipeline/scripts/smoke-cost-summary.sh +0 -139
  381. package/pipeline/scripts/smoke-cross-phase-cohesion.sh +0 -128
  382. package/pipeline/scripts/smoke-description-tr.sh +0 -82
  383. package/pipeline/scripts/smoke-dev-critic.sh +0 -144
  384. package/pipeline/scripts/smoke-diff-explain.sh +0 -147
  385. package/pipeline/scripts/smoke-diff-risk.sh +0 -190
  386. package/pipeline/scripts/smoke-dynamic-skill-loading.sh +0 -160
  387. package/pipeline/scripts/smoke-eval-live.sh +0 -136
  388. package/pipeline/scripts/smoke-evidence-gate.sh +0 -93
  389. package/pipeline/scripts/smoke-extract-conventions.sh +0 -163
  390. package/pipeline/scripts/smoke-fetchers-offline.sh +0 -448
  391. package/pipeline/scripts/smoke-figma-dispatch.sh +0 -112
  392. package/pipeline/scripts/smoke-gate-hooks.sh +0 -74
  393. package/pipeline/scripts/smoke-gc-tmp.sh +0 -130
  394. package/pipeline/scripts/smoke-gc-worktrees.sh +0 -125
  395. package/pipeline/scripts/smoke-generate-issue.sh +0 -120
  396. package/pipeline/scripts/smoke-handoff-contract.sh +0 -92
  397. package/pipeline/scripts/smoke-identity-isolation.sh +0 -70
  398. package/pipeline/scripts/smoke-install-layout.sh +0 -248
  399. package/pipeline/scripts/smoke-intent-guard.sh +0 -86
  400. package/pipeline/scripts/smoke-issue-comment-template.sh +0 -86
  401. package/pipeline/scripts/smoke-issue-jira-triad.sh +0 -120
  402. package/pipeline/scripts/smoke-keychain.sh +0 -158
  403. package/pipeline/scripts/smoke-language-axis.sh +0 -109
  404. package/pipeline/scripts/smoke-learning-curve.sh +0 -61
  405. package/pipeline/scripts/smoke-learnings-ledger.sh +0 -86
  406. package/pipeline/scripts/smoke-lib-scripts.sh +0 -448
  407. package/pipeline/scripts/smoke-mcp-gate.sh +0 -68
  408. package/pipeline/scripts/smoke-md-links.sh +0 -8
  409. package/pipeline/scripts/smoke-md2confluence.sh +0 -126
  410. package/pipeline/scripts/smoke-metrics-cache-ratio.sh +0 -72
  411. package/pipeline/scripts/smoke-migrate-state.sh +0 -102
  412. package/pipeline/scripts/smoke-mode-dispatch-drift.sh +0 -161
  413. package/pipeline/scripts/smoke-model-fallback.sh +0 -89
  414. package/pipeline/scripts/smoke-multi-repo-integration.sh +0 -116
  415. package/pipeline/scripts/smoke-multi-repo-worktree.sh +0 -61
  416. package/pipeline/scripts/smoke-no-mcp-in-dev-phases.sh +0 -115
  417. package/pipeline/scripts/smoke-no-token-prompt.sh +0 -85
  418. package/pipeline/scripts/smoke-pack-contents.sh +0 -140
  419. package/pipeline/scripts/smoke-pat-audit.sh +0 -128
  420. package/pipeline/scripts/smoke-per-repo-memory.sh +0 -156
  421. package/pipeline/scripts/smoke-phase-0-multi-repo.sh +0 -170
  422. package/pipeline/scripts/smoke-phase-6-multi.sh +0 -79
  423. package/pipeline/scripts/smoke-phase-banner.sh +0 -101
  424. package/pipeline/scripts/smoke-phase-tracker.sh +0 -324
  425. package/pipeline/scripts/smoke-phase0-bridge-contract.sh +0 -241
  426. package/pipeline/scripts/smoke-phase4-gates.sh +0 -45
  427. package/pipeline/scripts/smoke-phase4-triage.sh +0 -229
  428. package/pipeline/scripts/smoke-plan-approval-gate.sh +0 -71
  429. package/pipeline/scripts/smoke-plan-safety.sh +0 -139
  430. package/pipeline/scripts/smoke-plan-todos.sh +0 -196
  431. package/pipeline/scripts/smoke-plugin-validate.sh +0 -64
  432. package/pipeline/scripts/smoke-pr-review-actions.sh +0 -152
  433. package/pipeline/scripts/smoke-pre-commit.sh +0 -170
  434. package/pipeline/scripts/smoke-pref-migration.sh +0 -226
  435. package/pipeline/scripts/smoke-prefs-language.sh +0 -134
  436. package/pipeline/scripts/smoke-progress-contract.sh +0 -127
  437. package/pipeline/scripts/smoke-prune-logs.sh +0 -137
  438. package/pipeline/scripts/smoke-purge.sh +0 -138
  439. package/pipeline/scripts/smoke-push-retry.sh +0 -75
  440. package/pipeline/scripts/smoke-repo-map.sh +0 -300
  441. package/pipeline/scripts/smoke-review-readiness.sh +0 -92
  442. package/pipeline/scripts/smoke-review-watch.sh +0 -146
  443. package/pipeline/scripts/smoke-routines.sh +0 -84
  444. package/pipeline/scripts/smoke-run-aggregator.sh +0 -216
  445. package/pipeline/scripts/smoke-run-metrics.sh +0 -50
  446. package/pipeline/scripts/smoke-search.sh +0 -187
  447. package/pipeline/scripts/smoke-shadow-git.sh +0 -224
  448. package/pipeline/scripts/smoke-skill-authoring.sh +0 -137
  449. package/pipeline/scripts/smoke-skill-language.sh +0 -83
  450. package/pipeline/scripts/smoke-skill-manifest.sh +0 -138
  451. package/pipeline/scripts/smoke-skill-scan.sh +0 -198
  452. package/pipeline/scripts/smoke-source-parity.sh +0 -85
  453. package/pipeline/scripts/smoke-subagent-validators.sh +0 -108
  454. package/pipeline/scripts/smoke-sync-parity.sh +0 -92
  455. package/pipeline/scripts/smoke-tasklist-ordering.sh +0 -112
  456. package/pipeline/scripts/smoke-telemetry.sh +0 -147
  457. package/pipeline/scripts/smoke-test-gap.sh +0 -183
  458. package/pipeline/scripts/smoke-token-budget.sh +0 -67
  459. package/pipeline/scripts/smoke-token-preflight.sh +0 -82
  460. package/pipeline/scripts/smoke-tracker-contract.sh +0 -191
  461. package/pipeline/scripts/smoke-tracker-tokens-invocation.sh +0 -73
  462. package/pipeline/scripts/smoke-triage-memory.sh +0 -174
  463. package/pipeline/scripts/smoke-update-check.sh +0 -135
  464. package/pipeline/scripts/smoke-url-enrichment.sh +0 -70
  465. package/pipeline/scripts/smoke-validate-analysis-doc.sh +0 -161
  466. package/pipeline/scripts/smoke-validator-contradiction.sh +0 -67
  467. package/pipeline/scripts/smoke-validator-gates.sh +0 -164
  468. package/pipeline/scripts/smoke-vercel-deploy-redact.sh +0 -129
  469. package/pipeline/scripts/smoke-verify-by-test.sh +0 -148
  470. package/pipeline/scripts/smoke-wiki-integration.sh +0 -122
  471. package/pipeline/scripts/smoke-work-summary.sh +0 -163
  472. package/pipeline/scripts/smoke-workflow-audit.sh +0 -69
  473. package/pipeline/scripts/smoke-worktree-path-convention.sh +0 -86
  474. package/pipeline/scripts/smoke-wrapper-preservation.sh +0 -68
  475. package/pipeline/scripts/smoke-write-state.sh +0 -115
  476. package/pipeline/scripts/sync-parity-check.sh +0 -135
  477. package/pipeline/scripts/test-gap-rules/android.json +0 -25
  478. package/pipeline/scripts/test-gap-rules/ios.json +0 -29
  479. package/pipeline/scripts/test-gap-rules/node.json +0 -17
  480. package/pipeline/scripts/test-gap-rules/python.json +0 -19
  481. package/pipeline/scripts/validate-schemas.mjs +0 -88
@@ -1,297 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- /**
4
- * @file eval-golden-tasks-live.mjs - v7.8.0 Paket A
5
- *
6
- * Opt-in live evaluation harness for golden-task fixtures. Unlike the
7
- * always-on contract harness (`eval-golden-tasks.mjs`) which only validates
8
- * schema shape, this harness optionally invokes a real model via the local
9
- * `claude` CLI and compares the actual output against the fixture's expected
10
- * scope (stack, blocker count, deferral count).
11
- *
12
- * **Default behavior is dry-run** - no model is invoked unless `--live` is
13
- * passed AND `MULTI_AGENT_LIVE_EVAL=1` is set in the environment. This
14
- * double-gate prevents accidental cost.
15
- *
16
- * Cost guard:
17
- * - Per-case budget cap (default $1 USD) - case skipped if exceeded
18
- * - Run-wide max-cases cap (default 1) - even with --live, only N cases run
19
- * - Run-wide hard ceiling: budget * max_cases (default $1 total)
20
- *
21
- * Zero-dep: shells out to the user's installed `claude` CLI rather than the
22
- * Anthropic SDK. Honors ADR-4 and uses the user's existing CLI auth.
23
- *
24
- * Usage:
25
- * node eval-golden-tasks-live.mjs # dry-run (default)
26
- * node eval-golden-tasks-live.mjs --live # blocked unless env set
27
- * MULTI_AGENT_LIVE_EVAL=1 node eval-golden-tasks-live.mjs --live --case 01-...
28
- * node eval-golden-tasks-live.mjs --budget=0.50 --max-cases=2
29
- * node eval-golden-tasks-live.mjs --json # machine-readable summary
30
- *
31
- * Exit codes:
32
- * 0 all run cases produced output matching expected scope
33
- * 1 one or more cases mismatched expected
34
- * 2 setup/usage error
35
- * 3 budget exhausted before run completed
36
- *
37
- * @module pipeline/scripts/eval-golden-tasks-live
38
- */
39
-
40
- import { readdirSync, readFileSync, existsSync } from "node:fs";
41
- import { join, dirname, resolve } from "node:path";
42
- import { fileURLToPath } from "node:url";
43
- import { spawnSync } from "node:child_process";
44
-
45
- const here = dirname(fileURLToPath(import.meta.url));
46
- const root = resolve(here, "..", "..");
47
- const evalDir = join(root, "pipeline", "eval", "golden-tasks");
48
-
49
- const argv = process.argv.slice(2);
50
- const flags = {};
51
- for (let i = 0; i < argv.length; i++) {
52
- const a = argv[i];
53
- if (a.startsWith("--")) {
54
- const [key, ...rest] = a.slice(2).split("=");
55
- if (rest.length) flags[key] = rest.join("=");
56
- else if (argv[i + 1] && !argv[i + 1].startsWith("--")) {
57
- flags[key] = argv[i + 1];
58
- i++;
59
- } else flags[key] = true;
60
- }
61
- }
62
-
63
- if (flags.help || flags.h) {
64
- console.log(`Usage: eval-golden-tasks-live.mjs [options]
65
-
66
- Modes:
67
- (default) dry-run: list what would run, no model invocation
68
- --live live: call \`claude -p\` for real (requires
69
- MULTI_AGENT_LIVE_EVAL=1 in env)
70
-
71
- Selection:
72
- --case <slug> run only this case (e.g. 01-ios-bugfix-darkmode)
73
- --max-cases <N> cap total cases to run (default: 1)
74
-
75
- Cost guard:
76
- --budget <usd> per-case budget cap (default: 1.00 USD)
77
- --total-budget <usd> run-wide ceiling (default: budget * max-cases)
78
-
79
- Output:
80
- --json machine-readable JSON summary
81
- `);
82
- process.exit(0);
83
- }
84
-
85
- const isLive = !!flags.live;
86
- const liveEnvOK = process.env.MULTI_AGENT_LIVE_EVAL === "1";
87
- const budget = parseFloat(flags.budget || "1.00");
88
- const maxCases = parseInt(flags["max-cases"] || "1", 10);
89
- const totalBudget = parseFloat(flags["total-budget"] || String(budget * maxCases));
90
- const asJson = !!flags.json;
91
-
92
- if (isLive && !liveEnvOK) {
93
- process.stderr.write(
94
- "eval-golden-tasks-live: --live requires MULTI_AGENT_LIVE_EVAL=1 in env (cost gate).\n",
95
- );
96
- process.exit(2);
97
- }
98
-
99
- if (!Number.isFinite(budget) || budget <= 0) {
100
- process.stderr.write(`eval-golden-tasks-live: invalid --budget ${flags.budget}\n`);
101
- process.exit(2);
102
- }
103
-
104
- function listCases() {
105
- if (!existsSync(evalDir)) return [];
106
- return readdirSync(evalDir, { withFileTypes: true })
107
- .filter((e) => e.isDirectory() && /^\d{2}-/.test(e.name))
108
- .map((e) => e.name)
109
- .sort();
110
- }
111
-
112
- function loadCase(slug) {
113
- const caseDir = join(evalDir, slug);
114
- const taskFile = join(caseDir, "task.json");
115
- if (!existsSync(taskFile)) return null;
116
- let task;
117
- try {
118
- task = JSON.parse(readFileSync(taskFile, "utf-8"));
119
- } catch (e) {
120
- process.stderr.write(`[eval-live] malformed task.json for "${slug}": ${e.message}\n`);
121
- return null;
122
- }
123
- return { slug, dir: caseDir, task };
124
- }
125
-
126
- function hasClaudeCLI() {
127
- const r = spawnSync("which", ["claude"], { encoding: "utf-8" });
128
- return r.status === 0 && (r.stdout || "").trim().length > 0;
129
- }
130
-
131
- /**
132
- * Live invocation - wraps the task as a prompt for `claude -p`. Returns the
133
- * raw stdout. The current iteration captures only the analysis-phase output
134
- * by asking for a JSON-only response per the analysis schema. Future work
135
- * extends this to full Phase 1/2/4 multi-call evaluation.
136
- */
137
- function invokeLive(taskJson) {
138
- const prompt = [
139
- "Perform Phase 1 (Analysis) of the multi-agent-pipeline for the task below.",
140
- "Return ONLY the analysis JSON (matching pipeline/schemas/analysis-output.schema.json).",
141
- "Do not include explanatory prose, markdown fences, or commentary.",
142
- "",
143
- "Task:",
144
- JSON.stringify(taskJson, null, 2),
145
- ].join("\n");
146
-
147
- const r = spawnSync("claude", ["-p", prompt], {
148
- encoding: "utf-8",
149
- timeout: 5 * 60 * 1000,
150
- });
151
- if (r.status !== 0) {
152
- return { ok: false, error: `claude exit ${r.status}: ${r.stderr || ""}` };
153
- }
154
- return { ok: true, raw: r.stdout || "" };
155
- }
156
-
157
- function tryParseJson(s) {
158
- try {
159
- return { ok: true, value: JSON.parse(s) };
160
- } catch (e) {
161
- // Try to extract first {...} block
162
- const m = /\{[\s\S]*\}/.exec(s);
163
- if (m) {
164
- try {
165
- return { ok: true, value: JSON.parse(m[0]) };
166
- } catch {
167
- /* fallthrough */
168
- }
169
- }
170
- return { ok: false, error: e.message };
171
- }
172
- }
173
-
174
- function evaluateOne(c) {
175
- const result = {
176
- case: c.slug,
177
- mode: isLive ? "live" : "dry-run",
178
- expected: {
179
- stack: c.task.expectedStack,
180
- blockers: c.task.expectedBlockers,
181
- deferrals: c.task.expectedDeferrals,
182
- },
183
- actual: null,
184
- pass: null,
185
- notes: [],
186
- };
187
-
188
- if (!isLive) {
189
- result.pass = "skipped-dry-run";
190
- result.notes.push("would invoke claude -p with task prompt");
191
- return result;
192
- }
193
-
194
- if (!hasClaudeCLI()) {
195
- result.pass = "skipped-no-cli";
196
- result.notes.push("claude CLI not on PATH");
197
- return result;
198
- }
199
-
200
- const live = invokeLive(c.task);
201
- if (!live.ok) {
202
- result.pass = false;
203
- result.notes.push(`live invocation failed: ${live.error}`);
204
- return result;
205
- }
206
-
207
- const parsed = tryParseJson(live.raw);
208
- if (!parsed.ok) {
209
- result.pass = false;
210
- result.notes.push(`could not parse model output as JSON: ${parsed.error}`);
211
- return result;
212
- }
213
-
214
- const actual = parsed.value;
215
- result.actual = {
216
- stack: actual?.stack?.primary ?? actual?.stack ?? null,
217
- };
218
- if (actual?.stack?.primary === c.task.expectedStack) {
219
- result.pass = true;
220
- result.notes.push(`stack match: ${actual.stack.primary}`);
221
- } else {
222
- result.pass = false;
223
- result.notes.push(
224
- `stack mismatch: expected ${c.task.expectedStack}, got ${result.actual.stack}`,
225
- );
226
- }
227
- return result;
228
- }
229
-
230
- // ─────────────────────────────────────────────────────────────────────────
231
- const allCases = listCases();
232
- let selected = allCases;
233
- if (flags.case) {
234
- selected = selected.filter((s) => s === flags.case);
235
- if (selected.length === 0) {
236
- process.stderr.write(`eval-golden-tasks-live: case not found: ${flags.case}\n`);
237
- process.exit(2);
238
- }
239
- }
240
- selected = selected.slice(0, maxCases);
241
-
242
- const results = [];
243
- let projectedSpend = 0;
244
-
245
- for (const slug of selected) {
246
- const c = loadCase(slug);
247
- if (!c) {
248
- results.push({ case: slug, pass: false, notes: ["fixture missing"] });
249
- continue;
250
- }
251
- if (projectedSpend + budget > totalBudget) {
252
- results.push({
253
- case: slug,
254
- pass: "skipped-budget-exhausted",
255
- notes: [`would exceed total-budget ${totalBudget} (already projected ${projectedSpend})`],
256
- });
257
- continue;
258
- }
259
- const r = evaluateOne(c);
260
- if (r.pass !== "skipped-dry-run" && r.pass !== "skipped-no-cli") {
261
- projectedSpend += budget;
262
- }
263
- results.push(r);
264
- }
265
-
266
- const summary = {
267
- mode: isLive ? "live" : "dry-run",
268
- cases_total: allCases.length,
269
- cases_run: results.length,
270
- cases_passed: results.filter((r) => r.pass === true).length,
271
- cases_failed: results.filter((r) => r.pass === false).length,
272
- cases_skipped: results.filter((r) => typeof r.pass === "string" && r.pass.startsWith("skipped")).length,
273
- budget_per_case: budget,
274
- total_budget: totalBudget,
275
- projected_spend: projectedSpend,
276
- };
277
-
278
- if (asJson) {
279
- console.log(JSON.stringify({ summary, results }, null, 2));
280
- } else {
281
- console.log("");
282
- console.log(`eval-golden-tasks-live (${summary.mode})`);
283
- console.log(` cases: ${summary.cases_run}/${summary.cases_total} run, ${summary.cases_passed} passed, ${summary.cases_failed} failed, ${summary.cases_skipped} skipped`);
284
- console.log(` budget: $${budget.toFixed(2)}/case · total cap $${totalBudget.toFixed(2)} · projected spend $${projectedSpend.toFixed(2)}`);
285
- console.log("");
286
- for (const r of results) {
287
- const icon = r.pass === true ? "✓" : r.pass === false ? "✗" : "·";
288
- console.log(` ${icon} ${r.case} [${r.pass}]`);
289
- for (const n of r.notes) console.log(` ${n}`);
290
- }
291
- console.log("");
292
- }
293
-
294
- const hasFailure = results.some((r) => r.pass === false);
295
- const budgetExhausted = results.some((r) => r.pass === "skipped-budget-exhausted");
296
-
297
- process.exit(hasFailure ? 1 : budgetExhausted ? 3 : 0);
@@ -1,212 +0,0 @@
1
- #!/usr/bin/env node
2
- // eval-golden-tasks.mjs - contract regression check for whole-pipeline fixtures.
3
- //
4
- // Walks pipeline/eval/golden-tasks/<NN>-<name>/ subdirectories. For each case:
5
- // 1. Reads task.json (input) + expected/phase-{1,2,4-review,4-triage}.json.
6
- // 2. Validates each expected/ file against its paired validator script
7
- // (validate-analysis.mjs / validate-planning.mjs / validate-reviewer.mjs /
8
- // validate-triage.mjs).
9
- // 3. Cross-file consistency: Phase 2 tasks[].files[] ⊆ Phase 1
10
- // touchedAreas[].path. Phase 4 review findings[].file referenced in
11
- // Phase 2 planned files (or flagged as deferred in triage).
12
- // 4. Stack match: phase-1.stack.primary === task.expectedStack.
13
- // 5. Blocker count match: triage.accepted[].filter(severity=blocking).length
14
- // === task.expectedBlockers. Same for deferred bucket.
15
- //
16
- // What this catches:
17
- // - Schema drift (any of the 4 phase schemas changes in a breaking way).
18
- // - Fixture-vs-spec divergence (someone updates a schema without re-syncing
19
- // the golden fixtures).
20
- // - Coverage gap (a triage that loses findings vs. reviewer input).
21
- //
22
- // What this does NOT catch:
23
- // - Real model output quality. This is a CONTRACT test, not a model eval.
24
- // A separate multi-model harness (with API keys, non-deterministic) is
25
- // future work. See pipeline/eval/golden-tasks/README.md.
26
- //
27
- // Usage:
28
- // node pipeline/scripts/eval-golden-tasks.mjs # all cases
29
- // node pipeline/scripts/eval-golden-tasks.mjs --case 01-... # one case
30
- // node pipeline/scripts/eval-golden-tasks.mjs --json # CI-friendly
31
- //
32
- // Exit codes: 0 all pass, 1 one or more fail, 2 usage/setup error.
33
-
34
- import { readdirSync, readFileSync, statSync, existsSync } from "node:fs";
35
- import { dirname, join, resolve } from "node:path";
36
- import { fileURLToPath } from "node:url";
37
- import { spawnSync } from "node:child_process";
38
-
39
- const here = dirname(fileURLToPath(import.meta.url));
40
- const root = resolve(here, "..", "..");
41
- const evalDir = join(root, "pipeline", "eval", "golden-tasks");
42
-
43
- const validators = {
44
- analysis: join(root, "pipeline", "scripts", "validate-analysis.mjs"),
45
- planning: join(root, "pipeline", "scripts", "validate-planning.mjs"),
46
- reviewer: join(root, "pipeline", "scripts", "validate-reviewer.mjs"),
47
- triage: join(root, "pipeline", "scripts", "validate-triage.mjs"),
48
- };
49
-
50
- const args = process.argv.slice(2);
51
- const opts = { json: false, case: null };
52
- for (let i = 0; i < args.length; i++) {
53
- if (args[i] === "--json") opts.json = true;
54
- else if (args[i] === "--case") opts.case = args[++i];
55
- else if (args[i] === "-h" || args[i] === "--help") {
56
- console.log("Usage: eval-golden-tasks.mjs [--case <name>] [--json]");
57
- process.exit(0);
58
- }
59
- }
60
-
61
- function listCases() {
62
- if (!existsSync(evalDir)) return [];
63
- return readdirSync(evalDir)
64
- .filter((f) => f !== "README.md" && statSync(join(evalDir, f)).isDirectory())
65
- .sort();
66
- }
67
-
68
- function runValidator(scriptPath, payloadPath) {
69
- // validate-reviewer.mjs reads stdin with `-`; others take a path.
70
- // For simplicity, all validators accept either stdin or file path. Pass path.
71
- const r = spawnSync("node", [scriptPath, payloadPath], { encoding: "utf-8" });
72
- return { code: r.status, stdout: r.stdout || "", stderr: r.stderr || "" };
73
- }
74
-
75
- function runCase(name) {
76
- const dir = join(evalDir, name);
77
- const errors = [];
78
-
79
- const taskPath = join(dir, "task.json");
80
- const analysisPath = join(dir, "expected", "phase-1-analysis.json");
81
- const planPath = join(dir, "expected", "phase-2-plan.json");
82
- const reviewPath = join(dir, "expected", "phase-4-review.json");
83
- const triagePath = join(dir, "expected", "phase-4-triage.json");
84
-
85
- for (const [label, p] of [
86
- ["task.json", taskPath],
87
- ["phase-1-analysis.json", analysisPath],
88
- ["phase-2-plan.json", planPath],
89
- ["phase-4-review.json", reviewPath],
90
- ["phase-4-triage.json", triagePath],
91
- ]) {
92
- if (!existsSync(p)) {
93
- errors.push(`missing ${label}`);
94
- }
95
- }
96
- if (errors.length) return { name, ok: false, errors };
97
-
98
- let task, analysis, plan, review, triage;
99
- try {
100
- task = JSON.parse(readFileSync(taskPath, "utf-8"));
101
- analysis = JSON.parse(readFileSync(analysisPath, "utf-8"));
102
- plan = JSON.parse(readFileSync(planPath, "utf-8"));
103
- review = JSON.parse(readFileSync(reviewPath, "utf-8"));
104
- triage = JSON.parse(readFileSync(triagePath, "utf-8"));
105
- } catch (e) {
106
- return { name, ok: false, errors: [`JSON parse: ${e.message}`] };
107
- }
108
-
109
- // --- schema checks ---
110
- const a = runValidator(validators.analysis, analysisPath);
111
- if (a.code !== 0) errors.push(`phase-1-analysis fails validate-analysis (exit ${a.code}): ${a.stderr.trim()}`);
112
-
113
- const p = runValidator(validators.planning, planPath);
114
- if (p.code !== 0) errors.push(`phase-2-plan fails validate-planning (exit ${p.code}): ${p.stderr.trim()}`);
115
-
116
- // phase-4-review is an array (one entry per reviewer) - validate each
117
- if (!Array.isArray(review)) {
118
- errors.push("phase-4-review.json must be a JSON array (one entry per reviewer)");
119
- } else {
120
- review.forEach((entry, idx) => {
121
- // write temp file for each reviewer entry
122
- const tmpPayload = JSON.stringify(entry);
123
- const r = spawnSync("node", [validators.reviewer, "-"], {
124
- input: tmpPayload,
125
- encoding: "utf-8",
126
- });
127
- if (r.status !== 0) errors.push(`phase-4-review[${idx}] fails validate-reviewer (exit ${r.status}): ${(r.stderr || "").trim()}`);
128
- });
129
- }
130
-
131
- const t = runValidator(validators.triage, triagePath);
132
- // validate-triage exit codes: 0 valid+clean, 2 contradiction, 3 correction.
133
- // For fixtures we accept 0 only.
134
- if (t.code !== 0) errors.push(`phase-4-triage fails validate-triage (exit ${t.code}): ${t.stderr.trim()}`);
135
-
136
- // --- cross-file consistency ---
137
-
138
- // 1. Stack match
139
- if (analysis.stack?.primary !== task.expectedStack) {
140
- errors.push(`stack mismatch: task.expectedStack=${task.expectedStack}, phase-1.stack.primary=${analysis.stack?.primary}`);
141
- }
142
-
143
- // 2. plan files ⊆ analysis touchedAreas paths
144
- const touched = new Set((analysis.touchedAreas || []).map((a) => a.path));
145
- for (const task of plan.tasks || []) {
146
- for (const f of task.files || []) {
147
- if (!touched.has(f)) {
148
- errors.push(`plan task ${task.id} references file not in Phase 1 touchedAreas: ${f}`);
149
- }
150
- }
151
- }
152
-
153
- // 3. every reviewer finding traces to a planned file OR appears in triage
154
- // (as accepted / deferred / rejected).
155
- const plannedFiles = new Set();
156
- for (const task of plan.tasks || []) for (const f of task.files || []) plannedFiles.add(f);
157
-
158
- const triageAll = [
159
- ...(triage.accepted || []).map((x) => x),
160
- ...(triage.deferred || []).map((x) => x.finding),
161
- ...(triage.rejected || []).map((x) => x.finding),
162
- ].filter((f) => f && typeof f === "object");
163
- const triageKeys = new Set(triageAll.map((f) => `${f.file}::${f.line}::${f.issue}`));
164
-
165
- for (const rev of (Array.isArray(review) ? review : [])) {
166
- for (const f of rev.findings || []) {
167
- const key = `${f.file}::${f.line}::${f.issue}`;
168
- const inPlan = plannedFiles.has(f.file);
169
- const inTriage = triageKeys.has(key);
170
- if (!inPlan && !inTriage) {
171
- errors.push(`review finding not in plan+triage: ${key}`);
172
- }
173
- }
174
- }
175
-
176
- // 4. blocker + deferral counts match task.expected*
177
- const acceptedBlockers = (triage.accepted || []).filter((x) => x.severity === "blocking").length;
178
- if (typeof task.expectedBlockers === "number" && acceptedBlockers !== task.expectedBlockers) {
179
- errors.push(`expectedBlockers=${task.expectedBlockers}, triage.accepted blocking count=${acceptedBlockers}`);
180
- }
181
- const deferredCount = (triage.deferred || []).length;
182
- if (typeof task.expectedDeferrals === "number" && deferredCount !== task.expectedDeferrals) {
183
- errors.push(`expectedDeferrals=${task.expectedDeferrals}, triage.deferred count=${deferredCount}`);
184
- }
185
-
186
- return { name, ok: errors.length === 0, errors };
187
- }
188
-
189
- // --- main ---
190
-
191
- const cases = opts.case ? [opts.case] : listCases();
192
- if (cases.length === 0) {
193
- console.error("No golden-task fixtures found under pipeline/eval/golden-tasks/");
194
- process.exit(2);
195
- }
196
-
197
- const results = cases.map(runCase);
198
- const passed = results.filter((r) => r.ok).length;
199
- const failed = results.length - passed;
200
-
201
- if (opts.json) {
202
- console.log(JSON.stringify({ passed, failed, results }, null, 2));
203
- } else {
204
- for (const r of results) {
205
- const mark = r.ok ? "\x1b[32m✓\x1b[0m" : "\x1b[31m✗\x1b[0m";
206
- console.log(`${mark} ${r.name.padEnd(38)} ${r.ok ? "" : "FAIL"}`);
207
- if (!r.ok) for (const e of r.errors) console.log(` - ${e}`);
208
- }
209
- console.log(`\n══ eval-golden-tasks: ${passed}/${results.length} passed ══`);
210
- }
211
-
212
- process.exit(failed > 0 ? 1 : 0);
@@ -1,103 +0,0 @@
1
- #!/usr/bin/env node
2
-
3
- /**
4
- * @file eval-intent.mjs - measured accuracy for the Phase 0 intent guard.
5
- *
6
- * Runs every labeled case in pipeline/eval/intent-cases.json through
7
- * pipeline/lib/classify-intent.sh and reports accuracy. This turns the
8
- * classifier from "a heuristic we hope works" into a number with a regression
9
- * gate: a dangerous miss is a clear task read as `question` (work skipped) or a
10
- * clear question read as `task` (a worktree spun up for nothing). `ambiguous`
11
- * is the safe middle and is only correct when the case labels it so.
12
- *
13
- * Exit 0 if accuracy >= threshold (default 0.95), else 1. Override with
14
- * --min <0..1>. Prints a JSON summary + the mismatches.
15
- *
16
- * @module pipeline/scripts/eval-intent
17
- */
18
-
19
- import { readFileSync } from "node:fs";
20
- import { execFileSync } from "node:child_process";
21
- import { join, dirname } from "node:path";
22
- import { fileURLToPath } from "node:url";
23
-
24
- const HERE = dirname(fileURLToPath(import.meta.url));
25
- const ROOT = join(HERE, "..", "..");
26
- const CLASSIFY = join(ROOT, "pipeline", "lib", "classify-intent.sh");
27
- const CASES = join(ROOT, "pipeline", "eval", "intent-cases.json");
28
-
29
- const argv = process.argv.slice(2);
30
- let minAccuracy = 0.95;
31
- for (let i = 0; i < argv.length; i++) {
32
- if (argv[i] === "--min") {
33
- minAccuracy = Number(argv[i + 1]);
34
- // A typo'd `--min` with no/NaN value would set the threshold to NaN, and
35
- // `accuracy < NaN` is always false - silently turning the gate into a
36
- // no-op. Fail loudly instead.
37
- if (!Number.isFinite(minAccuracy) || minAccuracy < 0 || minAccuracy > 1) {
38
- console.error(`eval-intent: --min needs a number in [0,1], got: ${argv[i + 1]}`);
39
- process.exit(2);
40
- }
41
- }
42
- }
43
-
44
- function classify(input) {
45
- try {
46
- return execFileSync("bash", [CLASSIFY, input], { encoding: "utf8" }).trim();
47
- } catch {
48
- return "ERROR";
49
- }
50
- }
51
-
52
- const { cases } = JSON.parse(readFileSync(CASES, "utf8"));
53
- if (!Array.isArray(cases) || cases.length === 0) {
54
- console.error("eval-intent: no cases found");
55
- process.exit(2);
56
- }
57
-
58
- // Operationally-safe match: only TWO outcomes are dangerous, and they are what
59
- // the gate measures -
60
- // - a `task` read as `question` -> real work silently skipped
61
- // - a `question` read as `task` -> a worktree spun up for nothing
62
- // `ambiguous` proceeds as a task in Phase 0, so it is SAFE for a task/ambiguous
63
- // case but UNSAFE for a question case (ambiguous still avoids the worktree, so
64
- // it is acceptable for a question too - only `task` is the dangerous read).
65
- function isSafe(expected, got) {
66
- if (got === expected) return true;
67
- if (expected === "task") return got === "ambiguous"; // proceeds as task anyway
68
- if (expected === "ambiguous") return got === "task"; // both proceed to dev
69
- if (expected === "question") return got === "ambiguous"; // still no worktree spun up
70
- return false;
71
- }
72
-
73
- const mismatches = []; // operationally unsafe (gate-failing)
74
- const exactMisses = []; // not exact, but operationally safe (informational)
75
- let safe = 0;
76
- let exact = 0;
77
- for (const c of cases) {
78
- const got = classify(c.input);
79
- if (got === c.expected) exact++;
80
- else exactMisses.push({ input: c.input, expected: c.expected, got });
81
- if (isSafe(c.expected, got)) safe++;
82
- else mismatches.push({ input: c.input, expected: c.expected, got, danger: true });
83
- }
84
-
85
- const accuracy = safe / cases.length;
86
- const summary = {
87
- total: cases.length,
88
- safeMatches: safe,
89
- safeAccuracy: Math.round(accuracy * 1000) / 1000,
90
- exactMatches: exact,
91
- exactAccuracy: Math.round((exact / cases.length) * 1000) / 1000,
92
- minAccuracy,
93
- dangerousMisclassifications: mismatches,
94
- safeButInexact: exactMisses,
95
- };
96
- console.log(JSON.stringify(summary, null, 2));
97
-
98
- if (accuracy < minAccuracy) {
99
- console.error(`\neval-intent: safe accuracy ${(accuracy * 100).toFixed(1)}% < required ${(minAccuracy * 100).toFixed(0)}% (${mismatches.length} dangerous misclassification(s))`);
100
- process.exit(1);
101
- }
102
- console.error(`\neval-intent: safe ${(accuracy * 100).toFixed(1)}% (${safe}/${cases.length}), exact ${(exact / cases.length * 100).toFixed(1)}% - pass`);
103
- process.exit(0);