@claude-flow/cli 3.42.2 → 3.42.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (459) hide show
  1. package/.claude/agents/analysis/analyze-code-quality.md +178 -178
  2. package/.claude/agents/analysis/code-analyzer.md +209 -209
  3. package/.claude/agents/analysis/code-review/analyze-code-quality.md +178 -178
  4. package/.claude/agents/architecture/arch-system-design.md +156 -156
  5. package/.claude/agents/architecture/system-design/arch-system-design.md +154 -154
  6. package/.claude/agents/browser/browser-agent.yaml +182 -182
  7. package/.claude/agents/consensus/byzantine-coordinator.md +62 -62
  8. package/.claude/agents/consensus/crdt-synchronizer.md +996 -996
  9. package/.claude/agents/consensus/gossip-coordinator.md +62 -62
  10. package/.claude/agents/consensus/performance-benchmarker.md +850 -850
  11. package/.claude/agents/consensus/quorum-manager.md +822 -822
  12. package/.claude/agents/consensus/raft-manager.md +62 -62
  13. package/.claude/agents/consensus/security-manager.md +621 -621
  14. package/.claude/agents/core/planner.md +374 -374
  15. package/.claude/agents/custom/test-long-runner.md +44 -44
  16. package/.claude/agents/data/data-ml-model.md +444 -444
  17. package/.claude/agents/data/ml/data-ml-model.md +192 -192
  18. package/.claude/agents/development/backend/dev-backend-api.md +141 -141
  19. package/.claude/agents/development/dev-backend-api.md +344 -344
  20. package/.claude/agents/devops/ci-cd/ops-cicd-github.md +163 -163
  21. package/.claude/agents/devops/ops-cicd-github.md +164 -164
  22. package/.claude/agents/documentation/api-docs/docs-api-openapi.md +173 -173
  23. package/.claude/agents/documentation/docs-api-openapi.md +354 -354
  24. package/.claude/agents/flow-nexus/app-store.md +87 -87
  25. package/.claude/agents/flow-nexus/authentication.md +68 -68
  26. package/.claude/agents/flow-nexus/challenges.md +80 -80
  27. package/.claude/agents/flow-nexus/neural-network.md +87 -87
  28. package/.claude/agents/flow-nexus/payments.md +82 -82
  29. package/.claude/agents/flow-nexus/sandbox.md +75 -75
  30. package/.claude/agents/flow-nexus/swarm.md +75 -75
  31. package/.claude/agents/flow-nexus/user-tools.md +95 -95
  32. package/.claude/agents/flow-nexus/workflow.md +83 -83
  33. package/.claude/agents/github/code-review-swarm.md +377 -377
  34. package/.claude/agents/github/github-modes.md +172 -172
  35. package/.claude/agents/github/issue-tracker.md +575 -575
  36. package/.claude/agents/github/multi-repo-swarm.md +552 -552
  37. package/.claude/agents/github/pr-manager.md +437 -437
  38. package/.claude/agents/github/project-board-sync.md +508 -508
  39. package/.claude/agents/github/release-manager.md +604 -604
  40. package/.claude/agents/github/release-swarm.md +582 -582
  41. package/.claude/agents/github/repo-architect.md +397 -397
  42. package/.claude/agents/github/swarm-issue.md +572 -572
  43. package/.claude/agents/github/swarm-pr.md +427 -427
  44. package/.claude/agents/github/sync-coordinator.md +451 -451
  45. package/.claude/agents/github/workflow-automation.md +902 -902
  46. package/.claude/agents/goal/agent.md +815 -815
  47. package/.claude/agents/optimization/benchmark-suite.md +664 -664
  48. package/.claude/agents/optimization/load-balancer.md +430 -430
  49. package/.claude/agents/optimization/performance-monitor.md +671 -671
  50. package/.claude/agents/optimization/resource-allocator.md +673 -673
  51. package/.claude/agents/optimization/topology-optimizer.md +807 -807
  52. package/.claude/agents/payments/agentic-payments.md +126 -126
  53. package/.claude/agents/sona/sona-learning-optimizer.md +74 -74
  54. package/.claude/agents/sparc/architecture.md +698 -698
  55. package/.claude/agents/sparc/pseudocode.md +519 -519
  56. package/.claude/agents/sparc/refinement.md +801 -801
  57. package/.claude/agents/sparc/specification.md +477 -477
  58. package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +224 -224
  59. package/.claude/agents/specialized/spec-mobile-react-native.md +226 -226
  60. package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
  61. package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
  62. package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
  63. package/.claude/agents/sublinear/performance-optimizer.md +367 -367
  64. package/.claude/agents/sublinear/trading-predictor.md +245 -245
  65. package/.claude/agents/swarm/adaptive-coordinator.md +1126 -1126
  66. package/.claude/agents/swarm/hierarchical-coordinator.md +709 -709
  67. package/.claude/agents/swarm/mesh-coordinator.md +962 -962
  68. package/.claude/agents/templates/automation-smart-agent.md +204 -204
  69. package/.claude/agents/templates/base-template-generator.md +289 -289
  70. package/.claude/agents/templates/coordinator-swarm-init.md +89 -89
  71. package/.claude/agents/templates/github-pr-manager.md +176 -176
  72. package/.claude/agents/templates/implementer-sparc-coder.md +258 -258
  73. package/.claude/agents/templates/memory-coordinator.md +186 -186
  74. package/.claude/agents/templates/orchestrator-task.md +138 -138
  75. package/.claude/agents/templates/performance-analyzer.md +198 -198
  76. package/.claude/agents/templates/sparc-coordinator.md +513 -513
  77. package/.claude/agents/testing/production-validator.md +394 -394
  78. package/.claude/agents/testing/tdd-london-swarm.md +243 -243
  79. package/.claude/agents/v3/aidefence-guardian.md +282 -282
  80. package/.claude/agents/v3/claims-authorizer.md +208 -208
  81. package/.claude/agents/v3/collective-intelligence-coordinator.md +993 -993
  82. package/.claude/agents/v3/ddd-domain-expert.md +220 -220
  83. package/.claude/agents/v3/injection-analyst.md +236 -236
  84. package/.claude/agents/v3/performance-engineer.md +1233 -1233
  85. package/.claude/agents/v3/pii-detector.md +151 -151
  86. package/.claude/agents/v3/reasoningbank-learner.md +213 -213
  87. package/.claude/agents/v3/security-architect-aidefence.md +410 -410
  88. package/.claude/agents/v3/security-architect.md +867 -867
  89. package/.claude/agents/v3/swarm-memory-manager.md +157 -157
  90. package/.claude/agents/v3/v3-integration-architect.md +205 -205
  91. package/.claude/commands/agents/README.md +50 -50
  92. package/.claude/commands/agents/agent-capabilities.md +140 -140
  93. package/.claude/commands/agents/agent-coordination.md +28 -28
  94. package/.claude/commands/agents/agent-spawning.md +28 -28
  95. package/.claude/commands/agents/agent-types.md +216 -216
  96. package/.claude/commands/agents/health.md +139 -139
  97. package/.claude/commands/agents/list.md +100 -100
  98. package/.claude/commands/agents/logs.md +130 -130
  99. package/.claude/commands/agents/metrics.md +122 -122
  100. package/.claude/commands/agents/pool.md +127 -127
  101. package/.claude/commands/agents/spawn.md +140 -140
  102. package/.claude/commands/agents/status.md +115 -115
  103. package/.claude/commands/agents/stop.md +102 -102
  104. package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
  105. package/.claude/commands/analysis/README.md +9 -9
  106. package/.claude/commands/analysis/bottleneck-detect.md +162 -162
  107. package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
  108. package/.claude/commands/analysis/performance-report.md +25 -25
  109. package/.claude/commands/analysis/token-efficiency.md +44 -44
  110. package/.claude/commands/analysis/token-usage.md +25 -25
  111. package/.claude/commands/automation/README.md +9 -9
  112. package/.claude/commands/automation/auto-agent.md +122 -122
  113. package/.claude/commands/automation/self-healing.md +105 -105
  114. package/.claude/commands/automation/session-memory.md +89 -89
  115. package/.claude/commands/automation/smart-agents.md +72 -72
  116. package/.claude/commands/automation/smart-spawn.md +25 -25
  117. package/.claude/commands/automation/workflow-select.md +25 -25
  118. package/.claude/commands/claude-flow-help.md +103 -103
  119. package/.claude/commands/claude-flow-memory.md +107 -107
  120. package/.claude/commands/claude-flow-swarm.md +205 -205
  121. package/.claude/commands/coordination/README.md +9 -9
  122. package/.claude/commands/coordination/agent-spawn.md +25 -25
  123. package/.claude/commands/coordination/init.md +44 -44
  124. package/.claude/commands/coordination/orchestrate.md +43 -43
  125. package/.claude/commands/coordination/spawn.md +45 -45
  126. package/.claude/commands/coordination/swarm-init.md +85 -85
  127. package/.claude/commands/coordination/task-orchestrate.md +25 -25
  128. package/.claude/commands/github/README.md +11 -11
  129. package/.claude/commands/github/code-review-swarm.md +513 -513
  130. package/.claude/commands/github/code-review.md +25 -25
  131. package/.claude/commands/github/github-modes.md +146 -146
  132. package/.claude/commands/github/github-swarm.md +121 -121
  133. package/.claude/commands/github/issue-tracker.md +291 -291
  134. package/.claude/commands/github/issue-triage.md +25 -25
  135. package/.claude/commands/github/multi-repo-swarm.md +518 -518
  136. package/.claude/commands/github/pr-enhance.md +26 -26
  137. package/.claude/commands/github/pr-manager.md +169 -169
  138. package/.claude/commands/github/project-board-sync.md +470 -470
  139. package/.claude/commands/github/release-manager.md +339 -339
  140. package/.claude/commands/github/release-swarm.md +543 -543
  141. package/.claude/commands/github/repo-analyze.md +25 -25
  142. package/.claude/commands/github/repo-architect.md +366 -366
  143. package/.claude/commands/github/swarm-issue.md +484 -484
  144. package/.claude/commands/github/swarm-pr.md +287 -287
  145. package/.claude/commands/github/sync-coordinator.md +302 -302
  146. package/.claude/commands/github/workflow-automation.md +441 -441
  147. package/.claude/commands/hive-mind/README.md +17 -17
  148. package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
  149. package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
  150. package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
  151. package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
  152. package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
  153. package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
  154. package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
  155. package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
  156. package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
  157. package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
  158. package/.claude/commands/hive-mind/hive-mind.md +27 -27
  159. package/.claude/commands/hooks/README.md +11 -11
  160. package/.claude/commands/hooks/overview.md +57 -57
  161. package/.claude/commands/hooks/post-edit.md +117 -117
  162. package/.claude/commands/hooks/post-task.md +112 -112
  163. package/.claude/commands/hooks/pre-edit.md +113 -113
  164. package/.claude/commands/hooks/pre-task.md +111 -111
  165. package/.claude/commands/hooks/session-end.md +118 -118
  166. package/.claude/commands/hooks/setup.md +102 -102
  167. package/.claude/commands/memory/README.md +9 -9
  168. package/.claude/commands/memory/memory-persist.md +25 -25
  169. package/.claude/commands/memory/memory-search.md +25 -25
  170. package/.claude/commands/memory/memory-usage.md +25 -25
  171. package/.claude/commands/memory/neural.md +47 -47
  172. package/.claude/commands/monitoring/README.md +9 -9
  173. package/.claude/commands/monitoring/agent-metrics.md +25 -25
  174. package/.claude/commands/monitoring/agents.md +44 -44
  175. package/.claude/commands/monitoring/real-time-view.md +25 -25
  176. package/.claude/commands/monitoring/status.md +46 -46
  177. package/.claude/commands/monitoring/swarm-monitor.md +25 -25
  178. package/.claude/commands/optimization/README.md +9 -9
  179. package/.claude/commands/optimization/auto-topology.md +61 -61
  180. package/.claude/commands/optimization/cache-manage.md +25 -25
  181. package/.claude/commands/optimization/parallel-execute.md +25 -25
  182. package/.claude/commands/optimization/parallel-execution.md +49 -49
  183. package/.claude/commands/optimization/topology-optimize.md +25 -25
  184. package/.claude/commands/pair/README.md +260 -260
  185. package/.claude/commands/pair/commands.md +545 -545
  186. package/.claude/commands/pair/config.md +509 -509
  187. package/.claude/commands/pair/examples.md +511 -511
  188. package/.claude/commands/pair/modes.md +347 -347
  189. package/.claude/commands/pair/session.md +406 -406
  190. package/.claude/commands/pair/start.md +208 -208
  191. package/.claude/commands/sparc/analyzer.md +51 -51
  192. package/.claude/commands/sparc/architect.md +53 -53
  193. package/.claude/commands/sparc/ask.md +97 -97
  194. package/.claude/commands/sparc/batch-executor.md +54 -54
  195. package/.claude/commands/sparc/code.md +89 -89
  196. package/.claude/commands/sparc/coder.md +54 -54
  197. package/.claude/commands/sparc/debug.md +83 -83
  198. package/.claude/commands/sparc/debugger.md +54 -54
  199. package/.claude/commands/sparc/designer.md +53 -53
  200. package/.claude/commands/sparc/devops.md +109 -109
  201. package/.claude/commands/sparc/docs-writer.md +80 -80
  202. package/.claude/commands/sparc/documenter.md +54 -54
  203. package/.claude/commands/sparc/innovator.md +54 -54
  204. package/.claude/commands/sparc/integration.md +83 -83
  205. package/.claude/commands/sparc/mcp.md +117 -117
  206. package/.claude/commands/sparc/memory-manager.md +54 -54
  207. package/.claude/commands/sparc/optimizer.md +54 -54
  208. package/.claude/commands/sparc/orchestrator.md +131 -131
  209. package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
  210. package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
  211. package/.claude/commands/sparc/researcher.md +54 -54
  212. package/.claude/commands/sparc/reviewer.md +54 -54
  213. package/.claude/commands/sparc/security-review.md +80 -80
  214. package/.claude/commands/sparc/sparc-modes.md +174 -174
  215. package/.claude/commands/sparc/sparc.md +111 -111
  216. package/.claude/commands/sparc/spec-pseudocode.md +80 -80
  217. package/.claude/commands/sparc/supabase-admin.md +348 -348
  218. package/.claude/commands/sparc/swarm-coordinator.md +54 -54
  219. package/.claude/commands/sparc/tdd.md +54 -54
  220. package/.claude/commands/sparc/tester.md +54 -54
  221. package/.claude/commands/sparc/tutorial.md +79 -79
  222. package/.claude/commands/sparc/workflow-manager.md +54 -54
  223. package/.claude/commands/sparc.md +166 -166
  224. package/.claude/commands/stream-chain/pipeline.md +120 -120
  225. package/.claude/commands/stream-chain/run.md +69 -69
  226. package/.claude/commands/swarm/README.md +15 -15
  227. package/.claude/commands/swarm/analysis.md +95 -95
  228. package/.claude/commands/swarm/development.md +96 -96
  229. package/.claude/commands/swarm/examples.md +168 -168
  230. package/.claude/commands/swarm/maintenance.md +102 -102
  231. package/.claude/commands/swarm/optimization.md +117 -117
  232. package/.claude/commands/swarm/research.md +136 -136
  233. package/.claude/commands/swarm/swarm-analysis.md +8 -8
  234. package/.claude/commands/swarm/swarm-background.md +8 -8
  235. package/.claude/commands/swarm/swarm-init.md +19 -19
  236. package/.claude/commands/swarm/swarm-modes.md +8 -8
  237. package/.claude/commands/swarm/swarm-monitor.md +8 -8
  238. package/.claude/commands/swarm/swarm-spawn.md +19 -19
  239. package/.claude/commands/swarm/swarm-status.md +8 -8
  240. package/.claude/commands/swarm/swarm-strategies.md +8 -8
  241. package/.claude/commands/swarm/swarm.md +87 -87
  242. package/.claude/commands/swarm/testing.md +131 -131
  243. package/.claude/commands/training/README.md +9 -9
  244. package/.claude/commands/training/model-update.md +25 -25
  245. package/.claude/commands/training/neural-patterns.md +107 -107
  246. package/.claude/commands/training/neural-train.md +75 -75
  247. package/.claude/commands/training/pattern-learn.md +25 -25
  248. package/.claude/commands/training/specialization.md +62 -62
  249. package/.claude/commands/truth/start.md +142 -142
  250. package/.claude/commands/verify/check.md +49 -49
  251. package/.claude/commands/verify/start.md +127 -127
  252. package/.claude/commands/workflows/README.md +9 -9
  253. package/.claude/commands/workflows/development.md +77 -77
  254. package/.claude/commands/workflows/research.md +62 -62
  255. package/.claude/commands/workflows/workflow-create.md +25 -25
  256. package/.claude/commands/workflows/workflow-execute.md +25 -25
  257. package/.claude/commands/workflows/workflow-export.md +25 -25
  258. package/.claude/eval/human-relevance-frozen-v1.json +17 -17
  259. package/.claude/evolve-proof/generation-0.json +211 -211
  260. package/.claude/evolve-proof/real-generation-0.json +406 -406
  261. package/.claude/evolve-proof/real-generation-1.json +406 -406
  262. package/.claude/helpers/.helpers-version +1 -1
  263. package/.claude/helpers/README.md +96 -96
  264. package/.claude/helpers/adr-compliance.sh +186 -186
  265. package/.claude/helpers/auto-commit.sh +178 -178
  266. package/.claude/helpers/auto-memory-hook.mjs +430 -430
  267. package/.claude/helpers/checkpoint-manager.sh +251 -251
  268. package/.claude/helpers/daemon-manager.sh +252 -252
  269. package/.claude/helpers/ddd-tracker.sh +144 -144
  270. package/.claude/helpers/github-safe.js +156 -156
  271. package/.claude/helpers/github-setup.sh +45 -45
  272. package/.claude/helpers/guidance-hook.sh +13 -13
  273. package/.claude/helpers/guidance-hooks.sh +102 -102
  274. package/.claude/helpers/health-monitor.sh +108 -108
  275. package/.claude/helpers/helpers.manifest.json +6 -6
  276. package/.claude/helpers/hook-handler.cjs +606 -606
  277. package/.claude/helpers/intelligence.cjs +1169 -1169
  278. package/.claude/helpers/learning-hooks.sh +329 -329
  279. package/.claude/helpers/learning-optimizer.sh +127 -127
  280. package/.claude/helpers/learning-service.mjs +1144 -1144
  281. package/.claude/helpers/memory.js +83 -83
  282. package/.claude/helpers/metrics-db.mjs +503 -503
  283. package/.claude/helpers/pattern-consolidator.sh +86 -86
  284. package/.claude/helpers/perf-worker.sh +160 -160
  285. package/.claude/helpers/post-commit +16 -16
  286. package/.claude/helpers/pre-commit +26 -26
  287. package/.claude/helpers/quick-start.sh +19 -19
  288. package/.claude/helpers/router.js +105 -105
  289. package/.claude/helpers/security-scanner.sh +127 -127
  290. package/.claude/helpers/session.js +157 -157
  291. package/.claude/helpers/setup-mcp.sh +18 -18
  292. package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
  293. package/.claude/helpers/statusline-hook.sh +21 -21
  294. package/.claude/helpers/statusline.cjs +1290 -1290
  295. package/.claude/helpers/statusline.js +340 -340
  296. package/.claude/helpers/swarm-comms.sh +353 -353
  297. package/.claude/helpers/swarm-hooks.sh +761 -761
  298. package/.claude/helpers/swarm-monitor.sh +210 -210
  299. package/.claude/helpers/sync-v3-metrics.sh +245 -245
  300. package/.claude/helpers/update-v3-progress.sh +165 -165
  301. package/.claude/helpers/v3-quick-status.sh +57 -57
  302. package/.claude/helpers/v3.sh +110 -110
  303. package/.claude/helpers/validate-v3-config.sh +215 -215
  304. package/.claude/helpers/worker-manager.sh +170 -170
  305. package/.claude/proven-config.json +41 -41
  306. package/.claude/proven-config.manifest.json +37 -37
  307. package/.claude/proven-config.signed.json +41 -41
  308. package/.claude/settings.json +182 -182
  309. package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
  310. package/.claude/skills/agentdb-learning/SKILL.md +545 -545
  311. package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
  312. package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
  313. package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
  314. package/.claude/skills/browser/SKILL.md +204 -204
  315. package/.claude/skills/dual-mode/README.md +71 -71
  316. package/.claude/skills/dual-mode/dual-collect.md +103 -103
  317. package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
  318. package/.claude/skills/dual-mode/dual-spawn.md +81 -81
  319. package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
  320. package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
  321. package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
  322. package/.claude/skills/github-code-review/SKILL.md +1125 -1125
  323. package/.claude/skills/github-multi-repo/SKILL.md +862 -862
  324. package/.claude/skills/github-project-management/SKILL.md +1262 -1262
  325. package/.claude/skills/github-release-management/SKILL.md +1064 -1064
  326. package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
  327. package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
  328. package/.claude/skills/pair-programming/SKILL.md +1202 -1202
  329. package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
  330. package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
  331. package/.claude/skills/skill-builder/SKILL.md +910 -910
  332. package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
  333. package/.claude/skills/stream-chain/SKILL.md +560 -560
  334. package/.claude/skills/swarm-advanced/SKILL.md +970 -970
  335. package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
  336. package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
  337. package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
  338. package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
  339. package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
  340. package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
  341. package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
  342. package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
  343. package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
  344. package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
  345. package/.claude/skills/verification-quality/SKILL.md +691 -691
  346. package/README.md +422 -422
  347. package/bin/cli.js +338 -338
  348. package/bin/mcp-server.js +224 -224
  349. package/bin/preinstall.cjs +2 -2
  350. package/catalog-manifest.json +2 -2
  351. package/dist/src/benchmarks/gaia-critic.js +24 -24
  352. package/dist/src/business-pods/bbs-budget-tracker.js +53 -53
  353. package/dist/src/commands/completions.js +409 -409
  354. package/dist/src/commands/daemon.js +44 -44
  355. package/dist/src/commands/doctor.js +4 -4
  356. package/dist/src/commands/embeddings.js +26 -26
  357. package/dist/src/commands/hive-mind.js +97 -97
  358. package/dist/src/commands/hooks.js +9 -9
  359. package/dist/src/commands/init.js +75 -75
  360. package/dist/src/commands/ruvector/backup.js +23 -23
  361. package/dist/src/commands/ruvector/benchmark.js +31 -31
  362. package/dist/src/commands/ruvector/import.js +14 -14
  363. package/dist/src/commands/ruvector/init.js +115 -115
  364. package/dist/src/commands/ruvector/migrate.js +99 -99
  365. package/dist/src/commands/ruvector/optimize.js +51 -51
  366. package/dist/src/commands/ruvector/setup.js +624 -624
  367. package/dist/src/commands/ruvector/status.js +38 -38
  368. package/dist/src/config/proven-config.js +2 -2
  369. package/dist/src/init/claudemd-generator.js +273 -273
  370. package/dist/src/init/executor.js +453 -453
  371. package/dist/src/init/helper-signing.js +2 -2
  372. package/dist/src/init/helpers-generator.js +917 -751
  373. package/dist/src/init/statusline-generator.js +24 -24
  374. package/dist/src/mcp-tools/agentdb-tools.js +15 -15
  375. package/dist/src/mcp-tools/browser-intent-tools.js +19 -19
  376. package/dist/src/mcp-tools/seraphina-tools.js +4 -4
  377. package/dist/src/memory/graph-edge-writer.js +22 -22
  378. package/dist/src/memory/memory-bridge.js +114 -114
  379. package/dist/src/memory/memory-initializer.js +416 -416
  380. package/dist/src/memory/rabitq-index.js +5 -5
  381. package/dist/src/proxy/verify.js +2 -2
  382. package/dist/src/runtime/headless.js +28 -28
  383. package/dist/src/ruvector/diskann-backend.d.ts +78 -0
  384. package/dist/src/ruvector/diskann-backend.js +310 -0
  385. package/dist/src/services/distill-tuning.js +7 -7
  386. package/dist/src/services/headless-worker-executor.js +84 -84
  387. package/dist/src/services/memory-distillation.js +18 -18
  388. package/dist/src/transfer/deploy-seraphine.js +23 -23
  389. package/node_modules/@claude-flow/codex/.agents/skills/github-automation/SKILL.md +32 -32
  390. package/node_modules/@claude-flow/codex/.agents/skills/memory-management/SKILL.md +45 -45
  391. package/node_modules/@claude-flow/codex/.agents/skills/performance-analysis/SKILL.md +32 -32
  392. package/node_modules/@claude-flow/codex/.agents/skills/security-audit/SKILL.md +46 -46
  393. package/node_modules/@claude-flow/codex/.agents/skills/sparc-methodology/SKILL.md +46 -46
  394. package/node_modules/@claude-flow/codex/.agents/skills/swarm-orchestration/SKILL.md +53 -53
  395. package/node_modules/@claude-flow/codex/README.md +1044 -1044
  396. package/node_modules/@claude-flow/codex/dist/dual-mode/orchestrator.js +13 -13
  397. package/node_modules/@claude-flow/codex/dist/generators/agents-md.js +664 -664
  398. package/node_modules/@claude-flow/codex/dist/generators/config-toml.js +455 -455
  399. package/node_modules/@claude-flow/codex/dist/generators/skill-md.js +45 -45
  400. package/node_modules/@claude-flow/codex/dist/initializer.js +167 -167
  401. package/node_modules/@claude-flow/codex/dist/templates/index.js +15 -15
  402. package/node_modules/@claude-flow/mcp/README.md +429 -429
  403. package/node_modules/@claude-flow/plugin-agent-federation/README.md +49 -49
  404. package/node_modules/@claude-flow/security/README.md +292 -292
  405. package/node_modules/@claude-flow/security/dist/credential-generator.js +9 -9
  406. package/node_modules/@claude-flow/security/dist/input-validator.d.ts +6 -6
  407. package/node_modules/@claude-flow/security/dist/oauth/callback-server.js +9 -9
  408. package/package.json +181 -181
  409. package/plugins/ruflo-metaharness/.claude-plugin/plugin.json +32 -32
  410. package/plugins/ruflo-metaharness/README.md +72 -72
  411. package/plugins/ruflo-metaharness/agents/metaharness-architect.md +58 -58
  412. package/plugins/ruflo-metaharness/commands/ruflo-metaharness.md +50 -50
  413. package/plugins/ruflo-metaharness/scripts/_darwin.mjs +210 -210
  414. package/plugins/ruflo-metaharness/scripts/_harness.mjs +334 -334
  415. package/plugins/ruflo-metaharness/scripts/_invoke.mjs +230 -230
  416. package/plugins/ruflo-metaharness/scripts/_redblue.mjs +143 -143
  417. package/plugins/ruflo-metaharness/scripts/_similarity.mjs +161 -161
  418. package/plugins/ruflo-metaharness/scripts/_spike-similarity.mjs +223 -223
  419. package/plugins/ruflo-metaharness/scripts/audit-list.mjs +158 -158
  420. package/plugins/ruflo-metaharness/scripts/audit-trend.mjs +272 -272
  421. package/plugins/ruflo-metaharness/scripts/bench-parse-mcp-scan.mjs +146 -146
  422. package/plugins/ruflo-metaharness/scripts/bench-recordpair-overhead.mjs +186 -186
  423. package/plugins/ruflo-metaharness/scripts/bench-similarity.mjs +177 -177
  424. package/plugins/ruflo-metaharness/scripts/bench.mjs +95 -95
  425. package/plugins/ruflo-metaharness/scripts/drift-from-history.mjs +363 -363
  426. package/plugins/ruflo-metaharness/scripts/evolve.mjs +404 -404
  427. package/plugins/ruflo-metaharness/scripts/genome.mjs +105 -105
  428. package/plugins/ruflo-metaharness/scripts/gepa.mjs +153 -153
  429. package/plugins/ruflo-metaharness/scripts/learn.mjs +127 -127
  430. package/plugins/ruflo-metaharness/scripts/mcp-scan.mjs +110 -110
  431. package/plugins/ruflo-metaharness/scripts/mint.mjs +126 -126
  432. package/plugins/ruflo-metaharness/scripts/oia-audit.mjs +228 -228
  433. package/plugins/ruflo-metaharness/scripts/redblue.mjs +286 -286
  434. package/plugins/ruflo-metaharness/scripts/router-parallel-analyze.mjs +250 -250
  435. package/plugins/ruflo-metaharness/scripts/score.mjs +92 -92
  436. package/plugins/ruflo-metaharness/scripts/security-bench.mjs +174 -174
  437. package/plugins/ruflo-metaharness/scripts/similarity.mjs +158 -158
  438. package/plugins/ruflo-metaharness/scripts/smoke.sh +2422 -2422
  439. package/plugins/ruflo-metaharness/scripts/test-graceful-degradation.mjs +165 -165
  440. package/plugins/ruflo-metaharness/scripts/test-mcp-tools.mjs +498 -498
  441. package/plugins/ruflo-metaharness/scripts/test-parallel-pipeline.mjs +204 -204
  442. package/plugins/ruflo-metaharness/scripts/test-pipeline-roundtrip.mjs +586 -586
  443. package/plugins/ruflo-metaharness/scripts/test-similarity.mjs +372 -372
  444. package/plugins/ruflo-metaharness/scripts/test-with-openrouter.mjs +229 -229
  445. package/plugins/ruflo-metaharness/scripts/threat-model.mjs +62 -62
  446. package/plugins/ruflo-metaharness/skills/harness-bench/SKILL.md +64 -64
  447. package/plugins/ruflo-metaharness/skills/harness-drift-from-history/SKILL.md +65 -65
  448. package/plugins/ruflo-metaharness/skills/harness-evolve/SKILL.md +131 -131
  449. package/plugins/ruflo-metaharness/skills/harness-genome/SKILL.md +57 -57
  450. package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +65 -65
  451. package/plugins/ruflo-metaharness/skills/harness-learn/SKILL.md +65 -65
  452. package/plugins/ruflo-metaharness/skills/harness-mcp-scan/SKILL.md +49 -49
  453. package/plugins/ruflo-metaharness/skills/harness-mint/SKILL.md +72 -72
  454. package/plugins/ruflo-metaharness/skills/harness-oia-audit/SKILL.md +79 -79
  455. package/plugins/ruflo-metaharness/skills/harness-score/SKILL.md +66 -66
  456. package/plugins/ruflo-metaharness/skills/harness-security-bench/SKILL.md +101 -101
  457. package/plugins/ruflo-metaharness/skills/harness-similarity/SKILL.md +67 -67
  458. package/plugins/ruflo-metaharness/skills/harness-threat-model/SKILL.md +41 -41
  459. package/scripts/postinstall.cjs +153 -153
@@ -1,498 +1,498 @@
1
- #!/usr/bin/env node
2
- // test-mcp-tools.mjs — runtime test for the iter-20/21 MCP tool registry.
3
- //
4
- // tsc proves the metaharness-tools.ts module COMPILES; structural smoke
5
- // proves the source DECLARES the right tool names. Neither proves the
6
- // HANDLERS actually run without throwing. This test imports the compiled
7
- // module and invokes every tool's handler with minimal inputs.
8
- //
9
- // CONTRACT EACH TOOL MUST SATISFY
10
- // - handler is callable as `await tool.handler({ ... })`
11
- // - returns an object with keys: success, data, degraded, exitCode
12
- // - never throws (even with bad/missing optional dep — graceful)
13
- // - handler honors the 120s subprocess timeout (no hang)
14
- //
15
- // USAGE
16
- // node scripts/test-mcp-tools.mjs # default
17
- // node scripts/test-mcp-tools.mjs --format json
18
- //
19
- // EXIT CODES
20
- // 0 all tools satisfy the contract
21
- // 1 at least one tool failed
22
- // 2 setup error (compiled dist not present)
23
-
24
- import { existsSync } from 'node:fs';
25
- import { dirname, join, resolve } from 'node:path';
26
- import { fileURLToPath } from 'node:url';
27
-
28
- const SCRIPTS_DIR = dirname(fileURLToPath(import.meta.url));
29
- const ARGS = (() => {
30
- const a = { format: 'table' };
31
- for (let i = 2; i < process.argv.length; i++) {
32
- if (process.argv[i] === '--format') a.format = process.argv[++i];
33
- }
34
- return a;
35
- })();
36
-
37
- let passed = 0, failed = 0;
38
- const failures = [];
39
-
40
- function assert(cond, label) {
41
- if (cond) { console.log(` ✓ ${label}`); passed++; }
42
- else { console.log(` ✗ ${label}`); failures.push(label); failed++; }
43
- }
44
-
45
- async function main() {
46
- // Locate the compiled dist of metaharness-tools.
47
- const distPath = resolve(SCRIPTS_DIR, '..', '..', '..',
48
- 'v3', '@claude-flow', 'cli', 'dist', 'src', 'mcp-tools', 'metaharness-tools.js');
49
-
50
- if (!existsSync(distPath)) {
51
- console.log(`# test-mcp-tools — SKIPPED`);
52
- console.log('');
53
- console.log(`Compiled dist not present: ${distPath}`);
54
- console.log(`Build the CLI first:`);
55
- console.log(` cd v3/@claude-flow/cli && npm run build`);
56
- console.log('');
57
- console.log(`Exit 0 — this script is meaningfully runnable only post-build.`);
58
- process.exit(0);
59
- }
60
-
61
- let mod;
62
- try {
63
- mod = await import(distPath);
64
- } catch (e) {
65
- console.error(`test-mcp-tools: failed to import ${distPath}: ${e.message}`);
66
- process.exit(2);
67
- }
68
-
69
- const tools = mod.metaharnessTools;
70
- console.log(`# test-mcp-tools — runtime contract\n`);
71
-
72
- // ──────────────────────────────────────────────────────────────────
73
- // PHASE 1 — module exports the right shape
74
- // ──────────────────────────────────────────────────────────────────
75
- console.log('Phase 1 — module shape');
76
- assert(Array.isArray(tools), 'metaharnessTools is an array');
77
- assert(tools.length === 16, `16 tools registered (got ${tools.length})`);
78
-
79
- const expectedNames = new Set([
80
- 'metaharness_score',
81
- 'metaharness_genome',
82
- 'metaharness_mcp_scan',
83
- 'metaharness_threat_model',
84
- 'metaharness_oia_audit',
85
- 'metaharness_audit_list',
86
- 'metaharness_audit_trend',
87
- // iter 36 — ADR-152 §3.1 production
88
- 'metaharness_similarity',
89
- // iter 54 — one-command drift detection (composes audit-list + oia-audit + audit-trend)
90
- 'metaharness_drift_from_history',
91
- // ADR-153 — bench suites + evolve driver + security-focused bench
92
- 'metaharness_bench',
93
- 'metaharness_evolve',
94
- 'metaharness_security_bench',
95
- // @metaharness/redblue@~0.1.4 — adversarial red/blue LLM testing
96
- 'metaharness_redblue',
97
- // metaharness@0.3.0 — upstream ADR-235 GEPA learning run
98
- 'metaharness_learn',
99
- // @metaharness/darwin@0.8.0 — GEPA library surface (genome ops)
100
- 'metaharness_gepa',
101
- // ADR-322 — in-process governed evaluation and atomic promotion
102
- 'metaharness_flywheel',
103
- ]);
104
- const actualNames = new Set(tools.map((t) => t.name));
105
- for (const name of expectedNames) {
106
- assert(actualNames.has(name), `${name} registered`);
107
- }
108
-
109
- // ──────────────────────────────────────────────────────────────────
110
- // PHASE 2 — every tool has the required MCP shape
111
- // ──────────────────────────────────────────────────────────────────
112
- console.log('\nPhase 2 — per-tool shape');
113
- for (const tool of tools) {
114
- const ok = typeof tool.name === 'string'
115
- && typeof tool.description === 'string'
116
- && typeof tool.category === 'string'
117
- && typeof tool.handler === 'function'
118
- && typeof tool.inputSchema === 'object';
119
- assert(ok, `${tool.name} has {name, description, category, handler, inputSchema}`);
120
- assert(tool.category === 'metaharness', `${tool.name} category === 'metaharness'`);
121
- }
122
-
123
- // ──────────────────────────────────────────────────────────────────
124
- // PHASE 3 — handlers callable + return contract shape
125
- //
126
- // We invoke each handler with minimal valid input. The handlers may
127
- // succeed (if metaharness is installed) or report degraded (if not).
128
- // EITHER way, they must return { success, data, degraded, exitCode }
129
- // without throwing.
130
- // ──────────────────────────────────────────────────────────────────
131
- console.log('\nPhase 3 — handler invocations (allow up to 30s each)');
132
- for (const tool of tools) {
133
- // Construct minimal valid input per tool.
134
- let input = {};
135
- if (tool.name === 'metaharness_audit_trend') {
136
- // Requires baselineKey + currentKey — use fake keys that won't
137
- // resolve so we exercise the not-found path.
138
- input = { baselineKey: 'audit-fake-base', currentKey: 'audit-fake-curr' };
139
- }
140
- if (tool.name === 'metaharness_similarity') {
141
- // Needs --a/--b OR --a-key/--b-key. Use fake mem keys to exercise
142
- // the graceful not-found path (matches audit_trend convention).
143
- input = { aKey: 'harness-fake-a', bKey: 'harness-fake-b' };
144
- }
145
- if (tool.name === 'metaharness_drift_from_history') {
146
- // iter 54 — composes 3 subprocesses, needs more time than the default.
147
- input = { dryRun: true, threshold: 0.5 };
148
- }
149
- if (tool.name === 'metaharness_oia_audit') {
150
- // iter 128 — composite audit runs 5 sub-audits (oia-manifest +
151
- // threat-model + mcp-scan + score + genome) in parallel. Each
152
- // shells out via npx. --dry-run skips memory persistence so the
153
- // test doesn't pollute namespaces.
154
- input = { dryRun: true };
155
- }
156
- if (tool.name === 'metaharness_redblue') {
157
- // `attack` preview is the fastest path that exercises the upstream
158
- // binary without needing OPENROUTER_API_KEY or running any model
159
- // calls. Count=1 keeps cold-cache npx fetch the dominant cost.
160
- input = { subcommand: 'attack', family: 'prompt', count: 1 };
161
- }
162
- if (tool.name === 'metaharness_learn') {
163
- // No repo checkout in CI → structured {status:"checkout-required"}
164
- // exit-0 path. $0: without run=true upstream never spends anyway.
165
- input = {};
166
- }
167
- if (tool.name === 'metaharness_gepa') {
168
- // op is required; `genome` loads + validates the SHIPPED cand-6
169
- // genome — pure-local library call once darwin is cached.
170
- input = { op: 'genome' };
171
- }
172
-
173
- // iter 124 → 130 — timeouts have crept up as CI cold-cache npx
174
- // warmup costs got measured. Final budgets:
175
- // default : 60s
176
- // chain-tools : 180s (drift_from_history + oia_audit + audit_list)
177
- // iter 131 — bumped chain-tool budget 90s → 180s. audit_list still
178
- // timed out at 90s in CI; locally it runs in ~4s, but CI's
179
- // `npx @claude-flow/cli@latest memory list` invocation pays both
180
- // the npx fetch AND a full CLI startup (which loads agentic-flow +
181
- // ONNX). 180s gives 30x headroom over the local cost.
182
- const isChainTool = tool.name === 'metaharness_drift_from_history'
183
- || tool.name === 'metaharness_oia_audit'
184
- || tool.name === 'metaharness_audit_list'
185
- // redblue: `attack prompt --count 1` is preview-only (no model
186
- // calls) but the cold-cache `npx -y @metaharness/redblue@~0.1.4`
187
- // fetch can take 30-60s. 180s gives 3x headroom.
188
- || tool.name === 'metaharness_redblue'
189
- // learn: cold-cache `npx -y metaharness@latest` fetch dominates.
190
- // gepa: one-time `npm install --prefix ~/.ruflo/darwin-cache-*`
191
- // fallback install can take 30-60s on cold cache.
192
- || tool.name === 'metaharness_learn'
193
- || tool.name === 'metaharness_gepa';
194
- const timeoutMs = isChainTool ? 180_000 : 60_000;
195
- const handlerPromise = tool.handler(input);
196
- const timeoutPromise = new Promise((_, reject) =>
197
- setTimeout(() => reject(new Error(`${timeoutMs / 1000}s handler timeout`)), timeoutMs));
198
-
199
- let result;
200
- let threw = false;
201
- try {
202
- result = await Promise.race([handlerPromise, timeoutPromise]);
203
- } catch (e) {
204
- threw = true;
205
- console.log(` [${tool.name}] handler threw: ${e.message.slice(0, 80)}`);
206
- }
207
-
208
- assert(!threw, `${tool.name} handler did not throw`);
209
- if (!threw && result) {
210
- assert(typeof result === 'object', `${tool.name} returns object`);
211
- assert('success' in result, `${tool.name} result has 'success'`);
212
- assert('data' in result, `${tool.name} result has 'data'`);
213
- assert('degraded' in result, `${tool.name} result has 'degraded'`);
214
- assert('exitCode' in result, `${tool.name} result has 'exitCode'`);
215
- }
216
- }
217
-
218
- // ──────────────────────────────────────────────────────────────────
219
- // PHASE 4 — POSITIVE-CASE data-shape validation (iter 43)
220
- //
221
- // Iter 37 verified the {success, data, degraded, exitCode} envelope.
222
- // It did NOT verify that data.X contains the right keys when success
223
- // is genuinely true — leaving room for iter 42-style bugs where a
224
- // handler returns valid-looking degraded JSON while silently
225
- // misrouting input. This phase invokes each handler with VALID
226
- // inputs and asserts the expected output shape.
227
- //
228
- // Tools that depend on `npx metaharness` (score/genome/mcp-scan/
229
- // threat-model/oia-audit/audit-list/audit-trend) are SKIPPED in this
230
- // phase when the optional dep isn't installed — they're covered by
231
- // the no-metaharness-smoke workflow's drill. The similarity tool
232
- // has no @metaharness/* dep, so its positive case ALWAYS runs.
233
- // ──────────────────────────────────────────────────────────────────
234
- console.log('\nPhase 4 — positive-case data shape (iter 43)');
235
-
236
- const { writeFileSync, mkdtempSync, mkdirSync } = await import('node:fs');
237
- const { tmpdir } = await import('node:os');
238
- const { join: pjoin } = await import('node:path');
239
- const tmp = mkdtempSync(pjoin(tmpdir(), 'mcp-positive-'));
240
-
241
- // #2626 — upstream deliberately exits 2 for a valid `blocked` genome.
242
- // The wrapper must preserve that report as data instead of converting it
243
- // into an input/system error at the MCP boundary.
244
- const genomeTool = tools.find((t) => t.name === 'metaharness_genome');
245
- if (genomeTool) {
246
- const blockedRepo = pjoin(tmp, 'blocked-repo');
247
- mkdirSync(blockedRepo);
248
- const r = await genomeTool.handler({ path: blockedRepo });
249
- if (!r.degraded) {
250
- assert(r.success === true,
251
- 'genome blocked verdict: wrapper succeeds with a valid report (#2626)');
252
- assert(r.exitCode === 0,
253
- 'genome blocked verdict: wrapper exitCode === 0 (#2626)');
254
- assert(r.data?.verdict === 'blocked',
255
- `genome blocked verdict: data.verdict === blocked (got ${r.data?.verdict})`);
256
- assert(r.data?.verdictExitCode === 2,
257
- `genome blocked verdict: preserves upstream exit 2 (got ${r.data?.verdictExitCode})`);
258
- assert(typeof r.data?.risk_score === 'number' && r.data.risk_score >= 0.7,
259
- `genome blocked verdict: risk_score >= 0.7 (got ${r.data?.risk_score})`);
260
- } else {
261
- console.log(` ⊘ genome: metaharness absent — graceful skip`);
262
- }
263
- }
264
-
265
- // metaharness_similarity — full positive case (no @metaharness/* needed)
266
- const simTool = tools.find((t) => t.name === 'metaharness_similarity');
267
- if (simTool) {
268
- const aPath = pjoin(tmp, 'a.json');
269
- const bPath = pjoin(tmp, 'b.json');
270
- writeFileSync(aPath, JSON.stringify({
271
- score: { harnessFit: 78, compileConfidence: 92, taskCoverage: 65, toolSafety: 88, memoryUsefulness: 70, estCostPerRunUsd: 0.04, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:legal' },
272
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','w'], risk_score: 0.45, test_confidence: 0.7, publish_readiness: 0.6 },
273
- }));
274
- writeFileSync(bPath, JSON.stringify({
275
- score: { harnessFit: 75, compileConfidence: 90, taskCoverage: 70, toolSafety: 90, memoryUsefulness: 72, estCostPerRunUsd: 0.05, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:support' },
276
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','q','r'], risk_score: 0.40, test_confidence: 0.75, publish_readiness: 0.65 },
277
- }));
278
- const r = await simTool.handler({ aFile: aPath, bFile: bPath });
279
- assert(r.degraded === false, 'similarity positive case: degraded === false');
280
- assert(r.success === true, 'similarity positive case: success === true');
281
- assert(r.exitCode === 0, 'similarity positive case: exitCode === 0');
282
- const d = r.data ?? {};
283
- assert(typeof d.overall === 'number', 'similarity data has numeric `overall`');
284
- assert(typeof d.components === 'object' && d.components !== null,
285
- 'similarity data has `components` object');
286
- assert(typeof d.components?.cosine === 'number',
287
- 'similarity components.cosine numeric');
288
- assert(typeof d.components?.categorical === 'number',
289
- 'similarity components.categorical numeric');
290
- assert(typeof d.components?.jaccard === 'number',
291
- 'similarity components.jaccard numeric');
292
- assert(typeof d.weights === 'object' && d.weights !== null,
293
- 'similarity data has `weights` object');
294
- assert(d.adr === 'ADR-152', 'similarity data tagged adr=ADR-152');
295
- // Regression anchor — same fixtures as iter-35 spike with non-matching topologies
296
- assert(d.overall > 0 && d.overall < 1,
297
- `similarity overall in (0, 1) — got ${d.overall}`);
298
-
299
- // Per-dimension variant
300
- const rPD = await simTool.handler({ aFile: aPath, bFile: bPath, perDimension: true });
301
- assert(typeof rPD.data?.perDimension === 'object',
302
- 'similarity perDimension=true populates breakdown');
303
-
304
- // Alert-below variant exercises non-zero exit
305
- const rAlert = await simTool.handler({ aFile: aPath, bFile: bPath, alertBelow: 0.99 });
306
- assert(rAlert.data?.alert?.triggered === true,
307
- 'similarity alertBelow=0.99 triggers alert');
308
- assert(rAlert.exitCode === 1, 'similarity alertBelow=0.99 → exitCode 1');
309
- // iter 44 — success semantic anchor (was true under the pre-iter-44
310
- // `!degraded` rule; now false because exitCode !== 0 dominates).
311
- assert(rAlert.success === false,
312
- 'similarity alertBelow=0.99 → success === false (iter 44 fix)');
313
- }
314
-
315
- // metaharness_mcp_scan — positive case post iter-50 parser landing.
316
- // Until iter 50, mcp_scan's data field was an alert-only object with
317
- // no structured findings. After iter 50, findings[] is always present
318
- // (parsed from upstream text) and summary{overallSeverity, totalCount}
319
- // accompanies it.
320
- const scanTool = tools.find((t) => t.name === 'metaharness_mcp_scan');
321
- if (scanTool) {
322
- // Run against ruflo itself — guaranteed to produce at least the
323
- // INFO finding the iter-50 parser test verified manually.
324
- const r = await scanTool.handler({ path: '.', failOn: 'high' });
325
- // Either succeeds with structured findings, or gracefully degrades
326
- // if metaharness isn't installed in this environment.
327
- if (!r.degraded) {
328
- assert(r.success === true, 'mcp_scan positive: success === true');
329
- assert(r.exitCode === 0, 'mcp_scan positive: exitCode === 0');
330
- assert(Array.isArray(r.data?.findings),
331
- 'mcp_scan positive: data.findings is an array (iter 50 fix)');
332
- // Cwd-dependent: when scanning a dir without .mcp/servers.json the
333
- // upstream emits no findings. Only verify shape contract when array
334
- // is populated — the array-presence assertion above is the
335
- // load-bearing one for iter 50.
336
- if (r.data?.findings.length > 0) {
337
- const first = r.data.findings[0];
338
- assert(typeof first?.severity === 'string',
339
- 'mcp_scan positive: first finding has string severity');
340
- assert(typeof first?.message === 'string',
341
- 'mcp_scan positive: first finding has string message');
342
- }
343
- // summary may be null if the upstream produced no Result: line —
344
- // verify the field's presence (null OR object) but only deep-check
345
- // when populated.
346
- if (r.data?.summary) {
347
- assert(typeof r.data.summary.totalCount === 'number',
348
- 'mcp_scan positive: data.summary.totalCount is numeric (when summary present)');
349
- }
350
- } else {
351
- console.log(` ⊘ mcp_scan: metaharness absent — graceful skip`);
352
- }
353
- }
354
-
355
- // metaharness_audit_trend — positive case via file inputs
356
- const trendTool = tools.find((t) => t.name === 'metaharness_audit_trend');
357
- if (trendTool) {
358
- const basePath = pjoin(tmp, 'base.json');
359
- const currPath = pjoin(tmp, 'curr.json');
360
- const fingerprint = {
361
- score: { harnessFit: 80, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
362
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['a','b','c'], risk_score: 0.3, test_confidence: 0.85, publish_readiness: 0.9 },
363
- };
364
- writeFileSync(basePath, JSON.stringify({
365
- startedAt: '2026-06-15T00:00:00Z',
366
- composite: { worst: 'clean' },
367
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
368
- fingerprint,
369
- }));
370
- writeFileSync(currPath, JSON.stringify({
371
- startedAt: '2026-06-16T00:00:00Z',
372
- composite: { worst: 'clean' },
373
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
374
- fingerprint,
375
- }));
376
- // audit_trend tool only supports keys, not files at the MCP layer.
377
- // Document its actual wrapper semantics so future-us doesn't get
378
- // surprised:
379
- // - bad keys → script exits 2 with stderr (no JSON payload)
380
- // - runScript() can't parse a {degraded:true} marker, so it
381
- // returns degraded:false / success:true / exitCode:2
382
- // This is a real wrapper bug (success should not be true when
383
- // exit!=0 AND no JSON came back), tracked separately. Asserting
384
- // current behavior here protects against silent semantic shifts.
385
- // iter 46 — file-input path. audit_trend now accepts baselineFile/currentFile.
386
- const rFiles = await trendTool.handler({ baselineFile: basePath, currentFile: currPath });
387
- assert(rFiles.success === true,
388
- 'audit_trend file-input path: success === true (iter 46)');
389
- assert(rFiles.exitCode === 0, 'audit_trend file-input path: exitCode === 0');
390
- assert(typeof rFiles.data?.delta === 'object',
391
- 'audit_trend file-input path: data.delta object present');
392
- assert(rFiles.data?.delta?.structuralDistance?.verdict === 'near-identical',
393
- `audit_trend file-input path: identical fingerprints → near-identical (got ${rFiles.data?.delta?.structuralDistance?.verdict})`);
394
-
395
- // iter 54 — metaharness_drift_from_history positive case
396
- const driftTool = tools.find((t) => t.name === 'metaharness_drift_from_history');
397
- if (driftTool) {
398
- // iter 71 — verify iter-66/67 fast-path flags are now MCP-callable
399
- // Synthesize a baseline file on disk; pass via the new baselineFile input.
400
- const baselinePath = pjoin(tmp, 'drift-baseline.json');
401
- writeFileSync(baselinePath, JSON.stringify({
402
- startedAt: '2026-06-16T00:00:00Z',
403
- composite: { worst: 'clean' },
404
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
405
- fingerprint: {
406
- score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
407
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
408
- },
409
- }));
410
- const rFastFast = await driftTool.handler({
411
- path: '.', dryRun: true, threshold: 0.5, baselineFile: baselinePath,
412
- });
413
- if (!rFastFast.degraded) {
414
- assert(rFastFast.data?.timing?.usedBaselineFile === true,
415
- 'drift_from_history MCP-layer: baselineFile fastpath fires (iter 71)');
416
- assert(rFastFast.data?.timing?.skippedAuditList === true,
417
- 'drift_from_history MCP-layer: skippedAuditList=true via baselineFile (iter 71)');
418
- }
419
-
420
- // iter 85 — verify iter-78's alertOnNewSeverity MCP input plumbs
421
- // through. baselineFile has no findings; current ruflo audit has
422
- // 1 INFO finding. With alertOnNewSeverity='info' the gate fires
423
- // and surfaces in the response.
424
- const baselineNoFindings = pjoin(tmp, 'drift-baseline-no-findings.json');
425
- writeFileSync(baselineNoFindings, JSON.stringify({
426
- startedAt: '2026-06-16T00:00:00Z',
427
- composite: { worst: 'clean' },
428
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
429
- fingerprint: {
430
- score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
431
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
432
- },
433
- }));
434
- const rSevAlert = await driftTool.handler({
435
- path: '.', dryRun: true, threshold: 0.5,
436
- baselineFile: baselineNoFindings,
437
- alertOnNewSeverity: 'info',
438
- });
439
- if (!rSevAlert.degraded) {
440
- assert(rSevAlert.data?.alert?.newSeverityThreshold === 'info',
441
- 'drift_from_history MCP-layer: alertOnNewSeverity echoed in payload (iter 85)');
442
- // Triggered AND exit code reflects (only if the audit actually had findings)
443
- if (rSevAlert.data?.alert?.triggered === true) {
444
- assert(rSevAlert.exitCode === 1,
445
- `drift_from_history MCP-layer: alertOnNewSeverity exitCode=1 when triggered (got ${rSevAlert.exitCode})`);
446
- assert(rSevAlert.success === false,
447
- 'drift_from_history MCP-layer: success===false when alert fires (iter 44 fix)');
448
- }
449
- }
450
-
451
- const r54 = await driftTool.handler({ path: '.', dryRun: true, threshold: 0.5 });
452
- if (!r54.degraded) {
453
- assert(typeof r54.data === 'object' && r54.data !== null,
454
- 'drift_from_history positive: data is an object');
455
- // Either it produced the structured drift report OR the no-history error
456
- const isOk = r54.data?.command === 'drift-from-history';
457
- const isNoHistory = typeof r54.data?.error === 'string' && r54.data.error.includes('no audit records');
458
- assert(isOk || isNoHistory,
459
- `drift_from_history positive: structured report OR no-history error (got ${JSON.stringify(r54.data).slice(0,80)})`);
460
- if (isOk) {
461
- assert(typeof r54.data.baseline?.key === 'string',
462
- 'drift_from_history: baseline.key is a string');
463
- assert(typeof r54.data.alert?.threshold === 'number',
464
- 'drift_from_history: alert.threshold echoed numerically');
465
- }
466
- } else {
467
- console.log(` ⊘ drift_from_history: degraded (metaharness or memory absent)`);
468
- }
469
- }
470
-
471
- const r = await trendTool.handler({ baselineKey: 'missing-X', currentKey: 'missing-Y' });
472
- assert(r.exitCode === 2,
473
- 'audit_trend bad-keys path exits 2 (script-level guard fires)');
474
- assert(r.data === null || r.data === undefined,
475
- 'audit_trend bad-keys path: data null (no JSON emitted on stderr exit)');
476
- // iter 44 — success semantic anchor. Pre-iter-44 wrapper returned
477
- // success:true for this case (because no degraded marker). Now
478
- // returns false because exitCode !== 0.
479
- assert(r.success === false,
480
- 'audit_trend bad-keys path: success === false (iter 44 fix)');
481
- }
482
-
483
- // Cleanup
484
- try { (await import('node:fs')).rmSync(tmp, { recursive: true, force: true }); } catch { /* ignore */ }
485
-
486
- console.log(`\n${passed} passed, ${failed} failed`);
487
- if (failed > 0) {
488
- console.log('\nFailures:');
489
- for (const f of failures) console.log(` - ${f}`);
490
- process.exit(1);
491
- }
492
- console.log('\n✓ All 15 MCP tools satisfy the runtime contract.');
493
- }
494
-
495
- main().catch((e) => {
496
- console.error('test-mcp-tools crashed:', e.message || e);
497
- process.exit(2);
498
- });
1
+ #!/usr/bin/env node
2
+ // test-mcp-tools.mjs — runtime test for the iter-20/21 MCP tool registry.
3
+ //
4
+ // tsc proves the metaharness-tools.ts module COMPILES; structural smoke
5
+ // proves the source DECLARES the right tool names. Neither proves the
6
+ // HANDLERS actually run without throwing. This test imports the compiled
7
+ // module and invokes every tool's handler with minimal inputs.
8
+ //
9
+ // CONTRACT EACH TOOL MUST SATISFY
10
+ // - handler is callable as `await tool.handler({ ... })`
11
+ // - returns an object with keys: success, data, degraded, exitCode
12
+ // - never throws (even with bad/missing optional dep — graceful)
13
+ // - handler honors the 120s subprocess timeout (no hang)
14
+ //
15
+ // USAGE
16
+ // node scripts/test-mcp-tools.mjs # default
17
+ // node scripts/test-mcp-tools.mjs --format json
18
+ //
19
+ // EXIT CODES
20
+ // 0 all tools satisfy the contract
21
+ // 1 at least one tool failed
22
+ // 2 setup error (compiled dist not present)
23
+
24
+ import { existsSync } from 'node:fs';
25
+ import { dirname, join, resolve } from 'node:path';
26
+ import { fileURLToPath } from 'node:url';
27
+
28
+ const SCRIPTS_DIR = dirname(fileURLToPath(import.meta.url));
29
+ const ARGS = (() => {
30
+ const a = { format: 'table' };
31
+ for (let i = 2; i < process.argv.length; i++) {
32
+ if (process.argv[i] === '--format') a.format = process.argv[++i];
33
+ }
34
+ return a;
35
+ })();
36
+
37
+ let passed = 0, failed = 0;
38
+ const failures = [];
39
+
40
+ function assert(cond, label) {
41
+ if (cond) { console.log(` ✓ ${label}`); passed++; }
42
+ else { console.log(` ✗ ${label}`); failures.push(label); failed++; }
43
+ }
44
+
45
+ async function main() {
46
+ // Locate the compiled dist of metaharness-tools.
47
+ const distPath = resolve(SCRIPTS_DIR, '..', '..', '..',
48
+ 'v3', '@claude-flow', 'cli', 'dist', 'src', 'mcp-tools', 'metaharness-tools.js');
49
+
50
+ if (!existsSync(distPath)) {
51
+ console.log(`# test-mcp-tools — SKIPPED`);
52
+ console.log('');
53
+ console.log(`Compiled dist not present: ${distPath}`);
54
+ console.log(`Build the CLI first:`);
55
+ console.log(` cd v3/@claude-flow/cli && npm run build`);
56
+ console.log('');
57
+ console.log(`Exit 0 — this script is meaningfully runnable only post-build.`);
58
+ process.exit(0);
59
+ }
60
+
61
+ let mod;
62
+ try {
63
+ mod = await import(distPath);
64
+ } catch (e) {
65
+ console.error(`test-mcp-tools: failed to import ${distPath}: ${e.message}`);
66
+ process.exit(2);
67
+ }
68
+
69
+ const tools = mod.metaharnessTools;
70
+ console.log(`# test-mcp-tools — runtime contract\n`);
71
+
72
+ // ──────────────────────────────────────────────────────────────────
73
+ // PHASE 1 — module exports the right shape
74
+ // ──────────────────────────────────────────────────────────────────
75
+ console.log('Phase 1 — module shape');
76
+ assert(Array.isArray(tools), 'metaharnessTools is an array');
77
+ assert(tools.length === 16, `16 tools registered (got ${tools.length})`);
78
+
79
+ const expectedNames = new Set([
80
+ 'metaharness_score',
81
+ 'metaharness_genome',
82
+ 'metaharness_mcp_scan',
83
+ 'metaharness_threat_model',
84
+ 'metaharness_oia_audit',
85
+ 'metaharness_audit_list',
86
+ 'metaharness_audit_trend',
87
+ // iter 36 — ADR-152 §3.1 production
88
+ 'metaharness_similarity',
89
+ // iter 54 — one-command drift detection (composes audit-list + oia-audit + audit-trend)
90
+ 'metaharness_drift_from_history',
91
+ // ADR-153 — bench suites + evolve driver + security-focused bench
92
+ 'metaharness_bench',
93
+ 'metaharness_evolve',
94
+ 'metaharness_security_bench',
95
+ // @metaharness/redblue@~0.1.4 — adversarial red/blue LLM testing
96
+ 'metaharness_redblue',
97
+ // metaharness@0.3.0 — upstream ADR-235 GEPA learning run
98
+ 'metaharness_learn',
99
+ // @metaharness/darwin@0.8.0 — GEPA library surface (genome ops)
100
+ 'metaharness_gepa',
101
+ // ADR-322 — in-process governed evaluation and atomic promotion
102
+ 'metaharness_flywheel',
103
+ ]);
104
+ const actualNames = new Set(tools.map((t) => t.name));
105
+ for (const name of expectedNames) {
106
+ assert(actualNames.has(name), `${name} registered`);
107
+ }
108
+
109
+ // ──────────────────────────────────────────────────────────────────
110
+ // PHASE 2 — every tool has the required MCP shape
111
+ // ──────────────────────────────────────────────────────────────────
112
+ console.log('\nPhase 2 — per-tool shape');
113
+ for (const tool of tools) {
114
+ const ok = typeof tool.name === 'string'
115
+ && typeof tool.description === 'string'
116
+ && typeof tool.category === 'string'
117
+ && typeof tool.handler === 'function'
118
+ && typeof tool.inputSchema === 'object';
119
+ assert(ok, `${tool.name} has {name, description, category, handler, inputSchema}`);
120
+ assert(tool.category === 'metaharness', `${tool.name} category === 'metaharness'`);
121
+ }
122
+
123
+ // ──────────────────────────────────────────────────────────────────
124
+ // PHASE 3 — handlers callable + return contract shape
125
+ //
126
+ // We invoke each handler with minimal valid input. The handlers may
127
+ // succeed (if metaharness is installed) or report degraded (if not).
128
+ // EITHER way, they must return { success, data, degraded, exitCode }
129
+ // without throwing.
130
+ // ──────────────────────────────────────────────────────────────────
131
+ console.log('\nPhase 3 — handler invocations (allow up to 30s each)');
132
+ for (const tool of tools) {
133
+ // Construct minimal valid input per tool.
134
+ let input = {};
135
+ if (tool.name === 'metaharness_audit_trend') {
136
+ // Requires baselineKey + currentKey — use fake keys that won't
137
+ // resolve so we exercise the not-found path.
138
+ input = { baselineKey: 'audit-fake-base', currentKey: 'audit-fake-curr' };
139
+ }
140
+ if (tool.name === 'metaharness_similarity') {
141
+ // Needs --a/--b OR --a-key/--b-key. Use fake mem keys to exercise
142
+ // the graceful not-found path (matches audit_trend convention).
143
+ input = { aKey: 'harness-fake-a', bKey: 'harness-fake-b' };
144
+ }
145
+ if (tool.name === 'metaharness_drift_from_history') {
146
+ // iter 54 — composes 3 subprocesses, needs more time than the default.
147
+ input = { dryRun: true, threshold: 0.5 };
148
+ }
149
+ if (tool.name === 'metaharness_oia_audit') {
150
+ // iter 128 — composite audit runs 5 sub-audits (oia-manifest +
151
+ // threat-model + mcp-scan + score + genome) in parallel. Each
152
+ // shells out via npx. --dry-run skips memory persistence so the
153
+ // test doesn't pollute namespaces.
154
+ input = { dryRun: true };
155
+ }
156
+ if (tool.name === 'metaharness_redblue') {
157
+ // `attack` preview is the fastest path that exercises the upstream
158
+ // binary without needing OPENROUTER_API_KEY or running any model
159
+ // calls. Count=1 keeps cold-cache npx fetch the dominant cost.
160
+ input = { subcommand: 'attack', family: 'prompt', count: 1 };
161
+ }
162
+ if (tool.name === 'metaharness_learn') {
163
+ // No repo checkout in CI → structured {status:"checkout-required"}
164
+ // exit-0 path. $0: without run=true upstream never spends anyway.
165
+ input = {};
166
+ }
167
+ if (tool.name === 'metaharness_gepa') {
168
+ // op is required; `genome` loads + validates the SHIPPED cand-6
169
+ // genome — pure-local library call once darwin is cached.
170
+ input = { op: 'genome' };
171
+ }
172
+
173
+ // iter 124 → 130 — timeouts have crept up as CI cold-cache npx
174
+ // warmup costs got measured. Final budgets:
175
+ // default : 60s
176
+ // chain-tools : 180s (drift_from_history + oia_audit + audit_list)
177
+ // iter 131 — bumped chain-tool budget 90s → 180s. audit_list still
178
+ // timed out at 90s in CI; locally it runs in ~4s, but CI's
179
+ // `npx @claude-flow/cli@latest memory list` invocation pays both
180
+ // the npx fetch AND a full CLI startup (which loads agentic-flow +
181
+ // ONNX). 180s gives 30x headroom over the local cost.
182
+ const isChainTool = tool.name === 'metaharness_drift_from_history'
183
+ || tool.name === 'metaharness_oia_audit'
184
+ || tool.name === 'metaharness_audit_list'
185
+ // redblue: `attack prompt --count 1` is preview-only (no model
186
+ // calls) but the cold-cache `npx -y @metaharness/redblue@~0.1.4`
187
+ // fetch can take 30-60s. 180s gives 3x headroom.
188
+ || tool.name === 'metaharness_redblue'
189
+ // learn: cold-cache `npx -y metaharness@latest` fetch dominates.
190
+ // gepa: one-time `npm install --prefix ~/.ruflo/darwin-cache-*`
191
+ // fallback install can take 30-60s on cold cache.
192
+ || tool.name === 'metaharness_learn'
193
+ || tool.name === 'metaharness_gepa';
194
+ const timeoutMs = isChainTool ? 180_000 : 60_000;
195
+ const handlerPromise = tool.handler(input);
196
+ const timeoutPromise = new Promise((_, reject) =>
197
+ setTimeout(() => reject(new Error(`${timeoutMs / 1000}s handler timeout`)), timeoutMs));
198
+
199
+ let result;
200
+ let threw = false;
201
+ try {
202
+ result = await Promise.race([handlerPromise, timeoutPromise]);
203
+ } catch (e) {
204
+ threw = true;
205
+ console.log(` [${tool.name}] handler threw: ${e.message.slice(0, 80)}`);
206
+ }
207
+
208
+ assert(!threw, `${tool.name} handler did not throw`);
209
+ if (!threw && result) {
210
+ assert(typeof result === 'object', `${tool.name} returns object`);
211
+ assert('success' in result, `${tool.name} result has 'success'`);
212
+ assert('data' in result, `${tool.name} result has 'data'`);
213
+ assert('degraded' in result, `${tool.name} result has 'degraded'`);
214
+ assert('exitCode' in result, `${tool.name} result has 'exitCode'`);
215
+ }
216
+ }
217
+
218
+ // ──────────────────────────────────────────────────────────────────
219
+ // PHASE 4 — POSITIVE-CASE data-shape validation (iter 43)
220
+ //
221
+ // Iter 37 verified the {success, data, degraded, exitCode} envelope.
222
+ // It did NOT verify that data.X contains the right keys when success
223
+ // is genuinely true — leaving room for iter 42-style bugs where a
224
+ // handler returns valid-looking degraded JSON while silently
225
+ // misrouting input. This phase invokes each handler with VALID
226
+ // inputs and asserts the expected output shape.
227
+ //
228
+ // Tools that depend on `npx metaharness` (score/genome/mcp-scan/
229
+ // threat-model/oia-audit/audit-list/audit-trend) are SKIPPED in this
230
+ // phase when the optional dep isn't installed — they're covered by
231
+ // the no-metaharness-smoke workflow's drill. The similarity tool
232
+ // has no @metaharness/* dep, so its positive case ALWAYS runs.
233
+ // ──────────────────────────────────────────────────────────────────
234
+ console.log('\nPhase 4 — positive-case data shape (iter 43)');
235
+
236
+ const { writeFileSync, mkdtempSync, mkdirSync } = await import('node:fs');
237
+ const { tmpdir } = await import('node:os');
238
+ const { join: pjoin } = await import('node:path');
239
+ const tmp = mkdtempSync(pjoin(tmpdir(), 'mcp-positive-'));
240
+
241
+ // #2626 — upstream deliberately exits 2 for a valid `blocked` genome.
242
+ // The wrapper must preserve that report as data instead of converting it
243
+ // into an input/system error at the MCP boundary.
244
+ const genomeTool = tools.find((t) => t.name === 'metaharness_genome');
245
+ if (genomeTool) {
246
+ const blockedRepo = pjoin(tmp, 'blocked-repo');
247
+ mkdirSync(blockedRepo);
248
+ const r = await genomeTool.handler({ path: blockedRepo });
249
+ if (!r.degraded) {
250
+ assert(r.success === true,
251
+ 'genome blocked verdict: wrapper succeeds with a valid report (#2626)');
252
+ assert(r.exitCode === 0,
253
+ 'genome blocked verdict: wrapper exitCode === 0 (#2626)');
254
+ assert(r.data?.verdict === 'blocked',
255
+ `genome blocked verdict: data.verdict === blocked (got ${r.data?.verdict})`);
256
+ assert(r.data?.verdictExitCode === 2,
257
+ `genome blocked verdict: preserves upstream exit 2 (got ${r.data?.verdictExitCode})`);
258
+ assert(typeof r.data?.risk_score === 'number' && r.data.risk_score >= 0.7,
259
+ `genome blocked verdict: risk_score >= 0.7 (got ${r.data?.risk_score})`);
260
+ } else {
261
+ console.log(` ⊘ genome: metaharness absent — graceful skip`);
262
+ }
263
+ }
264
+
265
+ // metaharness_similarity — full positive case (no @metaharness/* needed)
266
+ const simTool = tools.find((t) => t.name === 'metaharness_similarity');
267
+ if (simTool) {
268
+ const aPath = pjoin(tmp, 'a.json');
269
+ const bPath = pjoin(tmp, 'b.json');
270
+ writeFileSync(aPath, JSON.stringify({
271
+ score: { harnessFit: 78, compileConfidence: 92, taskCoverage: 65, toolSafety: 88, memoryUsefulness: 70, estCostPerRunUsd: 0.04, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:legal' },
272
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','w'], risk_score: 0.45, test_confidence: 0.7, publish_readiness: 0.6 },
273
+ }));
274
+ writeFileSync(bPath, JSON.stringify({
275
+ score: { harnessFit: 75, compileConfidence: 90, taskCoverage: 70, toolSafety: 90, memoryUsefulness: 72, estCostPerRunUsd: 0.05, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:support' },
276
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','q','r'], risk_score: 0.40, test_confidence: 0.75, publish_readiness: 0.65 },
277
+ }));
278
+ const r = await simTool.handler({ aFile: aPath, bFile: bPath });
279
+ assert(r.degraded === false, 'similarity positive case: degraded === false');
280
+ assert(r.success === true, 'similarity positive case: success === true');
281
+ assert(r.exitCode === 0, 'similarity positive case: exitCode === 0');
282
+ const d = r.data ?? {};
283
+ assert(typeof d.overall === 'number', 'similarity data has numeric `overall`');
284
+ assert(typeof d.components === 'object' && d.components !== null,
285
+ 'similarity data has `components` object');
286
+ assert(typeof d.components?.cosine === 'number',
287
+ 'similarity components.cosine numeric');
288
+ assert(typeof d.components?.categorical === 'number',
289
+ 'similarity components.categorical numeric');
290
+ assert(typeof d.components?.jaccard === 'number',
291
+ 'similarity components.jaccard numeric');
292
+ assert(typeof d.weights === 'object' && d.weights !== null,
293
+ 'similarity data has `weights` object');
294
+ assert(d.adr === 'ADR-152', 'similarity data tagged adr=ADR-152');
295
+ // Regression anchor — same fixtures as iter-35 spike with non-matching topologies
296
+ assert(d.overall > 0 && d.overall < 1,
297
+ `similarity overall in (0, 1) — got ${d.overall}`);
298
+
299
+ // Per-dimension variant
300
+ const rPD = await simTool.handler({ aFile: aPath, bFile: bPath, perDimension: true });
301
+ assert(typeof rPD.data?.perDimension === 'object',
302
+ 'similarity perDimension=true populates breakdown');
303
+
304
+ // Alert-below variant exercises non-zero exit
305
+ const rAlert = await simTool.handler({ aFile: aPath, bFile: bPath, alertBelow: 0.99 });
306
+ assert(rAlert.data?.alert?.triggered === true,
307
+ 'similarity alertBelow=0.99 triggers alert');
308
+ assert(rAlert.exitCode === 1, 'similarity alertBelow=0.99 → exitCode 1');
309
+ // iter 44 — success semantic anchor (was true under the pre-iter-44
310
+ // `!degraded` rule; now false because exitCode !== 0 dominates).
311
+ assert(rAlert.success === false,
312
+ 'similarity alertBelow=0.99 → success === false (iter 44 fix)');
313
+ }
314
+
315
+ // metaharness_mcp_scan — positive case post iter-50 parser landing.
316
+ // Until iter 50, mcp_scan's data field was an alert-only object with
317
+ // no structured findings. After iter 50, findings[] is always present
318
+ // (parsed from upstream text) and summary{overallSeverity, totalCount}
319
+ // accompanies it.
320
+ const scanTool = tools.find((t) => t.name === 'metaharness_mcp_scan');
321
+ if (scanTool) {
322
+ // Run against ruflo itself — guaranteed to produce at least the
323
+ // INFO finding the iter-50 parser test verified manually.
324
+ const r = await scanTool.handler({ path: '.', failOn: 'high' });
325
+ // Either succeeds with structured findings, or gracefully degrades
326
+ // if metaharness isn't installed in this environment.
327
+ if (!r.degraded) {
328
+ assert(r.success === true, 'mcp_scan positive: success === true');
329
+ assert(r.exitCode === 0, 'mcp_scan positive: exitCode === 0');
330
+ assert(Array.isArray(r.data?.findings),
331
+ 'mcp_scan positive: data.findings is an array (iter 50 fix)');
332
+ // Cwd-dependent: when scanning a dir without .mcp/servers.json the
333
+ // upstream emits no findings. Only verify shape contract when array
334
+ // is populated — the array-presence assertion above is the
335
+ // load-bearing one for iter 50.
336
+ if (r.data?.findings.length > 0) {
337
+ const first = r.data.findings[0];
338
+ assert(typeof first?.severity === 'string',
339
+ 'mcp_scan positive: first finding has string severity');
340
+ assert(typeof first?.message === 'string',
341
+ 'mcp_scan positive: first finding has string message');
342
+ }
343
+ // summary may be null if the upstream produced no Result: line —
344
+ // verify the field's presence (null OR object) but only deep-check
345
+ // when populated.
346
+ if (r.data?.summary) {
347
+ assert(typeof r.data.summary.totalCount === 'number',
348
+ 'mcp_scan positive: data.summary.totalCount is numeric (when summary present)');
349
+ }
350
+ } else {
351
+ console.log(` ⊘ mcp_scan: metaharness absent — graceful skip`);
352
+ }
353
+ }
354
+
355
+ // metaharness_audit_trend — positive case via file inputs
356
+ const trendTool = tools.find((t) => t.name === 'metaharness_audit_trend');
357
+ if (trendTool) {
358
+ const basePath = pjoin(tmp, 'base.json');
359
+ const currPath = pjoin(tmp, 'curr.json');
360
+ const fingerprint = {
361
+ score: { harnessFit: 80, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
362
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['a','b','c'], risk_score: 0.3, test_confidence: 0.85, publish_readiness: 0.9 },
363
+ };
364
+ writeFileSync(basePath, JSON.stringify({
365
+ startedAt: '2026-06-15T00:00:00Z',
366
+ composite: { worst: 'clean' },
367
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
368
+ fingerprint,
369
+ }));
370
+ writeFileSync(currPath, JSON.stringify({
371
+ startedAt: '2026-06-16T00:00:00Z',
372
+ composite: { worst: 'clean' },
373
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
374
+ fingerprint,
375
+ }));
376
+ // audit_trend tool only supports keys, not files at the MCP layer.
377
+ // Document its actual wrapper semantics so future-us doesn't get
378
+ // surprised:
379
+ // - bad keys → script exits 2 with stderr (no JSON payload)
380
+ // - runScript() can't parse a {degraded:true} marker, so it
381
+ // returns degraded:false / success:true / exitCode:2
382
+ // This is a real wrapper bug (success should not be true when
383
+ // exit!=0 AND no JSON came back), tracked separately. Asserting
384
+ // current behavior here protects against silent semantic shifts.
385
+ // iter 46 — file-input path. audit_trend now accepts baselineFile/currentFile.
386
+ const rFiles = await trendTool.handler({ baselineFile: basePath, currentFile: currPath });
387
+ assert(rFiles.success === true,
388
+ 'audit_trend file-input path: success === true (iter 46)');
389
+ assert(rFiles.exitCode === 0, 'audit_trend file-input path: exitCode === 0');
390
+ assert(typeof rFiles.data?.delta === 'object',
391
+ 'audit_trend file-input path: data.delta object present');
392
+ assert(rFiles.data?.delta?.structuralDistance?.verdict === 'near-identical',
393
+ `audit_trend file-input path: identical fingerprints → near-identical (got ${rFiles.data?.delta?.structuralDistance?.verdict})`);
394
+
395
+ // iter 54 — metaharness_drift_from_history positive case
396
+ const driftTool = tools.find((t) => t.name === 'metaharness_drift_from_history');
397
+ if (driftTool) {
398
+ // iter 71 — verify iter-66/67 fast-path flags are now MCP-callable
399
+ // Synthesize a baseline file on disk; pass via the new baselineFile input.
400
+ const baselinePath = pjoin(tmp, 'drift-baseline.json');
401
+ writeFileSync(baselinePath, JSON.stringify({
402
+ startedAt: '2026-06-16T00:00:00Z',
403
+ composite: { worst: 'clean' },
404
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
405
+ fingerprint: {
406
+ score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
407
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
408
+ },
409
+ }));
410
+ const rFastFast = await driftTool.handler({
411
+ path: '.', dryRun: true, threshold: 0.5, baselineFile: baselinePath,
412
+ });
413
+ if (!rFastFast.degraded) {
414
+ assert(rFastFast.data?.timing?.usedBaselineFile === true,
415
+ 'drift_from_history MCP-layer: baselineFile fastpath fires (iter 71)');
416
+ assert(rFastFast.data?.timing?.skippedAuditList === true,
417
+ 'drift_from_history MCP-layer: skippedAuditList=true via baselineFile (iter 71)');
418
+ }
419
+
420
+ // iter 85 — verify iter-78's alertOnNewSeverity MCP input plumbs
421
+ // through. baselineFile has no findings; current ruflo audit has
422
+ // 1 INFO finding. With alertOnNewSeverity='info' the gate fires
423
+ // and surfaces in the response.
424
+ const baselineNoFindings = pjoin(tmp, 'drift-baseline-no-findings.json');
425
+ writeFileSync(baselineNoFindings, JSON.stringify({
426
+ startedAt: '2026-06-16T00:00:00Z',
427
+ composite: { worst: 'clean' },
428
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
429
+ fingerprint: {
430
+ score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
431
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
432
+ },
433
+ }));
434
+ const rSevAlert = await driftTool.handler({
435
+ path: '.', dryRun: true, threshold: 0.5,
436
+ baselineFile: baselineNoFindings,
437
+ alertOnNewSeverity: 'info',
438
+ });
439
+ if (!rSevAlert.degraded) {
440
+ assert(rSevAlert.data?.alert?.newSeverityThreshold === 'info',
441
+ 'drift_from_history MCP-layer: alertOnNewSeverity echoed in payload (iter 85)');
442
+ // Triggered AND exit code reflects (only if the audit actually had findings)
443
+ if (rSevAlert.data?.alert?.triggered === true) {
444
+ assert(rSevAlert.exitCode === 1,
445
+ `drift_from_history MCP-layer: alertOnNewSeverity exitCode=1 when triggered (got ${rSevAlert.exitCode})`);
446
+ assert(rSevAlert.success === false,
447
+ 'drift_from_history MCP-layer: success===false when alert fires (iter 44 fix)');
448
+ }
449
+ }
450
+
451
+ const r54 = await driftTool.handler({ path: '.', dryRun: true, threshold: 0.5 });
452
+ if (!r54.degraded) {
453
+ assert(typeof r54.data === 'object' && r54.data !== null,
454
+ 'drift_from_history positive: data is an object');
455
+ // Either it produced the structured drift report OR the no-history error
456
+ const isOk = r54.data?.command === 'drift-from-history';
457
+ const isNoHistory = typeof r54.data?.error === 'string' && r54.data.error.includes('no audit records');
458
+ assert(isOk || isNoHistory,
459
+ `drift_from_history positive: structured report OR no-history error (got ${JSON.stringify(r54.data).slice(0,80)})`);
460
+ if (isOk) {
461
+ assert(typeof r54.data.baseline?.key === 'string',
462
+ 'drift_from_history: baseline.key is a string');
463
+ assert(typeof r54.data.alert?.threshold === 'number',
464
+ 'drift_from_history: alert.threshold echoed numerically');
465
+ }
466
+ } else {
467
+ console.log(` ⊘ drift_from_history: degraded (metaharness or memory absent)`);
468
+ }
469
+ }
470
+
471
+ const r = await trendTool.handler({ baselineKey: 'missing-X', currentKey: 'missing-Y' });
472
+ assert(r.exitCode === 2,
473
+ 'audit_trend bad-keys path exits 2 (script-level guard fires)');
474
+ assert(r.data === null || r.data === undefined,
475
+ 'audit_trend bad-keys path: data null (no JSON emitted on stderr exit)');
476
+ // iter 44 — success semantic anchor. Pre-iter-44 wrapper returned
477
+ // success:true for this case (because no degraded marker). Now
478
+ // returns false because exitCode !== 0.
479
+ assert(r.success === false,
480
+ 'audit_trend bad-keys path: success === false (iter 44 fix)');
481
+ }
482
+
483
+ // Cleanup
484
+ try { (await import('node:fs')).rmSync(tmp, { recursive: true, force: true }); } catch { /* ignore */ }
485
+
486
+ console.log(`\n${passed} passed, ${failed} failed`);
487
+ if (failed > 0) {
488
+ console.log('\nFailures:');
489
+ for (const f of failures) console.log(` - ${f}`);
490
+ process.exit(1);
491
+ }
492
+ console.log('\n✓ All 15 MCP tools satisfy the runtime contract.');
493
+ }
494
+
495
+ main().catch((e) => {
496
+ console.error('test-mcp-tools crashed:', e.message || e);
497
+ process.exit(2);
498
+ });