@claude-flow/cli 3.32.9 → 3.32.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (447) hide show
  1. package/.claude/.proven-config-version +1 -0
  2. package/.claude/agents/analysis/analyze-code-quality.md +178 -178
  3. package/.claude/agents/analysis/code-analyzer.md +209 -209
  4. package/.claude/agents/analysis/code-review/analyze-code-quality.md +178 -178
  5. package/.claude/agents/architecture/arch-system-design.md +156 -156
  6. package/.claude/agents/architecture/system-design/arch-system-design.md +154 -154
  7. package/.claude/agents/browser/browser-agent.yaml +182 -182
  8. package/.claude/agents/consensus/byzantine-coordinator.md +62 -62
  9. package/.claude/agents/consensus/crdt-synchronizer.md +996 -996
  10. package/.claude/agents/consensus/gossip-coordinator.md +62 -62
  11. package/.claude/agents/consensus/performance-benchmarker.md +850 -850
  12. package/.claude/agents/consensus/quorum-manager.md +822 -822
  13. package/.claude/agents/consensus/raft-manager.md +62 -62
  14. package/.claude/agents/consensus/security-manager.md +621 -621
  15. package/.claude/agents/core/planner.md +374 -374
  16. package/.claude/agents/custom/test-long-runner.md +44 -44
  17. package/.claude/agents/data/data-ml-model.md +444 -444
  18. package/.claude/agents/data/ml/data-ml-model.md +192 -192
  19. package/.claude/agents/development/backend/dev-backend-api.md +141 -141
  20. package/.claude/agents/development/dev-backend-api.md +344 -344
  21. package/.claude/agents/devops/ci-cd/ops-cicd-github.md +163 -163
  22. package/.claude/agents/devops/ops-cicd-github.md +164 -164
  23. package/.claude/agents/documentation/api-docs/docs-api-openapi.md +173 -173
  24. package/.claude/agents/documentation/docs-api-openapi.md +354 -354
  25. package/.claude/agents/flow-nexus/app-store.md +87 -87
  26. package/.claude/agents/flow-nexus/authentication.md +68 -68
  27. package/.claude/agents/flow-nexus/challenges.md +80 -80
  28. package/.claude/agents/flow-nexus/neural-network.md +87 -87
  29. package/.claude/agents/flow-nexus/payments.md +82 -82
  30. package/.claude/agents/flow-nexus/sandbox.md +75 -75
  31. package/.claude/agents/flow-nexus/swarm.md +75 -75
  32. package/.claude/agents/flow-nexus/user-tools.md +95 -95
  33. package/.claude/agents/flow-nexus/workflow.md +83 -83
  34. package/.claude/agents/github/code-review-swarm.md +377 -377
  35. package/.claude/agents/github/github-modes.md +172 -172
  36. package/.claude/agents/github/issue-tracker.md +575 -575
  37. package/.claude/agents/github/multi-repo-swarm.md +552 -552
  38. package/.claude/agents/github/pr-manager.md +437 -437
  39. package/.claude/agents/github/project-board-sync.md +508 -508
  40. package/.claude/agents/github/release-manager.md +604 -604
  41. package/.claude/agents/github/release-swarm.md +582 -582
  42. package/.claude/agents/github/repo-architect.md +397 -397
  43. package/.claude/agents/github/swarm-issue.md +572 -572
  44. package/.claude/agents/github/swarm-pr.md +427 -427
  45. package/.claude/agents/github/sync-coordinator.md +451 -451
  46. package/.claude/agents/github/workflow-automation.md +902 -902
  47. package/.claude/agents/goal/agent.md +815 -815
  48. package/.claude/agents/optimization/benchmark-suite.md +664 -664
  49. package/.claude/agents/optimization/load-balancer.md +430 -430
  50. package/.claude/agents/optimization/performance-monitor.md +671 -671
  51. package/.claude/agents/optimization/resource-allocator.md +673 -673
  52. package/.claude/agents/optimization/topology-optimizer.md +807 -807
  53. package/.claude/agents/payments/agentic-payments.md +126 -126
  54. package/.claude/agents/sona/sona-learning-optimizer.md +74 -74
  55. package/.claude/agents/sparc/architecture.md +698 -698
  56. package/.claude/agents/sparc/pseudocode.md +519 -519
  57. package/.claude/agents/sparc/refinement.md +801 -801
  58. package/.claude/agents/sparc/specification.md +477 -477
  59. package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +224 -224
  60. package/.claude/agents/specialized/spec-mobile-react-native.md +226 -226
  61. package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
  62. package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
  63. package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
  64. package/.claude/agents/sublinear/performance-optimizer.md +367 -367
  65. package/.claude/agents/sublinear/trading-predictor.md +245 -245
  66. package/.claude/agents/swarm/adaptive-coordinator.md +1126 -1126
  67. package/.claude/agents/swarm/hierarchical-coordinator.md +709 -709
  68. package/.claude/agents/swarm/mesh-coordinator.md +962 -962
  69. package/.claude/agents/templates/automation-smart-agent.md +204 -204
  70. package/.claude/agents/templates/base-template-generator.md +289 -289
  71. package/.claude/agents/templates/coordinator-swarm-init.md +89 -89
  72. package/.claude/agents/templates/github-pr-manager.md +176 -176
  73. package/.claude/agents/templates/implementer-sparc-coder.md +258 -258
  74. package/.claude/agents/templates/memory-coordinator.md +186 -186
  75. package/.claude/agents/templates/orchestrator-task.md +138 -138
  76. package/.claude/agents/templates/performance-analyzer.md +198 -198
  77. package/.claude/agents/templates/sparc-coordinator.md +513 -513
  78. package/.claude/agents/testing/production-validator.md +394 -394
  79. package/.claude/agents/testing/tdd-london-swarm.md +243 -243
  80. package/.claude/agents/v3/aidefence-guardian.md +282 -282
  81. package/.claude/agents/v3/claims-authorizer.md +208 -208
  82. package/.claude/agents/v3/collective-intelligence-coordinator.md +993 -993
  83. package/.claude/agents/v3/ddd-domain-expert.md +220 -220
  84. package/.claude/agents/v3/injection-analyst.md +236 -236
  85. package/.claude/agents/v3/performance-engineer.md +1233 -1233
  86. package/.claude/agents/v3/pii-detector.md +151 -151
  87. package/.claude/agents/v3/reasoningbank-learner.md +213 -213
  88. package/.claude/agents/v3/security-architect-aidefence.md +410 -410
  89. package/.claude/agents/v3/security-architect.md +867 -867
  90. package/.claude/agents/v3/swarm-memory-manager.md +157 -157
  91. package/.claude/agents/v3/v3-integration-architect.md +205 -205
  92. package/.claude/commands/agents/README.md +50 -50
  93. package/.claude/commands/agents/agent-capabilities.md +140 -140
  94. package/.claude/commands/agents/agent-coordination.md +28 -28
  95. package/.claude/commands/agents/agent-spawning.md +28 -28
  96. package/.claude/commands/agents/agent-types.md +216 -216
  97. package/.claude/commands/agents/health.md +139 -139
  98. package/.claude/commands/agents/list.md +100 -100
  99. package/.claude/commands/agents/logs.md +130 -130
  100. package/.claude/commands/agents/metrics.md +122 -122
  101. package/.claude/commands/agents/pool.md +127 -127
  102. package/.claude/commands/agents/spawn.md +140 -140
  103. package/.claude/commands/agents/status.md +115 -115
  104. package/.claude/commands/agents/stop.md +102 -102
  105. package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
  106. package/.claude/commands/analysis/README.md +9 -9
  107. package/.claude/commands/analysis/bottleneck-detect.md +162 -162
  108. package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
  109. package/.claude/commands/analysis/performance-report.md +25 -25
  110. package/.claude/commands/analysis/token-efficiency.md +44 -44
  111. package/.claude/commands/analysis/token-usage.md +25 -25
  112. package/.claude/commands/automation/README.md +9 -9
  113. package/.claude/commands/automation/auto-agent.md +122 -122
  114. package/.claude/commands/automation/self-healing.md +105 -105
  115. package/.claude/commands/automation/session-memory.md +89 -89
  116. package/.claude/commands/automation/smart-agents.md +72 -72
  117. package/.claude/commands/automation/smart-spawn.md +25 -25
  118. package/.claude/commands/automation/workflow-select.md +25 -25
  119. package/.claude/commands/claude-flow-help.md +103 -103
  120. package/.claude/commands/claude-flow-memory.md +107 -107
  121. package/.claude/commands/claude-flow-swarm.md +205 -205
  122. package/.claude/commands/coordination/README.md +9 -9
  123. package/.claude/commands/coordination/agent-spawn.md +25 -25
  124. package/.claude/commands/coordination/init.md +44 -44
  125. package/.claude/commands/coordination/orchestrate.md +43 -43
  126. package/.claude/commands/coordination/spawn.md +45 -45
  127. package/.claude/commands/coordination/swarm-init.md +85 -85
  128. package/.claude/commands/coordination/task-orchestrate.md +25 -25
  129. package/.claude/commands/github/README.md +11 -11
  130. package/.claude/commands/github/code-review-swarm.md +513 -513
  131. package/.claude/commands/github/code-review.md +25 -25
  132. package/.claude/commands/github/github-modes.md +146 -146
  133. package/.claude/commands/github/github-swarm.md +121 -121
  134. package/.claude/commands/github/issue-tracker.md +291 -291
  135. package/.claude/commands/github/issue-triage.md +25 -25
  136. package/.claude/commands/github/multi-repo-swarm.md +518 -518
  137. package/.claude/commands/github/pr-enhance.md +26 -26
  138. package/.claude/commands/github/pr-manager.md +169 -169
  139. package/.claude/commands/github/project-board-sync.md +470 -470
  140. package/.claude/commands/github/release-manager.md +339 -339
  141. package/.claude/commands/github/release-swarm.md +543 -543
  142. package/.claude/commands/github/repo-analyze.md +25 -25
  143. package/.claude/commands/github/repo-architect.md +366 -366
  144. package/.claude/commands/github/swarm-issue.md +484 -484
  145. package/.claude/commands/github/swarm-pr.md +287 -287
  146. package/.claude/commands/github/sync-coordinator.md +302 -302
  147. package/.claude/commands/github/workflow-automation.md +441 -441
  148. package/.claude/commands/hive-mind/README.md +17 -17
  149. package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
  150. package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
  151. package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
  152. package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
  153. package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
  154. package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
  155. package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
  156. package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
  157. package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
  158. package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
  159. package/.claude/commands/hive-mind/hive-mind.md +27 -27
  160. package/.claude/commands/hooks/README.md +11 -11
  161. package/.claude/commands/hooks/overview.md +57 -57
  162. package/.claude/commands/hooks/post-edit.md +117 -117
  163. package/.claude/commands/hooks/post-task.md +112 -112
  164. package/.claude/commands/hooks/pre-edit.md +113 -113
  165. package/.claude/commands/hooks/pre-task.md +111 -111
  166. package/.claude/commands/hooks/session-end.md +118 -118
  167. package/.claude/commands/hooks/setup.md +102 -102
  168. package/.claude/commands/memory/README.md +9 -9
  169. package/.claude/commands/memory/memory-persist.md +25 -25
  170. package/.claude/commands/memory/memory-search.md +25 -25
  171. package/.claude/commands/memory/memory-usage.md +25 -25
  172. package/.claude/commands/memory/neural.md +47 -47
  173. package/.claude/commands/monitoring/README.md +9 -9
  174. package/.claude/commands/monitoring/agent-metrics.md +25 -25
  175. package/.claude/commands/monitoring/agents.md +44 -44
  176. package/.claude/commands/monitoring/real-time-view.md +25 -25
  177. package/.claude/commands/monitoring/status.md +46 -46
  178. package/.claude/commands/monitoring/swarm-monitor.md +25 -25
  179. package/.claude/commands/optimization/README.md +9 -9
  180. package/.claude/commands/optimization/auto-topology.md +61 -61
  181. package/.claude/commands/optimization/cache-manage.md +25 -25
  182. package/.claude/commands/optimization/parallel-execute.md +25 -25
  183. package/.claude/commands/optimization/parallel-execution.md +49 -49
  184. package/.claude/commands/optimization/topology-optimize.md +25 -25
  185. package/.claude/commands/pair/README.md +260 -260
  186. package/.claude/commands/pair/commands.md +545 -545
  187. package/.claude/commands/pair/config.md +509 -509
  188. package/.claude/commands/pair/examples.md +511 -511
  189. package/.claude/commands/pair/modes.md +347 -347
  190. package/.claude/commands/pair/session.md +406 -406
  191. package/.claude/commands/pair/start.md +208 -208
  192. package/.claude/commands/sparc/analyzer.md +51 -51
  193. package/.claude/commands/sparc/architect.md +53 -53
  194. package/.claude/commands/sparc/ask.md +97 -97
  195. package/.claude/commands/sparc/batch-executor.md +54 -54
  196. package/.claude/commands/sparc/code.md +89 -89
  197. package/.claude/commands/sparc/coder.md +54 -54
  198. package/.claude/commands/sparc/debug.md +83 -83
  199. package/.claude/commands/sparc/debugger.md +54 -54
  200. package/.claude/commands/sparc/designer.md +53 -53
  201. package/.claude/commands/sparc/devops.md +109 -109
  202. package/.claude/commands/sparc/docs-writer.md +80 -80
  203. package/.claude/commands/sparc/documenter.md +54 -54
  204. package/.claude/commands/sparc/innovator.md +54 -54
  205. package/.claude/commands/sparc/integration.md +83 -83
  206. package/.claude/commands/sparc/mcp.md +117 -117
  207. package/.claude/commands/sparc/memory-manager.md +54 -54
  208. package/.claude/commands/sparc/optimizer.md +54 -54
  209. package/.claude/commands/sparc/orchestrator.md +131 -131
  210. package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
  211. package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
  212. package/.claude/commands/sparc/researcher.md +54 -54
  213. package/.claude/commands/sparc/reviewer.md +54 -54
  214. package/.claude/commands/sparc/security-review.md +80 -80
  215. package/.claude/commands/sparc/sparc-modes.md +174 -174
  216. package/.claude/commands/sparc/sparc.md +111 -111
  217. package/.claude/commands/sparc/spec-pseudocode.md +80 -80
  218. package/.claude/commands/sparc/supabase-admin.md +348 -348
  219. package/.claude/commands/sparc/swarm-coordinator.md +54 -54
  220. package/.claude/commands/sparc/tdd.md +54 -54
  221. package/.claude/commands/sparc/tester.md +54 -54
  222. package/.claude/commands/sparc/tutorial.md +79 -79
  223. package/.claude/commands/sparc/workflow-manager.md +54 -54
  224. package/.claude/commands/sparc.md +166 -166
  225. package/.claude/commands/stream-chain/pipeline.md +120 -120
  226. package/.claude/commands/stream-chain/run.md +69 -69
  227. package/.claude/commands/swarm/README.md +15 -15
  228. package/.claude/commands/swarm/analysis.md +95 -95
  229. package/.claude/commands/swarm/development.md +96 -96
  230. package/.claude/commands/swarm/examples.md +168 -168
  231. package/.claude/commands/swarm/maintenance.md +102 -102
  232. package/.claude/commands/swarm/optimization.md +117 -117
  233. package/.claude/commands/swarm/research.md +136 -136
  234. package/.claude/commands/swarm/swarm-analysis.md +8 -8
  235. package/.claude/commands/swarm/swarm-background.md +8 -8
  236. package/.claude/commands/swarm/swarm-init.md +19 -19
  237. package/.claude/commands/swarm/swarm-modes.md +8 -8
  238. package/.claude/commands/swarm/swarm-monitor.md +8 -8
  239. package/.claude/commands/swarm/swarm-spawn.md +19 -19
  240. package/.claude/commands/swarm/swarm-status.md +8 -8
  241. package/.claude/commands/swarm/swarm-strategies.md +8 -8
  242. package/.claude/commands/swarm/swarm.md +87 -87
  243. package/.claude/commands/swarm/testing.md +131 -131
  244. package/.claude/commands/training/README.md +9 -9
  245. package/.claude/commands/training/model-update.md +25 -25
  246. package/.claude/commands/training/neural-patterns.md +107 -107
  247. package/.claude/commands/training/neural-train.md +75 -75
  248. package/.claude/commands/training/pattern-learn.md +25 -25
  249. package/.claude/commands/training/specialization.md +62 -62
  250. package/.claude/commands/truth/start.md +142 -142
  251. package/.claude/commands/verify/check.md +49 -49
  252. package/.claude/commands/verify/start.md +127 -127
  253. package/.claude/commands/workflows/README.md +9 -9
  254. package/.claude/commands/workflows/development.md +77 -77
  255. package/.claude/commands/workflows/research.md +62 -62
  256. package/.claude/commands/workflows/workflow-create.md +25 -25
  257. package/.claude/commands/workflows/workflow-execute.md +25 -25
  258. package/.claude/commands/workflows/workflow-export.md +25 -25
  259. package/.claude/eval/human-relevance-frozen-v1.json +17 -17
  260. package/.claude/evolve-proof/generation-0.json +211 -211
  261. package/.claude/evolve-proof/real-generation-0.json +406 -406
  262. package/.claude/evolve-proof/real-generation-1.json +406 -406
  263. package/.claude/helpers/.helpers-version +1 -1
  264. package/.claude/helpers/README.md +96 -96
  265. package/.claude/helpers/adr-compliance.sh +186 -186
  266. package/.claude/helpers/auto-commit.sh +178 -178
  267. package/.claude/helpers/auto-memory-hook.mjs +0 -0
  268. package/.claude/helpers/checkpoint-manager.sh +251 -251
  269. package/.claude/helpers/daemon-manager.sh +252 -252
  270. package/.claude/helpers/ddd-tracker.sh +144 -144
  271. package/.claude/helpers/github-safe.js +156 -156
  272. package/.claude/helpers/github-setup.sh +45 -45
  273. package/.claude/helpers/guidance-hook.sh +13 -13
  274. package/.claude/helpers/guidance-hooks.sh +102 -102
  275. package/.claude/helpers/health-monitor.sh +108 -108
  276. package/.claude/helpers/helpers.manifest.json +2 -2
  277. package/.claude/helpers/hook-handler.cjs +0 -0
  278. package/.claude/helpers/intelligence.cjs +0 -0
  279. package/.claude/helpers/learning-hooks.sh +329 -329
  280. package/.claude/helpers/learning-optimizer.sh +127 -127
  281. package/.claude/helpers/learning-service.mjs +1144 -1144
  282. package/.claude/helpers/memory.js +83 -83
  283. package/.claude/helpers/metrics-db.mjs +503 -503
  284. package/.claude/helpers/pattern-consolidator.sh +86 -86
  285. package/.claude/helpers/perf-worker.sh +160 -160
  286. package/.claude/helpers/post-commit +16 -16
  287. package/.claude/helpers/pre-commit +26 -26
  288. package/.claude/helpers/quick-start.sh +19 -19
  289. package/.claude/helpers/router.js +105 -105
  290. package/.claude/helpers/security-scanner.sh +127 -127
  291. package/.claude/helpers/session.js +157 -157
  292. package/.claude/helpers/setup-mcp.sh +18 -18
  293. package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
  294. package/.claude/helpers/statusline-hook.sh +21 -21
  295. package/.claude/helpers/statusline.cjs +0 -0
  296. package/.claude/helpers/statusline.js +340 -340
  297. package/.claude/helpers/swarm-comms.sh +353 -353
  298. package/.claude/helpers/swarm-hooks.sh +761 -761
  299. package/.claude/helpers/swarm-monitor.sh +210 -210
  300. package/.claude/helpers/sync-v3-metrics.sh +245 -245
  301. package/.claude/helpers/update-v3-progress.sh +165 -165
  302. package/.claude/helpers/v3-quick-status.sh +57 -57
  303. package/.claude/helpers/v3.sh +110 -110
  304. package/.claude/helpers/validate-v3-config.sh +215 -215
  305. package/.claude/helpers/worker-manager.sh +170 -170
  306. package/.claude/proven-config.json +42 -0
  307. package/.claude/proven-config.manifest.json +37 -37
  308. package/.claude/proven-config.signed.json +41 -41
  309. package/.claude/settings.json +182 -182
  310. package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
  311. package/.claude/skills/agentdb-learning/SKILL.md +545 -545
  312. package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
  313. package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
  314. package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
  315. package/.claude/skills/browser/SKILL.md +204 -204
  316. package/.claude/skills/dual-mode/README.md +71 -71
  317. package/.claude/skills/dual-mode/dual-collect.md +103 -103
  318. package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
  319. package/.claude/skills/dual-mode/dual-spawn.md +81 -81
  320. package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
  321. package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
  322. package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
  323. package/.claude/skills/github-code-review/SKILL.md +1125 -1125
  324. package/.claude/skills/github-multi-repo/SKILL.md +862 -862
  325. package/.claude/skills/github-project-management/SKILL.md +1262 -1262
  326. package/.claude/skills/github-release-management/SKILL.md +1064 -1064
  327. package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
  328. package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
  329. package/.claude/skills/pair-programming/SKILL.md +1202 -1202
  330. package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
  331. package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
  332. package/.claude/skills/skill-builder/SKILL.md +910 -910
  333. package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
  334. package/.claude/skills/stream-chain/SKILL.md +560 -560
  335. package/.claude/skills/swarm-advanced/SKILL.md +970 -970
  336. package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
  337. package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
  338. package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
  339. package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
  340. package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
  341. package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
  342. package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
  343. package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
  344. package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
  345. package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
  346. package/.claude/skills/verification-quality/SKILL.md +691 -691
  347. package/README.md +419 -419
  348. package/bin/cli.js +314 -314
  349. package/bin/mcp-server.js +224 -224
  350. package/bin/preinstall.cjs +2 -2
  351. package/catalog-manifest.json +2 -2
  352. package/dist/src/autopilot-state.js +24 -7
  353. package/dist/src/benchmarks/gaia-critic.js +24 -24
  354. package/dist/src/business-pods/bbs-budget-tracker.js +53 -53
  355. package/dist/src/commands/completions.js +409 -409
  356. package/dist/src/commands/daemon.js +44 -44
  357. package/dist/src/commands/embeddings.js +26 -26
  358. package/dist/src/commands/hive-mind.js +97 -97
  359. package/dist/src/commands/hooks.js +31 -10
  360. package/dist/src/commands/init.js +202 -34
  361. package/dist/src/commands/memory.js +12 -1
  362. package/dist/src/commands/ruvector/backup.js +23 -23
  363. package/dist/src/commands/ruvector/benchmark.js +31 -31
  364. package/dist/src/commands/ruvector/import.js +14 -14
  365. package/dist/src/commands/ruvector/init.js +115 -115
  366. package/dist/src/commands/ruvector/migrate.js +99 -99
  367. package/dist/src/commands/ruvector/optimize.js +51 -51
  368. package/dist/src/commands/ruvector/setup.js +624 -624
  369. package/dist/src/commands/ruvector/status.js +38 -38
  370. package/dist/src/config/proven-config.js +2 -2
  371. package/dist/src/funnel/disclosure.js +13 -2
  372. package/dist/src/funnel/messages.d.ts +12 -10
  373. package/dist/src/funnel/messages.js +83 -11
  374. package/dist/src/init/claudemd-generator.js +231 -231
  375. package/dist/src/init/executor.js +453 -453
  376. package/dist/src/init/helper-signing.js +2 -2
  377. package/dist/src/init/helpers-generator.js +751 -751
  378. package/dist/src/init/statusline-generator.js +24 -24
  379. package/dist/src/mcp-tools/agentdb-tools.js +15 -15
  380. package/dist/src/mcp-tools/browser-intent-tools.js +19 -19
  381. package/dist/src/mcp-tools/browser-tools.js +8 -0
  382. package/dist/src/mcp-tools/hooks-tools.js +21 -0
  383. package/dist/src/mcp-tools/memory-tools.js +4 -3
  384. package/dist/src/memory/graph-edge-writer.js +22 -22
  385. package/dist/src/memory/memory-bridge.js +248 -158
  386. package/dist/src/memory/memory-initializer.js +407 -407
  387. package/dist/src/memory/rabitq-index.js +5 -5
  388. package/dist/src/parser.js +25 -9
  389. package/dist/src/proxy/verify.js +2 -2
  390. package/dist/src/runtime/headless.js +28 -28
  391. package/dist/src/services/distill-tuning.js +7 -7
  392. package/dist/src/services/headless-worker-executor.js +84 -84
  393. package/dist/src/services/memory-distillation.js +4 -4
  394. package/dist/src/services/worker-daemon.js +7 -4
  395. package/dist/src/transfer/deploy-seraphine.js +23 -23
  396. package/package.json +137 -137
  397. package/plugins/ruflo-metaharness/.claude-plugin/plugin.json +32 -32
  398. package/plugins/ruflo-metaharness/README.md +72 -72
  399. package/plugins/ruflo-metaharness/agents/metaharness-architect.md +58 -58
  400. package/plugins/ruflo-metaharness/commands/ruflo-metaharness.md +48 -48
  401. package/plugins/ruflo-metaharness/scripts/_darwin.mjs +210 -210
  402. package/plugins/ruflo-metaharness/scripts/_harness.mjs +330 -330
  403. package/plugins/ruflo-metaharness/scripts/_invoke.mjs +231 -231
  404. package/plugins/ruflo-metaharness/scripts/_redblue.mjs +143 -143
  405. package/plugins/ruflo-metaharness/scripts/_similarity.mjs +161 -161
  406. package/plugins/ruflo-metaharness/scripts/_spike-similarity.mjs +223 -223
  407. package/plugins/ruflo-metaharness/scripts/audit-list.mjs +158 -158
  408. package/plugins/ruflo-metaharness/scripts/audit-trend.mjs +272 -272
  409. package/plugins/ruflo-metaharness/scripts/bench-parse-mcp-scan.mjs +146 -146
  410. package/plugins/ruflo-metaharness/scripts/bench-recordpair-overhead.mjs +186 -186
  411. package/plugins/ruflo-metaharness/scripts/bench-similarity.mjs +177 -177
  412. package/plugins/ruflo-metaharness/scripts/bench.mjs +95 -95
  413. package/plugins/ruflo-metaharness/scripts/drift-from-history.mjs +363 -363
  414. package/plugins/ruflo-metaharness/scripts/evolve.mjs +404 -404
  415. package/plugins/ruflo-metaharness/scripts/genome.mjs +80 -80
  416. package/plugins/ruflo-metaharness/scripts/gepa.mjs +153 -153
  417. package/plugins/ruflo-metaharness/scripts/learn.mjs +127 -127
  418. package/plugins/ruflo-metaharness/scripts/mcp-scan.mjs +111 -111
  419. package/plugins/ruflo-metaharness/scripts/mint.mjs +126 -126
  420. package/plugins/ruflo-metaharness/scripts/oia-audit.mjs +228 -228
  421. package/plugins/ruflo-metaharness/scripts/redblue.mjs +286 -286
  422. package/plugins/ruflo-metaharness/scripts/router-parallel-analyze.mjs +250 -250
  423. package/plugins/ruflo-metaharness/scripts/score.mjs +92 -92
  424. package/plugins/ruflo-metaharness/scripts/security-bench.mjs +174 -174
  425. package/plugins/ruflo-metaharness/scripts/similarity.mjs +158 -158
  426. package/plugins/ruflo-metaharness/scripts/smoke.sh +2356 -2356
  427. package/plugins/ruflo-metaharness/scripts/test-graceful-degradation.mjs +165 -165
  428. package/plugins/ruflo-metaharness/scripts/test-mcp-tools.mjs +472 -472
  429. package/plugins/ruflo-metaharness/scripts/test-parallel-pipeline.mjs +204 -204
  430. package/plugins/ruflo-metaharness/scripts/test-pipeline-roundtrip.mjs +586 -586
  431. package/plugins/ruflo-metaharness/scripts/test-similarity.mjs +334 -334
  432. package/plugins/ruflo-metaharness/scripts/test-with-openrouter.mjs +229 -229
  433. package/plugins/ruflo-metaharness/scripts/threat-model.mjs +59 -59
  434. package/plugins/ruflo-metaharness/skills/harness-bench/SKILL.md +64 -64
  435. package/plugins/ruflo-metaharness/skills/harness-drift-from-history/SKILL.md +65 -65
  436. package/plugins/ruflo-metaharness/skills/harness-evolve/SKILL.md +131 -131
  437. package/plugins/ruflo-metaharness/skills/harness-genome/SKILL.md +54 -54
  438. package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +65 -65
  439. package/plugins/ruflo-metaharness/skills/harness-learn/SKILL.md +65 -65
  440. package/plugins/ruflo-metaharness/skills/harness-mcp-scan/SKILL.md +49 -49
  441. package/plugins/ruflo-metaharness/skills/harness-mint/SKILL.md +72 -72
  442. package/plugins/ruflo-metaharness/skills/harness-oia-audit/SKILL.md +79 -79
  443. package/plugins/ruflo-metaharness/skills/harness-score/SKILL.md +66 -66
  444. package/plugins/ruflo-metaharness/skills/harness-security-bench/SKILL.md +101 -101
  445. package/plugins/ruflo-metaharness/skills/harness-similarity/SKILL.md +67 -67
  446. package/plugins/ruflo-metaharness/skills/harness-threat-model/SKILL.md +41 -41
  447. package/scripts/postinstall.cjs +153 -153
@@ -1,472 +1,472 @@
1
- #!/usr/bin/env node
2
- // test-mcp-tools.mjs — runtime test for the iter-20/21 MCP tool registry.
3
- //
4
- // tsc proves the metaharness-tools.ts module COMPILES; structural smoke
5
- // proves the source DECLARES the right tool names. Neither proves the
6
- // HANDLERS actually run without throwing. This test imports the compiled
7
- // module and invokes every tool's handler with minimal inputs.
8
- //
9
- // CONTRACT EACH TOOL MUST SATISFY
10
- // - handler is callable as `await tool.handler({ ... })`
11
- // - returns an object with keys: success, data, degraded, exitCode
12
- // - never throws (even with bad/missing optional dep — graceful)
13
- // - handler honors the 120s subprocess timeout (no hang)
14
- //
15
- // USAGE
16
- // node scripts/test-mcp-tools.mjs # default
17
- // node scripts/test-mcp-tools.mjs --format json
18
- //
19
- // EXIT CODES
20
- // 0 all tools satisfy the contract
21
- // 1 at least one tool failed
22
- // 2 setup error (compiled dist not present)
23
-
24
- import { existsSync } from 'node:fs';
25
- import { dirname, join, resolve } from 'node:path';
26
- import { fileURLToPath } from 'node:url';
27
-
28
- const SCRIPTS_DIR = dirname(fileURLToPath(import.meta.url));
29
- const ARGS = (() => {
30
- const a = { format: 'table' };
31
- for (let i = 2; i < process.argv.length; i++) {
32
- if (process.argv[i] === '--format') a.format = process.argv[++i];
33
- }
34
- return a;
35
- })();
36
-
37
- let passed = 0, failed = 0;
38
- const failures = [];
39
-
40
- function assert(cond, label) {
41
- if (cond) { console.log(` ✓ ${label}`); passed++; }
42
- else { console.log(` ✗ ${label}`); failures.push(label); failed++; }
43
- }
44
-
45
- async function main() {
46
- // Locate the compiled dist of metaharness-tools.
47
- const distPath = resolve(SCRIPTS_DIR, '..', '..', '..',
48
- 'v3', '@claude-flow', 'cli', 'dist', 'src', 'mcp-tools', 'metaharness-tools.js');
49
-
50
- if (!existsSync(distPath)) {
51
- console.log(`# test-mcp-tools — SKIPPED`);
52
- console.log('');
53
- console.log(`Compiled dist not present: ${distPath}`);
54
- console.log(`Build the CLI first:`);
55
- console.log(` cd v3/@claude-flow/cli && npm run build`);
56
- console.log('');
57
- console.log(`Exit 0 — this script is meaningfully runnable only post-build.`);
58
- process.exit(0);
59
- }
60
-
61
- let mod;
62
- try {
63
- mod = await import(distPath);
64
- } catch (e) {
65
- console.error(`test-mcp-tools: failed to import ${distPath}: ${e.message}`);
66
- process.exit(2);
67
- }
68
-
69
- const tools = mod.metaharnessTools;
70
- console.log(`# test-mcp-tools — runtime contract\n`);
71
-
72
- // ──────────────────────────────────────────────────────────────────
73
- // PHASE 1 — module exports the right shape
74
- // ──────────────────────────────────────────────────────────────────
75
- console.log('Phase 1 — module shape');
76
- assert(Array.isArray(tools), 'metaharnessTools is an array');
77
- assert(tools.length === 15, `15 tools registered (got ${tools.length})`);
78
-
79
- const expectedNames = new Set([
80
- 'metaharness_score',
81
- 'metaharness_genome',
82
- 'metaharness_mcp_scan',
83
- 'metaharness_threat_model',
84
- 'metaharness_oia_audit',
85
- 'metaharness_audit_list',
86
- 'metaharness_audit_trend',
87
- // iter 36 — ADR-152 §3.1 production
88
- 'metaharness_similarity',
89
- // iter 54 — one-command drift detection (composes audit-list + oia-audit + audit-trend)
90
- 'metaharness_drift_from_history',
91
- // ADR-153 — bench suites + evolve driver + security-focused bench
92
- 'metaharness_bench',
93
- 'metaharness_evolve',
94
- 'metaharness_security_bench',
95
- // @metaharness/redblue@~0.1.4 — adversarial red/blue LLM testing
96
- 'metaharness_redblue',
97
- // metaharness@0.3.0 — upstream ADR-235 GEPA learning run
98
- 'metaharness_learn',
99
- // @metaharness/darwin@0.8.0 — GEPA library surface (genome ops)
100
- 'metaharness_gepa',
101
- ]);
102
- const actualNames = new Set(tools.map((t) => t.name));
103
- for (const name of expectedNames) {
104
- assert(actualNames.has(name), `${name} registered`);
105
- }
106
-
107
- // ──────────────────────────────────────────────────────────────────
108
- // PHASE 2 — every tool has the required MCP shape
109
- // ──────────────────────────────────────────────────────────────────
110
- console.log('\nPhase 2 — per-tool shape');
111
- for (const tool of tools) {
112
- const ok = typeof tool.name === 'string'
113
- && typeof tool.description === 'string'
114
- && typeof tool.category === 'string'
115
- && typeof tool.handler === 'function'
116
- && typeof tool.inputSchema === 'object';
117
- assert(ok, `${tool.name} has {name, description, category, handler, inputSchema}`);
118
- assert(tool.category === 'metaharness', `${tool.name} category === 'metaharness'`);
119
- }
120
-
121
- // ──────────────────────────────────────────────────────────────────
122
- // PHASE 3 — handlers callable + return contract shape
123
- //
124
- // We invoke each handler with minimal valid input. The handlers may
125
- // succeed (if metaharness is installed) or report degraded (if not).
126
- // EITHER way, they must return { success, data, degraded, exitCode }
127
- // without throwing.
128
- // ──────────────────────────────────────────────────────────────────
129
- console.log('\nPhase 3 — handler invocations (allow up to 30s each)');
130
- for (const tool of tools) {
131
- // Construct minimal valid input per tool.
132
- let input = {};
133
- if (tool.name === 'metaharness_audit_trend') {
134
- // Requires baselineKey + currentKey — use fake keys that won't
135
- // resolve so we exercise the not-found path.
136
- input = { baselineKey: 'audit-fake-base', currentKey: 'audit-fake-curr' };
137
- }
138
- if (tool.name === 'metaharness_similarity') {
139
- // Needs --a/--b OR --a-key/--b-key. Use fake mem keys to exercise
140
- // the graceful not-found path (matches audit_trend convention).
141
- input = { aKey: 'harness-fake-a', bKey: 'harness-fake-b' };
142
- }
143
- if (tool.name === 'metaharness_drift_from_history') {
144
- // iter 54 — composes 3 subprocesses, needs more time than the default.
145
- input = { dryRun: true, threshold: 0.5 };
146
- }
147
- if (tool.name === 'metaharness_oia_audit') {
148
- // iter 128 — composite audit runs 5 sub-audits (oia-manifest +
149
- // threat-model + mcp-scan + score + genome) in parallel. Each
150
- // shells out via npx. --dry-run skips memory persistence so the
151
- // test doesn't pollute namespaces.
152
- input = { dryRun: true };
153
- }
154
- if (tool.name === 'metaharness_redblue') {
155
- // `attack` preview is the fastest path that exercises the upstream
156
- // binary without needing OPENROUTER_API_KEY or running any model
157
- // calls. Count=1 keeps cold-cache npx fetch the dominant cost.
158
- input = { subcommand: 'attack', family: 'prompt', count: 1 };
159
- }
160
- if (tool.name === 'metaharness_learn') {
161
- // No repo checkout in CI → structured {status:"checkout-required"}
162
- // exit-0 path. $0: without run=true upstream never spends anyway.
163
- input = {};
164
- }
165
- if (tool.name === 'metaharness_gepa') {
166
- // op is required; `genome` loads + validates the SHIPPED cand-6
167
- // genome — pure-local library call once darwin is cached.
168
- input = { op: 'genome' };
169
- }
170
-
171
- // iter 124 → 130 — timeouts have crept up as CI cold-cache npx
172
- // warmup costs got measured. Final budgets:
173
- // default : 60s
174
- // chain-tools : 180s (drift_from_history + oia_audit + audit_list)
175
- // iter 131 — bumped chain-tool budget 90s → 180s. audit_list still
176
- // timed out at 90s in CI; locally it runs in ~4s, but CI's
177
- // `npx @claude-flow/cli@latest memory list` invocation pays both
178
- // the npx fetch AND a full CLI startup (which loads agentic-flow +
179
- // ONNX). 180s gives 30x headroom over the local cost.
180
- const isChainTool = tool.name === 'metaharness_drift_from_history'
181
- || tool.name === 'metaharness_oia_audit'
182
- || tool.name === 'metaharness_audit_list'
183
- // redblue: `attack prompt --count 1` is preview-only (no model
184
- // calls) but the cold-cache `npx -y @metaharness/redblue@~0.1.4`
185
- // fetch can take 30-60s. 180s gives 3x headroom.
186
- || tool.name === 'metaharness_redblue'
187
- // learn: cold-cache `npx -y metaharness@latest` fetch dominates.
188
- // gepa: one-time `npm install --prefix ~/.ruflo/darwin-cache-*`
189
- // fallback install can take 30-60s on cold cache.
190
- || tool.name === 'metaharness_learn'
191
- || tool.name === 'metaharness_gepa';
192
- const timeoutMs = isChainTool ? 180_000 : 60_000;
193
- const handlerPromise = tool.handler(input);
194
- const timeoutPromise = new Promise((_, reject) =>
195
- setTimeout(() => reject(new Error(`${timeoutMs / 1000}s handler timeout`)), timeoutMs));
196
-
197
- let result;
198
- let threw = false;
199
- try {
200
- result = await Promise.race([handlerPromise, timeoutPromise]);
201
- } catch (e) {
202
- threw = true;
203
- console.log(` [${tool.name}] handler threw: ${e.message.slice(0, 80)}`);
204
- }
205
-
206
- assert(!threw, `${tool.name} handler did not throw`);
207
- if (!threw && result) {
208
- assert(typeof result === 'object', `${tool.name} returns object`);
209
- assert('success' in result, `${tool.name} result has 'success'`);
210
- assert('data' in result, `${tool.name} result has 'data'`);
211
- assert('degraded' in result, `${tool.name} result has 'degraded'`);
212
- assert('exitCode' in result, `${tool.name} result has 'exitCode'`);
213
- }
214
- }
215
-
216
- // ──────────────────────────────────────────────────────────────────
217
- // PHASE 4 — POSITIVE-CASE data-shape validation (iter 43)
218
- //
219
- // Iter 37 verified the {success, data, degraded, exitCode} envelope.
220
- // It did NOT verify that data.X contains the right keys when success
221
- // is genuinely true — leaving room for iter 42-style bugs where a
222
- // handler returns valid-looking degraded JSON while silently
223
- // misrouting input. This phase invokes each handler with VALID
224
- // inputs and asserts the expected output shape.
225
- //
226
- // Tools that depend on `npx metaharness` (score/genome/mcp-scan/
227
- // threat-model/oia-audit/audit-list/audit-trend) are SKIPPED in this
228
- // phase when the optional dep isn't installed — they're covered by
229
- // the no-metaharness-smoke workflow's drill. The similarity tool
230
- // has no @metaharness/* dep, so its positive case ALWAYS runs.
231
- // ──────────────────────────────────────────────────────────────────
232
- console.log('\nPhase 4 — positive-case data shape (iter 43)');
233
-
234
- const { writeFileSync, mkdtempSync } = await import('node:fs');
235
- const { tmpdir } = await import('node:os');
236
- const { join: pjoin } = await import('node:path');
237
- const tmp = mkdtempSync(pjoin(tmpdir(), 'mcp-positive-'));
238
-
239
- // metaharness_similarity — full positive case (no @metaharness/* needed)
240
- const simTool = tools.find((t) => t.name === 'metaharness_similarity');
241
- if (simTool) {
242
- const aPath = pjoin(tmp, 'a.json');
243
- const bPath = pjoin(tmp, 'b.json');
244
- writeFileSync(aPath, JSON.stringify({
245
- score: { harnessFit: 78, compileConfidence: 92, taskCoverage: 65, toolSafety: 88, memoryUsefulness: 70, estCostPerRunUsd: 0.04, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:legal' },
246
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','w'], risk_score: 0.45, test_confidence: 0.7, publish_readiness: 0.6 },
247
- }));
248
- writeFileSync(bPath, JSON.stringify({
249
- score: { harnessFit: 75, compileConfidence: 90, taskCoverage: 70, toolSafety: 90, memoryUsefulness: 72, estCostPerRunUsd: 0.05, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:support' },
250
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','q','r'], risk_score: 0.40, test_confidence: 0.75, publish_readiness: 0.65 },
251
- }));
252
- const r = await simTool.handler({ aFile: aPath, bFile: bPath });
253
- assert(r.degraded === false, 'similarity positive case: degraded === false');
254
- assert(r.success === true, 'similarity positive case: success === true');
255
- assert(r.exitCode === 0, 'similarity positive case: exitCode === 0');
256
- const d = r.data ?? {};
257
- assert(typeof d.overall === 'number', 'similarity data has numeric `overall`');
258
- assert(typeof d.components === 'object' && d.components !== null,
259
- 'similarity data has `components` object');
260
- assert(typeof d.components?.cosine === 'number',
261
- 'similarity components.cosine numeric');
262
- assert(typeof d.components?.categorical === 'number',
263
- 'similarity components.categorical numeric');
264
- assert(typeof d.components?.jaccard === 'number',
265
- 'similarity components.jaccard numeric');
266
- assert(typeof d.weights === 'object' && d.weights !== null,
267
- 'similarity data has `weights` object');
268
- assert(d.adr === 'ADR-152', 'similarity data tagged adr=ADR-152');
269
- // Regression anchor — same fixtures as iter-35 spike with non-matching topologies
270
- assert(d.overall > 0 && d.overall < 1,
271
- `similarity overall in (0, 1) — got ${d.overall}`);
272
-
273
- // Per-dimension variant
274
- const rPD = await simTool.handler({ aFile: aPath, bFile: bPath, perDimension: true });
275
- assert(typeof rPD.data?.perDimension === 'object',
276
- 'similarity perDimension=true populates breakdown');
277
-
278
- // Alert-below variant exercises non-zero exit
279
- const rAlert = await simTool.handler({ aFile: aPath, bFile: bPath, alertBelow: 0.99 });
280
- assert(rAlert.data?.alert?.triggered === true,
281
- 'similarity alertBelow=0.99 triggers alert');
282
- assert(rAlert.exitCode === 1, 'similarity alertBelow=0.99 → exitCode 1');
283
- // iter 44 — success semantic anchor (was true under the pre-iter-44
284
- // `!degraded` rule; now false because exitCode !== 0 dominates).
285
- assert(rAlert.success === false,
286
- 'similarity alertBelow=0.99 → success === false (iter 44 fix)');
287
- }
288
-
289
- // metaharness_mcp_scan — positive case post iter-50 parser landing.
290
- // Until iter 50, mcp_scan's data field was an alert-only object with
291
- // no structured findings. After iter 50, findings[] is always present
292
- // (parsed from upstream text) and summary{overallSeverity, totalCount}
293
- // accompanies it.
294
- const scanTool = tools.find((t) => t.name === 'metaharness_mcp_scan');
295
- if (scanTool) {
296
- // Run against ruflo itself — guaranteed to produce at least the
297
- // INFO finding the iter-50 parser test verified manually.
298
- const r = await scanTool.handler({ path: '.', failOn: 'high' });
299
- // Either succeeds with structured findings, or gracefully degrades
300
- // if metaharness isn't installed in this environment.
301
- if (!r.degraded) {
302
- assert(r.success === true, 'mcp_scan positive: success === true');
303
- assert(r.exitCode === 0, 'mcp_scan positive: exitCode === 0');
304
- assert(Array.isArray(r.data?.findings),
305
- 'mcp_scan positive: data.findings is an array (iter 50 fix)');
306
- // Cwd-dependent: when scanning a dir without .mcp/servers.json the
307
- // upstream emits no findings. Only verify shape contract when array
308
- // is populated — the array-presence assertion above is the
309
- // load-bearing one for iter 50.
310
- if (r.data?.findings.length > 0) {
311
- const first = r.data.findings[0];
312
- assert(typeof first?.severity === 'string',
313
- 'mcp_scan positive: first finding has string severity');
314
- assert(typeof first?.message === 'string',
315
- 'mcp_scan positive: first finding has string message');
316
- }
317
- // summary may be null if the upstream produced no Result: line —
318
- // verify the field's presence (null OR object) but only deep-check
319
- // when populated.
320
- if (r.data?.summary) {
321
- assert(typeof r.data.summary.totalCount === 'number',
322
- 'mcp_scan positive: data.summary.totalCount is numeric (when summary present)');
323
- }
324
- } else {
325
- console.log(` ⊘ mcp_scan: metaharness absent — graceful skip`);
326
- }
327
- }
328
-
329
- // metaharness_audit_trend — positive case via file inputs
330
- const trendTool = tools.find((t) => t.name === 'metaharness_audit_trend');
331
- if (trendTool) {
332
- const basePath = pjoin(tmp, 'base.json');
333
- const currPath = pjoin(tmp, 'curr.json');
334
- const fingerprint = {
335
- score: { harnessFit: 80, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
336
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['a','b','c'], risk_score: 0.3, test_confidence: 0.85, publish_readiness: 0.9 },
337
- };
338
- writeFileSync(basePath, JSON.stringify({
339
- startedAt: '2026-06-15T00:00:00Z',
340
- composite: { worst: 'clean' },
341
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
342
- fingerprint,
343
- }));
344
- writeFileSync(currPath, JSON.stringify({
345
- startedAt: '2026-06-16T00:00:00Z',
346
- composite: { worst: 'clean' },
347
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
348
- fingerprint,
349
- }));
350
- // audit_trend tool only supports keys, not files at the MCP layer.
351
- // Document its actual wrapper semantics so future-us doesn't get
352
- // surprised:
353
- // - bad keys → script exits 2 with stderr (no JSON payload)
354
- // - runScript() can't parse a {degraded:true} marker, so it
355
- // returns degraded:false / success:true / exitCode:2
356
- // This is a real wrapper bug (success should not be true when
357
- // exit!=0 AND no JSON came back), tracked separately. Asserting
358
- // current behavior here protects against silent semantic shifts.
359
- // iter 46 — file-input path. audit_trend now accepts baselineFile/currentFile.
360
- const rFiles = await trendTool.handler({ baselineFile: basePath, currentFile: currPath });
361
- assert(rFiles.success === true,
362
- 'audit_trend file-input path: success === true (iter 46)');
363
- assert(rFiles.exitCode === 0, 'audit_trend file-input path: exitCode === 0');
364
- assert(typeof rFiles.data?.delta === 'object',
365
- 'audit_trend file-input path: data.delta object present');
366
- assert(rFiles.data?.delta?.structuralDistance?.verdict === 'near-identical',
367
- `audit_trend file-input path: identical fingerprints → near-identical (got ${rFiles.data?.delta?.structuralDistance?.verdict})`);
368
-
369
- // iter 54 — metaharness_drift_from_history positive case
370
- const driftTool = tools.find((t) => t.name === 'metaharness_drift_from_history');
371
- if (driftTool) {
372
- // iter 71 — verify iter-66/67 fast-path flags are now MCP-callable
373
- // Synthesize a baseline file on disk; pass via the new baselineFile input.
374
- const baselinePath = pjoin(tmp, 'drift-baseline.json');
375
- writeFileSync(baselinePath, JSON.stringify({
376
- startedAt: '2026-06-16T00:00:00Z',
377
- composite: { worst: 'clean' },
378
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
379
- fingerprint: {
380
- score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
381
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
382
- },
383
- }));
384
- const rFastFast = await driftTool.handler({
385
- path: '.', dryRun: true, threshold: 0.5, baselineFile: baselinePath,
386
- });
387
- if (!rFastFast.degraded) {
388
- assert(rFastFast.data?.timing?.usedBaselineFile === true,
389
- 'drift_from_history MCP-layer: baselineFile fastpath fires (iter 71)');
390
- assert(rFastFast.data?.timing?.skippedAuditList === true,
391
- 'drift_from_history MCP-layer: skippedAuditList=true via baselineFile (iter 71)');
392
- }
393
-
394
- // iter 85 — verify iter-78's alertOnNewSeverity MCP input plumbs
395
- // through. baselineFile has no findings; current ruflo audit has
396
- // 1 INFO finding. With alertOnNewSeverity='info' the gate fires
397
- // and surfaces in the response.
398
- const baselineNoFindings = pjoin(tmp, 'drift-baseline-no-findings.json');
399
- writeFileSync(baselineNoFindings, JSON.stringify({
400
- startedAt: '2026-06-16T00:00:00Z',
401
- composite: { worst: 'clean' },
402
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
403
- fingerprint: {
404
- score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
405
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
406
- },
407
- }));
408
- const rSevAlert = await driftTool.handler({
409
- path: '.', dryRun: true, threshold: 0.5,
410
- baselineFile: baselineNoFindings,
411
- alertOnNewSeverity: 'info',
412
- });
413
- if (!rSevAlert.degraded) {
414
- assert(rSevAlert.data?.alert?.newSeverityThreshold === 'info',
415
- 'drift_from_history MCP-layer: alertOnNewSeverity echoed in payload (iter 85)');
416
- // Triggered AND exit code reflects (only if the audit actually had findings)
417
- if (rSevAlert.data?.alert?.triggered === true) {
418
- assert(rSevAlert.exitCode === 1,
419
- `drift_from_history MCP-layer: alertOnNewSeverity exitCode=1 when triggered (got ${rSevAlert.exitCode})`);
420
- assert(rSevAlert.success === false,
421
- 'drift_from_history MCP-layer: success===false when alert fires (iter 44 fix)');
422
- }
423
- }
424
-
425
- const r54 = await driftTool.handler({ path: '.', dryRun: true, threshold: 0.5 });
426
- if (!r54.degraded) {
427
- assert(typeof r54.data === 'object' && r54.data !== null,
428
- 'drift_from_history positive: data is an object');
429
- // Either it produced the structured drift report OR the no-history error
430
- const isOk = r54.data?.command === 'drift-from-history';
431
- const isNoHistory = typeof r54.data?.error === 'string' && r54.data.error.includes('no audit records');
432
- assert(isOk || isNoHistory,
433
- `drift_from_history positive: structured report OR no-history error (got ${JSON.stringify(r54.data).slice(0,80)})`);
434
- if (isOk) {
435
- assert(typeof r54.data.baseline?.key === 'string',
436
- 'drift_from_history: baseline.key is a string');
437
- assert(typeof r54.data.alert?.threshold === 'number',
438
- 'drift_from_history: alert.threshold echoed numerically');
439
- }
440
- } else {
441
- console.log(` ⊘ drift_from_history: degraded (metaharness or memory absent)`);
442
- }
443
- }
444
-
445
- const r = await trendTool.handler({ baselineKey: 'missing-X', currentKey: 'missing-Y' });
446
- assert(r.exitCode === 2,
447
- 'audit_trend bad-keys path exits 2 (script-level guard fires)');
448
- assert(r.data === null || r.data === undefined,
449
- 'audit_trend bad-keys path: data null (no JSON emitted on stderr exit)');
450
- // iter 44 — success semantic anchor. Pre-iter-44 wrapper returned
451
- // success:true for this case (because no degraded marker). Now
452
- // returns false because exitCode !== 0.
453
- assert(r.success === false,
454
- 'audit_trend bad-keys path: success === false (iter 44 fix)');
455
- }
456
-
457
- // Cleanup
458
- try { (await import('node:fs')).rmSync(tmp, { recursive: true, force: true }); } catch { /* ignore */ }
459
-
460
- console.log(`\n${passed} passed, ${failed} failed`);
461
- if (failed > 0) {
462
- console.log('\nFailures:');
463
- for (const f of failures) console.log(` - ${f}`);
464
- process.exit(1);
465
- }
466
- console.log('\n✓ All 15 MCP tools satisfy the runtime contract.');
467
- }
468
-
469
- main().catch((e) => {
470
- console.error('test-mcp-tools crashed:', e.message || e);
471
- process.exit(2);
472
- });
1
+ #!/usr/bin/env node
2
+ // test-mcp-tools.mjs — runtime test for the iter-20/21 MCP tool registry.
3
+ //
4
+ // tsc proves the metaharness-tools.ts module COMPILES; structural smoke
5
+ // proves the source DECLARES the right tool names. Neither proves the
6
+ // HANDLERS actually run without throwing. This test imports the compiled
7
+ // module and invokes every tool's handler with minimal inputs.
8
+ //
9
+ // CONTRACT EACH TOOL MUST SATISFY
10
+ // - handler is callable as `await tool.handler({ ... })`
11
+ // - returns an object with keys: success, data, degraded, exitCode
12
+ // - never throws (even with bad/missing optional dep — graceful)
13
+ // - handler honors the 120s subprocess timeout (no hang)
14
+ //
15
+ // USAGE
16
+ // node scripts/test-mcp-tools.mjs # default
17
+ // node scripts/test-mcp-tools.mjs --format json
18
+ //
19
+ // EXIT CODES
20
+ // 0 all tools satisfy the contract
21
+ // 1 at least one tool failed
22
+ // 2 setup error (compiled dist not present)
23
+
24
+ import { existsSync } from 'node:fs';
25
+ import { dirname, join, resolve } from 'node:path';
26
+ import { fileURLToPath } from 'node:url';
27
+
28
+ const SCRIPTS_DIR = dirname(fileURLToPath(import.meta.url));
29
+ const ARGS = (() => {
30
+ const a = { format: 'table' };
31
+ for (let i = 2; i < process.argv.length; i++) {
32
+ if (process.argv[i] === '--format') a.format = process.argv[++i];
33
+ }
34
+ return a;
35
+ })();
36
+
37
+ let passed = 0, failed = 0;
38
+ const failures = [];
39
+
40
+ function assert(cond, label) {
41
+ if (cond) { console.log(` ✓ ${label}`); passed++; }
42
+ else { console.log(` ✗ ${label}`); failures.push(label); failed++; }
43
+ }
44
+
45
+ async function main() {
46
+ // Locate the compiled dist of metaharness-tools.
47
+ const distPath = resolve(SCRIPTS_DIR, '..', '..', '..',
48
+ 'v3', '@claude-flow', 'cli', 'dist', 'src', 'mcp-tools', 'metaharness-tools.js');
49
+
50
+ if (!existsSync(distPath)) {
51
+ console.log(`# test-mcp-tools — SKIPPED`);
52
+ console.log('');
53
+ console.log(`Compiled dist not present: ${distPath}`);
54
+ console.log(`Build the CLI first:`);
55
+ console.log(` cd v3/@claude-flow/cli && npm run build`);
56
+ console.log('');
57
+ console.log(`Exit 0 — this script is meaningfully runnable only post-build.`);
58
+ process.exit(0);
59
+ }
60
+
61
+ let mod;
62
+ try {
63
+ mod = await import(distPath);
64
+ } catch (e) {
65
+ console.error(`test-mcp-tools: failed to import ${distPath}: ${e.message}`);
66
+ process.exit(2);
67
+ }
68
+
69
+ const tools = mod.metaharnessTools;
70
+ console.log(`# test-mcp-tools — runtime contract\n`);
71
+
72
+ // ──────────────────────────────────────────────────────────────────
73
+ // PHASE 1 — module exports the right shape
74
+ // ──────────────────────────────────────────────────────────────────
75
+ console.log('Phase 1 — module shape');
76
+ assert(Array.isArray(tools), 'metaharnessTools is an array');
77
+ assert(tools.length === 15, `15 tools registered (got ${tools.length})`);
78
+
79
+ const expectedNames = new Set([
80
+ 'metaharness_score',
81
+ 'metaharness_genome',
82
+ 'metaharness_mcp_scan',
83
+ 'metaharness_threat_model',
84
+ 'metaharness_oia_audit',
85
+ 'metaharness_audit_list',
86
+ 'metaharness_audit_trend',
87
+ // iter 36 — ADR-152 §3.1 production
88
+ 'metaharness_similarity',
89
+ // iter 54 — one-command drift detection (composes audit-list + oia-audit + audit-trend)
90
+ 'metaharness_drift_from_history',
91
+ // ADR-153 — bench suites + evolve driver + security-focused bench
92
+ 'metaharness_bench',
93
+ 'metaharness_evolve',
94
+ 'metaharness_security_bench',
95
+ // @metaharness/redblue@~0.1.4 — adversarial red/blue LLM testing
96
+ 'metaharness_redblue',
97
+ // metaharness@0.3.0 — upstream ADR-235 GEPA learning run
98
+ 'metaharness_learn',
99
+ // @metaharness/darwin@0.8.0 — GEPA library surface (genome ops)
100
+ 'metaharness_gepa',
101
+ ]);
102
+ const actualNames = new Set(tools.map((t) => t.name));
103
+ for (const name of expectedNames) {
104
+ assert(actualNames.has(name), `${name} registered`);
105
+ }
106
+
107
+ // ──────────────────────────────────────────────────────────────────
108
+ // PHASE 2 — every tool has the required MCP shape
109
+ // ──────────────────────────────────────────────────────────────────
110
+ console.log('\nPhase 2 — per-tool shape');
111
+ for (const tool of tools) {
112
+ const ok = typeof tool.name === 'string'
113
+ && typeof tool.description === 'string'
114
+ && typeof tool.category === 'string'
115
+ && typeof tool.handler === 'function'
116
+ && typeof tool.inputSchema === 'object';
117
+ assert(ok, `${tool.name} has {name, description, category, handler, inputSchema}`);
118
+ assert(tool.category === 'metaharness', `${tool.name} category === 'metaharness'`);
119
+ }
120
+
121
+ // ──────────────────────────────────────────────────────────────────
122
+ // PHASE 3 — handlers callable + return contract shape
123
+ //
124
+ // We invoke each handler with minimal valid input. The handlers may
125
+ // succeed (if metaharness is installed) or report degraded (if not).
126
+ // EITHER way, they must return { success, data, degraded, exitCode }
127
+ // without throwing.
128
+ // ──────────────────────────────────────────────────────────────────
129
+ console.log('\nPhase 3 — handler invocations (allow up to 30s each)');
130
+ for (const tool of tools) {
131
+ // Construct minimal valid input per tool.
132
+ let input = {};
133
+ if (tool.name === 'metaharness_audit_trend') {
134
+ // Requires baselineKey + currentKey — use fake keys that won't
135
+ // resolve so we exercise the not-found path.
136
+ input = { baselineKey: 'audit-fake-base', currentKey: 'audit-fake-curr' };
137
+ }
138
+ if (tool.name === 'metaharness_similarity') {
139
+ // Needs --a/--b OR --a-key/--b-key. Use fake mem keys to exercise
140
+ // the graceful not-found path (matches audit_trend convention).
141
+ input = { aKey: 'harness-fake-a', bKey: 'harness-fake-b' };
142
+ }
143
+ if (tool.name === 'metaharness_drift_from_history') {
144
+ // iter 54 — composes 3 subprocesses, needs more time than the default.
145
+ input = { dryRun: true, threshold: 0.5 };
146
+ }
147
+ if (tool.name === 'metaharness_oia_audit') {
148
+ // iter 128 — composite audit runs 5 sub-audits (oia-manifest +
149
+ // threat-model + mcp-scan + score + genome) in parallel. Each
150
+ // shells out via npx. --dry-run skips memory persistence so the
151
+ // test doesn't pollute namespaces.
152
+ input = { dryRun: true };
153
+ }
154
+ if (tool.name === 'metaharness_redblue') {
155
+ // `attack` preview is the fastest path that exercises the upstream
156
+ // binary without needing OPENROUTER_API_KEY or running any model
157
+ // calls. Count=1 keeps cold-cache npx fetch the dominant cost.
158
+ input = { subcommand: 'attack', family: 'prompt', count: 1 };
159
+ }
160
+ if (tool.name === 'metaharness_learn') {
161
+ // No repo checkout in CI → structured {status:"checkout-required"}
162
+ // exit-0 path. $0: without run=true upstream never spends anyway.
163
+ input = {};
164
+ }
165
+ if (tool.name === 'metaharness_gepa') {
166
+ // op is required; `genome` loads + validates the SHIPPED cand-6
167
+ // genome — pure-local library call once darwin is cached.
168
+ input = { op: 'genome' };
169
+ }
170
+
171
+ // iter 124 → 130 — timeouts have crept up as CI cold-cache npx
172
+ // warmup costs got measured. Final budgets:
173
+ // default : 60s
174
+ // chain-tools : 180s (drift_from_history + oia_audit + audit_list)
175
+ // iter 131 — bumped chain-tool budget 90s → 180s. audit_list still
176
+ // timed out at 90s in CI; locally it runs in ~4s, but CI's
177
+ // `npx @claude-flow/cli@latest memory list` invocation pays both
178
+ // the npx fetch AND a full CLI startup (which loads agentic-flow +
179
+ // ONNX). 180s gives 30x headroom over the local cost.
180
+ const isChainTool = tool.name === 'metaharness_drift_from_history'
181
+ || tool.name === 'metaharness_oia_audit'
182
+ || tool.name === 'metaharness_audit_list'
183
+ // redblue: `attack prompt --count 1` is preview-only (no model
184
+ // calls) but the cold-cache `npx -y @metaharness/redblue@~0.1.4`
185
+ // fetch can take 30-60s. 180s gives 3x headroom.
186
+ || tool.name === 'metaharness_redblue'
187
+ // learn: cold-cache `npx -y metaharness@latest` fetch dominates.
188
+ // gepa: one-time `npm install --prefix ~/.ruflo/darwin-cache-*`
189
+ // fallback install can take 30-60s on cold cache.
190
+ || tool.name === 'metaharness_learn'
191
+ || tool.name === 'metaharness_gepa';
192
+ const timeoutMs = isChainTool ? 180_000 : 60_000;
193
+ const handlerPromise = tool.handler(input);
194
+ const timeoutPromise = new Promise((_, reject) =>
195
+ setTimeout(() => reject(new Error(`${timeoutMs / 1000}s handler timeout`)), timeoutMs));
196
+
197
+ let result;
198
+ let threw = false;
199
+ try {
200
+ result = await Promise.race([handlerPromise, timeoutPromise]);
201
+ } catch (e) {
202
+ threw = true;
203
+ console.log(` [${tool.name}] handler threw: ${e.message.slice(0, 80)}`);
204
+ }
205
+
206
+ assert(!threw, `${tool.name} handler did not throw`);
207
+ if (!threw && result) {
208
+ assert(typeof result === 'object', `${tool.name} returns object`);
209
+ assert('success' in result, `${tool.name} result has 'success'`);
210
+ assert('data' in result, `${tool.name} result has 'data'`);
211
+ assert('degraded' in result, `${tool.name} result has 'degraded'`);
212
+ assert('exitCode' in result, `${tool.name} result has 'exitCode'`);
213
+ }
214
+ }
215
+
216
+ // ──────────────────────────────────────────────────────────────────
217
+ // PHASE 4 — POSITIVE-CASE data-shape validation (iter 43)
218
+ //
219
+ // Iter 37 verified the {success, data, degraded, exitCode} envelope.
220
+ // It did NOT verify that data.X contains the right keys when success
221
+ // is genuinely true — leaving room for iter 42-style bugs where a
222
+ // handler returns valid-looking degraded JSON while silently
223
+ // misrouting input. This phase invokes each handler with VALID
224
+ // inputs and asserts the expected output shape.
225
+ //
226
+ // Tools that depend on `npx metaharness` (score/genome/mcp-scan/
227
+ // threat-model/oia-audit/audit-list/audit-trend) are SKIPPED in this
228
+ // phase when the optional dep isn't installed — they're covered by
229
+ // the no-metaharness-smoke workflow's drill. The similarity tool
230
+ // has no @metaharness/* dep, so its positive case ALWAYS runs.
231
+ // ──────────────────────────────────────────────────────────────────
232
+ console.log('\nPhase 4 — positive-case data shape (iter 43)');
233
+
234
+ const { writeFileSync, mkdtempSync } = await import('node:fs');
235
+ const { tmpdir } = await import('node:os');
236
+ const { join: pjoin } = await import('node:path');
237
+ const tmp = mkdtempSync(pjoin(tmpdir(), 'mcp-positive-'));
238
+
239
+ // metaharness_similarity — full positive case (no @metaharness/* needed)
240
+ const simTool = tools.find((t) => t.name === 'metaharness_similarity');
241
+ if (simTool) {
242
+ const aPath = pjoin(tmp, 'a.json');
243
+ const bPath = pjoin(tmp, 'b.json');
244
+ writeFileSync(aPath, JSON.stringify({
245
+ score: { harnessFit: 78, compileConfidence: 92, taskCoverage: 65, toolSafety: 88, memoryUsefulness: 70, estCostPerRunUsd: 0.04, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:legal' },
246
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','w'], risk_score: 0.45, test_confidence: 0.7, publish_readiness: 0.6 },
247
+ }));
248
+ writeFileSync(bPath, JSON.stringify({
249
+ score: { harnessFit: 75, compileConfidence: 90, taskCoverage: 70, toolSafety: 90, memoryUsefulness: 72, estCostPerRunUsd: 0.05, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:support' },
250
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','q','r'], risk_score: 0.40, test_confidence: 0.75, publish_readiness: 0.65 },
251
+ }));
252
+ const r = await simTool.handler({ aFile: aPath, bFile: bPath });
253
+ assert(r.degraded === false, 'similarity positive case: degraded === false');
254
+ assert(r.success === true, 'similarity positive case: success === true');
255
+ assert(r.exitCode === 0, 'similarity positive case: exitCode === 0');
256
+ const d = r.data ?? {};
257
+ assert(typeof d.overall === 'number', 'similarity data has numeric `overall`');
258
+ assert(typeof d.components === 'object' && d.components !== null,
259
+ 'similarity data has `components` object');
260
+ assert(typeof d.components?.cosine === 'number',
261
+ 'similarity components.cosine numeric');
262
+ assert(typeof d.components?.categorical === 'number',
263
+ 'similarity components.categorical numeric');
264
+ assert(typeof d.components?.jaccard === 'number',
265
+ 'similarity components.jaccard numeric');
266
+ assert(typeof d.weights === 'object' && d.weights !== null,
267
+ 'similarity data has `weights` object');
268
+ assert(d.adr === 'ADR-152', 'similarity data tagged adr=ADR-152');
269
+ // Regression anchor — same fixtures as iter-35 spike with non-matching topologies
270
+ assert(d.overall > 0 && d.overall < 1,
271
+ `similarity overall in (0, 1) — got ${d.overall}`);
272
+
273
+ // Per-dimension variant
274
+ const rPD = await simTool.handler({ aFile: aPath, bFile: bPath, perDimension: true });
275
+ assert(typeof rPD.data?.perDimension === 'object',
276
+ 'similarity perDimension=true populates breakdown');
277
+
278
+ // Alert-below variant exercises non-zero exit
279
+ const rAlert = await simTool.handler({ aFile: aPath, bFile: bPath, alertBelow: 0.99 });
280
+ assert(rAlert.data?.alert?.triggered === true,
281
+ 'similarity alertBelow=0.99 triggers alert');
282
+ assert(rAlert.exitCode === 1, 'similarity alertBelow=0.99 → exitCode 1');
283
+ // iter 44 — success semantic anchor (was true under the pre-iter-44
284
+ // `!degraded` rule; now false because exitCode !== 0 dominates).
285
+ assert(rAlert.success === false,
286
+ 'similarity alertBelow=0.99 → success === false (iter 44 fix)');
287
+ }
288
+
289
+ // metaharness_mcp_scan — positive case post iter-50 parser landing.
290
+ // Until iter 50, mcp_scan's data field was an alert-only object with
291
+ // no structured findings. After iter 50, findings[] is always present
292
+ // (parsed from upstream text) and summary{overallSeverity, totalCount}
293
+ // accompanies it.
294
+ const scanTool = tools.find((t) => t.name === 'metaharness_mcp_scan');
295
+ if (scanTool) {
296
+ // Run against ruflo itself — guaranteed to produce at least the
297
+ // INFO finding the iter-50 parser test verified manually.
298
+ const r = await scanTool.handler({ path: '.', failOn: 'high' });
299
+ // Either succeeds with structured findings, or gracefully degrades
300
+ // if metaharness isn't installed in this environment.
301
+ if (!r.degraded) {
302
+ assert(r.success === true, 'mcp_scan positive: success === true');
303
+ assert(r.exitCode === 0, 'mcp_scan positive: exitCode === 0');
304
+ assert(Array.isArray(r.data?.findings),
305
+ 'mcp_scan positive: data.findings is an array (iter 50 fix)');
306
+ // Cwd-dependent: when scanning a dir without .mcp/servers.json the
307
+ // upstream emits no findings. Only verify shape contract when array
308
+ // is populated — the array-presence assertion above is the
309
+ // load-bearing one for iter 50.
310
+ if (r.data?.findings.length > 0) {
311
+ const first = r.data.findings[0];
312
+ assert(typeof first?.severity === 'string',
313
+ 'mcp_scan positive: first finding has string severity');
314
+ assert(typeof first?.message === 'string',
315
+ 'mcp_scan positive: first finding has string message');
316
+ }
317
+ // summary may be null if the upstream produced no Result: line —
318
+ // verify the field's presence (null OR object) but only deep-check
319
+ // when populated.
320
+ if (r.data?.summary) {
321
+ assert(typeof r.data.summary.totalCount === 'number',
322
+ 'mcp_scan positive: data.summary.totalCount is numeric (when summary present)');
323
+ }
324
+ } else {
325
+ console.log(` ⊘ mcp_scan: metaharness absent — graceful skip`);
326
+ }
327
+ }
328
+
329
+ // metaharness_audit_trend — positive case via file inputs
330
+ const trendTool = tools.find((t) => t.name === 'metaharness_audit_trend');
331
+ if (trendTool) {
332
+ const basePath = pjoin(tmp, 'base.json');
333
+ const currPath = pjoin(tmp, 'curr.json');
334
+ const fingerprint = {
335
+ score: { harnessFit: 80, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
336
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['a','b','c'], risk_score: 0.3, test_confidence: 0.85, publish_readiness: 0.9 },
337
+ };
338
+ writeFileSync(basePath, JSON.stringify({
339
+ startedAt: '2026-06-15T00:00:00Z',
340
+ composite: { worst: 'clean' },
341
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
342
+ fingerprint,
343
+ }));
344
+ writeFileSync(currPath, JSON.stringify({
345
+ startedAt: '2026-06-16T00:00:00Z',
346
+ composite: { worst: 'clean' },
347
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
348
+ fingerprint,
349
+ }));
350
+ // audit_trend tool only supports keys, not files at the MCP layer.
351
+ // Document its actual wrapper semantics so future-us doesn't get
352
+ // surprised:
353
+ // - bad keys → script exits 2 with stderr (no JSON payload)
354
+ // - runScript() can't parse a {degraded:true} marker, so it
355
+ // returns degraded:false / success:true / exitCode:2
356
+ // This is a real wrapper bug (success should not be true when
357
+ // exit!=0 AND no JSON came back), tracked separately. Asserting
358
+ // current behavior here protects against silent semantic shifts.
359
+ // iter 46 — file-input path. audit_trend now accepts baselineFile/currentFile.
360
+ const rFiles = await trendTool.handler({ baselineFile: basePath, currentFile: currPath });
361
+ assert(rFiles.success === true,
362
+ 'audit_trend file-input path: success === true (iter 46)');
363
+ assert(rFiles.exitCode === 0, 'audit_trend file-input path: exitCode === 0');
364
+ assert(typeof rFiles.data?.delta === 'object',
365
+ 'audit_trend file-input path: data.delta object present');
366
+ assert(rFiles.data?.delta?.structuralDistance?.verdict === 'near-identical',
367
+ `audit_trend file-input path: identical fingerprints → near-identical (got ${rFiles.data?.delta?.structuralDistance?.verdict})`);
368
+
369
+ // iter 54 — metaharness_drift_from_history positive case
370
+ const driftTool = tools.find((t) => t.name === 'metaharness_drift_from_history');
371
+ if (driftTool) {
372
+ // iter 71 — verify iter-66/67 fast-path flags are now MCP-callable
373
+ // Synthesize a baseline file on disk; pass via the new baselineFile input.
374
+ const baselinePath = pjoin(tmp, 'drift-baseline.json');
375
+ writeFileSync(baselinePath, JSON.stringify({
376
+ startedAt: '2026-06-16T00:00:00Z',
377
+ composite: { worst: 'clean' },
378
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
379
+ fingerprint: {
380
+ score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
381
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
382
+ },
383
+ }));
384
+ const rFastFast = await driftTool.handler({
385
+ path: '.', dryRun: true, threshold: 0.5, baselineFile: baselinePath,
386
+ });
387
+ if (!rFastFast.degraded) {
388
+ assert(rFastFast.data?.timing?.usedBaselineFile === true,
389
+ 'drift_from_history MCP-layer: baselineFile fastpath fires (iter 71)');
390
+ assert(rFastFast.data?.timing?.skippedAuditList === true,
391
+ 'drift_from_history MCP-layer: skippedAuditList=true via baselineFile (iter 71)');
392
+ }
393
+
394
+ // iter 85 — verify iter-78's alertOnNewSeverity MCP input plumbs
395
+ // through. baselineFile has no findings; current ruflo audit has
396
+ // 1 INFO finding. With alertOnNewSeverity='info' the gate fires
397
+ // and surfaces in the response.
398
+ const baselineNoFindings = pjoin(tmp, 'drift-baseline-no-findings.json');
399
+ writeFileSync(baselineNoFindings, JSON.stringify({
400
+ startedAt: '2026-06-16T00:00:00Z',
401
+ composite: { worst: 'clean' },
402
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
403
+ fingerprint: {
404
+ score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
405
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
406
+ },
407
+ }));
408
+ const rSevAlert = await driftTool.handler({
409
+ path: '.', dryRun: true, threshold: 0.5,
410
+ baselineFile: baselineNoFindings,
411
+ alertOnNewSeverity: 'info',
412
+ });
413
+ if (!rSevAlert.degraded) {
414
+ assert(rSevAlert.data?.alert?.newSeverityThreshold === 'info',
415
+ 'drift_from_history MCP-layer: alertOnNewSeverity echoed in payload (iter 85)');
416
+ // Triggered AND exit code reflects (only if the audit actually had findings)
417
+ if (rSevAlert.data?.alert?.triggered === true) {
418
+ assert(rSevAlert.exitCode === 1,
419
+ `drift_from_history MCP-layer: alertOnNewSeverity exitCode=1 when triggered (got ${rSevAlert.exitCode})`);
420
+ assert(rSevAlert.success === false,
421
+ 'drift_from_history MCP-layer: success===false when alert fires (iter 44 fix)');
422
+ }
423
+ }
424
+
425
+ const r54 = await driftTool.handler({ path: '.', dryRun: true, threshold: 0.5 });
426
+ if (!r54.degraded) {
427
+ assert(typeof r54.data === 'object' && r54.data !== null,
428
+ 'drift_from_history positive: data is an object');
429
+ // Either it produced the structured drift report OR the no-history error
430
+ const isOk = r54.data?.command === 'drift-from-history';
431
+ const isNoHistory = typeof r54.data?.error === 'string' && r54.data.error.includes('no audit records');
432
+ assert(isOk || isNoHistory,
433
+ `drift_from_history positive: structured report OR no-history error (got ${JSON.stringify(r54.data).slice(0,80)})`);
434
+ if (isOk) {
435
+ assert(typeof r54.data.baseline?.key === 'string',
436
+ 'drift_from_history: baseline.key is a string');
437
+ assert(typeof r54.data.alert?.threshold === 'number',
438
+ 'drift_from_history: alert.threshold echoed numerically');
439
+ }
440
+ } else {
441
+ console.log(` ⊘ drift_from_history: degraded (metaharness or memory absent)`);
442
+ }
443
+ }
444
+
445
+ const r = await trendTool.handler({ baselineKey: 'missing-X', currentKey: 'missing-Y' });
446
+ assert(r.exitCode === 2,
447
+ 'audit_trend bad-keys path exits 2 (script-level guard fires)');
448
+ assert(r.data === null || r.data === undefined,
449
+ 'audit_trend bad-keys path: data null (no JSON emitted on stderr exit)');
450
+ // iter 44 — success semantic anchor. Pre-iter-44 wrapper returned
451
+ // success:true for this case (because no degraded marker). Now
452
+ // returns false because exitCode !== 0.
453
+ assert(r.success === false,
454
+ 'audit_trend bad-keys path: success === false (iter 44 fix)');
455
+ }
456
+
457
+ // Cleanup
458
+ try { (await import('node:fs')).rmSync(tmp, { recursive: true, force: true }); } catch { /* ignore */ }
459
+
460
+ console.log(`\n${passed} passed, ${failed} failed`);
461
+ if (failed > 0) {
462
+ console.log('\nFailures:');
463
+ for (const f of failures) console.log(` - ${f}`);
464
+ process.exit(1);
465
+ }
466
+ console.log('\n✓ All 15 MCP tools satisfy the runtime contract.');
467
+ }
468
+
469
+ main().catch((e) => {
470
+ console.error('test-mcp-tools crashed:', e.message || e);
471
+ process.exit(2);
472
+ });