@claude-flow/cli 3.32.9 → 3.32.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (444) hide show
  1. package/.claude/agents/analysis/analyze-code-quality.md +178 -178
  2. package/.claude/agents/analysis/code-analyzer.md +209 -209
  3. package/.claude/agents/analysis/code-review/analyze-code-quality.md +178 -178
  4. package/.claude/agents/architecture/arch-system-design.md +156 -156
  5. package/.claude/agents/architecture/system-design/arch-system-design.md +154 -154
  6. package/.claude/agents/browser/browser-agent.yaml +182 -182
  7. package/.claude/agents/consensus/byzantine-coordinator.md +62 -62
  8. package/.claude/agents/consensus/crdt-synchronizer.md +996 -996
  9. package/.claude/agents/consensus/gossip-coordinator.md +62 -62
  10. package/.claude/agents/consensus/performance-benchmarker.md +850 -850
  11. package/.claude/agents/consensus/quorum-manager.md +822 -822
  12. package/.claude/agents/consensus/raft-manager.md +62 -62
  13. package/.claude/agents/consensus/security-manager.md +621 -621
  14. package/.claude/agents/core/planner.md +374 -374
  15. package/.claude/agents/custom/test-long-runner.md +44 -44
  16. package/.claude/agents/data/data-ml-model.md +444 -444
  17. package/.claude/agents/data/ml/data-ml-model.md +192 -192
  18. package/.claude/agents/development/backend/dev-backend-api.md +141 -141
  19. package/.claude/agents/development/dev-backend-api.md +344 -344
  20. package/.claude/agents/devops/ci-cd/ops-cicd-github.md +163 -163
  21. package/.claude/agents/devops/ops-cicd-github.md +164 -164
  22. package/.claude/agents/documentation/api-docs/docs-api-openapi.md +173 -173
  23. package/.claude/agents/documentation/docs-api-openapi.md +354 -354
  24. package/.claude/agents/flow-nexus/app-store.md +87 -87
  25. package/.claude/agents/flow-nexus/authentication.md +68 -68
  26. package/.claude/agents/flow-nexus/challenges.md +80 -80
  27. package/.claude/agents/flow-nexus/neural-network.md +87 -87
  28. package/.claude/agents/flow-nexus/payments.md +82 -82
  29. package/.claude/agents/flow-nexus/sandbox.md +75 -75
  30. package/.claude/agents/flow-nexus/swarm.md +75 -75
  31. package/.claude/agents/flow-nexus/user-tools.md +95 -95
  32. package/.claude/agents/flow-nexus/workflow.md +83 -83
  33. package/.claude/agents/github/code-review-swarm.md +377 -377
  34. package/.claude/agents/github/github-modes.md +172 -172
  35. package/.claude/agents/github/issue-tracker.md +575 -575
  36. package/.claude/agents/github/multi-repo-swarm.md +552 -552
  37. package/.claude/agents/github/pr-manager.md +437 -437
  38. package/.claude/agents/github/project-board-sync.md +508 -508
  39. package/.claude/agents/github/release-manager.md +604 -604
  40. package/.claude/agents/github/release-swarm.md +582 -582
  41. package/.claude/agents/github/repo-architect.md +397 -397
  42. package/.claude/agents/github/swarm-issue.md +572 -572
  43. package/.claude/agents/github/swarm-pr.md +427 -427
  44. package/.claude/agents/github/sync-coordinator.md +451 -451
  45. package/.claude/agents/github/workflow-automation.md +902 -902
  46. package/.claude/agents/goal/agent.md +815 -815
  47. package/.claude/agents/optimization/benchmark-suite.md +664 -664
  48. package/.claude/agents/optimization/load-balancer.md +430 -430
  49. package/.claude/agents/optimization/performance-monitor.md +671 -671
  50. package/.claude/agents/optimization/resource-allocator.md +673 -673
  51. package/.claude/agents/optimization/topology-optimizer.md +807 -807
  52. package/.claude/agents/payments/agentic-payments.md +126 -126
  53. package/.claude/agents/sona/sona-learning-optimizer.md +74 -74
  54. package/.claude/agents/sparc/architecture.md +698 -698
  55. package/.claude/agents/sparc/pseudocode.md +519 -519
  56. package/.claude/agents/sparc/refinement.md +801 -801
  57. package/.claude/agents/sparc/specification.md +477 -477
  58. package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +224 -224
  59. package/.claude/agents/specialized/spec-mobile-react-native.md +226 -226
  60. package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
  61. package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
  62. package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
  63. package/.claude/agents/sublinear/performance-optimizer.md +367 -367
  64. package/.claude/agents/sublinear/trading-predictor.md +245 -245
  65. package/.claude/agents/swarm/adaptive-coordinator.md +1126 -1126
  66. package/.claude/agents/swarm/hierarchical-coordinator.md +709 -709
  67. package/.claude/agents/swarm/mesh-coordinator.md +962 -962
  68. package/.claude/agents/templates/automation-smart-agent.md +204 -204
  69. package/.claude/agents/templates/base-template-generator.md +289 -289
  70. package/.claude/agents/templates/coordinator-swarm-init.md +89 -89
  71. package/.claude/agents/templates/github-pr-manager.md +176 -176
  72. package/.claude/agents/templates/implementer-sparc-coder.md +258 -258
  73. package/.claude/agents/templates/memory-coordinator.md +186 -186
  74. package/.claude/agents/templates/orchestrator-task.md +138 -138
  75. package/.claude/agents/templates/performance-analyzer.md +198 -198
  76. package/.claude/agents/templates/sparc-coordinator.md +513 -513
  77. package/.claude/agents/testing/production-validator.md +394 -394
  78. package/.claude/agents/testing/tdd-london-swarm.md +243 -243
  79. package/.claude/agents/v3/aidefence-guardian.md +282 -282
  80. package/.claude/agents/v3/claims-authorizer.md +208 -208
  81. package/.claude/agents/v3/collective-intelligence-coordinator.md +993 -993
  82. package/.claude/agents/v3/ddd-domain-expert.md +220 -220
  83. package/.claude/agents/v3/injection-analyst.md +236 -236
  84. package/.claude/agents/v3/performance-engineer.md +1233 -1233
  85. package/.claude/agents/v3/pii-detector.md +151 -151
  86. package/.claude/agents/v3/reasoningbank-learner.md +213 -213
  87. package/.claude/agents/v3/security-architect-aidefence.md +410 -410
  88. package/.claude/agents/v3/security-architect.md +867 -867
  89. package/.claude/agents/v3/swarm-memory-manager.md +157 -157
  90. package/.claude/agents/v3/v3-integration-architect.md +205 -205
  91. package/.claude/commands/agents/README.md +50 -50
  92. package/.claude/commands/agents/agent-capabilities.md +140 -140
  93. package/.claude/commands/agents/agent-coordination.md +28 -28
  94. package/.claude/commands/agents/agent-spawning.md +28 -28
  95. package/.claude/commands/agents/agent-types.md +216 -216
  96. package/.claude/commands/agents/health.md +139 -139
  97. package/.claude/commands/agents/list.md +100 -100
  98. package/.claude/commands/agents/logs.md +130 -130
  99. package/.claude/commands/agents/metrics.md +122 -122
  100. package/.claude/commands/agents/pool.md +127 -127
  101. package/.claude/commands/agents/spawn.md +140 -140
  102. package/.claude/commands/agents/status.md +115 -115
  103. package/.claude/commands/agents/stop.md +102 -102
  104. package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
  105. package/.claude/commands/analysis/README.md +9 -9
  106. package/.claude/commands/analysis/bottleneck-detect.md +162 -162
  107. package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
  108. package/.claude/commands/analysis/performance-report.md +25 -25
  109. package/.claude/commands/analysis/token-efficiency.md +44 -44
  110. package/.claude/commands/analysis/token-usage.md +25 -25
  111. package/.claude/commands/automation/README.md +9 -9
  112. package/.claude/commands/automation/auto-agent.md +122 -122
  113. package/.claude/commands/automation/self-healing.md +105 -105
  114. package/.claude/commands/automation/session-memory.md +89 -89
  115. package/.claude/commands/automation/smart-agents.md +72 -72
  116. package/.claude/commands/automation/smart-spawn.md +25 -25
  117. package/.claude/commands/automation/workflow-select.md +25 -25
  118. package/.claude/commands/claude-flow-help.md +103 -103
  119. package/.claude/commands/claude-flow-memory.md +107 -107
  120. package/.claude/commands/claude-flow-swarm.md +205 -205
  121. package/.claude/commands/coordination/README.md +9 -9
  122. package/.claude/commands/coordination/agent-spawn.md +25 -25
  123. package/.claude/commands/coordination/init.md +44 -44
  124. package/.claude/commands/coordination/orchestrate.md +43 -43
  125. package/.claude/commands/coordination/spawn.md +45 -45
  126. package/.claude/commands/coordination/swarm-init.md +85 -85
  127. package/.claude/commands/coordination/task-orchestrate.md +25 -25
  128. package/.claude/commands/github/README.md +11 -11
  129. package/.claude/commands/github/code-review-swarm.md +513 -513
  130. package/.claude/commands/github/code-review.md +25 -25
  131. package/.claude/commands/github/github-modes.md +146 -146
  132. package/.claude/commands/github/github-swarm.md +121 -121
  133. package/.claude/commands/github/issue-tracker.md +291 -291
  134. package/.claude/commands/github/issue-triage.md +25 -25
  135. package/.claude/commands/github/multi-repo-swarm.md +518 -518
  136. package/.claude/commands/github/pr-enhance.md +26 -26
  137. package/.claude/commands/github/pr-manager.md +169 -169
  138. package/.claude/commands/github/project-board-sync.md +470 -470
  139. package/.claude/commands/github/release-manager.md +339 -339
  140. package/.claude/commands/github/release-swarm.md +543 -543
  141. package/.claude/commands/github/repo-analyze.md +25 -25
  142. package/.claude/commands/github/repo-architect.md +366 -366
  143. package/.claude/commands/github/swarm-issue.md +484 -484
  144. package/.claude/commands/github/swarm-pr.md +287 -287
  145. package/.claude/commands/github/sync-coordinator.md +302 -302
  146. package/.claude/commands/github/workflow-automation.md +441 -441
  147. package/.claude/commands/hive-mind/README.md +17 -17
  148. package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
  149. package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
  150. package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
  151. package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
  152. package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
  153. package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
  154. package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
  155. package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
  156. package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
  157. package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
  158. package/.claude/commands/hive-mind/hive-mind.md +27 -27
  159. package/.claude/commands/hooks/README.md +11 -11
  160. package/.claude/commands/hooks/overview.md +57 -57
  161. package/.claude/commands/hooks/post-edit.md +117 -117
  162. package/.claude/commands/hooks/post-task.md +112 -112
  163. package/.claude/commands/hooks/pre-edit.md +113 -113
  164. package/.claude/commands/hooks/pre-task.md +111 -111
  165. package/.claude/commands/hooks/session-end.md +118 -118
  166. package/.claude/commands/hooks/setup.md +102 -102
  167. package/.claude/commands/memory/README.md +9 -9
  168. package/.claude/commands/memory/memory-persist.md +25 -25
  169. package/.claude/commands/memory/memory-search.md +25 -25
  170. package/.claude/commands/memory/memory-usage.md +25 -25
  171. package/.claude/commands/memory/neural.md +47 -47
  172. package/.claude/commands/monitoring/README.md +9 -9
  173. package/.claude/commands/monitoring/agent-metrics.md +25 -25
  174. package/.claude/commands/monitoring/agents.md +44 -44
  175. package/.claude/commands/monitoring/real-time-view.md +25 -25
  176. package/.claude/commands/monitoring/status.md +46 -46
  177. package/.claude/commands/monitoring/swarm-monitor.md +25 -25
  178. package/.claude/commands/optimization/README.md +9 -9
  179. package/.claude/commands/optimization/auto-topology.md +61 -61
  180. package/.claude/commands/optimization/cache-manage.md +25 -25
  181. package/.claude/commands/optimization/parallel-execute.md +25 -25
  182. package/.claude/commands/optimization/parallel-execution.md +49 -49
  183. package/.claude/commands/optimization/topology-optimize.md +25 -25
  184. package/.claude/commands/pair/README.md +260 -260
  185. package/.claude/commands/pair/commands.md +545 -545
  186. package/.claude/commands/pair/config.md +509 -509
  187. package/.claude/commands/pair/examples.md +511 -511
  188. package/.claude/commands/pair/modes.md +347 -347
  189. package/.claude/commands/pair/session.md +406 -406
  190. package/.claude/commands/pair/start.md +208 -208
  191. package/.claude/commands/sparc/analyzer.md +51 -51
  192. package/.claude/commands/sparc/architect.md +53 -53
  193. package/.claude/commands/sparc/ask.md +97 -97
  194. package/.claude/commands/sparc/batch-executor.md +54 -54
  195. package/.claude/commands/sparc/code.md +89 -89
  196. package/.claude/commands/sparc/coder.md +54 -54
  197. package/.claude/commands/sparc/debug.md +83 -83
  198. package/.claude/commands/sparc/debugger.md +54 -54
  199. package/.claude/commands/sparc/designer.md +53 -53
  200. package/.claude/commands/sparc/devops.md +109 -109
  201. package/.claude/commands/sparc/docs-writer.md +80 -80
  202. package/.claude/commands/sparc/documenter.md +54 -54
  203. package/.claude/commands/sparc/innovator.md +54 -54
  204. package/.claude/commands/sparc/integration.md +83 -83
  205. package/.claude/commands/sparc/mcp.md +117 -117
  206. package/.claude/commands/sparc/memory-manager.md +54 -54
  207. package/.claude/commands/sparc/optimizer.md +54 -54
  208. package/.claude/commands/sparc/orchestrator.md +131 -131
  209. package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
  210. package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
  211. package/.claude/commands/sparc/researcher.md +54 -54
  212. package/.claude/commands/sparc/reviewer.md +54 -54
  213. package/.claude/commands/sparc/security-review.md +80 -80
  214. package/.claude/commands/sparc/sparc-modes.md +174 -174
  215. package/.claude/commands/sparc/sparc.md +111 -111
  216. package/.claude/commands/sparc/spec-pseudocode.md +80 -80
  217. package/.claude/commands/sparc/supabase-admin.md +348 -348
  218. package/.claude/commands/sparc/swarm-coordinator.md +54 -54
  219. package/.claude/commands/sparc/tdd.md +54 -54
  220. package/.claude/commands/sparc/tester.md +54 -54
  221. package/.claude/commands/sparc/tutorial.md +79 -79
  222. package/.claude/commands/sparc/workflow-manager.md +54 -54
  223. package/.claude/commands/sparc.md +166 -166
  224. package/.claude/commands/stream-chain/pipeline.md +120 -120
  225. package/.claude/commands/stream-chain/run.md +69 -69
  226. package/.claude/commands/swarm/README.md +15 -15
  227. package/.claude/commands/swarm/analysis.md +95 -95
  228. package/.claude/commands/swarm/development.md +96 -96
  229. package/.claude/commands/swarm/examples.md +168 -168
  230. package/.claude/commands/swarm/maintenance.md +102 -102
  231. package/.claude/commands/swarm/optimization.md +117 -117
  232. package/.claude/commands/swarm/research.md +136 -136
  233. package/.claude/commands/swarm/swarm-analysis.md +8 -8
  234. package/.claude/commands/swarm/swarm-background.md +8 -8
  235. package/.claude/commands/swarm/swarm-init.md +19 -19
  236. package/.claude/commands/swarm/swarm-modes.md +8 -8
  237. package/.claude/commands/swarm/swarm-monitor.md +8 -8
  238. package/.claude/commands/swarm/swarm-spawn.md +19 -19
  239. package/.claude/commands/swarm/swarm-status.md +8 -8
  240. package/.claude/commands/swarm/swarm-strategies.md +8 -8
  241. package/.claude/commands/swarm/swarm.md +87 -87
  242. package/.claude/commands/swarm/testing.md +131 -131
  243. package/.claude/commands/training/README.md +9 -9
  244. package/.claude/commands/training/model-update.md +25 -25
  245. package/.claude/commands/training/neural-patterns.md +107 -107
  246. package/.claude/commands/training/neural-train.md +75 -75
  247. package/.claude/commands/training/pattern-learn.md +25 -25
  248. package/.claude/commands/training/specialization.md +62 -62
  249. package/.claude/commands/truth/start.md +142 -142
  250. package/.claude/commands/verify/check.md +49 -49
  251. package/.claude/commands/verify/start.md +127 -127
  252. package/.claude/commands/workflows/README.md +9 -9
  253. package/.claude/commands/workflows/development.md +77 -77
  254. package/.claude/commands/workflows/research.md +62 -62
  255. package/.claude/commands/workflows/workflow-create.md +25 -25
  256. package/.claude/commands/workflows/workflow-execute.md +25 -25
  257. package/.claude/commands/workflows/workflow-export.md +25 -25
  258. package/.claude/eval/human-relevance-frozen-v1.json +17 -17
  259. package/.claude/evolve-proof/generation-0.json +211 -211
  260. package/.claude/evolve-proof/real-generation-0.json +406 -406
  261. package/.claude/evolve-proof/real-generation-1.json +406 -406
  262. package/.claude/helpers/README.md +96 -96
  263. package/.claude/helpers/adr-compliance.sh +186 -186
  264. package/.claude/helpers/auto-commit.sh +178 -178
  265. package/.claude/helpers/auto-memory-hook.mjs +430 -430
  266. package/.claude/helpers/checkpoint-manager.sh +251 -251
  267. package/.claude/helpers/daemon-manager.sh +252 -252
  268. package/.claude/helpers/ddd-tracker.sh +144 -144
  269. package/.claude/helpers/github-safe.js +156 -156
  270. package/.claude/helpers/github-setup.sh +45 -45
  271. package/.claude/helpers/guidance-hook.sh +13 -13
  272. package/.claude/helpers/guidance-hooks.sh +102 -102
  273. package/.claude/helpers/health-monitor.sh +108 -108
  274. package/.claude/helpers/helpers.manifest.json +6 -6
  275. package/.claude/helpers/hook-handler.cjs +565 -565
  276. package/.claude/helpers/intelligence.cjs +1058 -1058
  277. package/.claude/helpers/learning-hooks.sh +329 -329
  278. package/.claude/helpers/learning-optimizer.sh +127 -127
  279. package/.claude/helpers/learning-service.mjs +1144 -1144
  280. package/.claude/helpers/memory.js +83 -83
  281. package/.claude/helpers/metrics-db.mjs +503 -503
  282. package/.claude/helpers/pattern-consolidator.sh +86 -86
  283. package/.claude/helpers/perf-worker.sh +160 -160
  284. package/.claude/helpers/post-commit +16 -16
  285. package/.claude/helpers/pre-commit +26 -26
  286. package/.claude/helpers/quick-start.sh +19 -19
  287. package/.claude/helpers/router.js +105 -105
  288. package/.claude/helpers/security-scanner.sh +127 -127
  289. package/.claude/helpers/session.js +157 -157
  290. package/.claude/helpers/setup-mcp.sh +18 -18
  291. package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
  292. package/.claude/helpers/statusline-hook.sh +21 -21
  293. package/.claude/helpers/statusline.cjs +1060 -1060
  294. package/.claude/helpers/statusline.js +340 -340
  295. package/.claude/helpers/swarm-comms.sh +353 -353
  296. package/.claude/helpers/swarm-hooks.sh +761 -761
  297. package/.claude/helpers/swarm-monitor.sh +210 -210
  298. package/.claude/helpers/sync-v3-metrics.sh +245 -245
  299. package/.claude/helpers/update-v3-progress.sh +165 -165
  300. package/.claude/helpers/v3-quick-status.sh +57 -57
  301. package/.claude/helpers/v3.sh +110 -110
  302. package/.claude/helpers/validate-v3-config.sh +215 -215
  303. package/.claude/helpers/worker-manager.sh +170 -170
  304. package/.claude/proven-config.manifest.json +37 -37
  305. package/.claude/proven-config.signed.json +41 -41
  306. package/.claude/settings.json +182 -182
  307. package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
  308. package/.claude/skills/agentdb-learning/SKILL.md +545 -545
  309. package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
  310. package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
  311. package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
  312. package/.claude/skills/browser/SKILL.md +204 -204
  313. package/.claude/skills/dual-mode/README.md +71 -71
  314. package/.claude/skills/dual-mode/dual-collect.md +103 -103
  315. package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
  316. package/.claude/skills/dual-mode/dual-spawn.md +81 -81
  317. package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
  318. package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
  319. package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
  320. package/.claude/skills/github-code-review/SKILL.md +1125 -1125
  321. package/.claude/skills/github-multi-repo/SKILL.md +862 -862
  322. package/.claude/skills/github-project-management/SKILL.md +1262 -1262
  323. package/.claude/skills/github-release-management/SKILL.md +1064 -1064
  324. package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
  325. package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
  326. package/.claude/skills/pair-programming/SKILL.md +1202 -1202
  327. package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
  328. package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
  329. package/.claude/skills/skill-builder/SKILL.md +910 -910
  330. package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
  331. package/.claude/skills/stream-chain/SKILL.md +560 -560
  332. package/.claude/skills/swarm-advanced/SKILL.md +970 -970
  333. package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
  334. package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
  335. package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
  336. package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
  337. package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
  338. package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
  339. package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
  340. package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
  341. package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
  342. package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
  343. package/.claude/skills/verification-quality/SKILL.md +691 -691
  344. package/README.md +419 -419
  345. package/bin/cli.js +314 -314
  346. package/bin/mcp-server.js +224 -224
  347. package/bin/preinstall.cjs +2 -2
  348. package/catalog-manifest.json +2 -2
  349. package/dist/src/autopilot-state.js +24 -7
  350. package/dist/src/benchmarks/gaia-critic.js +24 -24
  351. package/dist/src/business-pods/bbs-budget-tracker.js +53 -53
  352. package/dist/src/commands/completions.js +409 -409
  353. package/dist/src/commands/daemon.js +44 -44
  354. package/dist/src/commands/embeddings.js +26 -26
  355. package/dist/src/commands/hive-mind.js +97 -97
  356. package/dist/src/commands/hooks.js +31 -10
  357. package/dist/src/commands/init.js +202 -34
  358. package/dist/src/commands/memory.js +12 -1
  359. package/dist/src/commands/ruvector/backup.js +23 -23
  360. package/dist/src/commands/ruvector/benchmark.js +31 -31
  361. package/dist/src/commands/ruvector/import.js +14 -14
  362. package/dist/src/commands/ruvector/init.js +115 -115
  363. package/dist/src/commands/ruvector/migrate.js +99 -99
  364. package/dist/src/commands/ruvector/optimize.js +51 -51
  365. package/dist/src/commands/ruvector/setup.js +624 -624
  366. package/dist/src/commands/ruvector/status.js +38 -38
  367. package/dist/src/config/proven-config.js +2 -2
  368. package/dist/src/funnel/disclosure.js +13 -2
  369. package/dist/src/funnel/messages.d.ts +12 -10
  370. package/dist/src/funnel/messages.js +83 -11
  371. package/dist/src/init/claudemd-generator.js +231 -231
  372. package/dist/src/init/executor.js +453 -453
  373. package/dist/src/init/helper-signing.js +2 -2
  374. package/dist/src/init/helpers-generator.js +751 -751
  375. package/dist/src/init/statusline-generator.js +24 -24
  376. package/dist/src/mcp-tools/agentdb-tools.js +15 -15
  377. package/dist/src/mcp-tools/browser-intent-tools.js +19 -19
  378. package/dist/src/mcp-tools/browser-tools.js +8 -0
  379. package/dist/src/mcp-tools/hooks-tools.js +21 -0
  380. package/dist/src/mcp-tools/memory-tools.js +4 -3
  381. package/dist/src/memory/graph-edge-writer.js +22 -22
  382. package/dist/src/memory/memory-bridge.js +192 -123
  383. package/dist/src/memory/memory-initializer.js +407 -407
  384. package/dist/src/memory/rabitq-index.js +5 -5
  385. package/dist/src/parser.js +25 -9
  386. package/dist/src/proxy/verify.js +2 -2
  387. package/dist/src/runtime/headless.js +28 -28
  388. package/dist/src/services/distill-tuning.js +7 -7
  389. package/dist/src/services/headless-worker-executor.js +84 -84
  390. package/dist/src/services/memory-distillation.js +4 -4
  391. package/dist/src/services/worker-daemon.js +7 -4
  392. package/dist/src/transfer/deploy-seraphine.js +23 -23
  393. package/package.json +137 -137
  394. package/plugins/ruflo-metaharness/.claude-plugin/plugin.json +32 -32
  395. package/plugins/ruflo-metaharness/README.md +72 -72
  396. package/plugins/ruflo-metaharness/agents/metaharness-architect.md +58 -58
  397. package/plugins/ruflo-metaharness/commands/ruflo-metaharness.md +48 -48
  398. package/plugins/ruflo-metaharness/scripts/_darwin.mjs +210 -210
  399. package/plugins/ruflo-metaharness/scripts/_harness.mjs +330 -330
  400. package/plugins/ruflo-metaharness/scripts/_invoke.mjs +231 -231
  401. package/plugins/ruflo-metaharness/scripts/_redblue.mjs +143 -143
  402. package/plugins/ruflo-metaharness/scripts/_similarity.mjs +161 -161
  403. package/plugins/ruflo-metaharness/scripts/_spike-similarity.mjs +223 -223
  404. package/plugins/ruflo-metaharness/scripts/audit-list.mjs +158 -158
  405. package/plugins/ruflo-metaharness/scripts/audit-trend.mjs +272 -272
  406. package/plugins/ruflo-metaharness/scripts/bench-parse-mcp-scan.mjs +146 -146
  407. package/plugins/ruflo-metaharness/scripts/bench-recordpair-overhead.mjs +186 -186
  408. package/plugins/ruflo-metaharness/scripts/bench-similarity.mjs +177 -177
  409. package/plugins/ruflo-metaharness/scripts/bench.mjs +95 -95
  410. package/plugins/ruflo-metaharness/scripts/drift-from-history.mjs +363 -363
  411. package/plugins/ruflo-metaharness/scripts/evolve.mjs +404 -404
  412. package/plugins/ruflo-metaharness/scripts/genome.mjs +80 -80
  413. package/plugins/ruflo-metaharness/scripts/gepa.mjs +153 -153
  414. package/plugins/ruflo-metaharness/scripts/learn.mjs +127 -127
  415. package/plugins/ruflo-metaharness/scripts/mcp-scan.mjs +111 -111
  416. package/plugins/ruflo-metaharness/scripts/mint.mjs +126 -126
  417. package/plugins/ruflo-metaharness/scripts/oia-audit.mjs +228 -228
  418. package/plugins/ruflo-metaharness/scripts/redblue.mjs +286 -286
  419. package/plugins/ruflo-metaharness/scripts/router-parallel-analyze.mjs +250 -250
  420. package/plugins/ruflo-metaharness/scripts/score.mjs +92 -92
  421. package/plugins/ruflo-metaharness/scripts/security-bench.mjs +174 -174
  422. package/plugins/ruflo-metaharness/scripts/similarity.mjs +158 -158
  423. package/plugins/ruflo-metaharness/scripts/smoke.sh +2356 -2356
  424. package/plugins/ruflo-metaharness/scripts/test-graceful-degradation.mjs +165 -165
  425. package/plugins/ruflo-metaharness/scripts/test-mcp-tools.mjs +472 -472
  426. package/plugins/ruflo-metaharness/scripts/test-parallel-pipeline.mjs +204 -204
  427. package/plugins/ruflo-metaharness/scripts/test-pipeline-roundtrip.mjs +586 -586
  428. package/plugins/ruflo-metaharness/scripts/test-similarity.mjs +334 -334
  429. package/plugins/ruflo-metaharness/scripts/test-with-openrouter.mjs +229 -229
  430. package/plugins/ruflo-metaharness/scripts/threat-model.mjs +59 -59
  431. package/plugins/ruflo-metaharness/skills/harness-bench/SKILL.md +64 -64
  432. package/plugins/ruflo-metaharness/skills/harness-drift-from-history/SKILL.md +65 -65
  433. package/plugins/ruflo-metaharness/skills/harness-evolve/SKILL.md +131 -131
  434. package/plugins/ruflo-metaharness/skills/harness-genome/SKILL.md +54 -54
  435. package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +65 -65
  436. package/plugins/ruflo-metaharness/skills/harness-learn/SKILL.md +65 -65
  437. package/plugins/ruflo-metaharness/skills/harness-mcp-scan/SKILL.md +49 -49
  438. package/plugins/ruflo-metaharness/skills/harness-mint/SKILL.md +72 -72
  439. package/plugins/ruflo-metaharness/skills/harness-oia-audit/SKILL.md +79 -79
  440. package/plugins/ruflo-metaharness/skills/harness-score/SKILL.md +66 -66
  441. package/plugins/ruflo-metaharness/skills/harness-security-bench/SKILL.md +101 -101
  442. package/plugins/ruflo-metaharness/skills/harness-similarity/SKILL.md +67 -67
  443. package/plugins/ruflo-metaharness/skills/harness-threat-model/SKILL.md +41 -41
  444. package/scripts/postinstall.cjs +153 -153
@@ -1,472 +1,472 @@
1
- #!/usr/bin/env node
2
- // test-mcp-tools.mjs — runtime test for the iter-20/21 MCP tool registry.
3
- //
4
- // tsc proves the metaharness-tools.ts module COMPILES; structural smoke
5
- // proves the source DECLARES the right tool names. Neither proves the
6
- // HANDLERS actually run without throwing. This test imports the compiled
7
- // module and invokes every tool's handler with minimal inputs.
8
- //
9
- // CONTRACT EACH TOOL MUST SATISFY
10
- // - handler is callable as `await tool.handler({ ... })`
11
- // - returns an object with keys: success, data, degraded, exitCode
12
- // - never throws (even with bad/missing optional dep — graceful)
13
- // - handler honors the 120s subprocess timeout (no hang)
14
- //
15
- // USAGE
16
- // node scripts/test-mcp-tools.mjs # default
17
- // node scripts/test-mcp-tools.mjs --format json
18
- //
19
- // EXIT CODES
20
- // 0 all tools satisfy the contract
21
- // 1 at least one tool failed
22
- // 2 setup error (compiled dist not present)
23
-
24
- import { existsSync } from 'node:fs';
25
- import { dirname, join, resolve } from 'node:path';
26
- import { fileURLToPath } from 'node:url';
27
-
28
- const SCRIPTS_DIR = dirname(fileURLToPath(import.meta.url));
29
- const ARGS = (() => {
30
- const a = { format: 'table' };
31
- for (let i = 2; i < process.argv.length; i++) {
32
- if (process.argv[i] === '--format') a.format = process.argv[++i];
33
- }
34
- return a;
35
- })();
36
-
37
- let passed = 0, failed = 0;
38
- const failures = [];
39
-
40
- function assert(cond, label) {
41
- if (cond) { console.log(` ✓ ${label}`); passed++; }
42
- else { console.log(` ✗ ${label}`); failures.push(label); failed++; }
43
- }
44
-
45
- async function main() {
46
- // Locate the compiled dist of metaharness-tools.
47
- const distPath = resolve(SCRIPTS_DIR, '..', '..', '..',
48
- 'v3', '@claude-flow', 'cli', 'dist', 'src', 'mcp-tools', 'metaharness-tools.js');
49
-
50
- if (!existsSync(distPath)) {
51
- console.log(`# test-mcp-tools — SKIPPED`);
52
- console.log('');
53
- console.log(`Compiled dist not present: ${distPath}`);
54
- console.log(`Build the CLI first:`);
55
- console.log(` cd v3/@claude-flow/cli && npm run build`);
56
- console.log('');
57
- console.log(`Exit 0 — this script is meaningfully runnable only post-build.`);
58
- process.exit(0);
59
- }
60
-
61
- let mod;
62
- try {
63
- mod = await import(distPath);
64
- } catch (e) {
65
- console.error(`test-mcp-tools: failed to import ${distPath}: ${e.message}`);
66
- process.exit(2);
67
- }
68
-
69
- const tools = mod.metaharnessTools;
70
- console.log(`# test-mcp-tools — runtime contract\n`);
71
-
72
- // ──────────────────────────────────────────────────────────────────
73
- // PHASE 1 — module exports the right shape
74
- // ──────────────────────────────────────────────────────────────────
75
- console.log('Phase 1 — module shape');
76
- assert(Array.isArray(tools), 'metaharnessTools is an array');
77
- assert(tools.length === 15, `15 tools registered (got ${tools.length})`);
78
-
79
- const expectedNames = new Set([
80
- 'metaharness_score',
81
- 'metaharness_genome',
82
- 'metaharness_mcp_scan',
83
- 'metaharness_threat_model',
84
- 'metaharness_oia_audit',
85
- 'metaharness_audit_list',
86
- 'metaharness_audit_trend',
87
- // iter 36 — ADR-152 §3.1 production
88
- 'metaharness_similarity',
89
- // iter 54 — one-command drift detection (composes audit-list + oia-audit + audit-trend)
90
- 'metaharness_drift_from_history',
91
- // ADR-153 — bench suites + evolve driver + security-focused bench
92
- 'metaharness_bench',
93
- 'metaharness_evolve',
94
- 'metaharness_security_bench',
95
- // @metaharness/redblue@~0.1.4 — adversarial red/blue LLM testing
96
- 'metaharness_redblue',
97
- // metaharness@0.3.0 — upstream ADR-235 GEPA learning run
98
- 'metaharness_learn',
99
- // @metaharness/darwin@0.8.0 — GEPA library surface (genome ops)
100
- 'metaharness_gepa',
101
- ]);
102
- const actualNames = new Set(tools.map((t) => t.name));
103
- for (const name of expectedNames) {
104
- assert(actualNames.has(name), `${name} registered`);
105
- }
106
-
107
- // ──────────────────────────────────────────────────────────────────
108
- // PHASE 2 — every tool has the required MCP shape
109
- // ──────────────────────────────────────────────────────────────────
110
- console.log('\nPhase 2 — per-tool shape');
111
- for (const tool of tools) {
112
- const ok = typeof tool.name === 'string'
113
- && typeof tool.description === 'string'
114
- && typeof tool.category === 'string'
115
- && typeof tool.handler === 'function'
116
- && typeof tool.inputSchema === 'object';
117
- assert(ok, `${tool.name} has {name, description, category, handler, inputSchema}`);
118
- assert(tool.category === 'metaharness', `${tool.name} category === 'metaharness'`);
119
- }
120
-
121
- // ──────────────────────────────────────────────────────────────────
122
- // PHASE 3 — handlers callable + return contract shape
123
- //
124
- // We invoke each handler with minimal valid input. The handlers may
125
- // succeed (if metaharness is installed) or report degraded (if not).
126
- // EITHER way, they must return { success, data, degraded, exitCode }
127
- // without throwing.
128
- // ──────────────────────────────────────────────────────────────────
129
- console.log('\nPhase 3 — handler invocations (allow up to 30s each)');
130
- for (const tool of tools) {
131
- // Construct minimal valid input per tool.
132
- let input = {};
133
- if (tool.name === 'metaharness_audit_trend') {
134
- // Requires baselineKey + currentKey — use fake keys that won't
135
- // resolve so we exercise the not-found path.
136
- input = { baselineKey: 'audit-fake-base', currentKey: 'audit-fake-curr' };
137
- }
138
- if (tool.name === 'metaharness_similarity') {
139
- // Needs --a/--b OR --a-key/--b-key. Use fake mem keys to exercise
140
- // the graceful not-found path (matches audit_trend convention).
141
- input = { aKey: 'harness-fake-a', bKey: 'harness-fake-b' };
142
- }
143
- if (tool.name === 'metaharness_drift_from_history') {
144
- // iter 54 — composes 3 subprocesses, needs more time than the default.
145
- input = { dryRun: true, threshold: 0.5 };
146
- }
147
- if (tool.name === 'metaharness_oia_audit') {
148
- // iter 128 — composite audit runs 5 sub-audits (oia-manifest +
149
- // threat-model + mcp-scan + score + genome) in parallel. Each
150
- // shells out via npx. --dry-run skips memory persistence so the
151
- // test doesn't pollute namespaces.
152
- input = { dryRun: true };
153
- }
154
- if (tool.name === 'metaharness_redblue') {
155
- // `attack` preview is the fastest path that exercises the upstream
156
- // binary without needing OPENROUTER_API_KEY or running any model
157
- // calls. Count=1 keeps cold-cache npx fetch the dominant cost.
158
- input = { subcommand: 'attack', family: 'prompt', count: 1 };
159
- }
160
- if (tool.name === 'metaharness_learn') {
161
- // No repo checkout in CI → structured {status:"checkout-required"}
162
- // exit-0 path. $0: without run=true upstream never spends anyway.
163
- input = {};
164
- }
165
- if (tool.name === 'metaharness_gepa') {
166
- // op is required; `genome` loads + validates the SHIPPED cand-6
167
- // genome — pure-local library call once darwin is cached.
168
- input = { op: 'genome' };
169
- }
170
-
171
- // iter 124 → 130 — timeouts have crept up as CI cold-cache npx
172
- // warmup costs got measured. Final budgets:
173
- // default : 60s
174
- // chain-tools : 180s (drift_from_history + oia_audit + audit_list)
175
- // iter 131 — bumped chain-tool budget 90s → 180s. audit_list still
176
- // timed out at 90s in CI; locally it runs in ~4s, but CI's
177
- // `npx @claude-flow/cli@latest memory list` invocation pays both
178
- // the npx fetch AND a full CLI startup (which loads agentic-flow +
179
- // ONNX). 180s gives 30x headroom over the local cost.
180
- const isChainTool = tool.name === 'metaharness_drift_from_history'
181
- || tool.name === 'metaharness_oia_audit'
182
- || tool.name === 'metaharness_audit_list'
183
- // redblue: `attack prompt --count 1` is preview-only (no model
184
- // calls) but the cold-cache `npx -y @metaharness/redblue@~0.1.4`
185
- // fetch can take 30-60s. 180s gives 3x headroom.
186
- || tool.name === 'metaharness_redblue'
187
- // learn: cold-cache `npx -y metaharness@latest` fetch dominates.
188
- // gepa: one-time `npm install --prefix ~/.ruflo/darwin-cache-*`
189
- // fallback install can take 30-60s on cold cache.
190
- || tool.name === 'metaharness_learn'
191
- || tool.name === 'metaharness_gepa';
192
- const timeoutMs = isChainTool ? 180_000 : 60_000;
193
- const handlerPromise = tool.handler(input);
194
- const timeoutPromise = new Promise((_, reject) =>
195
- setTimeout(() => reject(new Error(`${timeoutMs / 1000}s handler timeout`)), timeoutMs));
196
-
197
- let result;
198
- let threw = false;
199
- try {
200
- result = await Promise.race([handlerPromise, timeoutPromise]);
201
- } catch (e) {
202
- threw = true;
203
- console.log(` [${tool.name}] handler threw: ${e.message.slice(0, 80)}`);
204
- }
205
-
206
- assert(!threw, `${tool.name} handler did not throw`);
207
- if (!threw && result) {
208
- assert(typeof result === 'object', `${tool.name} returns object`);
209
- assert('success' in result, `${tool.name} result has 'success'`);
210
- assert('data' in result, `${tool.name} result has 'data'`);
211
- assert('degraded' in result, `${tool.name} result has 'degraded'`);
212
- assert('exitCode' in result, `${tool.name} result has 'exitCode'`);
213
- }
214
- }
215
-
216
- // ──────────────────────────────────────────────────────────────────
217
- // PHASE 4 — POSITIVE-CASE data-shape validation (iter 43)
218
- //
219
- // Iter 37 verified the {success, data, degraded, exitCode} envelope.
220
- // It did NOT verify that data.X contains the right keys when success
221
- // is genuinely true — leaving room for iter 42-style bugs where a
222
- // handler returns valid-looking degraded JSON while silently
223
- // misrouting input. This phase invokes each handler with VALID
224
- // inputs and asserts the expected output shape.
225
- //
226
- // Tools that depend on `npx metaharness` (score/genome/mcp-scan/
227
- // threat-model/oia-audit/audit-list/audit-trend) are SKIPPED in this
228
- // phase when the optional dep isn't installed — they're covered by
229
- // the no-metaharness-smoke workflow's drill. The similarity tool
230
- // has no @metaharness/* dep, so its positive case ALWAYS runs.
231
- // ──────────────────────────────────────────────────────────────────
232
- console.log('\nPhase 4 — positive-case data shape (iter 43)');
233
-
234
- const { writeFileSync, mkdtempSync } = await import('node:fs');
235
- const { tmpdir } = await import('node:os');
236
- const { join: pjoin } = await import('node:path');
237
- const tmp = mkdtempSync(pjoin(tmpdir(), 'mcp-positive-'));
238
-
239
- // metaharness_similarity — full positive case (no @metaharness/* needed)
240
- const simTool = tools.find((t) => t.name === 'metaharness_similarity');
241
- if (simTool) {
242
- const aPath = pjoin(tmp, 'a.json');
243
- const bPath = pjoin(tmp, 'b.json');
244
- writeFileSync(aPath, JSON.stringify({
245
- score: { harnessFit: 78, compileConfidence: 92, taskCoverage: 65, toolSafety: 88, memoryUsefulness: 70, estCostPerRunUsd: 0.04, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:legal' },
246
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','w'], risk_score: 0.45, test_confidence: 0.7, publish_readiness: 0.6 },
247
- }));
248
- writeFileSync(bPath, JSON.stringify({
249
- score: { harnessFit: 75, compileConfidence: 90, taskCoverage: 70, toolSafety: 90, memoryUsefulness: 72, estCostPerRunUsd: 0.05, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:support' },
250
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','q','r'], risk_score: 0.40, test_confidence: 0.75, publish_readiness: 0.65 },
251
- }));
252
- const r = await simTool.handler({ aFile: aPath, bFile: bPath });
253
- assert(r.degraded === false, 'similarity positive case: degraded === false');
254
- assert(r.success === true, 'similarity positive case: success === true');
255
- assert(r.exitCode === 0, 'similarity positive case: exitCode === 0');
256
- const d = r.data ?? {};
257
- assert(typeof d.overall === 'number', 'similarity data has numeric `overall`');
258
- assert(typeof d.components === 'object' && d.components !== null,
259
- 'similarity data has `components` object');
260
- assert(typeof d.components?.cosine === 'number',
261
- 'similarity components.cosine numeric');
262
- assert(typeof d.components?.categorical === 'number',
263
- 'similarity components.categorical numeric');
264
- assert(typeof d.components?.jaccard === 'number',
265
- 'similarity components.jaccard numeric');
266
- assert(typeof d.weights === 'object' && d.weights !== null,
267
- 'similarity data has `weights` object');
268
- assert(d.adr === 'ADR-152', 'similarity data tagged adr=ADR-152');
269
- // Regression anchor — same fixtures as iter-35 spike with non-matching topologies
270
- assert(d.overall > 0 && d.overall < 1,
271
- `similarity overall in (0, 1) — got ${d.overall}`);
272
-
273
- // Per-dimension variant
274
- const rPD = await simTool.handler({ aFile: aPath, bFile: bPath, perDimension: true });
275
- assert(typeof rPD.data?.perDimension === 'object',
276
- 'similarity perDimension=true populates breakdown');
277
-
278
- // Alert-below variant exercises non-zero exit
279
- const rAlert = await simTool.handler({ aFile: aPath, bFile: bPath, alertBelow: 0.99 });
280
- assert(rAlert.data?.alert?.triggered === true,
281
- 'similarity alertBelow=0.99 triggers alert');
282
- assert(rAlert.exitCode === 1, 'similarity alertBelow=0.99 → exitCode 1');
283
- // iter 44 — success semantic anchor (was true under the pre-iter-44
284
- // `!degraded` rule; now false because exitCode !== 0 dominates).
285
- assert(rAlert.success === false,
286
- 'similarity alertBelow=0.99 → success === false (iter 44 fix)');
287
- }
288
-
289
- // metaharness_mcp_scan — positive case post iter-50 parser landing.
290
- // Until iter 50, mcp_scan's data field was an alert-only object with
291
- // no structured findings. After iter 50, findings[] is always present
292
- // (parsed from upstream text) and summary{overallSeverity, totalCount}
293
- // accompanies it.
294
- const scanTool = tools.find((t) => t.name === 'metaharness_mcp_scan');
295
- if (scanTool) {
296
- // Run against ruflo itself — guaranteed to produce at least the
297
- // INFO finding the iter-50 parser test verified manually.
298
- const r = await scanTool.handler({ path: '.', failOn: 'high' });
299
- // Either succeeds with structured findings, or gracefully degrades
300
- // if metaharness isn't installed in this environment.
301
- if (!r.degraded) {
302
- assert(r.success === true, 'mcp_scan positive: success === true');
303
- assert(r.exitCode === 0, 'mcp_scan positive: exitCode === 0');
304
- assert(Array.isArray(r.data?.findings),
305
- 'mcp_scan positive: data.findings is an array (iter 50 fix)');
306
- // Cwd-dependent: when scanning a dir without .mcp/servers.json the
307
- // upstream emits no findings. Only verify shape contract when array
308
- // is populated — the array-presence assertion above is the
309
- // load-bearing one for iter 50.
310
- if (r.data?.findings.length > 0) {
311
- const first = r.data.findings[0];
312
- assert(typeof first?.severity === 'string',
313
- 'mcp_scan positive: first finding has string severity');
314
- assert(typeof first?.message === 'string',
315
- 'mcp_scan positive: first finding has string message');
316
- }
317
- // summary may be null if the upstream produced no Result: line —
318
- // verify the field's presence (null OR object) but only deep-check
319
- // when populated.
320
- if (r.data?.summary) {
321
- assert(typeof r.data.summary.totalCount === 'number',
322
- 'mcp_scan positive: data.summary.totalCount is numeric (when summary present)');
323
- }
324
- } else {
325
- console.log(` ⊘ mcp_scan: metaharness absent — graceful skip`);
326
- }
327
- }
328
-
329
- // metaharness_audit_trend — positive case via file inputs
330
- const trendTool = tools.find((t) => t.name === 'metaharness_audit_trend');
331
- if (trendTool) {
332
- const basePath = pjoin(tmp, 'base.json');
333
- const currPath = pjoin(tmp, 'curr.json');
334
- const fingerprint = {
335
- score: { harnessFit: 80, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
336
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['a','b','c'], risk_score: 0.3, test_confidence: 0.85, publish_readiness: 0.9 },
337
- };
338
- writeFileSync(basePath, JSON.stringify({
339
- startedAt: '2026-06-15T00:00:00Z',
340
- composite: { worst: 'clean' },
341
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
342
- fingerprint,
343
- }));
344
- writeFileSync(currPath, JSON.stringify({
345
- startedAt: '2026-06-16T00:00:00Z',
346
- composite: { worst: 'clean' },
347
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
348
- fingerprint,
349
- }));
350
- // audit_trend tool only supports keys, not files at the MCP layer.
351
- // Document its actual wrapper semantics so future-us doesn't get
352
- // surprised:
353
- // - bad keys → script exits 2 with stderr (no JSON payload)
354
- // - runScript() can't parse a {degraded:true} marker, so it
355
- // returns degraded:false / success:true / exitCode:2
356
- // This is a real wrapper bug (success should not be true when
357
- // exit!=0 AND no JSON came back), tracked separately. Asserting
358
- // current behavior here protects against silent semantic shifts.
359
- // iter 46 — file-input path. audit_trend now accepts baselineFile/currentFile.
360
- const rFiles = await trendTool.handler({ baselineFile: basePath, currentFile: currPath });
361
- assert(rFiles.success === true,
362
- 'audit_trend file-input path: success === true (iter 46)');
363
- assert(rFiles.exitCode === 0, 'audit_trend file-input path: exitCode === 0');
364
- assert(typeof rFiles.data?.delta === 'object',
365
- 'audit_trend file-input path: data.delta object present');
366
- assert(rFiles.data?.delta?.structuralDistance?.verdict === 'near-identical',
367
- `audit_trend file-input path: identical fingerprints → near-identical (got ${rFiles.data?.delta?.structuralDistance?.verdict})`);
368
-
369
- // iter 54 — metaharness_drift_from_history positive case
370
- const driftTool = tools.find((t) => t.name === 'metaharness_drift_from_history');
371
- if (driftTool) {
372
- // iter 71 — verify iter-66/67 fast-path flags are now MCP-callable
373
- // Synthesize a baseline file on disk; pass via the new baselineFile input.
374
- const baselinePath = pjoin(tmp, 'drift-baseline.json');
375
- writeFileSync(baselinePath, JSON.stringify({
376
- startedAt: '2026-06-16T00:00:00Z',
377
- composite: { worst: 'clean' },
378
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
379
- fingerprint: {
380
- score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
381
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
382
- },
383
- }));
384
- const rFastFast = await driftTool.handler({
385
- path: '.', dryRun: true, threshold: 0.5, baselineFile: baselinePath,
386
- });
387
- if (!rFastFast.degraded) {
388
- assert(rFastFast.data?.timing?.usedBaselineFile === true,
389
- 'drift_from_history MCP-layer: baselineFile fastpath fires (iter 71)');
390
- assert(rFastFast.data?.timing?.skippedAuditList === true,
391
- 'drift_from_history MCP-layer: skippedAuditList=true via baselineFile (iter 71)');
392
- }
393
-
394
- // iter 85 — verify iter-78's alertOnNewSeverity MCP input plumbs
395
- // through. baselineFile has no findings; current ruflo audit has
396
- // 1 INFO finding. With alertOnNewSeverity='info' the gate fires
397
- // and surfaces in the response.
398
- const baselineNoFindings = pjoin(tmp, 'drift-baseline-no-findings.json');
399
- writeFileSync(baselineNoFindings, JSON.stringify({
400
- startedAt: '2026-06-16T00:00:00Z',
401
- composite: { worst: 'clean' },
402
- components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
403
- fingerprint: {
404
- score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
405
- genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
406
- },
407
- }));
408
- const rSevAlert = await driftTool.handler({
409
- path: '.', dryRun: true, threshold: 0.5,
410
- baselineFile: baselineNoFindings,
411
- alertOnNewSeverity: 'info',
412
- });
413
- if (!rSevAlert.degraded) {
414
- assert(rSevAlert.data?.alert?.newSeverityThreshold === 'info',
415
- 'drift_from_history MCP-layer: alertOnNewSeverity echoed in payload (iter 85)');
416
- // Triggered AND exit code reflects (only if the audit actually had findings)
417
- if (rSevAlert.data?.alert?.triggered === true) {
418
- assert(rSevAlert.exitCode === 1,
419
- `drift_from_history MCP-layer: alertOnNewSeverity exitCode=1 when triggered (got ${rSevAlert.exitCode})`);
420
- assert(rSevAlert.success === false,
421
- 'drift_from_history MCP-layer: success===false when alert fires (iter 44 fix)');
422
- }
423
- }
424
-
425
- const r54 = await driftTool.handler({ path: '.', dryRun: true, threshold: 0.5 });
426
- if (!r54.degraded) {
427
- assert(typeof r54.data === 'object' && r54.data !== null,
428
- 'drift_from_history positive: data is an object');
429
- // Either it produced the structured drift report OR the no-history error
430
- const isOk = r54.data?.command === 'drift-from-history';
431
- const isNoHistory = typeof r54.data?.error === 'string' && r54.data.error.includes('no audit records');
432
- assert(isOk || isNoHistory,
433
- `drift_from_history positive: structured report OR no-history error (got ${JSON.stringify(r54.data).slice(0,80)})`);
434
- if (isOk) {
435
- assert(typeof r54.data.baseline?.key === 'string',
436
- 'drift_from_history: baseline.key is a string');
437
- assert(typeof r54.data.alert?.threshold === 'number',
438
- 'drift_from_history: alert.threshold echoed numerically');
439
- }
440
- } else {
441
- console.log(` ⊘ drift_from_history: degraded (metaharness or memory absent)`);
442
- }
443
- }
444
-
445
- const r = await trendTool.handler({ baselineKey: 'missing-X', currentKey: 'missing-Y' });
446
- assert(r.exitCode === 2,
447
- 'audit_trend bad-keys path exits 2 (script-level guard fires)');
448
- assert(r.data === null || r.data === undefined,
449
- 'audit_trend bad-keys path: data null (no JSON emitted on stderr exit)');
450
- // iter 44 — success semantic anchor. Pre-iter-44 wrapper returned
451
- // success:true for this case (because no degraded marker). Now
452
- // returns false because exitCode !== 0.
453
- assert(r.success === false,
454
- 'audit_trend bad-keys path: success === false (iter 44 fix)');
455
- }
456
-
457
- // Cleanup
458
- try { (await import('node:fs')).rmSync(tmp, { recursive: true, force: true }); } catch { /* ignore */ }
459
-
460
- console.log(`\n${passed} passed, ${failed} failed`);
461
- if (failed > 0) {
462
- console.log('\nFailures:');
463
- for (const f of failures) console.log(` - ${f}`);
464
- process.exit(1);
465
- }
466
- console.log('\n✓ All 15 MCP tools satisfy the runtime contract.');
467
- }
468
-
469
- main().catch((e) => {
470
- console.error('test-mcp-tools crashed:', e.message || e);
471
- process.exit(2);
472
- });
1
+ #!/usr/bin/env node
2
+ // test-mcp-tools.mjs — runtime test for the iter-20/21 MCP tool registry.
3
+ //
4
+ // tsc proves the metaharness-tools.ts module COMPILES; structural smoke
5
+ // proves the source DECLARES the right tool names. Neither proves the
6
+ // HANDLERS actually run without throwing. This test imports the compiled
7
+ // module and invokes every tool's handler with minimal inputs.
8
+ //
9
+ // CONTRACT EACH TOOL MUST SATISFY
10
+ // - handler is callable as `await tool.handler({ ... })`
11
+ // - returns an object with keys: success, data, degraded, exitCode
12
+ // - never throws (even with bad/missing optional dep — graceful)
13
+ // - handler honors the 120s subprocess timeout (no hang)
14
+ //
15
+ // USAGE
16
+ // node scripts/test-mcp-tools.mjs # default
17
+ // node scripts/test-mcp-tools.mjs --format json
18
+ //
19
+ // EXIT CODES
20
+ // 0 all tools satisfy the contract
21
+ // 1 at least one tool failed
22
+ // 2 setup error (compiled dist not present)
23
+
24
+ import { existsSync } from 'node:fs';
25
+ import { dirname, join, resolve } from 'node:path';
26
+ import { fileURLToPath } from 'node:url';
27
+
28
+ const SCRIPTS_DIR = dirname(fileURLToPath(import.meta.url));
29
+ const ARGS = (() => {
30
+ const a = { format: 'table' };
31
+ for (let i = 2; i < process.argv.length; i++) {
32
+ if (process.argv[i] === '--format') a.format = process.argv[++i];
33
+ }
34
+ return a;
35
+ })();
36
+
37
+ let passed = 0, failed = 0;
38
+ const failures = [];
39
+
40
+ function assert(cond, label) {
41
+ if (cond) { console.log(` ✓ ${label}`); passed++; }
42
+ else { console.log(` ✗ ${label}`); failures.push(label); failed++; }
43
+ }
44
+
45
+ async function main() {
46
+ // Locate the compiled dist of metaharness-tools.
47
+ const distPath = resolve(SCRIPTS_DIR, '..', '..', '..',
48
+ 'v3', '@claude-flow', 'cli', 'dist', 'src', 'mcp-tools', 'metaharness-tools.js');
49
+
50
+ if (!existsSync(distPath)) {
51
+ console.log(`# test-mcp-tools — SKIPPED`);
52
+ console.log('');
53
+ console.log(`Compiled dist not present: ${distPath}`);
54
+ console.log(`Build the CLI first:`);
55
+ console.log(` cd v3/@claude-flow/cli && npm run build`);
56
+ console.log('');
57
+ console.log(`Exit 0 — this script is meaningfully runnable only post-build.`);
58
+ process.exit(0);
59
+ }
60
+
61
+ let mod;
62
+ try {
63
+ mod = await import(distPath);
64
+ } catch (e) {
65
+ console.error(`test-mcp-tools: failed to import ${distPath}: ${e.message}`);
66
+ process.exit(2);
67
+ }
68
+
69
+ const tools = mod.metaharnessTools;
70
+ console.log(`# test-mcp-tools — runtime contract\n`);
71
+
72
+ // ──────────────────────────────────────────────────────────────────
73
+ // PHASE 1 — module exports the right shape
74
+ // ──────────────────────────────────────────────────────────────────
75
+ console.log('Phase 1 — module shape');
76
+ assert(Array.isArray(tools), 'metaharnessTools is an array');
77
+ assert(tools.length === 15, `15 tools registered (got ${tools.length})`);
78
+
79
+ const expectedNames = new Set([
80
+ 'metaharness_score',
81
+ 'metaharness_genome',
82
+ 'metaharness_mcp_scan',
83
+ 'metaharness_threat_model',
84
+ 'metaharness_oia_audit',
85
+ 'metaharness_audit_list',
86
+ 'metaharness_audit_trend',
87
+ // iter 36 — ADR-152 §3.1 production
88
+ 'metaharness_similarity',
89
+ // iter 54 — one-command drift detection (composes audit-list + oia-audit + audit-trend)
90
+ 'metaharness_drift_from_history',
91
+ // ADR-153 — bench suites + evolve driver + security-focused bench
92
+ 'metaharness_bench',
93
+ 'metaharness_evolve',
94
+ 'metaharness_security_bench',
95
+ // @metaharness/redblue@~0.1.4 — adversarial red/blue LLM testing
96
+ 'metaharness_redblue',
97
+ // metaharness@0.3.0 — upstream ADR-235 GEPA learning run
98
+ 'metaharness_learn',
99
+ // @metaharness/darwin@0.8.0 — GEPA library surface (genome ops)
100
+ 'metaharness_gepa',
101
+ ]);
102
+ const actualNames = new Set(tools.map((t) => t.name));
103
+ for (const name of expectedNames) {
104
+ assert(actualNames.has(name), `${name} registered`);
105
+ }
106
+
107
+ // ──────────────────────────────────────────────────────────────────
108
+ // PHASE 2 — every tool has the required MCP shape
109
+ // ──────────────────────────────────────────────────────────────────
110
+ console.log('\nPhase 2 — per-tool shape');
111
+ for (const tool of tools) {
112
+ const ok = typeof tool.name === 'string'
113
+ && typeof tool.description === 'string'
114
+ && typeof tool.category === 'string'
115
+ && typeof tool.handler === 'function'
116
+ && typeof tool.inputSchema === 'object';
117
+ assert(ok, `${tool.name} has {name, description, category, handler, inputSchema}`);
118
+ assert(tool.category === 'metaharness', `${tool.name} category === 'metaharness'`);
119
+ }
120
+
121
+ // ──────────────────────────────────────────────────────────────────
122
+ // PHASE 3 — handlers callable + return contract shape
123
+ //
124
+ // We invoke each handler with minimal valid input. The handlers may
125
+ // succeed (if metaharness is installed) or report degraded (if not).
126
+ // EITHER way, they must return { success, data, degraded, exitCode }
127
+ // without throwing.
128
+ // ──────────────────────────────────────────────────────────────────
129
+ console.log('\nPhase 3 — handler invocations (allow up to 30s each)');
130
+ for (const tool of tools) {
131
+ // Construct minimal valid input per tool.
132
+ let input = {};
133
+ if (tool.name === 'metaharness_audit_trend') {
134
+ // Requires baselineKey + currentKey — use fake keys that won't
135
+ // resolve so we exercise the not-found path.
136
+ input = { baselineKey: 'audit-fake-base', currentKey: 'audit-fake-curr' };
137
+ }
138
+ if (tool.name === 'metaharness_similarity') {
139
+ // Needs --a/--b OR --a-key/--b-key. Use fake mem keys to exercise
140
+ // the graceful not-found path (matches audit_trend convention).
141
+ input = { aKey: 'harness-fake-a', bKey: 'harness-fake-b' };
142
+ }
143
+ if (tool.name === 'metaharness_drift_from_history') {
144
+ // iter 54 — composes 3 subprocesses, needs more time than the default.
145
+ input = { dryRun: true, threshold: 0.5 };
146
+ }
147
+ if (tool.name === 'metaharness_oia_audit') {
148
+ // iter 128 — composite audit runs 5 sub-audits (oia-manifest +
149
+ // threat-model + mcp-scan + score + genome) in parallel. Each
150
+ // shells out via npx. --dry-run skips memory persistence so the
151
+ // test doesn't pollute namespaces.
152
+ input = { dryRun: true };
153
+ }
154
+ if (tool.name === 'metaharness_redblue') {
155
+ // `attack` preview is the fastest path that exercises the upstream
156
+ // binary without needing OPENROUTER_API_KEY or running any model
157
+ // calls. Count=1 keeps cold-cache npx fetch the dominant cost.
158
+ input = { subcommand: 'attack', family: 'prompt', count: 1 };
159
+ }
160
+ if (tool.name === 'metaharness_learn') {
161
+ // No repo checkout in CI → structured {status:"checkout-required"}
162
+ // exit-0 path. $0: without run=true upstream never spends anyway.
163
+ input = {};
164
+ }
165
+ if (tool.name === 'metaharness_gepa') {
166
+ // op is required; `genome` loads + validates the SHIPPED cand-6
167
+ // genome — pure-local library call once darwin is cached.
168
+ input = { op: 'genome' };
169
+ }
170
+
171
+ // iter 124 → 130 — timeouts have crept up as CI cold-cache npx
172
+ // warmup costs got measured. Final budgets:
173
+ // default : 60s
174
+ // chain-tools : 180s (drift_from_history + oia_audit + audit_list)
175
+ // iter 131 — bumped chain-tool budget 90s → 180s. audit_list still
176
+ // timed out at 90s in CI; locally it runs in ~4s, but CI's
177
+ // `npx @claude-flow/cli@latest memory list` invocation pays both
178
+ // the npx fetch AND a full CLI startup (which loads agentic-flow +
179
+ // ONNX). 180s gives 30x headroom over the local cost.
180
+ const isChainTool = tool.name === 'metaharness_drift_from_history'
181
+ || tool.name === 'metaharness_oia_audit'
182
+ || tool.name === 'metaharness_audit_list'
183
+ // redblue: `attack prompt --count 1` is preview-only (no model
184
+ // calls) but the cold-cache `npx -y @metaharness/redblue@~0.1.4`
185
+ // fetch can take 30-60s. 180s gives 3x headroom.
186
+ || tool.name === 'metaharness_redblue'
187
+ // learn: cold-cache `npx -y metaharness@latest` fetch dominates.
188
+ // gepa: one-time `npm install --prefix ~/.ruflo/darwin-cache-*`
189
+ // fallback install can take 30-60s on cold cache.
190
+ || tool.name === 'metaharness_learn'
191
+ || tool.name === 'metaharness_gepa';
192
+ const timeoutMs = isChainTool ? 180_000 : 60_000;
193
+ const handlerPromise = tool.handler(input);
194
+ const timeoutPromise = new Promise((_, reject) =>
195
+ setTimeout(() => reject(new Error(`${timeoutMs / 1000}s handler timeout`)), timeoutMs));
196
+
197
+ let result;
198
+ let threw = false;
199
+ try {
200
+ result = await Promise.race([handlerPromise, timeoutPromise]);
201
+ } catch (e) {
202
+ threw = true;
203
+ console.log(` [${tool.name}] handler threw: ${e.message.slice(0, 80)}`);
204
+ }
205
+
206
+ assert(!threw, `${tool.name} handler did not throw`);
207
+ if (!threw && result) {
208
+ assert(typeof result === 'object', `${tool.name} returns object`);
209
+ assert('success' in result, `${tool.name} result has 'success'`);
210
+ assert('data' in result, `${tool.name} result has 'data'`);
211
+ assert('degraded' in result, `${tool.name} result has 'degraded'`);
212
+ assert('exitCode' in result, `${tool.name} result has 'exitCode'`);
213
+ }
214
+ }
215
+
216
+ // ──────────────────────────────────────────────────────────────────
217
+ // PHASE 4 — POSITIVE-CASE data-shape validation (iter 43)
218
+ //
219
+ // Iter 37 verified the {success, data, degraded, exitCode} envelope.
220
+ // It did NOT verify that data.X contains the right keys when success
221
+ // is genuinely true — leaving room for iter 42-style bugs where a
222
+ // handler returns valid-looking degraded JSON while silently
223
+ // misrouting input. This phase invokes each handler with VALID
224
+ // inputs and asserts the expected output shape.
225
+ //
226
+ // Tools that depend on `npx metaharness` (score/genome/mcp-scan/
227
+ // threat-model/oia-audit/audit-list/audit-trend) are SKIPPED in this
228
+ // phase when the optional dep isn't installed — they're covered by
229
+ // the no-metaharness-smoke workflow's drill. The similarity tool
230
+ // has no @metaharness/* dep, so its positive case ALWAYS runs.
231
+ // ──────────────────────────────────────────────────────────────────
232
+ console.log('\nPhase 4 — positive-case data shape (iter 43)');
233
+
234
+ const { writeFileSync, mkdtempSync } = await import('node:fs');
235
+ const { tmpdir } = await import('node:os');
236
+ const { join: pjoin } = await import('node:path');
237
+ const tmp = mkdtempSync(pjoin(tmpdir(), 'mcp-positive-'));
238
+
239
+ // metaharness_similarity — full positive case (no @metaharness/* needed)
240
+ const simTool = tools.find((t) => t.name === 'metaharness_similarity');
241
+ if (simTool) {
242
+ const aPath = pjoin(tmp, 'a.json');
243
+ const bPath = pjoin(tmp, 'b.json');
244
+ writeFileSync(aPath, JSON.stringify({
245
+ score: { harnessFit: 78, compileConfidence: 92, taskCoverage: 65, toolSafety: 88, memoryUsefulness: 70, estCostPerRunUsd: 0.04, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:legal' },
246
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','w'], risk_score: 0.45, test_confidence: 0.7, publish_readiness: 0.6 },
247
+ }));
248
+ writeFileSync(bPath, JSON.stringify({
249
+ score: { harnessFit: 75, compileConfidence: 90, taskCoverage: 70, toolSafety: 90, memoryUsefulness: 72, estCostPerRunUsd: 0.05, recommendedMode: 'CLI + MCP', archetype: 'compliance-harness', template: 'vertical:support' },
250
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['x','y','z','q','r'], risk_score: 0.40, test_confidence: 0.75, publish_readiness: 0.65 },
251
+ }));
252
+ const r = await simTool.handler({ aFile: aPath, bFile: bPath });
253
+ assert(r.degraded === false, 'similarity positive case: degraded === false');
254
+ assert(r.success === true, 'similarity positive case: success === true');
255
+ assert(r.exitCode === 0, 'similarity positive case: exitCode === 0');
256
+ const d = r.data ?? {};
257
+ assert(typeof d.overall === 'number', 'similarity data has numeric `overall`');
258
+ assert(typeof d.components === 'object' && d.components !== null,
259
+ 'similarity data has `components` object');
260
+ assert(typeof d.components?.cosine === 'number',
261
+ 'similarity components.cosine numeric');
262
+ assert(typeof d.components?.categorical === 'number',
263
+ 'similarity components.categorical numeric');
264
+ assert(typeof d.components?.jaccard === 'number',
265
+ 'similarity components.jaccard numeric');
266
+ assert(typeof d.weights === 'object' && d.weights !== null,
267
+ 'similarity data has `weights` object');
268
+ assert(d.adr === 'ADR-152', 'similarity data tagged adr=ADR-152');
269
+ // Regression anchor — same fixtures as iter-35 spike with non-matching topologies
270
+ assert(d.overall > 0 && d.overall < 1,
271
+ `similarity overall in (0, 1) — got ${d.overall}`);
272
+
273
+ // Per-dimension variant
274
+ const rPD = await simTool.handler({ aFile: aPath, bFile: bPath, perDimension: true });
275
+ assert(typeof rPD.data?.perDimension === 'object',
276
+ 'similarity perDimension=true populates breakdown');
277
+
278
+ // Alert-below variant exercises non-zero exit
279
+ const rAlert = await simTool.handler({ aFile: aPath, bFile: bPath, alertBelow: 0.99 });
280
+ assert(rAlert.data?.alert?.triggered === true,
281
+ 'similarity alertBelow=0.99 triggers alert');
282
+ assert(rAlert.exitCode === 1, 'similarity alertBelow=0.99 → exitCode 1');
283
+ // iter 44 — success semantic anchor (was true under the pre-iter-44
284
+ // `!degraded` rule; now false because exitCode !== 0 dominates).
285
+ assert(rAlert.success === false,
286
+ 'similarity alertBelow=0.99 → success === false (iter 44 fix)');
287
+ }
288
+
289
+ // metaharness_mcp_scan — positive case post iter-50 parser landing.
290
+ // Until iter 50, mcp_scan's data field was an alert-only object with
291
+ // no structured findings. After iter 50, findings[] is always present
292
+ // (parsed from upstream text) and summary{overallSeverity, totalCount}
293
+ // accompanies it.
294
+ const scanTool = tools.find((t) => t.name === 'metaharness_mcp_scan');
295
+ if (scanTool) {
296
+ // Run against ruflo itself — guaranteed to produce at least the
297
+ // INFO finding the iter-50 parser test verified manually.
298
+ const r = await scanTool.handler({ path: '.', failOn: 'high' });
299
+ // Either succeeds with structured findings, or gracefully degrades
300
+ // if metaharness isn't installed in this environment.
301
+ if (!r.degraded) {
302
+ assert(r.success === true, 'mcp_scan positive: success === true');
303
+ assert(r.exitCode === 0, 'mcp_scan positive: exitCode === 0');
304
+ assert(Array.isArray(r.data?.findings),
305
+ 'mcp_scan positive: data.findings is an array (iter 50 fix)');
306
+ // Cwd-dependent: when scanning a dir without .mcp/servers.json the
307
+ // upstream emits no findings. Only verify shape contract when array
308
+ // is populated — the array-presence assertion above is the
309
+ // load-bearing one for iter 50.
310
+ if (r.data?.findings.length > 0) {
311
+ const first = r.data.findings[0];
312
+ assert(typeof first?.severity === 'string',
313
+ 'mcp_scan positive: first finding has string severity');
314
+ assert(typeof first?.message === 'string',
315
+ 'mcp_scan positive: first finding has string message');
316
+ }
317
+ // summary may be null if the upstream produced no Result: line —
318
+ // verify the field's presence (null OR object) but only deep-check
319
+ // when populated.
320
+ if (r.data?.summary) {
321
+ assert(typeof r.data.summary.totalCount === 'number',
322
+ 'mcp_scan positive: data.summary.totalCount is numeric (when summary present)');
323
+ }
324
+ } else {
325
+ console.log(` ⊘ mcp_scan: metaharness absent — graceful skip`);
326
+ }
327
+ }
328
+
329
+ // metaharness_audit_trend — positive case via file inputs
330
+ const trendTool = tools.find((t) => t.name === 'metaharness_audit_trend');
331
+ if (trendTool) {
332
+ const basePath = pjoin(tmp, 'base.json');
333
+ const currPath = pjoin(tmp, 'curr.json');
334
+ const fingerprint = {
335
+ score: { harnessFit: 80, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
336
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['a','b','c'], risk_score: 0.3, test_confidence: 0.85, publish_readiness: 0.9 },
337
+ };
338
+ writeFileSync(basePath, JSON.stringify({
339
+ startedAt: '2026-06-15T00:00:00Z',
340
+ composite: { worst: 'clean' },
341
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
342
+ fingerprint,
343
+ }));
344
+ writeFileSync(currPath, JSON.stringify({
345
+ startedAt: '2026-06-16T00:00:00Z',
346
+ composite: { worst: 'clean' },
347
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
348
+ fingerprint,
349
+ }));
350
+ // audit_trend tool only supports keys, not files at the MCP layer.
351
+ // Document its actual wrapper semantics so future-us doesn't get
352
+ // surprised:
353
+ // - bad keys → script exits 2 with stderr (no JSON payload)
354
+ // - runScript() can't parse a {degraded:true} marker, so it
355
+ // returns degraded:false / success:true / exitCode:2
356
+ // This is a real wrapper bug (success should not be true when
357
+ // exit!=0 AND no JSON came back), tracked separately. Asserting
358
+ // current behavior here protects against silent semantic shifts.
359
+ // iter 46 — file-input path. audit_trend now accepts baselineFile/currentFile.
360
+ const rFiles = await trendTool.handler({ baselineFile: basePath, currentFile: currPath });
361
+ assert(rFiles.success === true,
362
+ 'audit_trend file-input path: success === true (iter 46)');
363
+ assert(rFiles.exitCode === 0, 'audit_trend file-input path: exitCode === 0');
364
+ assert(typeof rFiles.data?.delta === 'object',
365
+ 'audit_trend file-input path: data.delta object present');
366
+ assert(rFiles.data?.delta?.structuralDistance?.verdict === 'near-identical',
367
+ `audit_trend file-input path: identical fingerprints → near-identical (got ${rFiles.data?.delta?.structuralDistance?.verdict})`);
368
+
369
+ // iter 54 — metaharness_drift_from_history positive case
370
+ const driftTool = tools.find((t) => t.name === 'metaharness_drift_from_history');
371
+ if (driftTool) {
372
+ // iter 71 — verify iter-66/67 fast-path flags are now MCP-callable
373
+ // Synthesize a baseline file on disk; pass via the new baselineFile input.
374
+ const baselinePath = pjoin(tmp, 'drift-baseline.json');
375
+ writeFileSync(baselinePath, JSON.stringify({
376
+ startedAt: '2026-06-16T00:00:00Z',
377
+ composite: { worst: 'clean' },
378
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
379
+ fingerprint: {
380
+ score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
381
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
382
+ },
383
+ }));
384
+ const rFastFast = await driftTool.handler({
385
+ path: '.', dryRun: true, threshold: 0.5, baselineFile: baselinePath,
386
+ });
387
+ if (!rFastFast.degraded) {
388
+ assert(rFastFast.data?.timing?.usedBaselineFile === true,
389
+ 'drift_from_history MCP-layer: baselineFile fastpath fires (iter 71)');
390
+ assert(rFastFast.data?.timing?.skippedAuditList === true,
391
+ 'drift_from_history MCP-layer: skippedAuditList=true via baselineFile (iter 71)');
392
+ }
393
+
394
+ // iter 85 — verify iter-78's alertOnNewSeverity MCP input plumbs
395
+ // through. baselineFile has no findings; current ruflo audit has
396
+ // 1 INFO finding. With alertOnNewSeverity='info' the gate fires
397
+ // and surfaces in the response.
398
+ const baselineNoFindings = pjoin(tmp, 'drift-baseline-no-findings.json');
399
+ writeFileSync(baselineNoFindings, JSON.stringify({
400
+ startedAt: '2026-06-16T00:00:00Z',
401
+ composite: { worst: 'clean' },
402
+ components: { oiaManifest: {}, threatModel: {}, mcpScan: { json: { findings: [] } } },
403
+ fingerprint: {
404
+ score: { harnessFit: 82, recommendedMode: 'CLI + MCP', archetype: 'typescript-sdk-harness', template: 'vertical:coding' },
405
+ genome: { repo_type: 'node_mcp_ci', agent_topology: ['m', 't'], risk_score: 0.3 },
406
+ },
407
+ }));
408
+ const rSevAlert = await driftTool.handler({
409
+ path: '.', dryRun: true, threshold: 0.5,
410
+ baselineFile: baselineNoFindings,
411
+ alertOnNewSeverity: 'info',
412
+ });
413
+ if (!rSevAlert.degraded) {
414
+ assert(rSevAlert.data?.alert?.newSeverityThreshold === 'info',
415
+ 'drift_from_history MCP-layer: alertOnNewSeverity echoed in payload (iter 85)');
416
+ // Triggered AND exit code reflects (only if the audit actually had findings)
417
+ if (rSevAlert.data?.alert?.triggered === true) {
418
+ assert(rSevAlert.exitCode === 1,
419
+ `drift_from_history MCP-layer: alertOnNewSeverity exitCode=1 when triggered (got ${rSevAlert.exitCode})`);
420
+ assert(rSevAlert.success === false,
421
+ 'drift_from_history MCP-layer: success===false when alert fires (iter 44 fix)');
422
+ }
423
+ }
424
+
425
+ const r54 = await driftTool.handler({ path: '.', dryRun: true, threshold: 0.5 });
426
+ if (!r54.degraded) {
427
+ assert(typeof r54.data === 'object' && r54.data !== null,
428
+ 'drift_from_history positive: data is an object');
429
+ // Either it produced the structured drift report OR the no-history error
430
+ const isOk = r54.data?.command === 'drift-from-history';
431
+ const isNoHistory = typeof r54.data?.error === 'string' && r54.data.error.includes('no audit records');
432
+ assert(isOk || isNoHistory,
433
+ `drift_from_history positive: structured report OR no-history error (got ${JSON.stringify(r54.data).slice(0,80)})`);
434
+ if (isOk) {
435
+ assert(typeof r54.data.baseline?.key === 'string',
436
+ 'drift_from_history: baseline.key is a string');
437
+ assert(typeof r54.data.alert?.threshold === 'number',
438
+ 'drift_from_history: alert.threshold echoed numerically');
439
+ }
440
+ } else {
441
+ console.log(` ⊘ drift_from_history: degraded (metaharness or memory absent)`);
442
+ }
443
+ }
444
+
445
+ const r = await trendTool.handler({ baselineKey: 'missing-X', currentKey: 'missing-Y' });
446
+ assert(r.exitCode === 2,
447
+ 'audit_trend bad-keys path exits 2 (script-level guard fires)');
448
+ assert(r.data === null || r.data === undefined,
449
+ 'audit_trend bad-keys path: data null (no JSON emitted on stderr exit)');
450
+ // iter 44 — success semantic anchor. Pre-iter-44 wrapper returned
451
+ // success:true for this case (because no degraded marker). Now
452
+ // returns false because exitCode !== 0.
453
+ assert(r.success === false,
454
+ 'audit_trend bad-keys path: success === false (iter 44 fix)');
455
+ }
456
+
457
+ // Cleanup
458
+ try { (await import('node:fs')).rmSync(tmp, { recursive: true, force: true }); } catch { /* ignore */ }
459
+
460
+ console.log(`\n${passed} passed, ${failed} failed`);
461
+ if (failed > 0) {
462
+ console.log('\nFailures:');
463
+ for (const f of failures) console.log(` - ${f}`);
464
+ process.exit(1);
465
+ }
466
+ console.log('\n✓ All 15 MCP tools satisfy the runtime contract.');
467
+ }
468
+
469
+ main().catch((e) => {
470
+ console.error('test-mcp-tools crashed:', e.message || e);
471
+ process.exit(2);
472
+ });