@claude-flow/cli 3.32.9 → 3.32.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (447) hide show
  1. package/.claude/.proven-config-version +1 -0
  2. package/.claude/agents/analysis/analyze-code-quality.md +178 -178
  3. package/.claude/agents/analysis/code-analyzer.md +209 -209
  4. package/.claude/agents/analysis/code-review/analyze-code-quality.md +178 -178
  5. package/.claude/agents/architecture/arch-system-design.md +156 -156
  6. package/.claude/agents/architecture/system-design/arch-system-design.md +154 -154
  7. package/.claude/agents/browser/browser-agent.yaml +182 -182
  8. package/.claude/agents/consensus/byzantine-coordinator.md +62 -62
  9. package/.claude/agents/consensus/crdt-synchronizer.md +996 -996
  10. package/.claude/agents/consensus/gossip-coordinator.md +62 -62
  11. package/.claude/agents/consensus/performance-benchmarker.md +850 -850
  12. package/.claude/agents/consensus/quorum-manager.md +822 -822
  13. package/.claude/agents/consensus/raft-manager.md +62 -62
  14. package/.claude/agents/consensus/security-manager.md +621 -621
  15. package/.claude/agents/core/planner.md +374 -374
  16. package/.claude/agents/custom/test-long-runner.md +44 -44
  17. package/.claude/agents/data/data-ml-model.md +444 -444
  18. package/.claude/agents/data/ml/data-ml-model.md +192 -192
  19. package/.claude/agents/development/backend/dev-backend-api.md +141 -141
  20. package/.claude/agents/development/dev-backend-api.md +344 -344
  21. package/.claude/agents/devops/ci-cd/ops-cicd-github.md +163 -163
  22. package/.claude/agents/devops/ops-cicd-github.md +164 -164
  23. package/.claude/agents/documentation/api-docs/docs-api-openapi.md +173 -173
  24. package/.claude/agents/documentation/docs-api-openapi.md +354 -354
  25. package/.claude/agents/flow-nexus/app-store.md +87 -87
  26. package/.claude/agents/flow-nexus/authentication.md +68 -68
  27. package/.claude/agents/flow-nexus/challenges.md +80 -80
  28. package/.claude/agents/flow-nexus/neural-network.md +87 -87
  29. package/.claude/agents/flow-nexus/payments.md +82 -82
  30. package/.claude/agents/flow-nexus/sandbox.md +75 -75
  31. package/.claude/agents/flow-nexus/swarm.md +75 -75
  32. package/.claude/agents/flow-nexus/user-tools.md +95 -95
  33. package/.claude/agents/flow-nexus/workflow.md +83 -83
  34. package/.claude/agents/github/code-review-swarm.md +377 -377
  35. package/.claude/agents/github/github-modes.md +172 -172
  36. package/.claude/agents/github/issue-tracker.md +575 -575
  37. package/.claude/agents/github/multi-repo-swarm.md +552 -552
  38. package/.claude/agents/github/pr-manager.md +437 -437
  39. package/.claude/agents/github/project-board-sync.md +508 -508
  40. package/.claude/agents/github/release-manager.md +604 -604
  41. package/.claude/agents/github/release-swarm.md +582 -582
  42. package/.claude/agents/github/repo-architect.md +397 -397
  43. package/.claude/agents/github/swarm-issue.md +572 -572
  44. package/.claude/agents/github/swarm-pr.md +427 -427
  45. package/.claude/agents/github/sync-coordinator.md +451 -451
  46. package/.claude/agents/github/workflow-automation.md +902 -902
  47. package/.claude/agents/goal/agent.md +815 -815
  48. package/.claude/agents/optimization/benchmark-suite.md +664 -664
  49. package/.claude/agents/optimization/load-balancer.md +430 -430
  50. package/.claude/agents/optimization/performance-monitor.md +671 -671
  51. package/.claude/agents/optimization/resource-allocator.md +673 -673
  52. package/.claude/agents/optimization/topology-optimizer.md +807 -807
  53. package/.claude/agents/payments/agentic-payments.md +126 -126
  54. package/.claude/agents/sona/sona-learning-optimizer.md +74 -74
  55. package/.claude/agents/sparc/architecture.md +698 -698
  56. package/.claude/agents/sparc/pseudocode.md +519 -519
  57. package/.claude/agents/sparc/refinement.md +801 -801
  58. package/.claude/agents/sparc/specification.md +477 -477
  59. package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +224 -224
  60. package/.claude/agents/specialized/spec-mobile-react-native.md +226 -226
  61. package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
  62. package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
  63. package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
  64. package/.claude/agents/sublinear/performance-optimizer.md +367 -367
  65. package/.claude/agents/sublinear/trading-predictor.md +245 -245
  66. package/.claude/agents/swarm/adaptive-coordinator.md +1126 -1126
  67. package/.claude/agents/swarm/hierarchical-coordinator.md +709 -709
  68. package/.claude/agents/swarm/mesh-coordinator.md +962 -962
  69. package/.claude/agents/templates/automation-smart-agent.md +204 -204
  70. package/.claude/agents/templates/base-template-generator.md +289 -289
  71. package/.claude/agents/templates/coordinator-swarm-init.md +89 -89
  72. package/.claude/agents/templates/github-pr-manager.md +176 -176
  73. package/.claude/agents/templates/implementer-sparc-coder.md +258 -258
  74. package/.claude/agents/templates/memory-coordinator.md +186 -186
  75. package/.claude/agents/templates/orchestrator-task.md +138 -138
  76. package/.claude/agents/templates/performance-analyzer.md +198 -198
  77. package/.claude/agents/templates/sparc-coordinator.md +513 -513
  78. package/.claude/agents/testing/production-validator.md +394 -394
  79. package/.claude/agents/testing/tdd-london-swarm.md +243 -243
  80. package/.claude/agents/v3/aidefence-guardian.md +282 -282
  81. package/.claude/agents/v3/claims-authorizer.md +208 -208
  82. package/.claude/agents/v3/collective-intelligence-coordinator.md +993 -993
  83. package/.claude/agents/v3/ddd-domain-expert.md +220 -220
  84. package/.claude/agents/v3/injection-analyst.md +236 -236
  85. package/.claude/agents/v3/performance-engineer.md +1233 -1233
  86. package/.claude/agents/v3/pii-detector.md +151 -151
  87. package/.claude/agents/v3/reasoningbank-learner.md +213 -213
  88. package/.claude/agents/v3/security-architect-aidefence.md +410 -410
  89. package/.claude/agents/v3/security-architect.md +867 -867
  90. package/.claude/agents/v3/swarm-memory-manager.md +157 -157
  91. package/.claude/agents/v3/v3-integration-architect.md +205 -205
  92. package/.claude/commands/agents/README.md +50 -50
  93. package/.claude/commands/agents/agent-capabilities.md +140 -140
  94. package/.claude/commands/agents/agent-coordination.md +28 -28
  95. package/.claude/commands/agents/agent-spawning.md +28 -28
  96. package/.claude/commands/agents/agent-types.md +216 -216
  97. package/.claude/commands/agents/health.md +139 -139
  98. package/.claude/commands/agents/list.md +100 -100
  99. package/.claude/commands/agents/logs.md +130 -130
  100. package/.claude/commands/agents/metrics.md +122 -122
  101. package/.claude/commands/agents/pool.md +127 -127
  102. package/.claude/commands/agents/spawn.md +140 -140
  103. package/.claude/commands/agents/status.md +115 -115
  104. package/.claude/commands/agents/stop.md +102 -102
  105. package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
  106. package/.claude/commands/analysis/README.md +9 -9
  107. package/.claude/commands/analysis/bottleneck-detect.md +162 -162
  108. package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
  109. package/.claude/commands/analysis/performance-report.md +25 -25
  110. package/.claude/commands/analysis/token-efficiency.md +44 -44
  111. package/.claude/commands/analysis/token-usage.md +25 -25
  112. package/.claude/commands/automation/README.md +9 -9
  113. package/.claude/commands/automation/auto-agent.md +122 -122
  114. package/.claude/commands/automation/self-healing.md +105 -105
  115. package/.claude/commands/automation/session-memory.md +89 -89
  116. package/.claude/commands/automation/smart-agents.md +72 -72
  117. package/.claude/commands/automation/smart-spawn.md +25 -25
  118. package/.claude/commands/automation/workflow-select.md +25 -25
  119. package/.claude/commands/claude-flow-help.md +103 -103
  120. package/.claude/commands/claude-flow-memory.md +107 -107
  121. package/.claude/commands/claude-flow-swarm.md +205 -205
  122. package/.claude/commands/coordination/README.md +9 -9
  123. package/.claude/commands/coordination/agent-spawn.md +25 -25
  124. package/.claude/commands/coordination/init.md +44 -44
  125. package/.claude/commands/coordination/orchestrate.md +43 -43
  126. package/.claude/commands/coordination/spawn.md +45 -45
  127. package/.claude/commands/coordination/swarm-init.md +85 -85
  128. package/.claude/commands/coordination/task-orchestrate.md +25 -25
  129. package/.claude/commands/github/README.md +11 -11
  130. package/.claude/commands/github/code-review-swarm.md +513 -513
  131. package/.claude/commands/github/code-review.md +25 -25
  132. package/.claude/commands/github/github-modes.md +146 -146
  133. package/.claude/commands/github/github-swarm.md +121 -121
  134. package/.claude/commands/github/issue-tracker.md +291 -291
  135. package/.claude/commands/github/issue-triage.md +25 -25
  136. package/.claude/commands/github/multi-repo-swarm.md +518 -518
  137. package/.claude/commands/github/pr-enhance.md +26 -26
  138. package/.claude/commands/github/pr-manager.md +169 -169
  139. package/.claude/commands/github/project-board-sync.md +470 -470
  140. package/.claude/commands/github/release-manager.md +339 -339
  141. package/.claude/commands/github/release-swarm.md +543 -543
  142. package/.claude/commands/github/repo-analyze.md +25 -25
  143. package/.claude/commands/github/repo-architect.md +366 -366
  144. package/.claude/commands/github/swarm-issue.md +484 -484
  145. package/.claude/commands/github/swarm-pr.md +287 -287
  146. package/.claude/commands/github/sync-coordinator.md +302 -302
  147. package/.claude/commands/github/workflow-automation.md +441 -441
  148. package/.claude/commands/hive-mind/README.md +17 -17
  149. package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
  150. package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
  151. package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
  152. package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
  153. package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
  154. package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
  155. package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
  156. package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
  157. package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
  158. package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
  159. package/.claude/commands/hive-mind/hive-mind.md +27 -27
  160. package/.claude/commands/hooks/README.md +11 -11
  161. package/.claude/commands/hooks/overview.md +57 -57
  162. package/.claude/commands/hooks/post-edit.md +117 -117
  163. package/.claude/commands/hooks/post-task.md +112 -112
  164. package/.claude/commands/hooks/pre-edit.md +113 -113
  165. package/.claude/commands/hooks/pre-task.md +111 -111
  166. package/.claude/commands/hooks/session-end.md +118 -118
  167. package/.claude/commands/hooks/setup.md +102 -102
  168. package/.claude/commands/memory/README.md +9 -9
  169. package/.claude/commands/memory/memory-persist.md +25 -25
  170. package/.claude/commands/memory/memory-search.md +25 -25
  171. package/.claude/commands/memory/memory-usage.md +25 -25
  172. package/.claude/commands/memory/neural.md +47 -47
  173. package/.claude/commands/monitoring/README.md +9 -9
  174. package/.claude/commands/monitoring/agent-metrics.md +25 -25
  175. package/.claude/commands/monitoring/agents.md +44 -44
  176. package/.claude/commands/monitoring/real-time-view.md +25 -25
  177. package/.claude/commands/monitoring/status.md +46 -46
  178. package/.claude/commands/monitoring/swarm-monitor.md +25 -25
  179. package/.claude/commands/optimization/README.md +9 -9
  180. package/.claude/commands/optimization/auto-topology.md +61 -61
  181. package/.claude/commands/optimization/cache-manage.md +25 -25
  182. package/.claude/commands/optimization/parallel-execute.md +25 -25
  183. package/.claude/commands/optimization/parallel-execution.md +49 -49
  184. package/.claude/commands/optimization/topology-optimize.md +25 -25
  185. package/.claude/commands/pair/README.md +260 -260
  186. package/.claude/commands/pair/commands.md +545 -545
  187. package/.claude/commands/pair/config.md +509 -509
  188. package/.claude/commands/pair/examples.md +511 -511
  189. package/.claude/commands/pair/modes.md +347 -347
  190. package/.claude/commands/pair/session.md +406 -406
  191. package/.claude/commands/pair/start.md +208 -208
  192. package/.claude/commands/sparc/analyzer.md +51 -51
  193. package/.claude/commands/sparc/architect.md +53 -53
  194. package/.claude/commands/sparc/ask.md +97 -97
  195. package/.claude/commands/sparc/batch-executor.md +54 -54
  196. package/.claude/commands/sparc/code.md +89 -89
  197. package/.claude/commands/sparc/coder.md +54 -54
  198. package/.claude/commands/sparc/debug.md +83 -83
  199. package/.claude/commands/sparc/debugger.md +54 -54
  200. package/.claude/commands/sparc/designer.md +53 -53
  201. package/.claude/commands/sparc/devops.md +109 -109
  202. package/.claude/commands/sparc/docs-writer.md +80 -80
  203. package/.claude/commands/sparc/documenter.md +54 -54
  204. package/.claude/commands/sparc/innovator.md +54 -54
  205. package/.claude/commands/sparc/integration.md +83 -83
  206. package/.claude/commands/sparc/mcp.md +117 -117
  207. package/.claude/commands/sparc/memory-manager.md +54 -54
  208. package/.claude/commands/sparc/optimizer.md +54 -54
  209. package/.claude/commands/sparc/orchestrator.md +131 -131
  210. package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
  211. package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
  212. package/.claude/commands/sparc/researcher.md +54 -54
  213. package/.claude/commands/sparc/reviewer.md +54 -54
  214. package/.claude/commands/sparc/security-review.md +80 -80
  215. package/.claude/commands/sparc/sparc-modes.md +174 -174
  216. package/.claude/commands/sparc/sparc.md +111 -111
  217. package/.claude/commands/sparc/spec-pseudocode.md +80 -80
  218. package/.claude/commands/sparc/supabase-admin.md +348 -348
  219. package/.claude/commands/sparc/swarm-coordinator.md +54 -54
  220. package/.claude/commands/sparc/tdd.md +54 -54
  221. package/.claude/commands/sparc/tester.md +54 -54
  222. package/.claude/commands/sparc/tutorial.md +79 -79
  223. package/.claude/commands/sparc/workflow-manager.md +54 -54
  224. package/.claude/commands/sparc.md +166 -166
  225. package/.claude/commands/stream-chain/pipeline.md +120 -120
  226. package/.claude/commands/stream-chain/run.md +69 -69
  227. package/.claude/commands/swarm/README.md +15 -15
  228. package/.claude/commands/swarm/analysis.md +95 -95
  229. package/.claude/commands/swarm/development.md +96 -96
  230. package/.claude/commands/swarm/examples.md +168 -168
  231. package/.claude/commands/swarm/maintenance.md +102 -102
  232. package/.claude/commands/swarm/optimization.md +117 -117
  233. package/.claude/commands/swarm/research.md +136 -136
  234. package/.claude/commands/swarm/swarm-analysis.md +8 -8
  235. package/.claude/commands/swarm/swarm-background.md +8 -8
  236. package/.claude/commands/swarm/swarm-init.md +19 -19
  237. package/.claude/commands/swarm/swarm-modes.md +8 -8
  238. package/.claude/commands/swarm/swarm-monitor.md +8 -8
  239. package/.claude/commands/swarm/swarm-spawn.md +19 -19
  240. package/.claude/commands/swarm/swarm-status.md +8 -8
  241. package/.claude/commands/swarm/swarm-strategies.md +8 -8
  242. package/.claude/commands/swarm/swarm.md +87 -87
  243. package/.claude/commands/swarm/testing.md +131 -131
  244. package/.claude/commands/training/README.md +9 -9
  245. package/.claude/commands/training/model-update.md +25 -25
  246. package/.claude/commands/training/neural-patterns.md +107 -107
  247. package/.claude/commands/training/neural-train.md +75 -75
  248. package/.claude/commands/training/pattern-learn.md +25 -25
  249. package/.claude/commands/training/specialization.md +62 -62
  250. package/.claude/commands/truth/start.md +142 -142
  251. package/.claude/commands/verify/check.md +49 -49
  252. package/.claude/commands/verify/start.md +127 -127
  253. package/.claude/commands/workflows/README.md +9 -9
  254. package/.claude/commands/workflows/development.md +77 -77
  255. package/.claude/commands/workflows/research.md +62 -62
  256. package/.claude/commands/workflows/workflow-create.md +25 -25
  257. package/.claude/commands/workflows/workflow-execute.md +25 -25
  258. package/.claude/commands/workflows/workflow-export.md +25 -25
  259. package/.claude/eval/human-relevance-frozen-v1.json +17 -17
  260. package/.claude/evolve-proof/generation-0.json +211 -211
  261. package/.claude/evolve-proof/real-generation-0.json +406 -406
  262. package/.claude/evolve-proof/real-generation-1.json +406 -406
  263. package/.claude/helpers/.helpers-version +1 -1
  264. package/.claude/helpers/README.md +96 -96
  265. package/.claude/helpers/adr-compliance.sh +186 -186
  266. package/.claude/helpers/auto-commit.sh +178 -178
  267. package/.claude/helpers/auto-memory-hook.mjs +0 -0
  268. package/.claude/helpers/checkpoint-manager.sh +251 -251
  269. package/.claude/helpers/daemon-manager.sh +252 -252
  270. package/.claude/helpers/ddd-tracker.sh +144 -144
  271. package/.claude/helpers/github-safe.js +156 -156
  272. package/.claude/helpers/github-setup.sh +45 -45
  273. package/.claude/helpers/guidance-hook.sh +13 -13
  274. package/.claude/helpers/guidance-hooks.sh +102 -102
  275. package/.claude/helpers/health-monitor.sh +108 -108
  276. package/.claude/helpers/helpers.manifest.json +2 -2
  277. package/.claude/helpers/hook-handler.cjs +0 -0
  278. package/.claude/helpers/intelligence.cjs +0 -0
  279. package/.claude/helpers/learning-hooks.sh +329 -329
  280. package/.claude/helpers/learning-optimizer.sh +127 -127
  281. package/.claude/helpers/learning-service.mjs +1144 -1144
  282. package/.claude/helpers/memory.js +83 -83
  283. package/.claude/helpers/metrics-db.mjs +503 -503
  284. package/.claude/helpers/pattern-consolidator.sh +86 -86
  285. package/.claude/helpers/perf-worker.sh +160 -160
  286. package/.claude/helpers/post-commit +16 -16
  287. package/.claude/helpers/pre-commit +26 -26
  288. package/.claude/helpers/quick-start.sh +19 -19
  289. package/.claude/helpers/router.js +105 -105
  290. package/.claude/helpers/security-scanner.sh +127 -127
  291. package/.claude/helpers/session.js +157 -157
  292. package/.claude/helpers/setup-mcp.sh +18 -18
  293. package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
  294. package/.claude/helpers/statusline-hook.sh +21 -21
  295. package/.claude/helpers/statusline.cjs +0 -0
  296. package/.claude/helpers/statusline.js +340 -340
  297. package/.claude/helpers/swarm-comms.sh +353 -353
  298. package/.claude/helpers/swarm-hooks.sh +761 -761
  299. package/.claude/helpers/swarm-monitor.sh +210 -210
  300. package/.claude/helpers/sync-v3-metrics.sh +245 -245
  301. package/.claude/helpers/update-v3-progress.sh +165 -165
  302. package/.claude/helpers/v3-quick-status.sh +57 -57
  303. package/.claude/helpers/v3.sh +110 -110
  304. package/.claude/helpers/validate-v3-config.sh +215 -215
  305. package/.claude/helpers/worker-manager.sh +170 -170
  306. package/.claude/proven-config.json +42 -0
  307. package/.claude/proven-config.manifest.json +37 -37
  308. package/.claude/proven-config.signed.json +41 -41
  309. package/.claude/settings.json +182 -182
  310. package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
  311. package/.claude/skills/agentdb-learning/SKILL.md +545 -545
  312. package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
  313. package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
  314. package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
  315. package/.claude/skills/browser/SKILL.md +204 -204
  316. package/.claude/skills/dual-mode/README.md +71 -71
  317. package/.claude/skills/dual-mode/dual-collect.md +103 -103
  318. package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
  319. package/.claude/skills/dual-mode/dual-spawn.md +81 -81
  320. package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
  321. package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
  322. package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
  323. package/.claude/skills/github-code-review/SKILL.md +1125 -1125
  324. package/.claude/skills/github-multi-repo/SKILL.md +862 -862
  325. package/.claude/skills/github-project-management/SKILL.md +1262 -1262
  326. package/.claude/skills/github-release-management/SKILL.md +1064 -1064
  327. package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
  328. package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
  329. package/.claude/skills/pair-programming/SKILL.md +1202 -1202
  330. package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
  331. package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
  332. package/.claude/skills/skill-builder/SKILL.md +910 -910
  333. package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
  334. package/.claude/skills/stream-chain/SKILL.md +560 -560
  335. package/.claude/skills/swarm-advanced/SKILL.md +970 -970
  336. package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
  337. package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
  338. package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
  339. package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
  340. package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
  341. package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
  342. package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
  343. package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
  344. package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
  345. package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
  346. package/.claude/skills/verification-quality/SKILL.md +691 -691
  347. package/README.md +419 -419
  348. package/bin/cli.js +314 -314
  349. package/bin/mcp-server.js +224 -224
  350. package/bin/preinstall.cjs +2 -2
  351. package/catalog-manifest.json +2 -2
  352. package/dist/src/autopilot-state.js +24 -7
  353. package/dist/src/benchmarks/gaia-critic.js +24 -24
  354. package/dist/src/business-pods/bbs-budget-tracker.js +53 -53
  355. package/dist/src/commands/completions.js +409 -409
  356. package/dist/src/commands/daemon.js +44 -44
  357. package/dist/src/commands/embeddings.js +26 -26
  358. package/dist/src/commands/hive-mind.js +97 -97
  359. package/dist/src/commands/hooks.js +31 -10
  360. package/dist/src/commands/init.js +202 -34
  361. package/dist/src/commands/memory.js +12 -1
  362. package/dist/src/commands/ruvector/backup.js +23 -23
  363. package/dist/src/commands/ruvector/benchmark.js +31 -31
  364. package/dist/src/commands/ruvector/import.js +14 -14
  365. package/dist/src/commands/ruvector/init.js +115 -115
  366. package/dist/src/commands/ruvector/migrate.js +99 -99
  367. package/dist/src/commands/ruvector/optimize.js +51 -51
  368. package/dist/src/commands/ruvector/setup.js +624 -624
  369. package/dist/src/commands/ruvector/status.js +38 -38
  370. package/dist/src/config/proven-config.js +2 -2
  371. package/dist/src/funnel/disclosure.js +13 -2
  372. package/dist/src/funnel/messages.d.ts +12 -10
  373. package/dist/src/funnel/messages.js +83 -11
  374. package/dist/src/init/claudemd-generator.js +231 -231
  375. package/dist/src/init/executor.js +453 -453
  376. package/dist/src/init/helper-signing.js +2 -2
  377. package/dist/src/init/helpers-generator.js +751 -751
  378. package/dist/src/init/statusline-generator.js +24 -24
  379. package/dist/src/mcp-tools/agentdb-tools.js +15 -15
  380. package/dist/src/mcp-tools/browser-intent-tools.js +19 -19
  381. package/dist/src/mcp-tools/browser-tools.js +8 -0
  382. package/dist/src/mcp-tools/hooks-tools.js +21 -0
  383. package/dist/src/mcp-tools/memory-tools.js +4 -3
  384. package/dist/src/memory/graph-edge-writer.js +22 -22
  385. package/dist/src/memory/memory-bridge.js +248 -158
  386. package/dist/src/memory/memory-initializer.js +407 -407
  387. package/dist/src/memory/rabitq-index.js +5 -5
  388. package/dist/src/parser.js +25 -9
  389. package/dist/src/proxy/verify.js +2 -2
  390. package/dist/src/runtime/headless.js +28 -28
  391. package/dist/src/services/distill-tuning.js +7 -7
  392. package/dist/src/services/headless-worker-executor.js +84 -84
  393. package/dist/src/services/memory-distillation.js +4 -4
  394. package/dist/src/services/worker-daemon.js +7 -4
  395. package/dist/src/transfer/deploy-seraphine.js +23 -23
  396. package/package.json +137 -137
  397. package/plugins/ruflo-metaharness/.claude-plugin/plugin.json +32 -32
  398. package/plugins/ruflo-metaharness/README.md +72 -72
  399. package/plugins/ruflo-metaharness/agents/metaharness-architect.md +58 -58
  400. package/plugins/ruflo-metaharness/commands/ruflo-metaharness.md +48 -48
  401. package/plugins/ruflo-metaharness/scripts/_darwin.mjs +210 -210
  402. package/plugins/ruflo-metaharness/scripts/_harness.mjs +330 -330
  403. package/plugins/ruflo-metaharness/scripts/_invoke.mjs +231 -231
  404. package/plugins/ruflo-metaharness/scripts/_redblue.mjs +143 -143
  405. package/plugins/ruflo-metaharness/scripts/_similarity.mjs +161 -161
  406. package/plugins/ruflo-metaharness/scripts/_spike-similarity.mjs +223 -223
  407. package/plugins/ruflo-metaharness/scripts/audit-list.mjs +158 -158
  408. package/plugins/ruflo-metaharness/scripts/audit-trend.mjs +272 -272
  409. package/plugins/ruflo-metaharness/scripts/bench-parse-mcp-scan.mjs +146 -146
  410. package/plugins/ruflo-metaharness/scripts/bench-recordpair-overhead.mjs +186 -186
  411. package/plugins/ruflo-metaharness/scripts/bench-similarity.mjs +177 -177
  412. package/plugins/ruflo-metaharness/scripts/bench.mjs +95 -95
  413. package/plugins/ruflo-metaharness/scripts/drift-from-history.mjs +363 -363
  414. package/plugins/ruflo-metaharness/scripts/evolve.mjs +404 -404
  415. package/plugins/ruflo-metaharness/scripts/genome.mjs +80 -80
  416. package/plugins/ruflo-metaharness/scripts/gepa.mjs +153 -153
  417. package/plugins/ruflo-metaharness/scripts/learn.mjs +127 -127
  418. package/plugins/ruflo-metaharness/scripts/mcp-scan.mjs +111 -111
  419. package/plugins/ruflo-metaharness/scripts/mint.mjs +126 -126
  420. package/plugins/ruflo-metaharness/scripts/oia-audit.mjs +228 -228
  421. package/plugins/ruflo-metaharness/scripts/redblue.mjs +286 -286
  422. package/plugins/ruflo-metaharness/scripts/router-parallel-analyze.mjs +250 -250
  423. package/plugins/ruflo-metaharness/scripts/score.mjs +92 -92
  424. package/plugins/ruflo-metaharness/scripts/security-bench.mjs +174 -174
  425. package/plugins/ruflo-metaharness/scripts/similarity.mjs +158 -158
  426. package/plugins/ruflo-metaharness/scripts/smoke.sh +2356 -2356
  427. package/plugins/ruflo-metaharness/scripts/test-graceful-degradation.mjs +165 -165
  428. package/plugins/ruflo-metaharness/scripts/test-mcp-tools.mjs +472 -472
  429. package/plugins/ruflo-metaharness/scripts/test-parallel-pipeline.mjs +204 -204
  430. package/plugins/ruflo-metaharness/scripts/test-pipeline-roundtrip.mjs +586 -586
  431. package/plugins/ruflo-metaharness/scripts/test-similarity.mjs +334 -334
  432. package/plugins/ruflo-metaharness/scripts/test-with-openrouter.mjs +229 -229
  433. package/plugins/ruflo-metaharness/scripts/threat-model.mjs +59 -59
  434. package/plugins/ruflo-metaharness/skills/harness-bench/SKILL.md +64 -64
  435. package/plugins/ruflo-metaharness/skills/harness-drift-from-history/SKILL.md +65 -65
  436. package/plugins/ruflo-metaharness/skills/harness-evolve/SKILL.md +131 -131
  437. package/plugins/ruflo-metaharness/skills/harness-genome/SKILL.md +54 -54
  438. package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +65 -65
  439. package/plugins/ruflo-metaharness/skills/harness-learn/SKILL.md +65 -65
  440. package/plugins/ruflo-metaharness/skills/harness-mcp-scan/SKILL.md +49 -49
  441. package/plugins/ruflo-metaharness/skills/harness-mint/SKILL.md +72 -72
  442. package/plugins/ruflo-metaharness/skills/harness-oia-audit/SKILL.md +79 -79
  443. package/plugins/ruflo-metaharness/skills/harness-score/SKILL.md +66 -66
  444. package/plugins/ruflo-metaharness/skills/harness-security-bench/SKILL.md +101 -101
  445. package/plugins/ruflo-metaharness/skills/harness-similarity/SKILL.md +67 -67
  446. package/plugins/ruflo-metaharness/skills/harness-threat-model/SKILL.md +41 -41
  447. package/scripts/postinstall.cjs +153 -153
@@ -1,79 +1,79 @@
1
- ---
2
- name: harness-oia-audit
3
- description: Composite Phase-2 audit worker (ADR-150). Bundles harness oia-manifest + threat-model + mcp-scan into one timestamped audit record stored in the `metaharness-audit` memory namespace. Designed for cron-scheduled drift detection.
4
- argument-hint: "[--path .] [--dry-run] [--alert-on-worst clean|low|medium|high] [--format table|json]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- The 13th worker (ADR-150 Phase 2) — runs three MetaHarness static
9
- surfaces in one shot, computes a composite worst-severity signal, and
10
- persists the audit record to memory so drift over time is visible.
11
-
12
- ## Algorithm
13
-
14
- Implementation: [`scripts/oia-audit.mjs`](../../scripts/oia-audit.mjs).
15
-
16
- 1. Run `harness oia-manifest <path>` — Open Infrastructure Architecture
17
- layer alignment (L1-L9).
18
- 2. Run `harness threat-model <path>` — categorized MCP-surface threat
19
- report with `worst: clean|low|medium|high`.
20
- 3. Run `harness mcp-scan <path>` — per-server/tool policy + permissions
21
- + dep findings.
22
- 4. Composite worst = `max(threatModel.worst, max(mcpScan.findings.severity))`.
23
- 5. Persist payload to memory namespace `metaharness-audit` with key
24
- `audit-<iso-timestamp>` (unless `--dry-run`).
25
- 6. `--alert-on-worst <severity>`: exit 1 if composite worst ≥ threshold.
26
-
27
- ## Graceful degradation
28
-
29
- When ALL three components report `metaharness-not-available`, the script
30
- emits the standard degraded payload and exits 0. When only some are
31
- degraded, each individual component carries its own `degraded: true`
32
- flag in the audit record — the audit still runs and persists what it
33
- could gather.
34
-
35
- ## CI / cron integration
36
-
37
- Designed for weekly cron in `.github/workflows/`:
38
-
39
- ```yaml
40
- on:
41
- schedule:
42
- - cron: '17 4 * * 0' # Sundays at 04:17 UTC
43
- jobs:
44
- oia-audit:
45
- runs-on: ubuntu-latest
46
- steps:
47
- - uses: actions/checkout@v4
48
- - uses: actions/setup-node@v4
49
- - run: node plugins/ruflo-metaharness/scripts/oia-audit.mjs --alert-on-worst high
50
- ```
51
-
52
- `--alert-on-worst high` fails the job on any HIGH-severity finding;
53
- drift below HIGH is logged but doesn't block.
54
-
55
- ## Memory namespace
56
-
57
- Each audit run stores under `metaharness-audit:audit-<iso-ts>`. To list
58
- recent audits:
59
-
60
- ```bash
61
- npx @claude-flow/cli@latest memory list --namespace metaharness-audit --limit 10
62
- ```
63
-
64
- To diff two audits (drift detection):
65
-
66
- ```bash
67
- A=$(npx ... memory retrieve --key audit-2026-06-01... --namespace metaharness-audit)
68
- B=$(npx ... memory retrieve --key audit-2026-06-15... --namespace metaharness-audit)
69
- # Compare composite.worst, components.threatModel.worst, etc.
70
- ```
71
-
72
- A future ADR can wire this into a dedicated `cost-diff`-style diff
73
- viewer specifically for audit drift.
74
-
75
- ## Pairs with
76
-
77
- - `harness-threat-model` — the underlying threat-model component
78
- - `harness-mcp-scan` — the underlying MCP-scan component
79
- - `harness-score` + `harness-genome` — readiness metrics (orthogonal to audit)
1
+ ---
2
+ name: harness-oia-audit
3
+ description: Composite Phase-2 audit worker (ADR-150). Bundles harness oia-manifest + threat-model + mcp-scan into one timestamped audit record stored in the `metaharness-audit` memory namespace. Designed for cron-scheduled drift detection.
4
+ argument-hint: "[--path .] [--dry-run] [--alert-on-worst clean|low|medium|high] [--format table|json]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ The 13th worker (ADR-150 Phase 2) — runs three MetaHarness static
9
+ surfaces in one shot, computes a composite worst-severity signal, and
10
+ persists the audit record to memory so drift over time is visible.
11
+
12
+ ## Algorithm
13
+
14
+ Implementation: [`scripts/oia-audit.mjs`](../../scripts/oia-audit.mjs).
15
+
16
+ 1. Run `harness oia-manifest <path>` — Open Infrastructure Architecture
17
+ layer alignment (L1-L9).
18
+ 2. Run `harness threat-model <path>` — categorized MCP-surface threat
19
+ report with `worst: clean|low|medium|high`.
20
+ 3. Run `harness mcp-scan <path>` — per-server/tool policy + permissions
21
+ + dep findings.
22
+ 4. Composite worst = `max(threatModel.worst, max(mcpScan.findings.severity))`.
23
+ 5. Persist payload to memory namespace `metaharness-audit` with key
24
+ `audit-<iso-timestamp>` (unless `--dry-run`).
25
+ 6. `--alert-on-worst <severity>`: exit 1 if composite worst ≥ threshold.
26
+
27
+ ## Graceful degradation
28
+
29
+ When ALL three components report `metaharness-not-available`, the script
30
+ emits the standard degraded payload and exits 0. When only some are
31
+ degraded, each individual component carries its own `degraded: true`
32
+ flag in the audit record — the audit still runs and persists what it
33
+ could gather.
34
+
35
+ ## CI / cron integration
36
+
37
+ Designed for weekly cron in `.github/workflows/`:
38
+
39
+ ```yaml
40
+ on:
41
+ schedule:
42
+ - cron: '17 4 * * 0' # Sundays at 04:17 UTC
43
+ jobs:
44
+ oia-audit:
45
+ runs-on: ubuntu-latest
46
+ steps:
47
+ - uses: actions/checkout@v4
48
+ - uses: actions/setup-node@v4
49
+ - run: node plugins/ruflo-metaharness/scripts/oia-audit.mjs --alert-on-worst high
50
+ ```
51
+
52
+ `--alert-on-worst high` fails the job on any HIGH-severity finding;
53
+ drift below HIGH is logged but doesn't block.
54
+
55
+ ## Memory namespace
56
+
57
+ Each audit run stores under `metaharness-audit:audit-<iso-ts>`. To list
58
+ recent audits:
59
+
60
+ ```bash
61
+ npx @claude-flow/cli@latest memory list --namespace metaharness-audit --limit 10
62
+ ```
63
+
64
+ To diff two audits (drift detection):
65
+
66
+ ```bash
67
+ A=$(npx ... memory retrieve --key audit-2026-06-01... --namespace metaharness-audit)
68
+ B=$(npx ... memory retrieve --key audit-2026-06-15... --namespace metaharness-audit)
69
+ # Compare composite.worst, components.threatModel.worst, etc.
70
+ ```
71
+
72
+ A future ADR can wire this into a dedicated `cost-diff`-style diff
73
+ viewer specifically for audit drift.
74
+
75
+ ## Pairs with
76
+
77
+ - `harness-threat-model` — the underlying threat-model component
78
+ - `harness-mcp-scan` — the underlying MCP-scan component
79
+ - `harness-score` + `harness-genome` — readiness metrics (orthogonal to audit)
@@ -1,66 +1,66 @@
1
- ---
2
- name: harness-score
3
- description: 5-dimension harness readiness scorecard from `metaharness score <path>`. Returns harnessFit / compileConfidence / taskCoverage / toolSafety / memoryUsefulness + estCostPerRunUsd + scaffoldReady. Pure-read; subprocess invocation; degrades gracefully when MetaHarness is absent (ADR-150 architectural constraint).
4
- argument-hint: "[--path .] [--alert-on-fit-below 70] [--format table|json]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- Surfaces the upstream `metaharness score` CLI as a ruflo skill. Use when
9
- Claude Code needs to assess whether a repo is ready for harness adoption
10
- before recommending the user run `npx ruflo init` or `harness-mint`.
11
-
12
- ## Algorithm
13
-
14
- Implementation: [`scripts/score.mjs`](../../scripts/score.mjs).
15
-
16
- 1. Shell out to `npx metaharness score <path> --json` (single subprocess,
17
- 60s hard timeout).
18
- 2. Parse the JSON shape: `{ harnessFit, compileConfidence, taskCoverage,
19
- toolSafety, memoryUsefulness, estCostPerRunUsd, recommendedMode,
20
- archetype, template, scaffoldReady, hardConstraints }`.
21
- 3. If `--alert-on-fit-below N`: exit 1 when `harnessFit < N`.
22
- 4. Output JSON (default) or markdown table.
23
-
24
- ## Phase-0 baseline (ruflo's own scorecard, measured 2026-06-16)
25
-
26
- | Dimension | Value |
27
- |---|---:|
28
- | harnessFit | 82/100 |
29
- | compileConfidence | 100 |
30
- | taskCoverage | 79 |
31
- | toolSafety | 100 |
32
- | memoryUsefulness | 40 |
33
- | estCostPerRunUsd | $0.048 |
34
- | recommendedMode | CLI + MCP |
35
- | archetype | typescript-sdk-harness |
36
- | template | vertical:coding |
37
- | scaffoldReady | true |
38
-
39
- Ruflo passes its own readiness check. `memoryUsefulness: 40` is the
40
- weakest dimension — track this as a leading indicator for future memory
41
- work in the AgentDB layer.
42
-
43
- ## CI integration
44
-
45
- ```bash
46
- node plugins/ruflo-metaharness/scripts/score.mjs --alert-on-fit-below 70 --format json
47
- ```
48
-
49
- Exit 1 fails the build. Pair with `harness-genome` for the full
50
- 7-section view.
51
-
52
- ## Graceful degradation (ADR-150 architectural constraint rule #3)
53
-
54
- When `metaharness` is not installed and `npx` can't fetch it (offline,
55
- no network, registry unreachable), the script emits:
56
-
57
- ```json
58
- {
59
- "degraded": true,
60
- "reason": "metaharness-not-available",
61
- "hint": "Install with `npm i -D metaharness@~0.3.0` (pinned range — this plugin never fetches @latest) or verify network access for the one-time cache install."
62
- }
63
- ```
64
-
65
- and exits 0. Ruflo continues to function — this is the architectural
66
- constraint in action.
1
+ ---
2
+ name: harness-score
3
+ description: 5-dimension harness readiness scorecard from `metaharness score <path>`. Returns harnessFit / compileConfidence / taskCoverage / toolSafety / memoryUsefulness + estCostPerRunUsd + scaffoldReady. Pure-read; subprocess invocation; degrades gracefully when MetaHarness is absent (ADR-150 architectural constraint).
4
+ argument-hint: "[--path .] [--alert-on-fit-below 70] [--format table|json]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ Surfaces the upstream `metaharness score` CLI as a ruflo skill. Use when
9
+ Claude Code needs to assess whether a repo is ready for harness adoption
10
+ before recommending the user run `npx ruflo init` or `harness-mint`.
11
+
12
+ ## Algorithm
13
+
14
+ Implementation: [`scripts/score.mjs`](../../scripts/score.mjs).
15
+
16
+ 1. Shell out to `npx metaharness score <path> --json` (single subprocess,
17
+ 60s hard timeout).
18
+ 2. Parse the JSON shape: `{ harnessFit, compileConfidence, taskCoverage,
19
+ toolSafety, memoryUsefulness, estCostPerRunUsd, recommendedMode,
20
+ archetype, template, scaffoldReady, hardConstraints }`.
21
+ 3. If `--alert-on-fit-below N`: exit 1 when `harnessFit < N`.
22
+ 4. Output JSON (default) or markdown table.
23
+
24
+ ## Phase-0 baseline (ruflo's own scorecard, measured 2026-06-16)
25
+
26
+ | Dimension | Value |
27
+ |---|---:|
28
+ | harnessFit | 82/100 |
29
+ | compileConfidence | 100 |
30
+ | taskCoverage | 79 |
31
+ | toolSafety | 100 |
32
+ | memoryUsefulness | 40 |
33
+ | estCostPerRunUsd | $0.048 |
34
+ | recommendedMode | CLI + MCP |
35
+ | archetype | typescript-sdk-harness |
36
+ | template | vertical:coding |
37
+ | scaffoldReady | true |
38
+
39
+ Ruflo passes its own readiness check. `memoryUsefulness: 40` is the
40
+ weakest dimension — track this as a leading indicator for future memory
41
+ work in the AgentDB layer.
42
+
43
+ ## CI integration
44
+
45
+ ```bash
46
+ node plugins/ruflo-metaharness/scripts/score.mjs --alert-on-fit-below 70 --format json
47
+ ```
48
+
49
+ Exit 1 fails the build. Pair with `harness-genome` for the full
50
+ 7-section view.
51
+
52
+ ## Graceful degradation (ADR-150 architectural constraint rule #3)
53
+
54
+ When `metaharness` is not installed and `npx` can't fetch it (offline,
55
+ no network, registry unreachable), the script emits:
56
+
57
+ ```json
58
+ {
59
+ "degraded": true,
60
+ "reason": "metaharness-not-available",
61
+ "hint": "Install with `npm i -D metaharness@~0.3.0` (pinned range — this plugin never fetches @latest) or verify network access for the one-time cache install."
62
+ }
63
+ ```
64
+
65
+ and exits 0. Ruflo continues to function — this is the architectural
66
+ constraint in action.
@@ -1,101 +1,101 @@
1
- ---
2
- name: harness-security-bench
3
- description: Run `@metaharness/darwin security bench` (upstream "Darwin Shield" / ADR-155) — evolves a champion security-detection harness against a 10-vuln / 9-decoy corpus and grades it on TPR/FPR/patch-pass/repro/unsafe vs four baselines (B0 static, B1 LLM-single-pass, B2 fixed-agent, B3 Darwin-champion). Closest reference implementation for ruflo's own ADR-155 nightly self-learning security harness (PR #2417). Degrades gracefully when @metaharness/darwin is absent.
4
- argument-hint: "[--population 2] [--cycles 1] [--seed N] [--alert-on-fail]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- Surfaces the upstream `metaharness-darwin security bench` command. **This is
9
- the upstream's own ADR-155 — Darwin Shield — and is the closest reference
10
- implementation for ruflo's nightly self-learning security harness ([#2417](https://github.com/ruvnet/ruflo/pull/2417)).**
11
-
12
- ## Why this matters for ruflo's ADR-155
13
-
14
- ruflo's ADR-155 proposes three learning loops (per-dimension confidence,
15
- severity calibration, auto-fix bid). Loop A trains on accumulated
16
- `(finding, dimension, human_outcome)` tuples — but the gradient signal is
17
- only sound if the underlying detection mechanism converges on a known-good
18
- corpus. Darwin Shield evolves exactly that mechanism on a 10-vuln/9-decoy
19
- ground-truth set. Running this nightly gives us:
20
-
21
- - **Empirical floor:** if Darwin Shield's champion can't reach
22
- TPR=1/FPR=0 on the bench corpus, our Loop A's reward signal is noise.
23
- - **Drift detection:** week-over-week champion fitness deltas surface
24
- when the security landscape (or our mutator policy) shifts.
25
- - **Baseline diversity:** the 4 baselines (B0–B3) give us 4 anchor
26
- points to weight per-dimension confidence against.
27
-
28
- ## Algorithm
29
-
30
- Implementation: [`scripts/security-bench.mjs`](../../scripts/security-bench.mjs).
31
-
32
- 1. Shell to `npx -y @metaharness/darwin@~0.8.0 metaharness-darwin security bench --population N --cycles N [--seed S]`.
33
- 2. Default timeout = `3s × 19 evaluations × population × cycles + 30s overhead`.
34
- At default `--population 2 --cycles 1` ≈ 144s; at `--population 4 --cycles 3` ≈ 12 min.
35
- 3. Parse the markdown report — overall PASS/FAIL plus per-gate
36
- pass/fail rows (gate examples: "TPR improvement ≥ 25% vs fixed",
37
- "FPR reduction ≥ 40%", "Patch-test pass rate ≥ 80%", "Reproduction
38
- success ≥ 90%", "Unsafe outputs = 0", "Cost increase ≤ 2× fixed",
39
- "Beyond SOTA: champion statistically beats previous champion",
40
- "Compounding: false-positive repeat-rate drop ≥ 35%").
41
- 4. Parse the baselines-vs-champion table (4 rows: fitness/TPR/FPR/patchPass/
42
- repro/unsafe/cost per harness).
43
- 5. Emit structured JSON. With `--alert-on-fail`, exit 1 when overall = FAIL.
44
-
45
- ## Output shape
46
-
47
- ```json
48
- {
49
- "success": true,
50
- "data": {
51
- "overall": { "ok": true, "icon": "✅" },
52
- "gates": {
53
- "total": 11,
54
- "passed": 11,
55
- "failed": 0,
56
- "details": [{ "ok": true, "criterion": "TPR improvement ≥ 25% vs fixed harness", "measured": "+150% (B2 0.4 → B3 1)" }, ...]
57
- },
58
- "baselines": [
59
- { "harness": "static-only", "fitness": 0.5665, "tpr": 0.3, "fpr": 1, "unsafe": 0, ... },
60
- { "harness": "LLM single-pass", "fitness": 0.1365, ... },
61
- { "harness": "fixed agent", "fitness": 0.598, ... },
62
- { "harness": "Darwin champion", "fitness": 0.93275, "tpr": 1, "fpr": 0, ... }
63
- ],
64
- "rawMarkdown": "...",
65
- "shape": { "population": 2, "cycles": 1, "seed": null },
66
- "durationMs": 142000
67
- }
68
- }
69
- ```
70
-
71
- ## Wiring into ADR-155 nightly harness
72
-
73
- The ADR-155 nightly workflow (per #2418 task `W1.5`) will spawn this as
74
- one of the active-pentest dimension's calls — its results become a
75
- trajectory record:
76
-
77
- ```jsonc
78
- {
79
- "dimension": "mcp-pentest",
80
- "subdimension": "darwin-shield-bench",
81
- "champion_fitness": 0.93275,
82
- "champion_tpr": 1, "champion_fpr": 0,
83
- "gates_passed": 11, "gates_failed": 0,
84
- "shape": { "population": 4, "cycles": 3 }
85
- }
86
- ```
87
-
88
- Loop A learns: if `darwin-shield-bench` consistently passes on the seeded
89
- corpus, weight findings caught only by `mcp-pentest` higher.
90
-
91
- ## Exit codes
92
-
93
- | Code | Meaning |
94
- |---|---|
95
- | 0 | Bench ran (overall PASS or FAIL — distinguish via JSON `overall.ok`), or degraded |
96
- | 1 | `--alert-on-fail` and `overall.ok === false` |
97
- | 2 | Config error or upstream infrastructure failure |
98
-
99
- ## Graceful degradation
100
-
101
- When `@metaharness/darwin` is absent, emits `{degraded: true, reason: 'metaharness-darwin-not-available'}` and exits 0.
1
+ ---
2
+ name: harness-security-bench
3
+ description: Run `@metaharness/darwin security bench` (upstream "Darwin Shield" / ADR-155) — evolves a champion security-detection harness against a 10-vuln / 9-decoy corpus and grades it on TPR/FPR/patch-pass/repro/unsafe vs four baselines (B0 static, B1 LLM-single-pass, B2 fixed-agent, B3 Darwin-champion). Closest reference implementation for ruflo's own ADR-155 nightly self-learning security harness (PR #2417). Degrades gracefully when @metaharness/darwin is absent.
4
+ argument-hint: "[--population 2] [--cycles 1] [--seed N] [--alert-on-fail]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ Surfaces the upstream `metaharness-darwin security bench` command. **This is
9
+ the upstream's own ADR-155 — Darwin Shield — and is the closest reference
10
+ implementation for ruflo's nightly self-learning security harness ([#2417](https://github.com/ruvnet/ruflo/pull/2417)).**
11
+
12
+ ## Why this matters for ruflo's ADR-155
13
+
14
+ ruflo's ADR-155 proposes three learning loops (per-dimension confidence,
15
+ severity calibration, auto-fix bid). Loop A trains on accumulated
16
+ `(finding, dimension, human_outcome)` tuples — but the gradient signal is
17
+ only sound if the underlying detection mechanism converges on a known-good
18
+ corpus. Darwin Shield evolves exactly that mechanism on a 10-vuln/9-decoy
19
+ ground-truth set. Running this nightly gives us:
20
+
21
+ - **Empirical floor:** if Darwin Shield's champion can't reach
22
+ TPR=1/FPR=0 on the bench corpus, our Loop A's reward signal is noise.
23
+ - **Drift detection:** week-over-week champion fitness deltas surface
24
+ when the security landscape (or our mutator policy) shifts.
25
+ - **Baseline diversity:** the 4 baselines (B0–B3) give us 4 anchor
26
+ points to weight per-dimension confidence against.
27
+
28
+ ## Algorithm
29
+
30
+ Implementation: [`scripts/security-bench.mjs`](../../scripts/security-bench.mjs).
31
+
32
+ 1. Shell to `npx -y @metaharness/darwin@~0.8.0 metaharness-darwin security bench --population N --cycles N [--seed S]`.
33
+ 2. Default timeout = `3s × 19 evaluations × population × cycles + 30s overhead`.
34
+ At default `--population 2 --cycles 1` ≈ 144s; at `--population 4 --cycles 3` ≈ 12 min.
35
+ 3. Parse the markdown report — overall PASS/FAIL plus per-gate
36
+ pass/fail rows (gate examples: "TPR improvement ≥ 25% vs fixed",
37
+ "FPR reduction ≥ 40%", "Patch-test pass rate ≥ 80%", "Reproduction
38
+ success ≥ 90%", "Unsafe outputs = 0", "Cost increase ≤ 2× fixed",
39
+ "Beyond SOTA: champion statistically beats previous champion",
40
+ "Compounding: false-positive repeat-rate drop ≥ 35%").
41
+ 4. Parse the baselines-vs-champion table (4 rows: fitness/TPR/FPR/patchPass/
42
+ repro/unsafe/cost per harness).
43
+ 5. Emit structured JSON. With `--alert-on-fail`, exit 1 when overall = FAIL.
44
+
45
+ ## Output shape
46
+
47
+ ```json
48
+ {
49
+ "success": true,
50
+ "data": {
51
+ "overall": { "ok": true, "icon": "✅" },
52
+ "gates": {
53
+ "total": 11,
54
+ "passed": 11,
55
+ "failed": 0,
56
+ "details": [{ "ok": true, "criterion": "TPR improvement ≥ 25% vs fixed harness", "measured": "+150% (B2 0.4 → B3 1)" }, ...]
57
+ },
58
+ "baselines": [
59
+ { "harness": "static-only", "fitness": 0.5665, "tpr": 0.3, "fpr": 1, "unsafe": 0, ... },
60
+ { "harness": "LLM single-pass", "fitness": 0.1365, ... },
61
+ { "harness": "fixed agent", "fitness": 0.598, ... },
62
+ { "harness": "Darwin champion", "fitness": 0.93275, "tpr": 1, "fpr": 0, ... }
63
+ ],
64
+ "rawMarkdown": "...",
65
+ "shape": { "population": 2, "cycles": 1, "seed": null },
66
+ "durationMs": 142000
67
+ }
68
+ }
69
+ ```
70
+
71
+ ## Wiring into ADR-155 nightly harness
72
+
73
+ The ADR-155 nightly workflow (per #2418 task `W1.5`) will spawn this as
74
+ one of the active-pentest dimension's calls — its results become a
75
+ trajectory record:
76
+
77
+ ```jsonc
78
+ {
79
+ "dimension": "mcp-pentest",
80
+ "subdimension": "darwin-shield-bench",
81
+ "champion_fitness": 0.93275,
82
+ "champion_tpr": 1, "champion_fpr": 0,
83
+ "gates_passed": 11, "gates_failed": 0,
84
+ "shape": { "population": 4, "cycles": 3 }
85
+ }
86
+ ```
87
+
88
+ Loop A learns: if `darwin-shield-bench` consistently passes on the seeded
89
+ corpus, weight findings caught only by `mcp-pentest` higher.
90
+
91
+ ## Exit codes
92
+
93
+ | Code | Meaning |
94
+ |---|---|
95
+ | 0 | Bench ran (overall PASS or FAIL — distinguish via JSON `overall.ok`), or degraded |
96
+ | 1 | `--alert-on-fail` and `overall.ok === false` |
97
+ | 2 | Config error or upstream infrastructure failure |
98
+
99
+ ## Graceful degradation
100
+
101
+ When `@metaharness/darwin` is absent, emits `{degraded: true, reason: 'metaharness-darwin-not-available'}` and exits 0.
@@ -1,67 +1,67 @@
1
- ---
2
- name: harness-similarity
3
- description: ADR-152 — weighted similarity between two harness fingerprints (genome + score JSON). Returns overall score in [0,1] plus per-component breakdown (cosine over 9 numerics, categorical agreement over 4 enums, jaccard over agent_topology). Unblocks ADR-151 §3.2 Recommender, §3.3 Drift Detection, §3.5 Plugin Compat. Pure-TS, no `@metaharness/*` dep — preserves ADR-150's four architectural constraints.
4
- argument-hint: "(--a a.json --b b.json | --a-key X --b-key Y) [--per-dimension] [--alert-below 0.5] [--format json|table]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- Surfaces the production similarity function from [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs) as a callable skill. Use when an agent needs to:
9
-
10
- - decide whether to fork an existing harness vs scaffold a new one
11
- - rank candidate templates against a target repo's genome
12
- - diff two harnesses produced by different teams to find duplicate work
13
- - generate the confidence number that ADR-151 §3.2's Recommender wraps
14
-
15
- ## Algorithm (from ADR-152 §Decision)
16
-
17
- ```
18
- overall = 0.60·cosine + 0.25·categorical + 0.15·jaccard
19
- ```
20
-
21
- - **cosine** — over a 9-dim numerical vector of normalized scorecard + genome dims
22
- - **categorical** — fraction of 4 enum fields that match (`repo_type`, `archetype`, `template`, `recommendedMode`)
23
- - **jaccard** — `|A ∩ B| / |A ∪ B|` over the `agent_topology[]` array
24
-
25
- The 3-component design is load-bearing: numerical cosine alone is too coarse (the iter-35 spike showed LEGAL vs DEVOPS at cosine=0.97 despite being unrelated verticals). Categorical + jaccard pull the composite to the correct ordering.
26
-
27
- ## Reference outputs (iter-35 spike fixtures)
28
-
29
- | Pair | overall | cosine | categorical | jaccard |
30
- |---|---:|---:|---:|---:|
31
- | `LEGAL` × `LEGAL` (self) | 1.0000 | 1.0000 | 1.0000 | 1.0000 |
32
- | `LEGAL` × `SUPPORT` | 0.8296 | 0.9987 | 0.7500 | 0.2857 |
33
- | `LEGAL` × `DEVOPS` | 0.5840 | 0.9734 | 0.0000 | 0.0000 |
34
-
35
- Both invariants from ADR-152 §"Smallest demonstrable spike" hold:
36
- 1. `similarity(X, X) === 1` exactly
37
- 2. `similarity(LEGAL, DEVOPS) < similarity(LEGAL, SUPPORT)` (vertical affinity)
38
-
39
- ## Architectural constraint inheritance (ADR-150)
40
-
41
- - **Removable** — pure-TS function, zero static `@metaharness/*` imports.
42
- - **Optional** — no new dep in `package.json`.
43
- - **Graceful** — malformed inputs emit `{ degraded: true, reason }` with exit code 2; never throws.
44
- - **CI-gate** — smoke step 17y locks the contract: module exports, spike fixtures reproduce, CLI dispatcher entry registered, MCP tool registered.
45
-
46
- ## Usage
47
-
48
- ```bash
49
- # File inputs
50
- npx ruflo metaharness similarity --a a.json --b b.json
51
-
52
- # Memory inputs (records persisted by oia-audit.mjs)
53
- npx ruflo metaharness similarity --a-key harness-X --b-key harness-Y
54
-
55
- # Per-dimension breakdown (used by ADR-151 §3.2 Recommender)
56
- npx ruflo metaharness similarity --a a.json --b b.json --per-dimension
57
-
58
- # Alert when too-dissimilar (used by ADR-151 §3.3 Drift Detection)
59
- npx ruflo metaharness similarity --a a.json --b b.json --alert-below 0.5
60
- ```
61
-
62
- ## Implementation
63
-
64
- Production module: [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs)
65
- CLI skill: [`scripts/similarity.mjs`](../../scripts/similarity.mjs)
66
- MCP tool: `mcp__plugin_ruflo-core_ruflo__metaharness_similarity` (registered in `v3/@claude-flow/cli/src/mcp-tools/metaharness-tools.ts`)
67
- Spike anchor: [`scripts/_spike-similarity.mjs`](../../scripts/_spike-similarity.mjs) (regression suite — invariants locked here)
1
+ ---
2
+ name: harness-similarity
3
+ description: ADR-152 — weighted similarity between two harness fingerprints (genome + score JSON). Returns overall score in [0,1] plus per-component breakdown (cosine over 9 numerics, categorical agreement over 4 enums, jaccard over agent_topology). Unblocks ADR-151 §3.2 Recommender, §3.3 Drift Detection, §3.5 Plugin Compat. Pure-TS, no `@metaharness/*` dep — preserves ADR-150's four architectural constraints.
4
+ argument-hint: "(--a a.json --b b.json | --a-key X --b-key Y) [--per-dimension] [--alert-below 0.5] [--format json|table]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ Surfaces the production similarity function from [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs) as a callable skill. Use when an agent needs to:
9
+
10
+ - decide whether to fork an existing harness vs scaffold a new one
11
+ - rank candidate templates against a target repo's genome
12
+ - diff two harnesses produced by different teams to find duplicate work
13
+ - generate the confidence number that ADR-151 §3.2's Recommender wraps
14
+
15
+ ## Algorithm (from ADR-152 §Decision)
16
+
17
+ ```
18
+ overall = 0.60·cosine + 0.25·categorical + 0.15·jaccard
19
+ ```
20
+
21
+ - **cosine** — over a 9-dim numerical vector of normalized scorecard + genome dims
22
+ - **categorical** — fraction of 4 enum fields that match (`repo_type`, `archetype`, `template`, `recommendedMode`)
23
+ - **jaccard** — `|A ∩ B| / |A ∪ B|` over the `agent_topology[]` array
24
+
25
+ The 3-component design is load-bearing: numerical cosine alone is too coarse (the iter-35 spike showed LEGAL vs DEVOPS at cosine=0.97 despite being unrelated verticals). Categorical + jaccard pull the composite to the correct ordering.
26
+
27
+ ## Reference outputs (iter-35 spike fixtures)
28
+
29
+ | Pair | overall | cosine | categorical | jaccard |
30
+ |---|---:|---:|---:|---:|
31
+ | `LEGAL` × `LEGAL` (self) | 1.0000 | 1.0000 | 1.0000 | 1.0000 |
32
+ | `LEGAL` × `SUPPORT` | 0.8296 | 0.9987 | 0.7500 | 0.2857 |
33
+ | `LEGAL` × `DEVOPS` | 0.5840 | 0.9734 | 0.0000 | 0.0000 |
34
+
35
+ Both invariants from ADR-152 §"Smallest demonstrable spike" hold:
36
+ 1. `similarity(X, X) === 1` exactly
37
+ 2. `similarity(LEGAL, DEVOPS) < similarity(LEGAL, SUPPORT)` (vertical affinity)
38
+
39
+ ## Architectural constraint inheritance (ADR-150)
40
+
41
+ - **Removable** — pure-TS function, zero static `@metaharness/*` imports.
42
+ - **Optional** — no new dep in `package.json`.
43
+ - **Graceful** — malformed inputs emit `{ degraded: true, reason }` with exit code 2; never throws.
44
+ - **CI-gate** — smoke step 17y locks the contract: module exports, spike fixtures reproduce, CLI dispatcher entry registered, MCP tool registered.
45
+
46
+ ## Usage
47
+
48
+ ```bash
49
+ # File inputs
50
+ npx ruflo metaharness similarity --a a.json --b b.json
51
+
52
+ # Memory inputs (records persisted by oia-audit.mjs)
53
+ npx ruflo metaharness similarity --a-key harness-X --b-key harness-Y
54
+
55
+ # Per-dimension breakdown (used by ADR-151 §3.2 Recommender)
56
+ npx ruflo metaharness similarity --a a.json --b b.json --per-dimension
57
+
58
+ # Alert when too-dissimilar (used by ADR-151 §3.3 Drift Detection)
59
+ npx ruflo metaharness similarity --a a.json --b b.json --alert-below 0.5
60
+ ```
61
+
62
+ ## Implementation
63
+
64
+ Production module: [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs)
65
+ CLI skill: [`scripts/similarity.mjs`](../../scripts/similarity.mjs)
66
+ MCP tool: `mcp__plugin_ruflo-core_ruflo__metaharness_similarity` (registered in `v3/@claude-flow/cli/src/mcp-tools/metaharness-tools.ts`)
67
+ Spike anchor: [`scripts/_spike-similarity.mjs`](../../scripts/_spike-similarity.mjs) (regression suite — invariants locked here)