@claude-flow/cli 3.42.2 → 3.42.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (459) hide show
  1. package/.claude/agents/analysis/analyze-code-quality.md +178 -178
  2. package/.claude/agents/analysis/code-analyzer.md +209 -209
  3. package/.claude/agents/analysis/code-review/analyze-code-quality.md +178 -178
  4. package/.claude/agents/architecture/arch-system-design.md +156 -156
  5. package/.claude/agents/architecture/system-design/arch-system-design.md +154 -154
  6. package/.claude/agents/browser/browser-agent.yaml +182 -182
  7. package/.claude/agents/consensus/byzantine-coordinator.md +62 -62
  8. package/.claude/agents/consensus/crdt-synchronizer.md +996 -996
  9. package/.claude/agents/consensus/gossip-coordinator.md +62 -62
  10. package/.claude/agents/consensus/performance-benchmarker.md +850 -850
  11. package/.claude/agents/consensus/quorum-manager.md +822 -822
  12. package/.claude/agents/consensus/raft-manager.md +62 -62
  13. package/.claude/agents/consensus/security-manager.md +621 -621
  14. package/.claude/agents/core/planner.md +374 -374
  15. package/.claude/agents/custom/test-long-runner.md +44 -44
  16. package/.claude/agents/data/data-ml-model.md +444 -444
  17. package/.claude/agents/data/ml/data-ml-model.md +192 -192
  18. package/.claude/agents/development/backend/dev-backend-api.md +141 -141
  19. package/.claude/agents/development/dev-backend-api.md +344 -344
  20. package/.claude/agents/devops/ci-cd/ops-cicd-github.md +163 -163
  21. package/.claude/agents/devops/ops-cicd-github.md +164 -164
  22. package/.claude/agents/documentation/api-docs/docs-api-openapi.md +173 -173
  23. package/.claude/agents/documentation/docs-api-openapi.md +354 -354
  24. package/.claude/agents/flow-nexus/app-store.md +87 -87
  25. package/.claude/agents/flow-nexus/authentication.md +68 -68
  26. package/.claude/agents/flow-nexus/challenges.md +80 -80
  27. package/.claude/agents/flow-nexus/neural-network.md +87 -87
  28. package/.claude/agents/flow-nexus/payments.md +82 -82
  29. package/.claude/agents/flow-nexus/sandbox.md +75 -75
  30. package/.claude/agents/flow-nexus/swarm.md +75 -75
  31. package/.claude/agents/flow-nexus/user-tools.md +95 -95
  32. package/.claude/agents/flow-nexus/workflow.md +83 -83
  33. package/.claude/agents/github/code-review-swarm.md +377 -377
  34. package/.claude/agents/github/github-modes.md +172 -172
  35. package/.claude/agents/github/issue-tracker.md +575 -575
  36. package/.claude/agents/github/multi-repo-swarm.md +552 -552
  37. package/.claude/agents/github/pr-manager.md +437 -437
  38. package/.claude/agents/github/project-board-sync.md +508 -508
  39. package/.claude/agents/github/release-manager.md +604 -604
  40. package/.claude/agents/github/release-swarm.md +582 -582
  41. package/.claude/agents/github/repo-architect.md +397 -397
  42. package/.claude/agents/github/swarm-issue.md +572 -572
  43. package/.claude/agents/github/swarm-pr.md +427 -427
  44. package/.claude/agents/github/sync-coordinator.md +451 -451
  45. package/.claude/agents/github/workflow-automation.md +902 -902
  46. package/.claude/agents/goal/agent.md +815 -815
  47. package/.claude/agents/optimization/benchmark-suite.md +664 -664
  48. package/.claude/agents/optimization/load-balancer.md +430 -430
  49. package/.claude/agents/optimization/performance-monitor.md +671 -671
  50. package/.claude/agents/optimization/resource-allocator.md +673 -673
  51. package/.claude/agents/optimization/topology-optimizer.md +807 -807
  52. package/.claude/agents/payments/agentic-payments.md +126 -126
  53. package/.claude/agents/sona/sona-learning-optimizer.md +74 -74
  54. package/.claude/agents/sparc/architecture.md +698 -698
  55. package/.claude/agents/sparc/pseudocode.md +519 -519
  56. package/.claude/agents/sparc/refinement.md +801 -801
  57. package/.claude/agents/sparc/specification.md +477 -477
  58. package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +224 -224
  59. package/.claude/agents/specialized/spec-mobile-react-native.md +226 -226
  60. package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
  61. package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
  62. package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
  63. package/.claude/agents/sublinear/performance-optimizer.md +367 -367
  64. package/.claude/agents/sublinear/trading-predictor.md +245 -245
  65. package/.claude/agents/swarm/adaptive-coordinator.md +1126 -1126
  66. package/.claude/agents/swarm/hierarchical-coordinator.md +709 -709
  67. package/.claude/agents/swarm/mesh-coordinator.md +962 -962
  68. package/.claude/agents/templates/automation-smart-agent.md +204 -204
  69. package/.claude/agents/templates/base-template-generator.md +289 -289
  70. package/.claude/agents/templates/coordinator-swarm-init.md +89 -89
  71. package/.claude/agents/templates/github-pr-manager.md +176 -176
  72. package/.claude/agents/templates/implementer-sparc-coder.md +258 -258
  73. package/.claude/agents/templates/memory-coordinator.md +186 -186
  74. package/.claude/agents/templates/orchestrator-task.md +138 -138
  75. package/.claude/agents/templates/performance-analyzer.md +198 -198
  76. package/.claude/agents/templates/sparc-coordinator.md +513 -513
  77. package/.claude/agents/testing/production-validator.md +394 -394
  78. package/.claude/agents/testing/tdd-london-swarm.md +243 -243
  79. package/.claude/agents/v3/aidefence-guardian.md +282 -282
  80. package/.claude/agents/v3/claims-authorizer.md +208 -208
  81. package/.claude/agents/v3/collective-intelligence-coordinator.md +993 -993
  82. package/.claude/agents/v3/ddd-domain-expert.md +220 -220
  83. package/.claude/agents/v3/injection-analyst.md +236 -236
  84. package/.claude/agents/v3/performance-engineer.md +1233 -1233
  85. package/.claude/agents/v3/pii-detector.md +151 -151
  86. package/.claude/agents/v3/reasoningbank-learner.md +213 -213
  87. package/.claude/agents/v3/security-architect-aidefence.md +410 -410
  88. package/.claude/agents/v3/security-architect.md +867 -867
  89. package/.claude/agents/v3/swarm-memory-manager.md +157 -157
  90. package/.claude/agents/v3/v3-integration-architect.md +205 -205
  91. package/.claude/commands/agents/README.md +50 -50
  92. package/.claude/commands/agents/agent-capabilities.md +140 -140
  93. package/.claude/commands/agents/agent-coordination.md +28 -28
  94. package/.claude/commands/agents/agent-spawning.md +28 -28
  95. package/.claude/commands/agents/agent-types.md +216 -216
  96. package/.claude/commands/agents/health.md +139 -139
  97. package/.claude/commands/agents/list.md +100 -100
  98. package/.claude/commands/agents/logs.md +130 -130
  99. package/.claude/commands/agents/metrics.md +122 -122
  100. package/.claude/commands/agents/pool.md +127 -127
  101. package/.claude/commands/agents/spawn.md +140 -140
  102. package/.claude/commands/agents/status.md +115 -115
  103. package/.claude/commands/agents/stop.md +102 -102
  104. package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
  105. package/.claude/commands/analysis/README.md +9 -9
  106. package/.claude/commands/analysis/bottleneck-detect.md +162 -162
  107. package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
  108. package/.claude/commands/analysis/performance-report.md +25 -25
  109. package/.claude/commands/analysis/token-efficiency.md +44 -44
  110. package/.claude/commands/analysis/token-usage.md +25 -25
  111. package/.claude/commands/automation/README.md +9 -9
  112. package/.claude/commands/automation/auto-agent.md +122 -122
  113. package/.claude/commands/automation/self-healing.md +105 -105
  114. package/.claude/commands/automation/session-memory.md +89 -89
  115. package/.claude/commands/automation/smart-agents.md +72 -72
  116. package/.claude/commands/automation/smart-spawn.md +25 -25
  117. package/.claude/commands/automation/workflow-select.md +25 -25
  118. package/.claude/commands/claude-flow-help.md +103 -103
  119. package/.claude/commands/claude-flow-memory.md +107 -107
  120. package/.claude/commands/claude-flow-swarm.md +205 -205
  121. package/.claude/commands/coordination/README.md +9 -9
  122. package/.claude/commands/coordination/agent-spawn.md +25 -25
  123. package/.claude/commands/coordination/init.md +44 -44
  124. package/.claude/commands/coordination/orchestrate.md +43 -43
  125. package/.claude/commands/coordination/spawn.md +45 -45
  126. package/.claude/commands/coordination/swarm-init.md +85 -85
  127. package/.claude/commands/coordination/task-orchestrate.md +25 -25
  128. package/.claude/commands/github/README.md +11 -11
  129. package/.claude/commands/github/code-review-swarm.md +513 -513
  130. package/.claude/commands/github/code-review.md +25 -25
  131. package/.claude/commands/github/github-modes.md +146 -146
  132. package/.claude/commands/github/github-swarm.md +121 -121
  133. package/.claude/commands/github/issue-tracker.md +291 -291
  134. package/.claude/commands/github/issue-triage.md +25 -25
  135. package/.claude/commands/github/multi-repo-swarm.md +518 -518
  136. package/.claude/commands/github/pr-enhance.md +26 -26
  137. package/.claude/commands/github/pr-manager.md +169 -169
  138. package/.claude/commands/github/project-board-sync.md +470 -470
  139. package/.claude/commands/github/release-manager.md +339 -339
  140. package/.claude/commands/github/release-swarm.md +543 -543
  141. package/.claude/commands/github/repo-analyze.md +25 -25
  142. package/.claude/commands/github/repo-architect.md +366 -366
  143. package/.claude/commands/github/swarm-issue.md +484 -484
  144. package/.claude/commands/github/swarm-pr.md +287 -287
  145. package/.claude/commands/github/sync-coordinator.md +302 -302
  146. package/.claude/commands/github/workflow-automation.md +441 -441
  147. package/.claude/commands/hive-mind/README.md +17 -17
  148. package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
  149. package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
  150. package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
  151. package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
  152. package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
  153. package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
  154. package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
  155. package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
  156. package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
  157. package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
  158. package/.claude/commands/hive-mind/hive-mind.md +27 -27
  159. package/.claude/commands/hooks/README.md +11 -11
  160. package/.claude/commands/hooks/overview.md +57 -57
  161. package/.claude/commands/hooks/post-edit.md +117 -117
  162. package/.claude/commands/hooks/post-task.md +112 -112
  163. package/.claude/commands/hooks/pre-edit.md +113 -113
  164. package/.claude/commands/hooks/pre-task.md +111 -111
  165. package/.claude/commands/hooks/session-end.md +118 -118
  166. package/.claude/commands/hooks/setup.md +102 -102
  167. package/.claude/commands/memory/README.md +9 -9
  168. package/.claude/commands/memory/memory-persist.md +25 -25
  169. package/.claude/commands/memory/memory-search.md +25 -25
  170. package/.claude/commands/memory/memory-usage.md +25 -25
  171. package/.claude/commands/memory/neural.md +47 -47
  172. package/.claude/commands/monitoring/README.md +9 -9
  173. package/.claude/commands/monitoring/agent-metrics.md +25 -25
  174. package/.claude/commands/monitoring/agents.md +44 -44
  175. package/.claude/commands/monitoring/real-time-view.md +25 -25
  176. package/.claude/commands/monitoring/status.md +46 -46
  177. package/.claude/commands/monitoring/swarm-monitor.md +25 -25
  178. package/.claude/commands/optimization/README.md +9 -9
  179. package/.claude/commands/optimization/auto-topology.md +61 -61
  180. package/.claude/commands/optimization/cache-manage.md +25 -25
  181. package/.claude/commands/optimization/parallel-execute.md +25 -25
  182. package/.claude/commands/optimization/parallel-execution.md +49 -49
  183. package/.claude/commands/optimization/topology-optimize.md +25 -25
  184. package/.claude/commands/pair/README.md +260 -260
  185. package/.claude/commands/pair/commands.md +545 -545
  186. package/.claude/commands/pair/config.md +509 -509
  187. package/.claude/commands/pair/examples.md +511 -511
  188. package/.claude/commands/pair/modes.md +347 -347
  189. package/.claude/commands/pair/session.md +406 -406
  190. package/.claude/commands/pair/start.md +208 -208
  191. package/.claude/commands/sparc/analyzer.md +51 -51
  192. package/.claude/commands/sparc/architect.md +53 -53
  193. package/.claude/commands/sparc/ask.md +97 -97
  194. package/.claude/commands/sparc/batch-executor.md +54 -54
  195. package/.claude/commands/sparc/code.md +89 -89
  196. package/.claude/commands/sparc/coder.md +54 -54
  197. package/.claude/commands/sparc/debug.md +83 -83
  198. package/.claude/commands/sparc/debugger.md +54 -54
  199. package/.claude/commands/sparc/designer.md +53 -53
  200. package/.claude/commands/sparc/devops.md +109 -109
  201. package/.claude/commands/sparc/docs-writer.md +80 -80
  202. package/.claude/commands/sparc/documenter.md +54 -54
  203. package/.claude/commands/sparc/innovator.md +54 -54
  204. package/.claude/commands/sparc/integration.md +83 -83
  205. package/.claude/commands/sparc/mcp.md +117 -117
  206. package/.claude/commands/sparc/memory-manager.md +54 -54
  207. package/.claude/commands/sparc/optimizer.md +54 -54
  208. package/.claude/commands/sparc/orchestrator.md +131 -131
  209. package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
  210. package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
  211. package/.claude/commands/sparc/researcher.md +54 -54
  212. package/.claude/commands/sparc/reviewer.md +54 -54
  213. package/.claude/commands/sparc/security-review.md +80 -80
  214. package/.claude/commands/sparc/sparc-modes.md +174 -174
  215. package/.claude/commands/sparc/sparc.md +111 -111
  216. package/.claude/commands/sparc/spec-pseudocode.md +80 -80
  217. package/.claude/commands/sparc/supabase-admin.md +348 -348
  218. package/.claude/commands/sparc/swarm-coordinator.md +54 -54
  219. package/.claude/commands/sparc/tdd.md +54 -54
  220. package/.claude/commands/sparc/tester.md +54 -54
  221. package/.claude/commands/sparc/tutorial.md +79 -79
  222. package/.claude/commands/sparc/workflow-manager.md +54 -54
  223. package/.claude/commands/sparc.md +166 -166
  224. package/.claude/commands/stream-chain/pipeline.md +120 -120
  225. package/.claude/commands/stream-chain/run.md +69 -69
  226. package/.claude/commands/swarm/README.md +15 -15
  227. package/.claude/commands/swarm/analysis.md +95 -95
  228. package/.claude/commands/swarm/development.md +96 -96
  229. package/.claude/commands/swarm/examples.md +168 -168
  230. package/.claude/commands/swarm/maintenance.md +102 -102
  231. package/.claude/commands/swarm/optimization.md +117 -117
  232. package/.claude/commands/swarm/research.md +136 -136
  233. package/.claude/commands/swarm/swarm-analysis.md +8 -8
  234. package/.claude/commands/swarm/swarm-background.md +8 -8
  235. package/.claude/commands/swarm/swarm-init.md +19 -19
  236. package/.claude/commands/swarm/swarm-modes.md +8 -8
  237. package/.claude/commands/swarm/swarm-monitor.md +8 -8
  238. package/.claude/commands/swarm/swarm-spawn.md +19 -19
  239. package/.claude/commands/swarm/swarm-status.md +8 -8
  240. package/.claude/commands/swarm/swarm-strategies.md +8 -8
  241. package/.claude/commands/swarm/swarm.md +87 -87
  242. package/.claude/commands/swarm/testing.md +131 -131
  243. package/.claude/commands/training/README.md +9 -9
  244. package/.claude/commands/training/model-update.md +25 -25
  245. package/.claude/commands/training/neural-patterns.md +107 -107
  246. package/.claude/commands/training/neural-train.md +75 -75
  247. package/.claude/commands/training/pattern-learn.md +25 -25
  248. package/.claude/commands/training/specialization.md +62 -62
  249. package/.claude/commands/truth/start.md +142 -142
  250. package/.claude/commands/verify/check.md +49 -49
  251. package/.claude/commands/verify/start.md +127 -127
  252. package/.claude/commands/workflows/README.md +9 -9
  253. package/.claude/commands/workflows/development.md +77 -77
  254. package/.claude/commands/workflows/research.md +62 -62
  255. package/.claude/commands/workflows/workflow-create.md +25 -25
  256. package/.claude/commands/workflows/workflow-execute.md +25 -25
  257. package/.claude/commands/workflows/workflow-export.md +25 -25
  258. package/.claude/eval/human-relevance-frozen-v1.json +17 -17
  259. package/.claude/evolve-proof/generation-0.json +211 -211
  260. package/.claude/evolve-proof/real-generation-0.json +406 -406
  261. package/.claude/evolve-proof/real-generation-1.json +406 -406
  262. package/.claude/helpers/.helpers-version +1 -1
  263. package/.claude/helpers/README.md +96 -96
  264. package/.claude/helpers/adr-compliance.sh +186 -186
  265. package/.claude/helpers/auto-commit.sh +178 -178
  266. package/.claude/helpers/auto-memory-hook.mjs +430 -430
  267. package/.claude/helpers/checkpoint-manager.sh +251 -251
  268. package/.claude/helpers/daemon-manager.sh +252 -252
  269. package/.claude/helpers/ddd-tracker.sh +144 -144
  270. package/.claude/helpers/github-safe.js +156 -156
  271. package/.claude/helpers/github-setup.sh +45 -45
  272. package/.claude/helpers/guidance-hook.sh +13 -13
  273. package/.claude/helpers/guidance-hooks.sh +102 -102
  274. package/.claude/helpers/health-monitor.sh +108 -108
  275. package/.claude/helpers/helpers.manifest.json +6 -6
  276. package/.claude/helpers/hook-handler.cjs +606 -606
  277. package/.claude/helpers/intelligence.cjs +1169 -1169
  278. package/.claude/helpers/learning-hooks.sh +329 -329
  279. package/.claude/helpers/learning-optimizer.sh +127 -127
  280. package/.claude/helpers/learning-service.mjs +1144 -1144
  281. package/.claude/helpers/memory.js +83 -83
  282. package/.claude/helpers/metrics-db.mjs +503 -503
  283. package/.claude/helpers/pattern-consolidator.sh +86 -86
  284. package/.claude/helpers/perf-worker.sh +160 -160
  285. package/.claude/helpers/post-commit +16 -16
  286. package/.claude/helpers/pre-commit +26 -26
  287. package/.claude/helpers/quick-start.sh +19 -19
  288. package/.claude/helpers/router.js +105 -105
  289. package/.claude/helpers/security-scanner.sh +127 -127
  290. package/.claude/helpers/session.js +157 -157
  291. package/.claude/helpers/setup-mcp.sh +18 -18
  292. package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
  293. package/.claude/helpers/statusline-hook.sh +21 -21
  294. package/.claude/helpers/statusline.cjs +1290 -1290
  295. package/.claude/helpers/statusline.js +340 -340
  296. package/.claude/helpers/swarm-comms.sh +353 -353
  297. package/.claude/helpers/swarm-hooks.sh +761 -761
  298. package/.claude/helpers/swarm-monitor.sh +210 -210
  299. package/.claude/helpers/sync-v3-metrics.sh +245 -245
  300. package/.claude/helpers/update-v3-progress.sh +165 -165
  301. package/.claude/helpers/v3-quick-status.sh +57 -57
  302. package/.claude/helpers/v3.sh +110 -110
  303. package/.claude/helpers/validate-v3-config.sh +215 -215
  304. package/.claude/helpers/worker-manager.sh +170 -170
  305. package/.claude/proven-config.json +41 -41
  306. package/.claude/proven-config.manifest.json +37 -37
  307. package/.claude/proven-config.signed.json +41 -41
  308. package/.claude/settings.json +182 -182
  309. package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
  310. package/.claude/skills/agentdb-learning/SKILL.md +545 -545
  311. package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
  312. package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
  313. package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
  314. package/.claude/skills/browser/SKILL.md +204 -204
  315. package/.claude/skills/dual-mode/README.md +71 -71
  316. package/.claude/skills/dual-mode/dual-collect.md +103 -103
  317. package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
  318. package/.claude/skills/dual-mode/dual-spawn.md +81 -81
  319. package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
  320. package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
  321. package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
  322. package/.claude/skills/github-code-review/SKILL.md +1125 -1125
  323. package/.claude/skills/github-multi-repo/SKILL.md +862 -862
  324. package/.claude/skills/github-project-management/SKILL.md +1262 -1262
  325. package/.claude/skills/github-release-management/SKILL.md +1064 -1064
  326. package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
  327. package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
  328. package/.claude/skills/pair-programming/SKILL.md +1202 -1202
  329. package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
  330. package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
  331. package/.claude/skills/skill-builder/SKILL.md +910 -910
  332. package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
  333. package/.claude/skills/stream-chain/SKILL.md +560 -560
  334. package/.claude/skills/swarm-advanced/SKILL.md +970 -970
  335. package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
  336. package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
  337. package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
  338. package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
  339. package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
  340. package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
  341. package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
  342. package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
  343. package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
  344. package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
  345. package/.claude/skills/verification-quality/SKILL.md +691 -691
  346. package/README.md +422 -422
  347. package/bin/cli.js +338 -338
  348. package/bin/mcp-server.js +224 -224
  349. package/bin/preinstall.cjs +2 -2
  350. package/catalog-manifest.json +2 -2
  351. package/dist/src/benchmarks/gaia-critic.js +24 -24
  352. package/dist/src/business-pods/bbs-budget-tracker.js +53 -53
  353. package/dist/src/commands/completions.js +409 -409
  354. package/dist/src/commands/daemon.js +44 -44
  355. package/dist/src/commands/doctor.js +4 -4
  356. package/dist/src/commands/embeddings.js +26 -26
  357. package/dist/src/commands/hive-mind.js +97 -97
  358. package/dist/src/commands/hooks.js +9 -9
  359. package/dist/src/commands/init.js +75 -75
  360. package/dist/src/commands/ruvector/backup.js +23 -23
  361. package/dist/src/commands/ruvector/benchmark.js +31 -31
  362. package/dist/src/commands/ruvector/import.js +14 -14
  363. package/dist/src/commands/ruvector/init.js +115 -115
  364. package/dist/src/commands/ruvector/migrate.js +99 -99
  365. package/dist/src/commands/ruvector/optimize.js +51 -51
  366. package/dist/src/commands/ruvector/setup.js +624 -624
  367. package/dist/src/commands/ruvector/status.js +38 -38
  368. package/dist/src/config/proven-config.js +2 -2
  369. package/dist/src/init/claudemd-generator.js +273 -273
  370. package/dist/src/init/executor.js +453 -453
  371. package/dist/src/init/helper-signing.js +2 -2
  372. package/dist/src/init/helpers-generator.js +917 -751
  373. package/dist/src/init/statusline-generator.js +24 -24
  374. package/dist/src/mcp-tools/agentdb-tools.js +15 -15
  375. package/dist/src/mcp-tools/browser-intent-tools.js +19 -19
  376. package/dist/src/mcp-tools/seraphina-tools.js +4 -4
  377. package/dist/src/memory/graph-edge-writer.js +22 -22
  378. package/dist/src/memory/memory-bridge.js +114 -114
  379. package/dist/src/memory/memory-initializer.js +416 -416
  380. package/dist/src/memory/rabitq-index.js +5 -5
  381. package/dist/src/proxy/verify.js +2 -2
  382. package/dist/src/runtime/headless.js +28 -28
  383. package/dist/src/ruvector/diskann-backend.d.ts +78 -0
  384. package/dist/src/ruvector/diskann-backend.js +310 -0
  385. package/dist/src/services/distill-tuning.js +7 -7
  386. package/dist/src/services/headless-worker-executor.js +84 -84
  387. package/dist/src/services/memory-distillation.js +18 -18
  388. package/dist/src/transfer/deploy-seraphine.js +23 -23
  389. package/node_modules/@claude-flow/codex/.agents/skills/github-automation/SKILL.md +32 -32
  390. package/node_modules/@claude-flow/codex/.agents/skills/memory-management/SKILL.md +45 -45
  391. package/node_modules/@claude-flow/codex/.agents/skills/performance-analysis/SKILL.md +32 -32
  392. package/node_modules/@claude-flow/codex/.agents/skills/security-audit/SKILL.md +46 -46
  393. package/node_modules/@claude-flow/codex/.agents/skills/sparc-methodology/SKILL.md +46 -46
  394. package/node_modules/@claude-flow/codex/.agents/skills/swarm-orchestration/SKILL.md +53 -53
  395. package/node_modules/@claude-flow/codex/README.md +1044 -1044
  396. package/node_modules/@claude-flow/codex/dist/dual-mode/orchestrator.js +13 -13
  397. package/node_modules/@claude-flow/codex/dist/generators/agents-md.js +664 -664
  398. package/node_modules/@claude-flow/codex/dist/generators/config-toml.js +455 -455
  399. package/node_modules/@claude-flow/codex/dist/generators/skill-md.js +45 -45
  400. package/node_modules/@claude-flow/codex/dist/initializer.js +167 -167
  401. package/node_modules/@claude-flow/codex/dist/templates/index.js +15 -15
  402. package/node_modules/@claude-flow/mcp/README.md +429 -429
  403. package/node_modules/@claude-flow/plugin-agent-federation/README.md +49 -49
  404. package/node_modules/@claude-flow/security/README.md +292 -292
  405. package/node_modules/@claude-flow/security/dist/credential-generator.js +9 -9
  406. package/node_modules/@claude-flow/security/dist/input-validator.d.ts +6 -6
  407. package/node_modules/@claude-flow/security/dist/oauth/callback-server.js +9 -9
  408. package/package.json +181 -181
  409. package/plugins/ruflo-metaharness/.claude-plugin/plugin.json +32 -32
  410. package/plugins/ruflo-metaharness/README.md +72 -72
  411. package/plugins/ruflo-metaharness/agents/metaharness-architect.md +58 -58
  412. package/plugins/ruflo-metaharness/commands/ruflo-metaharness.md +50 -50
  413. package/plugins/ruflo-metaharness/scripts/_darwin.mjs +210 -210
  414. package/plugins/ruflo-metaharness/scripts/_harness.mjs +334 -334
  415. package/plugins/ruflo-metaharness/scripts/_invoke.mjs +230 -230
  416. package/plugins/ruflo-metaharness/scripts/_redblue.mjs +143 -143
  417. package/plugins/ruflo-metaharness/scripts/_similarity.mjs +161 -161
  418. package/plugins/ruflo-metaharness/scripts/_spike-similarity.mjs +223 -223
  419. package/plugins/ruflo-metaharness/scripts/audit-list.mjs +158 -158
  420. package/plugins/ruflo-metaharness/scripts/audit-trend.mjs +272 -272
  421. package/plugins/ruflo-metaharness/scripts/bench-parse-mcp-scan.mjs +146 -146
  422. package/plugins/ruflo-metaharness/scripts/bench-recordpair-overhead.mjs +186 -186
  423. package/plugins/ruflo-metaharness/scripts/bench-similarity.mjs +177 -177
  424. package/plugins/ruflo-metaharness/scripts/bench.mjs +95 -95
  425. package/plugins/ruflo-metaharness/scripts/drift-from-history.mjs +363 -363
  426. package/plugins/ruflo-metaharness/scripts/evolve.mjs +404 -404
  427. package/plugins/ruflo-metaharness/scripts/genome.mjs +105 -105
  428. package/plugins/ruflo-metaharness/scripts/gepa.mjs +153 -153
  429. package/plugins/ruflo-metaharness/scripts/learn.mjs +127 -127
  430. package/plugins/ruflo-metaharness/scripts/mcp-scan.mjs +110 -110
  431. package/plugins/ruflo-metaharness/scripts/mint.mjs +126 -126
  432. package/plugins/ruflo-metaharness/scripts/oia-audit.mjs +228 -228
  433. package/plugins/ruflo-metaharness/scripts/redblue.mjs +286 -286
  434. package/plugins/ruflo-metaharness/scripts/router-parallel-analyze.mjs +250 -250
  435. package/plugins/ruflo-metaharness/scripts/score.mjs +92 -92
  436. package/plugins/ruflo-metaharness/scripts/security-bench.mjs +174 -174
  437. package/plugins/ruflo-metaharness/scripts/similarity.mjs +158 -158
  438. package/plugins/ruflo-metaharness/scripts/smoke.sh +2422 -2422
  439. package/plugins/ruflo-metaharness/scripts/test-graceful-degradation.mjs +165 -165
  440. package/plugins/ruflo-metaharness/scripts/test-mcp-tools.mjs +498 -498
  441. package/plugins/ruflo-metaharness/scripts/test-parallel-pipeline.mjs +204 -204
  442. package/plugins/ruflo-metaharness/scripts/test-pipeline-roundtrip.mjs +586 -586
  443. package/plugins/ruflo-metaharness/scripts/test-similarity.mjs +372 -372
  444. package/plugins/ruflo-metaharness/scripts/test-with-openrouter.mjs +229 -229
  445. package/plugins/ruflo-metaharness/scripts/threat-model.mjs +62 -62
  446. package/plugins/ruflo-metaharness/skills/harness-bench/SKILL.md +64 -64
  447. package/plugins/ruflo-metaharness/skills/harness-drift-from-history/SKILL.md +65 -65
  448. package/plugins/ruflo-metaharness/skills/harness-evolve/SKILL.md +131 -131
  449. package/plugins/ruflo-metaharness/skills/harness-genome/SKILL.md +57 -57
  450. package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +65 -65
  451. package/plugins/ruflo-metaharness/skills/harness-learn/SKILL.md +65 -65
  452. package/plugins/ruflo-metaharness/skills/harness-mcp-scan/SKILL.md +49 -49
  453. package/plugins/ruflo-metaharness/skills/harness-mint/SKILL.md +72 -72
  454. package/plugins/ruflo-metaharness/skills/harness-oia-audit/SKILL.md +79 -79
  455. package/plugins/ruflo-metaharness/skills/harness-score/SKILL.md +66 -66
  456. package/plugins/ruflo-metaharness/skills/harness-security-bench/SKILL.md +101 -101
  457. package/plugins/ruflo-metaharness/skills/harness-similarity/SKILL.md +67 -67
  458. package/plugins/ruflo-metaharness/skills/harness-threat-model/SKILL.md +41 -41
  459. package/scripts/postinstall.cjs +153 -153
@@ -1,79 +1,79 @@
1
- ---
2
- name: harness-oia-audit
3
- description: Composite Phase-2 audit worker (ADR-150). Bundles harness oia-manifest + threat-model + mcp-scan into one timestamped audit record stored in the `metaharness-audit` memory namespace. Designed for cron-scheduled drift detection.
4
- argument-hint: "[--path .] [--dry-run] [--alert-on-worst clean|low|medium|high] [--format table|json]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- The 13th worker (ADR-150 Phase 2) — runs three MetaHarness static
9
- surfaces in one shot, computes a composite worst-severity signal, and
10
- persists the audit record to memory so drift over time is visible.
11
-
12
- ## Algorithm
13
-
14
- Implementation: [`scripts/oia-audit.mjs`](../../scripts/oia-audit.mjs).
15
-
16
- 1. Run `harness oia-manifest <path>` — Open Infrastructure Architecture
17
- layer alignment (L1-L9).
18
- 2. Run `harness threat-model <path>` — categorized MCP-surface threat
19
- report with `worst: clean|low|medium|high`.
20
- 3. Run `harness mcp-scan <path>` — per-server/tool policy + permissions
21
- + dep findings.
22
- 4. Composite worst = `max(threatModel.worst, max(mcpScan.findings.severity))`.
23
- 5. Persist payload to memory namespace `metaharness-audit` with key
24
- `audit-<iso-timestamp>` (unless `--dry-run`).
25
- 6. `--alert-on-worst <severity>`: exit 1 if composite worst ≥ threshold.
26
-
27
- ## Graceful degradation
28
-
29
- When ALL three components report `metaharness-not-available`, the script
30
- emits the standard degraded payload and exits 0. When only some are
31
- degraded, each individual component carries its own `degraded: true`
32
- flag in the audit record — the audit still runs and persists what it
33
- could gather.
34
-
35
- ## CI / cron integration
36
-
37
- Designed for weekly cron in `.github/workflows/`:
38
-
39
- ```yaml
40
- on:
41
- schedule:
42
- - cron: '17 4 * * 0' # Sundays at 04:17 UTC
43
- jobs:
44
- oia-audit:
45
- runs-on: ubuntu-latest
46
- steps:
47
- - uses: actions/checkout@v4
48
- - uses: actions/setup-node@v4
49
- - run: node plugins/ruflo-metaharness/scripts/oia-audit.mjs --alert-on-worst high
50
- ```
51
-
52
- `--alert-on-worst high` fails the job on any HIGH-severity finding;
53
- drift below HIGH is logged but doesn't block.
54
-
55
- ## Memory namespace
56
-
57
- Each audit run stores under `metaharness-audit:audit-<iso-ts>`. To list
58
- recent audits:
59
-
60
- ```bash
61
- npx @claude-flow/cli@latest memory list --namespace metaharness-audit --limit 10
62
- ```
63
-
64
- To diff two audits (drift detection):
65
-
66
- ```bash
67
- A=$(npx ... memory retrieve --key audit-2026-06-01... --namespace metaharness-audit)
68
- B=$(npx ... memory retrieve --key audit-2026-06-15... --namespace metaharness-audit)
69
- # Compare composite.worst, components.threatModel.worst, etc.
70
- ```
71
-
72
- A future ADR can wire this into a dedicated `cost-diff`-style diff
73
- viewer specifically for audit drift.
74
-
75
- ## Pairs with
76
-
77
- - `harness-threat-model` — the underlying threat-model component
78
- - `harness-mcp-scan` — the underlying MCP-scan component
79
- - `harness-score` + `harness-genome` — readiness metrics (orthogonal to audit)
1
+ ---
2
+ name: harness-oia-audit
3
+ description: Composite Phase-2 audit worker (ADR-150). Bundles harness oia-manifest + threat-model + mcp-scan into one timestamped audit record stored in the `metaharness-audit` memory namespace. Designed for cron-scheduled drift detection.
4
+ argument-hint: "[--path .] [--dry-run] [--alert-on-worst clean|low|medium|high] [--format table|json]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ The 13th worker (ADR-150 Phase 2) — runs three MetaHarness static
9
+ surfaces in one shot, computes a composite worst-severity signal, and
10
+ persists the audit record to memory so drift over time is visible.
11
+
12
+ ## Algorithm
13
+
14
+ Implementation: [`scripts/oia-audit.mjs`](../../scripts/oia-audit.mjs).
15
+
16
+ 1. Run `harness oia-manifest <path>` — Open Infrastructure Architecture
17
+ layer alignment (L1-L9).
18
+ 2. Run `harness threat-model <path>` — categorized MCP-surface threat
19
+ report with `worst: clean|low|medium|high`.
20
+ 3. Run `harness mcp-scan <path>` — per-server/tool policy + permissions
21
+ + dep findings.
22
+ 4. Composite worst = `max(threatModel.worst, max(mcpScan.findings.severity))`.
23
+ 5. Persist payload to memory namespace `metaharness-audit` with key
24
+ `audit-<iso-timestamp>` (unless `--dry-run`).
25
+ 6. `--alert-on-worst <severity>`: exit 1 if composite worst ≥ threshold.
26
+
27
+ ## Graceful degradation
28
+
29
+ When ALL three components report `metaharness-not-available`, the script
30
+ emits the standard degraded payload and exits 0. When only some are
31
+ degraded, each individual component carries its own `degraded: true`
32
+ flag in the audit record — the audit still runs and persists what it
33
+ could gather.
34
+
35
+ ## CI / cron integration
36
+
37
+ Designed for weekly cron in `.github/workflows/`:
38
+
39
+ ```yaml
40
+ on:
41
+ schedule:
42
+ - cron: '17 4 * * 0' # Sundays at 04:17 UTC
43
+ jobs:
44
+ oia-audit:
45
+ runs-on: ubuntu-latest
46
+ steps:
47
+ - uses: actions/checkout@v4
48
+ - uses: actions/setup-node@v4
49
+ - run: node plugins/ruflo-metaharness/scripts/oia-audit.mjs --alert-on-worst high
50
+ ```
51
+
52
+ `--alert-on-worst high` fails the job on any HIGH-severity finding;
53
+ drift below HIGH is logged but doesn't block.
54
+
55
+ ## Memory namespace
56
+
57
+ Each audit run stores under `metaharness-audit:audit-<iso-ts>`. To list
58
+ recent audits:
59
+
60
+ ```bash
61
+ npx @claude-flow/cli@latest memory list --namespace metaharness-audit --limit 10
62
+ ```
63
+
64
+ To diff two audits (drift detection):
65
+
66
+ ```bash
67
+ A=$(npx ... memory retrieve --key audit-2026-06-01... --namespace metaharness-audit)
68
+ B=$(npx ... memory retrieve --key audit-2026-06-15... --namespace metaharness-audit)
69
+ # Compare composite.worst, components.threatModel.worst, etc.
70
+ ```
71
+
72
+ A future ADR can wire this into a dedicated `cost-diff`-style diff
73
+ viewer specifically for audit drift.
74
+
75
+ ## Pairs with
76
+
77
+ - `harness-threat-model` — the underlying threat-model component
78
+ - `harness-mcp-scan` — the underlying MCP-scan component
79
+ - `harness-score` + `harness-genome` — readiness metrics (orthogonal to audit)
@@ -1,66 +1,66 @@
1
- ---
2
- name: harness-score
3
- description: 5-dimension harness readiness scorecard from `metaharness score <path>`. Returns harnessFit / compileConfidence / taskCoverage / toolSafety / memoryUsefulness + estCostPerRunUsd + scaffoldReady. Pure-read; subprocess invocation; degrades gracefully when MetaHarness is absent (ADR-150 architectural constraint).
4
- argument-hint: "[--path .] [--alert-on-fit-below 70] [--format table|json]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- Surfaces the upstream `metaharness score` CLI as a ruflo skill. Use when
9
- Claude Code needs to assess whether a repo is ready for harness adoption
10
- before recommending the user run `npx ruflo init` or `harness-mint`.
11
-
12
- ## Algorithm
13
-
14
- Implementation: [`scripts/score.mjs`](../../scripts/score.mjs).
15
-
16
- 1. Shell out to `npx metaharness score <path> --json` (single subprocess,
17
- 60s hard timeout).
18
- 2. Parse the JSON shape: `{ harnessFit, compileConfidence, taskCoverage,
19
- toolSafety, memoryUsefulness, estCostPerRunUsd, recommendedMode,
20
- archetype, template, scaffoldReady, hardConstraints }`.
21
- 3. If `--alert-on-fit-below N`: exit 1 when `harnessFit < N`.
22
- 4. Output JSON (default) or markdown table.
23
-
24
- ## Phase-0 baseline (ruflo's own scorecard, measured 2026-06-16)
25
-
26
- | Dimension | Value |
27
- |---|---:|
28
- | harnessFit | 82/100 |
29
- | compileConfidence | 100 |
30
- | taskCoverage | 79 |
31
- | toolSafety | 100 |
32
- | memoryUsefulness | 40 |
33
- | estCostPerRunUsd | $0.048 |
34
- | recommendedMode | CLI + MCP |
35
- | archetype | typescript-sdk-harness |
36
- | template | vertical:coding |
37
- | scaffoldReady | true |
38
-
39
- Ruflo passes its own readiness check. `memoryUsefulness: 40` is the
40
- weakest dimension — track this as a leading indicator for future memory
41
- work in the AgentDB layer.
42
-
43
- ## CI integration
44
-
45
- ```bash
46
- node plugins/ruflo-metaharness/scripts/score.mjs --alert-on-fit-below 70 --format json
47
- ```
48
-
49
- Exit 1 fails the build. Pair with `harness-genome` for the full
50
- 7-section view.
51
-
52
- ## Graceful degradation (ADR-150 architectural constraint rule #3)
53
-
54
- When `metaharness` is not installed and `npx` can't fetch it (offline,
55
- no network, registry unreachable), the script emits:
56
-
57
- ```json
58
- {
59
- "degraded": true,
60
- "reason": "metaharness-not-available",
61
- "hint": "Install with `npm i -D metaharness@~0.3.0` (pinned range — this plugin never fetches @latest) or verify network access for the one-time cache install."
62
- }
63
- ```
64
-
65
- and exits 0. Ruflo continues to function — this is the architectural
66
- constraint in action.
1
+ ---
2
+ name: harness-score
3
+ description: 5-dimension harness readiness scorecard from `metaharness score <path>`. Returns harnessFit / compileConfidence / taskCoverage / toolSafety / memoryUsefulness + estCostPerRunUsd + scaffoldReady. Pure-read; subprocess invocation; degrades gracefully when MetaHarness is absent (ADR-150 architectural constraint).
4
+ argument-hint: "[--path .] [--alert-on-fit-below 70] [--format table|json]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ Surfaces the upstream `metaharness score` CLI as a ruflo skill. Use when
9
+ Claude Code needs to assess whether a repo is ready for harness adoption
10
+ before recommending the user run `npx ruflo init` or `harness-mint`.
11
+
12
+ ## Algorithm
13
+
14
+ Implementation: [`scripts/score.mjs`](../../scripts/score.mjs).
15
+
16
+ 1. Shell out to `npx metaharness score <path> --json` (single subprocess,
17
+ 60s hard timeout).
18
+ 2. Parse the JSON shape: `{ harnessFit, compileConfidence, taskCoverage,
19
+ toolSafety, memoryUsefulness, estCostPerRunUsd, recommendedMode,
20
+ archetype, template, scaffoldReady, hardConstraints }`.
21
+ 3. If `--alert-on-fit-below N`: exit 1 when `harnessFit < N`.
22
+ 4. Output JSON (default) or markdown table.
23
+
24
+ ## Phase-0 baseline (ruflo's own scorecard, measured 2026-06-16)
25
+
26
+ | Dimension | Value |
27
+ |---|---:|
28
+ | harnessFit | 82/100 |
29
+ | compileConfidence | 100 |
30
+ | taskCoverage | 79 |
31
+ | toolSafety | 100 |
32
+ | memoryUsefulness | 40 |
33
+ | estCostPerRunUsd | $0.048 |
34
+ | recommendedMode | CLI + MCP |
35
+ | archetype | typescript-sdk-harness |
36
+ | template | vertical:coding |
37
+ | scaffoldReady | true |
38
+
39
+ Ruflo passes its own readiness check. `memoryUsefulness: 40` is the
40
+ weakest dimension — track this as a leading indicator for future memory
41
+ work in the AgentDB layer.
42
+
43
+ ## CI integration
44
+
45
+ ```bash
46
+ node plugins/ruflo-metaharness/scripts/score.mjs --alert-on-fit-below 70 --format json
47
+ ```
48
+
49
+ Exit 1 fails the build. Pair with `harness-genome` for the full
50
+ 7-section view.
51
+
52
+ ## Graceful degradation (ADR-150 architectural constraint rule #3)
53
+
54
+ When `metaharness` is not installed and `npx` can't fetch it (offline,
55
+ no network, registry unreachable), the script emits:
56
+
57
+ ```json
58
+ {
59
+ "degraded": true,
60
+ "reason": "metaharness-not-available",
61
+ "hint": "Install with `npm i -D metaharness@~0.3.0` (pinned range — this plugin never fetches @latest) or verify network access for the one-time cache install."
62
+ }
63
+ ```
64
+
65
+ and exits 0. Ruflo continues to function — this is the architectural
66
+ constraint in action.
@@ -1,101 +1,101 @@
1
- ---
2
- name: harness-security-bench
3
- description: Run `@metaharness/darwin security bench` (upstream "Darwin Shield" / ADR-155) — evolves a champion security-detection harness against a 10-vuln / 9-decoy corpus and grades it on TPR/FPR/patch-pass/repro/unsafe vs four baselines (B0 static, B1 LLM-single-pass, B2 fixed-agent, B3 Darwin-champion). Closest reference implementation for ruflo's own ADR-155 nightly self-learning security harness (PR #2417). Degrades gracefully when @metaharness/darwin is absent.
4
- argument-hint: "[--population 2] [--cycles 1] [--seed N] [--alert-on-fail]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- Surfaces the upstream `metaharness-darwin security bench` command. **This is
9
- the upstream's own ADR-155 — Darwin Shield — and is the closest reference
10
- implementation for ruflo's nightly self-learning security harness ([#2417](https://github.com/ruvnet/ruflo/pull/2417)).**
11
-
12
- ## Why this matters for ruflo's ADR-155
13
-
14
- ruflo's ADR-155 proposes three learning loops (per-dimension confidence,
15
- severity calibration, auto-fix bid). Loop A trains on accumulated
16
- `(finding, dimension, human_outcome)` tuples — but the gradient signal is
17
- only sound if the underlying detection mechanism converges on a known-good
18
- corpus. Darwin Shield evolves exactly that mechanism on a 10-vuln/9-decoy
19
- ground-truth set. Running this nightly gives us:
20
-
21
- - **Empirical floor:** if Darwin Shield's champion can't reach
22
- TPR=1/FPR=0 on the bench corpus, our Loop A's reward signal is noise.
23
- - **Drift detection:** week-over-week champion fitness deltas surface
24
- when the security landscape (or our mutator policy) shifts.
25
- - **Baseline diversity:** the 4 baselines (B0–B3) give us 4 anchor
26
- points to weight per-dimension confidence against.
27
-
28
- ## Algorithm
29
-
30
- Implementation: [`scripts/security-bench.mjs`](../../scripts/security-bench.mjs).
31
-
32
- 1. Shell to `npx -y @metaharness/darwin@~0.8.0 metaharness-darwin security bench --population N --cycles N [--seed S]`.
33
- 2. Default timeout = `3s × 19 evaluations × population × cycles + 30s overhead`.
34
- At default `--population 2 --cycles 1` ≈ 144s; at `--population 4 --cycles 3` ≈ 12 min.
35
- 3. Parse the markdown report — overall PASS/FAIL plus per-gate
36
- pass/fail rows (gate examples: "TPR improvement ≥ 25% vs fixed",
37
- "FPR reduction ≥ 40%", "Patch-test pass rate ≥ 80%", "Reproduction
38
- success ≥ 90%", "Unsafe outputs = 0", "Cost increase ≤ 2× fixed",
39
- "Beyond SOTA: champion statistically beats previous champion",
40
- "Compounding: false-positive repeat-rate drop ≥ 35%").
41
- 4. Parse the baselines-vs-champion table (4 rows: fitness/TPR/FPR/patchPass/
42
- repro/unsafe/cost per harness).
43
- 5. Emit structured JSON. With `--alert-on-fail`, exit 1 when overall = FAIL.
44
-
45
- ## Output shape
46
-
47
- ```json
48
- {
49
- "success": true,
50
- "data": {
51
- "overall": { "ok": true, "icon": "✅" },
52
- "gates": {
53
- "total": 11,
54
- "passed": 11,
55
- "failed": 0,
56
- "details": [{ "ok": true, "criterion": "TPR improvement ≥ 25% vs fixed harness", "measured": "+150% (B2 0.4 → B3 1)" }, ...]
57
- },
58
- "baselines": [
59
- { "harness": "static-only", "fitness": 0.5665, "tpr": 0.3, "fpr": 1, "unsafe": 0, ... },
60
- { "harness": "LLM single-pass", "fitness": 0.1365, ... },
61
- { "harness": "fixed agent", "fitness": 0.598, ... },
62
- { "harness": "Darwin champion", "fitness": 0.93275, "tpr": 1, "fpr": 0, ... }
63
- ],
64
- "rawMarkdown": "...",
65
- "shape": { "population": 2, "cycles": 1, "seed": null },
66
- "durationMs": 142000
67
- }
68
- }
69
- ```
70
-
71
- ## Wiring into ADR-155 nightly harness
72
-
73
- The ADR-155 nightly workflow (per #2418 task `W1.5`) will spawn this as
74
- one of the active-pentest dimension's calls — its results become a
75
- trajectory record:
76
-
77
- ```jsonc
78
- {
79
- "dimension": "mcp-pentest",
80
- "subdimension": "darwin-shield-bench",
81
- "champion_fitness": 0.93275,
82
- "champion_tpr": 1, "champion_fpr": 0,
83
- "gates_passed": 11, "gates_failed": 0,
84
- "shape": { "population": 4, "cycles": 3 }
85
- }
86
- ```
87
-
88
- Loop A learns: if `darwin-shield-bench` consistently passes on the seeded
89
- corpus, weight findings caught only by `mcp-pentest` higher.
90
-
91
- ## Exit codes
92
-
93
- | Code | Meaning |
94
- |---|---|
95
- | 0 | Bench ran (overall PASS or FAIL — distinguish via JSON `overall.ok`), or degraded |
96
- | 1 | `--alert-on-fail` and `overall.ok === false` |
97
- | 2 | Config error or upstream infrastructure failure |
98
-
99
- ## Graceful degradation
100
-
101
- When `@metaharness/darwin` is absent, emits `{degraded: true, reason: 'metaharness-darwin-not-available'}` and exits 0.
1
+ ---
2
+ name: harness-security-bench
3
+ description: Run `@metaharness/darwin security bench` (upstream "Darwin Shield" / ADR-155) — evolves a champion security-detection harness against a 10-vuln / 9-decoy corpus and grades it on TPR/FPR/patch-pass/repro/unsafe vs four baselines (B0 static, B1 LLM-single-pass, B2 fixed-agent, B3 Darwin-champion). Closest reference implementation for ruflo's own ADR-155 nightly self-learning security harness (PR #2417). Degrades gracefully when @metaharness/darwin is absent.
4
+ argument-hint: "[--population 2] [--cycles 1] [--seed N] [--alert-on-fail]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ Surfaces the upstream `metaharness-darwin security bench` command. **This is
9
+ the upstream's own ADR-155 — Darwin Shield — and is the closest reference
10
+ implementation for ruflo's nightly self-learning security harness ([#2417](https://github.com/ruvnet/ruflo/pull/2417)).**
11
+
12
+ ## Why this matters for ruflo's ADR-155
13
+
14
+ ruflo's ADR-155 proposes three learning loops (per-dimension confidence,
15
+ severity calibration, auto-fix bid). Loop A trains on accumulated
16
+ `(finding, dimension, human_outcome)` tuples — but the gradient signal is
17
+ only sound if the underlying detection mechanism converges on a known-good
18
+ corpus. Darwin Shield evolves exactly that mechanism on a 10-vuln/9-decoy
19
+ ground-truth set. Running this nightly gives us:
20
+
21
+ - **Empirical floor:** if Darwin Shield's champion can't reach
22
+ TPR=1/FPR=0 on the bench corpus, our Loop A's reward signal is noise.
23
+ - **Drift detection:** week-over-week champion fitness deltas surface
24
+ when the security landscape (or our mutator policy) shifts.
25
+ - **Baseline diversity:** the 4 baselines (B0–B3) give us 4 anchor
26
+ points to weight per-dimension confidence against.
27
+
28
+ ## Algorithm
29
+
30
+ Implementation: [`scripts/security-bench.mjs`](../../scripts/security-bench.mjs).
31
+
32
+ 1. Shell to `npx -y @metaharness/darwin@~0.8.0 metaharness-darwin security bench --population N --cycles N [--seed S]`.
33
+ 2. Default timeout = `3s × 19 evaluations × population × cycles + 30s overhead`.
34
+ At default `--population 2 --cycles 1` ≈ 144s; at `--population 4 --cycles 3` ≈ 12 min.
35
+ 3. Parse the markdown report — overall PASS/FAIL plus per-gate
36
+ pass/fail rows (gate examples: "TPR improvement ≥ 25% vs fixed",
37
+ "FPR reduction ≥ 40%", "Patch-test pass rate ≥ 80%", "Reproduction
38
+ success ≥ 90%", "Unsafe outputs = 0", "Cost increase ≤ 2× fixed",
39
+ "Beyond SOTA: champion statistically beats previous champion",
40
+ "Compounding: false-positive repeat-rate drop ≥ 35%").
41
+ 4. Parse the baselines-vs-champion table (4 rows: fitness/TPR/FPR/patchPass/
42
+ repro/unsafe/cost per harness).
43
+ 5. Emit structured JSON. With `--alert-on-fail`, exit 1 when overall = FAIL.
44
+
45
+ ## Output shape
46
+
47
+ ```json
48
+ {
49
+ "success": true,
50
+ "data": {
51
+ "overall": { "ok": true, "icon": "✅" },
52
+ "gates": {
53
+ "total": 11,
54
+ "passed": 11,
55
+ "failed": 0,
56
+ "details": [{ "ok": true, "criterion": "TPR improvement ≥ 25% vs fixed harness", "measured": "+150% (B2 0.4 → B3 1)" }, ...]
57
+ },
58
+ "baselines": [
59
+ { "harness": "static-only", "fitness": 0.5665, "tpr": 0.3, "fpr": 1, "unsafe": 0, ... },
60
+ { "harness": "LLM single-pass", "fitness": 0.1365, ... },
61
+ { "harness": "fixed agent", "fitness": 0.598, ... },
62
+ { "harness": "Darwin champion", "fitness": 0.93275, "tpr": 1, "fpr": 0, ... }
63
+ ],
64
+ "rawMarkdown": "...",
65
+ "shape": { "population": 2, "cycles": 1, "seed": null },
66
+ "durationMs": 142000
67
+ }
68
+ }
69
+ ```
70
+
71
+ ## Wiring into ADR-155 nightly harness
72
+
73
+ The ADR-155 nightly workflow (per #2418 task `W1.5`) will spawn this as
74
+ one of the active-pentest dimension's calls — its results become a
75
+ trajectory record:
76
+
77
+ ```jsonc
78
+ {
79
+ "dimension": "mcp-pentest",
80
+ "subdimension": "darwin-shield-bench",
81
+ "champion_fitness": 0.93275,
82
+ "champion_tpr": 1, "champion_fpr": 0,
83
+ "gates_passed": 11, "gates_failed": 0,
84
+ "shape": { "population": 4, "cycles": 3 }
85
+ }
86
+ ```
87
+
88
+ Loop A learns: if `darwin-shield-bench` consistently passes on the seeded
89
+ corpus, weight findings caught only by `mcp-pentest` higher.
90
+
91
+ ## Exit codes
92
+
93
+ | Code | Meaning |
94
+ |---|---|
95
+ | 0 | Bench ran (overall PASS or FAIL — distinguish via JSON `overall.ok`), or degraded |
96
+ | 1 | `--alert-on-fail` and `overall.ok === false` |
97
+ | 2 | Config error or upstream infrastructure failure |
98
+
99
+ ## Graceful degradation
100
+
101
+ When `@metaharness/darwin` is absent, emits `{degraded: true, reason: 'metaharness-darwin-not-available'}` and exits 0.
@@ -1,67 +1,67 @@
1
- ---
2
- name: harness-similarity
3
- description: ADR-152 — weighted similarity between two harness fingerprints (genome + score JSON). Returns overall score in [0,1] plus per-component breakdown (cosine over 9 numerics, categorical agreement over 4 enums, jaccard over agent_topology). Unblocks ADR-151 §3.2 Recommender, §3.3 Drift Detection, §3.5 Plugin Compat. Pure-TS, no `@metaharness/*` dep — preserves ADR-150's four architectural constraints.
4
- argument-hint: "(--a a.json --b b.json | --a-key X --b-key Y) [--per-dimension] [--alert-below 0.5] [--format json|table]"
5
- allowed-tools: Bash
6
- ---
7
-
8
- Surfaces the production similarity function from [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs) as a callable skill. Use when an agent needs to:
9
-
10
- - decide whether to fork an existing harness vs scaffold a new one
11
- - rank candidate templates against a target repo's genome
12
- - diff two harnesses produced by different teams to find duplicate work
13
- - generate the confidence number that ADR-151 §3.2's Recommender wraps
14
-
15
- ## Algorithm (from ADR-152 §Decision)
16
-
17
- ```
18
- overall = 0.60·cosine + 0.25·categorical + 0.15·jaccard
19
- ```
20
-
21
- - **cosine** — over a 9-dim numerical vector of normalized scorecard + genome dims
22
- - **categorical** — fraction of 4 enum fields that match (`repo_type`, `archetype`, `template`, `recommendedMode`)
23
- - **jaccard** — `|A ∩ B| / |A ∪ B|` over the `agent_topology[]` array
24
-
25
- The 3-component design is load-bearing: numerical cosine alone is too coarse (the iter-35 spike showed LEGAL vs DEVOPS at cosine=0.97 despite being unrelated verticals). Categorical + jaccard pull the composite to the correct ordering.
26
-
27
- ## Reference outputs (iter-35 spike fixtures)
28
-
29
- | Pair | overall | cosine | categorical | jaccard |
30
- |---|---:|---:|---:|---:|
31
- | `LEGAL` × `LEGAL` (self) | 1.0000 | 1.0000 | 1.0000 | 1.0000 |
32
- | `LEGAL` × `SUPPORT` | 0.8296 | 0.9987 | 0.7500 | 0.2857 |
33
- | `LEGAL` × `DEVOPS` | 0.5840 | 0.9734 | 0.0000 | 0.0000 |
34
-
35
- Both invariants from ADR-152 §"Smallest demonstrable spike" hold:
36
- 1. `similarity(X, X) === 1` exactly
37
- 2. `similarity(LEGAL, DEVOPS) < similarity(LEGAL, SUPPORT)` (vertical affinity)
38
-
39
- ## Architectural constraint inheritance (ADR-150)
40
-
41
- - **Removable** — pure-TS function, zero static `@metaharness/*` imports.
42
- - **Optional** — no new dep in `package.json`.
43
- - **Graceful** — malformed inputs emit `{ degraded: true, reason }` with exit code 2; never throws.
44
- - **CI-gate** — smoke step 17y locks the contract: module exports, spike fixtures reproduce, CLI dispatcher entry registered, MCP tool registered.
45
-
46
- ## Usage
47
-
48
- ```bash
49
- # File inputs
50
- npx ruflo metaharness similarity --a a.json --b b.json
51
-
52
- # Memory inputs (records persisted by oia-audit.mjs)
53
- npx ruflo metaharness similarity --a-key harness-X --b-key harness-Y
54
-
55
- # Per-dimension breakdown (used by ADR-151 §3.2 Recommender)
56
- npx ruflo metaharness similarity --a a.json --b b.json --per-dimension
57
-
58
- # Alert when too-dissimilar (used by ADR-151 §3.3 Drift Detection)
59
- npx ruflo metaharness similarity --a a.json --b b.json --alert-below 0.5
60
- ```
61
-
62
- ## Implementation
63
-
64
- Production module: [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs)
65
- CLI skill: [`scripts/similarity.mjs`](../../scripts/similarity.mjs)
66
- MCP tool: `mcp__plugin_ruflo-core_ruflo__metaharness_similarity` (registered in `v3/@claude-flow/cli/src/mcp-tools/metaharness-tools.ts`)
67
- Spike anchor: [`scripts/_spike-similarity.mjs`](../../scripts/_spike-similarity.mjs) (regression suite — invariants locked here)
1
+ ---
2
+ name: harness-similarity
3
+ description: ADR-152 — weighted similarity between two harness fingerprints (genome + score JSON). Returns overall score in [0,1] plus per-component breakdown (cosine over 9 numerics, categorical agreement over 4 enums, jaccard over agent_topology). Unblocks ADR-151 §3.2 Recommender, §3.3 Drift Detection, §3.5 Plugin Compat. Pure-TS, no `@metaharness/*` dep — preserves ADR-150's four architectural constraints.
4
+ argument-hint: "(--a a.json --b b.json | --a-key X --b-key Y) [--per-dimension] [--alert-below 0.5] [--format json|table]"
5
+ allowed-tools: Bash
6
+ ---
7
+
8
+ Surfaces the production similarity function from [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs) as a callable skill. Use when an agent needs to:
9
+
10
+ - decide whether to fork an existing harness vs scaffold a new one
11
+ - rank candidate templates against a target repo's genome
12
+ - diff two harnesses produced by different teams to find duplicate work
13
+ - generate the confidence number that ADR-151 §3.2's Recommender wraps
14
+
15
+ ## Algorithm (from ADR-152 §Decision)
16
+
17
+ ```
18
+ overall = 0.60·cosine + 0.25·categorical + 0.15·jaccard
19
+ ```
20
+
21
+ - **cosine** — over a 9-dim numerical vector of normalized scorecard + genome dims
22
+ - **categorical** — fraction of 4 enum fields that match (`repo_type`, `archetype`, `template`, `recommendedMode`)
23
+ - **jaccard** — `|A ∩ B| / |A ∪ B|` over the `agent_topology[]` array
24
+
25
+ The 3-component design is load-bearing: numerical cosine alone is too coarse (the iter-35 spike showed LEGAL vs DEVOPS at cosine=0.97 despite being unrelated verticals). Categorical + jaccard pull the composite to the correct ordering.
26
+
27
+ ## Reference outputs (iter-35 spike fixtures)
28
+
29
+ | Pair | overall | cosine | categorical | jaccard |
30
+ |---|---:|---:|---:|---:|
31
+ | `LEGAL` × `LEGAL` (self) | 1.0000 | 1.0000 | 1.0000 | 1.0000 |
32
+ | `LEGAL` × `SUPPORT` | 0.8296 | 0.9987 | 0.7500 | 0.2857 |
33
+ | `LEGAL` × `DEVOPS` | 0.5840 | 0.9734 | 0.0000 | 0.0000 |
34
+
35
+ Both invariants from ADR-152 §"Smallest demonstrable spike" hold:
36
+ 1. `similarity(X, X) === 1` exactly
37
+ 2. `similarity(LEGAL, DEVOPS) < similarity(LEGAL, SUPPORT)` (vertical affinity)
38
+
39
+ ## Architectural constraint inheritance (ADR-150)
40
+
41
+ - **Removable** — pure-TS function, zero static `@metaharness/*` imports.
42
+ - **Optional** — no new dep in `package.json`.
43
+ - **Graceful** — malformed inputs emit `{ degraded: true, reason }` with exit code 2; never throws.
44
+ - **CI-gate** — smoke step 17y locks the contract: module exports, spike fixtures reproduce, CLI dispatcher entry registered, MCP tool registered.
45
+
46
+ ## Usage
47
+
48
+ ```bash
49
+ # File inputs
50
+ npx ruflo metaharness similarity --a a.json --b b.json
51
+
52
+ # Memory inputs (records persisted by oia-audit.mjs)
53
+ npx ruflo metaharness similarity --a-key harness-X --b-key harness-Y
54
+
55
+ # Per-dimension breakdown (used by ADR-151 §3.2 Recommender)
56
+ npx ruflo metaharness similarity --a a.json --b b.json --per-dimension
57
+
58
+ # Alert when too-dissimilar (used by ADR-151 §3.3 Drift Detection)
59
+ npx ruflo metaharness similarity --a a.json --b b.json --alert-below 0.5
60
+ ```
61
+
62
+ ## Implementation
63
+
64
+ Production module: [`scripts/_similarity.mjs`](../../scripts/_similarity.mjs)
65
+ CLI skill: [`scripts/similarity.mjs`](../../scripts/similarity.mjs)
66
+ MCP tool: `mcp__plugin_ruflo-core_ruflo__metaharness_similarity` (registered in `v3/@claude-flow/cli/src/mcp-tools/metaharness-tools.ts`)
67
+ Spike anchor: [`scripts/_spike-similarity.mjs`](../../scripts/_spike-similarity.mjs) (regression suite — invariants locked here)