claude-flow 3.32.8 → 3.32.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/.proven-config-version +1 -0
- package/.claude/agents/MIGRATION_SUMMARY.md +221 -221
- package/.claude/agents/analysis/analyze-code-quality.md +57 -57
- package/.claude/agents/analysis/code-analyzer.md +188 -188
- package/.claude/agents/analysis/code-review/analyze-code-quality.md +57 -57
- package/.claude/agents/architecture/system-design/arch-system-design.md +35 -35
- package/.claude/agents/base-template-generator.md +41 -41
- package/.claude/agents/consensus/byzantine-coordinator.md +42 -42
- package/.claude/agents/consensus/crdt-synchronizer.md +976 -976
- package/.claude/agents/consensus/gossip-coordinator.md +42 -42
- package/.claude/agents/consensus/performance-benchmarker.md +830 -830
- package/.claude/agents/consensus/quorum-manager.md +802 -802
- package/.claude/agents/consensus/raft-manager.md +42 -42
- package/.claude/agents/consensus/security-manager.md +601 -601
- package/.claude/agents/core/coder.md +254 -254
- package/.claude/agents/core/planner.md +151 -151
- package/.claude/agents/core/researcher.md +173 -173
- package/.claude/agents/core/reviewer.md +308 -308
- package/.claude/agents/core/tester.md +299 -299
- package/.claude/agents/custom/test-long-runner.md +43 -43
- package/.claude/agents/data/ml/data-ml-model.md +75 -75
- package/.claude/agents/database-specialist.md +9 -9
- package/.claude/agents/development/backend/dev-backend-api.md +28 -28
- package/.claude/agents/development/dev-backend-api.md +177 -177
- package/.claude/agents/devops/ci-cd/ops-cicd-github.md +51 -51
- package/.claude/agents/documentation/api-docs/docs-api-openapi.md +62 -62
- package/.claude/agents/dual-mode/codex-coordinator.md +206 -206
- package/.claude/agents/dual-mode/codex-worker.md +190 -190
- package/.claude/agents/dual-mode/dual-orchestrator.md +253 -253
- package/.claude/agents/flow-nexus/app-store.md +87 -87
- package/.claude/agents/flow-nexus/authentication.md +68 -68
- package/.claude/agents/flow-nexus/challenges.md +80 -80
- package/.claude/agents/flow-nexus/neural-network.md +87 -87
- package/.claude/agents/flow-nexus/payments.md +82 -82
- package/.claude/agents/flow-nexus/sandbox.md +75 -75
- package/.claude/agents/flow-nexus/swarm.md +75 -75
- package/.claude/agents/flow-nexus/user-tools.md +95 -95
- package/.claude/agents/flow-nexus/workflow.md +83 -83
- package/.claude/agents/github/code-review-swarm.md +520 -520
- package/.claude/agents/github/github-modes.md +153 -153
- package/.claude/agents/github/issue-tracker.md +298 -298
- package/.claude/agents/github/multi-repo-swarm.md +524 -524
- package/.claude/agents/github/pr-manager.md +162 -162
- package/.claude/agents/github/project-board-sync.md +477 -477
- package/.claude/agents/github/release-manager.md +337 -337
- package/.claude/agents/github/release-swarm.md +550 -550
- package/.claude/agents/github/repo-architect.md +364 -364
- package/.claude/agents/github/swarm-issue.md +550 -550
- package/.claude/agents/github/swarm-pr.md +401 -401
- package/.claude/agents/github/sync-coordinator.md +424 -424
- package/.claude/agents/github/workflow-automation.md +604 -604
- package/.claude/agents/goal/agent.md +816 -816
- package/.claude/agents/goal/code-goal-planner.md +444 -444
- package/.claude/agents/goal/goal-planner.md +167 -167
- package/.claude/agents/hive-mind/collective-intelligence-coordinator.md +128 -128
- package/.claude/agents/hive-mind/queen-coordinator.md +201 -201
- package/.claude/agents/hive-mind/scout-explorer.md +240 -240
- package/.claude/agents/hive-mind/swarm-memory-manager.md +191 -191
- package/.claude/agents/hive-mind/worker-specialist.md +215 -215
- package/.claude/agents/neural/safla-neural.md +73 -73
- package/.claude/agents/optimization/benchmark-suite.md +662 -662
- package/.claude/agents/optimization/load-balancer.md +428 -428
- package/.claude/agents/optimization/performance-monitor.md +669 -669
- package/.claude/agents/optimization/resource-allocator.md +671 -671
- package/.claude/agents/optimization/topology-optimizer.md +805 -805
- package/.claude/agents/payments/agentic-payments.md +126 -126
- package/.claude/agents/project-coordinator.md +8 -8
- package/.claude/agents/python-specialist.md +9 -9
- package/.claude/agents/reasoning/agent.md +816 -816
- package/.claude/agents/reasoning/goal-planner.md +72 -72
- package/.claude/agents/security-auditor.md +9 -9
- package/.claude/agents/sona/sona-learning-optimizer.md +65 -65
- package/.claude/agents/sparc/architecture.md +452 -452
- package/.claude/agents/sparc/pseudocode.md +298 -298
- package/.claude/agents/sparc/refinement.md +503 -503
- package/.claude/agents/sparc/specification.md +257 -257
- package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +87 -87
- package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
- package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
- package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
- package/.claude/agents/sublinear/performance-optimizer.md +367 -367
- package/.claude/agents/sublinear/trading-predictor.md +245 -245
- package/.claude/agents/swarm/adaptive-coordinator.md +363 -363
- package/.claude/agents/swarm/hierarchical-coordinator.md +299 -299
- package/.claude/agents/swarm/mesh-coordinator.md +362 -362
- package/.claude/agents/templates/automation-smart-agent.md +184 -184
- package/.claude/agents/templates/coordinator-swarm-init.md +82 -82
- package/.claude/agents/templates/github-pr-manager.md +154 -154
- package/.claude/agents/templates/implementer-sparc-coder.md +242 -242
- package/.claude/agents/templates/memory-coordinator.md +162 -162
- package/.claude/agents/templates/migration-plan.md +723 -723
- package/.claude/agents/templates/orchestrator-task.md +119 -119
- package/.claude/agents/templates/performance-analyzer.md +178 -178
- package/.claude/agents/templates/sparc-coordinator.md +162 -162
- package/.claude/agents/testing/production-validator.md +372 -372
- package/.claude/agents/testing/tdd-london-swarm.md +221 -221
- package/.claude/agents/testing/unit/tdd-london-swarm.md +221 -221
- package/.claude/agents/testing/validation/production-validator.md +372 -372
- package/.claude/agents/typescript-specialist.md +9 -9
- package/.claude/agents/v3/database-specialist.md +9 -9
- package/.claude/agents/v3/project-coordinator.md +8 -8
- package/.claude/agents/v3/python-specialist.md +9 -9
- package/.claude/agents/v3/test-architect.md +9 -9
- package/.claude/agents/v3/typescript-specialist.md +9 -9
- package/.claude/agents/v3/v3-integration-architect.md +311 -311
- package/.claude/agents/v3/v3-memory-specialist.md +280 -280
- package/.claude/agents/v3/v3-performance-engineer.md +362 -362
- package/.claude/agents/v3/v3-queen-coordinator.md +62 -62
- package/.claude/agents/v3/v3-security-architect.md +139 -139
- package/.claude/checkpoints/1767754460.json +8 -8
- package/.claude/commands/agents/README.md +10 -10
- package/.claude/commands/agents/agent-capabilities.md +21 -21
- package/.claude/commands/agents/agent-coordination.md +28 -28
- package/.claude/commands/agents/agent-spawning.md +28 -28
- package/.claude/commands/agents/agent-types.md +26 -26
- package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
- package/.claude/commands/analysis/README.md +9 -9
- package/.claude/commands/analysis/bottleneck-detect.md +162 -162
- package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
- package/.claude/commands/analysis/performance-report.md +25 -25
- package/.claude/commands/analysis/token-efficiency.md +44 -44
- package/.claude/commands/analysis/token-usage.md +25 -25
- package/.claude/commands/automation/README.md +9 -9
- package/.claude/commands/automation/auto-agent.md +122 -122
- package/.claude/commands/automation/self-healing.md +105 -105
- package/.claude/commands/automation/session-memory.md +89 -89
- package/.claude/commands/automation/smart-agents.md +72 -72
- package/.claude/commands/automation/smart-spawn.md +25 -25
- package/.claude/commands/automation/workflow-select.md +25 -25
- package/.claude/commands/claude-flow-help.md +103 -103
- package/.claude/commands/claude-flow-memory.md +107 -107
- package/.claude/commands/claude-flow-swarm.md +205 -205
- package/.claude/commands/coordination/README.md +9 -9
- package/.claude/commands/coordination/agent-spawn.md +25 -25
- package/.claude/commands/coordination/init.md +44 -44
- package/.claude/commands/coordination/orchestrate.md +43 -43
- package/.claude/commands/coordination/spawn.md +45 -45
- package/.claude/commands/coordination/swarm-init.md +85 -85
- package/.claude/commands/coordination/task-orchestrate.md +25 -25
- package/.claude/commands/flow-nexus/app-store.md +123 -123
- package/.claude/commands/flow-nexus/challenges.md +119 -119
- package/.claude/commands/flow-nexus/login-registration.md +64 -64
- package/.claude/commands/flow-nexus/neural-network.md +133 -133
- package/.claude/commands/flow-nexus/payments.md +115 -115
- package/.claude/commands/flow-nexus/sandbox.md +82 -82
- package/.claude/commands/flow-nexus/swarm.md +86 -86
- package/.claude/commands/flow-nexus/user-tools.md +151 -151
- package/.claude/commands/flow-nexus/workflow.md +114 -114
- package/.claude/commands/github/README.md +11 -11
- package/.claude/commands/github/code-review-swarm.md +513 -513
- package/.claude/commands/github/code-review.md +25 -25
- package/.claude/commands/github/github-modes.md +146 -146
- package/.claude/commands/github/github-swarm.md +121 -121
- package/.claude/commands/github/issue-tracker.md +291 -291
- package/.claude/commands/github/issue-triage.md +25 -25
- package/.claude/commands/github/multi-repo-swarm.md +518 -518
- package/.claude/commands/github/pr-enhance.md +26 -26
- package/.claude/commands/github/pr-manager.md +169 -169
- package/.claude/commands/github/project-board-sync.md +470 -470
- package/.claude/commands/github/release-manager.md +337 -337
- package/.claude/commands/github/release-swarm.md +543 -543
- package/.claude/commands/github/repo-analyze.md +25 -25
- package/.claude/commands/github/repo-architect.md +366 -366
- package/.claude/commands/github/swarm-issue.md +481 -481
- package/.claude/commands/github/swarm-pr.md +284 -284
- package/.claude/commands/github/sync-coordinator.md +300 -300
- package/.claude/commands/github/workflow-automation.md +441 -441
- package/.claude/commands/hive-mind/README.md +17 -17
- package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
- package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
- package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
- package/.claude/commands/hive-mind/hive-mind.md +27 -27
- package/.claude/commands/hooks/README.md +11 -11
- package/.claude/commands/hooks/overview.md +57 -57
- package/.claude/commands/hooks/post-edit.md +117 -117
- package/.claude/commands/hooks/post-task.md +112 -112
- package/.claude/commands/hooks/pre-edit.md +113 -113
- package/.claude/commands/hooks/pre-task.md +111 -111
- package/.claude/commands/hooks/session-end.md +118 -118
- package/.claude/commands/hooks/setup.md +102 -102
- package/.claude/commands/memory/README.md +9 -9
- package/.claude/commands/memory/memory-persist.md +25 -25
- package/.claude/commands/memory/memory-search.md +25 -25
- package/.claude/commands/memory/memory-usage.md +25 -25
- package/.claude/commands/memory/neural.md +47 -47
- package/.claude/commands/monitoring/README.md +9 -9
- package/.claude/commands/monitoring/agent-metrics.md +25 -25
- package/.claude/commands/monitoring/agents.md +44 -44
- package/.claude/commands/monitoring/real-time-view.md +25 -25
- package/.claude/commands/monitoring/status.md +46 -46
- package/.claude/commands/monitoring/swarm-monitor.md +25 -25
- package/.claude/commands/optimization/README.md +9 -9
- package/.claude/commands/optimization/auto-topology.md +61 -61
- package/.claude/commands/optimization/cache-manage.md +25 -25
- package/.claude/commands/optimization/parallel-execute.md +25 -25
- package/.claude/commands/optimization/parallel-execution.md +49 -49
- package/.claude/commands/optimization/topology-optimize.md +25 -25
- package/.claude/commands/pair/README.md +260 -260
- package/.claude/commands/pair/commands.md +545 -545
- package/.claude/commands/pair/config.md +509 -509
- package/.claude/commands/pair/examples.md +511 -511
- package/.claude/commands/pair/modes.md +347 -347
- package/.claude/commands/pair/session.md +406 -406
- package/.claude/commands/pair/start.md +208 -208
- package/.claude/commands/sparc/analyzer.md +51 -51
- package/.claude/commands/sparc/architect.md +53 -53
- package/.claude/commands/sparc/ask.md +97 -97
- package/.claude/commands/sparc/batch-executor.md +54 -54
- package/.claude/commands/sparc/code.md +89 -89
- package/.claude/commands/sparc/coder.md +54 -54
- package/.claude/commands/sparc/debug.md +83 -83
- package/.claude/commands/sparc/debugger.md +54 -54
- package/.claude/commands/sparc/designer.md +53 -53
- package/.claude/commands/sparc/devops.md +109 -109
- package/.claude/commands/sparc/docs-writer.md +80 -80
- package/.claude/commands/sparc/documenter.md +54 -54
- package/.claude/commands/sparc/innovator.md +54 -54
- package/.claude/commands/sparc/integration.md +83 -83
- package/.claude/commands/sparc/mcp.md +117 -117
- package/.claude/commands/sparc/memory-manager.md +54 -54
- package/.claude/commands/sparc/optimizer.md +54 -54
- package/.claude/commands/sparc/orchestrator.md +131 -131
- package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
- package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
- package/.claude/commands/sparc/researcher.md +54 -54
- package/.claude/commands/sparc/reviewer.md +54 -54
- package/.claude/commands/sparc/security-review.md +80 -80
- package/.claude/commands/sparc/sparc-modes.md +174 -174
- package/.claude/commands/sparc/sparc.md +111 -111
- package/.claude/commands/sparc/spec-pseudocode.md +80 -80
- package/.claude/commands/sparc/supabase-admin.md +348 -348
- package/.claude/commands/sparc/swarm-coordinator.md +54 -54
- package/.claude/commands/sparc/tdd.md +54 -54
- package/.claude/commands/sparc/tester.md +54 -54
- package/.claude/commands/sparc/tutorial.md +79 -79
- package/.claude/commands/sparc/workflow-manager.md +54 -54
- package/.claude/commands/sparc.md +166 -166
- package/.claude/commands/stream-chain/pipeline.md +120 -120
- package/.claude/commands/stream-chain/run.md +69 -69
- package/.claude/commands/swarm/README.md +15 -15
- package/.claude/commands/swarm/analysis.md +95 -95
- package/.claude/commands/swarm/development.md +96 -96
- package/.claude/commands/swarm/examples.md +168 -168
- package/.claude/commands/swarm/maintenance.md +102 -102
- package/.claude/commands/swarm/optimization.md +117 -117
- package/.claude/commands/swarm/research.md +136 -136
- package/.claude/commands/swarm/swarm-analysis.md +8 -8
- package/.claude/commands/swarm/swarm-background.md +8 -8
- package/.claude/commands/swarm/swarm-init.md +19 -19
- package/.claude/commands/swarm/swarm-modes.md +8 -8
- package/.claude/commands/swarm/swarm-monitor.md +8 -8
- package/.claude/commands/swarm/swarm-spawn.md +19 -19
- package/.claude/commands/swarm/swarm-status.md +8 -8
- package/.claude/commands/swarm/swarm-strategies.md +8 -8
- package/.claude/commands/swarm/swarm.md +27 -27
- package/.claude/commands/swarm/testing.md +131 -131
- package/.claude/commands/training/README.md +9 -9
- package/.claude/commands/training/model-update.md +25 -25
- package/.claude/commands/training/neural-patterns.md +73 -73
- package/.claude/commands/training/neural-train.md +25 -25
- package/.claude/commands/training/pattern-learn.md +25 -25
- package/.claude/commands/training/specialization.md +62 -62
- package/.claude/commands/truth/start.md +142 -142
- package/.claude/commands/verify/check.md +49 -49
- package/.claude/commands/verify/start.md +127 -127
- package/.claude/commands/workflows/README.md +9 -9
- package/.claude/commands/workflows/development.md +77 -77
- package/.claude/commands/workflows/research.md +62 -62
- package/.claude/commands/workflows/workflow-create.md +25 -25
- package/.claude/commands/workflows/workflow-execute.md +25 -25
- package/.claude/commands/workflows/workflow-export.md +25 -25
- package/.claude/config/v3-dependency-optimization.json +265 -265
- package/.claude/config/v3-performance-targets.json +250 -250
- package/.claude/helpers/.LOCKED +2 -2
- package/.claude/helpers/.helpers-version +1 -0
- package/.claude/helpers/README.md +96 -96
- package/.claude/helpers/adr-compliance.sh +186 -186
- package/.claude/helpers/aggressive-microcompact.mjs +36 -36
- package/.claude/helpers/auto-commit.sh +178 -178
- package/.claude/helpers/auto-memory-hook.mjs +430 -430
- package/.claude/helpers/checkpoint-manager.sh +251 -251
- package/.claude/helpers/context-persistence-hook.mjs +2001 -2001
- package/.claude/helpers/daemon-manager.sh +252 -252
- package/.claude/helpers/ddd-tracker.sh +144 -144
- package/.claude/helpers/github-safe.js +156 -156
- package/.claude/helpers/github-setup.sh +45 -45
- package/.claude/helpers/guidance-hook.sh +13 -13
- package/.claude/helpers/guidance-hooks.sh +102 -102
- package/.claude/helpers/health-monitor.sh +108 -108
- package/.claude/helpers/helpers.manifest.json +13 -0
- package/.claude/helpers/hook-handler.cjs +464 -464
- package/.claude/helpers/intelligence.cjs +1058 -1058
- package/.claude/helpers/learning-hooks.sh +329 -329
- package/.claude/helpers/learning-optimizer.sh +127 -127
- package/.claude/helpers/learning-service.mjs +1144 -1144
- package/.claude/helpers/memory.cjs +84 -84
- package/.claude/helpers/metrics-db.mjs +503 -503
- package/.claude/helpers/patch-aggressive-prune.mjs +184 -184
- package/.claude/helpers/pattern-consolidator.sh +86 -86
- package/.claude/helpers/perf-worker.sh +160 -160
- package/.claude/helpers/quick-start.sh +19 -19
- package/.claude/helpers/router.cjs +62 -62
- package/.claude/helpers/security-scanner.sh +127 -127
- package/.claude/helpers/session.cjs +125 -125
- package/.claude/helpers/setup-mcp.sh +18 -18
- package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
- package/.claude/helpers/statusline.cjs +211 -2
- package/.claude/helpers/swarm-comms.sh +353 -353
- package/.claude/helpers/swarm-hooks.sh +761 -761
- package/.claude/helpers/swarm-monitor.sh +210 -210
- package/.claude/helpers/sync-v3-metrics.sh +245 -245
- package/.claude/helpers/update-v3-progress.sh +165 -165
- package/.claude/helpers/v3-quick-status.sh +57 -57
- package/.claude/helpers/v3.sh +110 -110
- package/.claude/helpers/validate-v3-config.sh +215 -215
- package/.claude/helpers/worker-manager.sh +170 -170
- package/.claude/mcp.json +12 -12
- package/.claude/proven-config.json +42 -0
- package/.claude/settings.json +284 -284
- package/.claude/settings.json.bak +526 -526
- package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
- package/.claude/skills/agentdb-learning/SKILL.md +545 -545
- package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
- package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
- package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
- package/.claude/skills/agentic-jujutsu/SKILL.md +645 -645
- package/.claude/skills/browser/SKILL.md +204 -204
- package/.claude/skills/dual-mode/README.md +71 -71
- package/.claude/skills/dual-mode/dual-collect.md +103 -103
- package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
- package/.claude/skills/dual-mode/dual-spawn.md +81 -81
- package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
- package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
- package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
- package/.claude/skills/github-code-review/SKILL.md +1125 -1125
- package/.claude/skills/github-multi-repo/SKILL.md +862 -862
- package/.claude/skills/github-project-management/SKILL.md +1262 -1262
- package/.claude/skills/github-release-management/SKILL.md +1064 -1064
- package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
- package/.claude/skills/hive-mind-advanced/SKILL.md +709 -709
- package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
- package/.claude/skills/pair-programming/SKILL.md +1202 -1202
- package/.claude/skills/performance-analysis/SKILL.md +560 -560
- package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
- package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
- package/.claude/skills/skill-builder/SKILL.md +910 -910
- package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
- package/.claude/skills/stream-chain/SKILL.md +560 -560
- package/.claude/skills/swarm-advanced/SKILL.md +970 -970
- package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
- package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
- package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
- package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
- package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
- package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
- package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
- package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
- package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
- package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
- package/.claude/skills/verification-quality/SKILL.md +691 -691
- package/.claude/skills/worker-benchmarks/SKILL.md +129 -129
- package/.claude/skills/worker-integration/SKILL.md +147 -147
- package/.claude/statusline-command.sh +176 -176
- package/.claude/statusline.mjs +109 -109
- package/.claude/statusline.sh +431 -431
- package/.claude/workflows/full-system-test.js +65 -65
- package/.claude/workflows/intelligence-system-hardening.js +120 -120
- package/.claude/workflows/plugin-contract-audit.js +91 -91
- package/.claude-plugin/README.md +720 -720
- package/.claude-plugin/docs/INSTALLATION.md +261 -261
- package/.claude-plugin/docs/PLUGIN_SUMMARY.md +361 -361
- package/.claude-plugin/docs/QUICKSTART.md +361 -361
- package/.claude-plugin/docs/STRUCTURE.md +128 -128
- package/.claude-plugin/hooks/hooks.json +79 -77
- package/.claude-plugin/marketplace.json +185 -185
- package/.claude-plugin/plugin.json +71 -71
- package/.claude-plugin/scripts/install.sh +234 -234
- package/.claude-plugin/scripts/ruflo-hook.cjs +166 -166
- package/.claude-plugin/scripts/ruflo-hook.sh +52 -52
- package/.claude-plugin/scripts/uninstall.sh +36 -36
- package/.claude-plugin/scripts/verify.sh +108 -108
- package/LICENSE +21 -21
- package/README.md +419 -419
- package/bin/cli.js +11 -11
- package/bin/npx-repair.js +7 -7
- package/bin/npx-safe-launch.js +9 -9
- package/package.json +192 -186
- package/v3/@claude-flow/cli/README.md +419 -419
- package/v3/@claude-flow/cli/bin/cli.js +314 -314
- package/v3/@claude-flow/cli/bin/mcp-server.js +224 -224
- package/v3/@claude-flow/cli/bin/preinstall.cjs +2 -2
- package/v3/@claude-flow/cli/catalog-manifest.json +2 -2
- package/v3/@claude-flow/cli/dist/src/autopilot-state.js +24 -7
- package/v3/@claude-flow/cli/dist/src/benchmarks/gaia-critic.js +24 -24
- package/v3/@claude-flow/cli/dist/src/business-pods/bbs-budget-tracker.js +53 -53
- package/v3/@claude-flow/cli/dist/src/commands/completions.js +409 -409
- package/v3/@claude-flow/cli/dist/src/commands/daemon.js +44 -44
- package/v3/@claude-flow/cli/dist/src/commands/doctor.js +267 -20
- package/v3/@claude-flow/cli/dist/src/commands/embeddings.js +26 -26
- package/v3/@claude-flow/cli/dist/src/commands/hive-mind.js +97 -97
- package/v3/@claude-flow/cli/dist/src/commands/hooks.js +74 -11
- package/v3/@claude-flow/cli/dist/src/commands/init.js +202 -34
- package/v3/@claude-flow/cli/dist/src/commands/memory.js +12 -1
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/backup.js +23 -23
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/benchmark.js +31 -31
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/import.js +14 -14
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/init.js +115 -115
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/migrate.js +99 -99
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/optimize.js +51 -51
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/setup.js +624 -624
- package/v3/@claude-flow/cli/dist/src/commands/ruvector/status.js +38 -38
- package/v3/@claude-flow/cli/dist/src/config/proven-config.js +2 -2
- package/v3/@claude-flow/cli/dist/src/funnel/disclosure.js +13 -2
- package/v3/@claude-flow/cli/dist/src/funnel/message-transport.d.ts +11 -4
- package/v3/@claude-flow/cli/dist/src/funnel/message-transport.js +11 -4
- package/v3/@claude-flow/cli/dist/src/funnel/messages.d.ts +12 -10
- package/v3/@claude-flow/cli/dist/src/funnel/messages.js +83 -11
- package/v3/@claude-flow/cli/dist/src/init/claudemd-generator.js +231 -231
- package/v3/@claude-flow/cli/dist/src/init/executor.js +453 -453
- package/v3/@claude-flow/cli/dist/src/init/helper-signing.js +2 -2
- package/v3/@claude-flow/cli/dist/src/init/helpers-generator.js +751 -751
- package/v3/@claude-flow/cli/dist/src/init/statusline-generator.js +24 -24
- package/v3/@claude-flow/cli/dist/src/mcp-tools/agentdb-tools.js +15 -15
- package/v3/@claude-flow/cli/dist/src/mcp-tools/browser-intent-tools.js +19 -19
- package/v3/@claude-flow/cli/dist/src/mcp-tools/browser-tools.js +8 -0
- package/v3/@claude-flow/cli/dist/src/mcp-tools/hooks-tools.js +21 -0
- package/v3/@claude-flow/cli/dist/src/mcp-tools/memory-tools.js +4 -3
- package/v3/@claude-flow/cli/dist/src/memory/graph-edge-writer.d.ts +9 -0
- package/v3/@claude-flow/cli/dist/src/memory/graph-edge-writer.js +35 -22
- package/v3/@claude-flow/cli/dist/src/memory/memory-bridge.js +189 -93
- package/v3/@claude-flow/cli/dist/src/memory/memory-initializer.js +479 -407
- package/v3/@claude-flow/cli/dist/src/memory/rabitq-index.js +5 -5
- package/v3/@claude-flow/cli/dist/src/parser.js +25 -9
- package/v3/@claude-flow/cli/dist/src/proxy/verify.js +2 -2
- package/v3/@claude-flow/cli/dist/src/runtime/headless.js +28 -28
- package/v3/@claude-flow/cli/dist/src/services/distill-tuning.js +7 -7
- package/v3/@claude-flow/cli/dist/src/services/headless-worker-executor.js +84 -84
- package/v3/@claude-flow/cli/dist/src/services/memory-distillation.js +4 -4
- package/v3/@claude-flow/cli/dist/src/services/worker-daemon.js +7 -4
- package/v3/@claude-flow/cli/dist/src/transfer/deploy-seraphine.js +23 -23
- package/v3/@claude-flow/cli/package.json +137 -135
- package/v3/@claude-flow/guidance/README.md +1195 -1195
- package/v3/@claude-flow/guidance/package.json +198 -198
- package/v3/@claude-flow/shared/README.md +323 -323
- package/v3/@claude-flow/shared/dist/events/event-store.js +31 -31
- package/v3/@claude-flow/shared/dist/hooks/safety/git-commit.js +3 -3
- package/v3/@claude-flow/shared/package.json +43 -43
- package/v3/README.md +493 -493
|
@@ -1,65 +1,65 @@
|
|
|
1
|
-
export const meta = {
|
|
2
|
-
name: 'full-system-test',
|
|
3
|
-
description: 'Full system test — CLI build + test suite + runtime smoke + all plugin smoke contracts, run in parallel, with a synthesized pass/fail report',
|
|
4
|
-
phases: [
|
|
5
|
-
{ title: 'Test', detail: 'parallel: build, unit tests, CLI runtime smoke, plugin contracts' },
|
|
6
|
-
{ title: 'Report', detail: 'synthesize a single green/red verdict' },
|
|
7
|
-
],
|
|
8
|
-
}
|
|
9
|
-
|
|
10
|
-
// args (optional): { skipTests?: boolean } — skip the (slow) vitest suite, run the rest
|
|
11
|
-
const skipTests = !!(args && args.skipTests)
|
|
12
|
-
|
|
13
|
-
const CLI = 'v3/@claude-flow/cli'
|
|
14
|
-
|
|
15
|
-
const DIM_SCHEMA = {
|
|
16
|
-
type: 'object', additionalProperties: false,
|
|
17
|
-
required: ['dimension', 'ok', 'summary'],
|
|
18
|
-
properties: {
|
|
19
|
-
dimension: { type: 'string' },
|
|
20
|
-
ok: { type: 'boolean' },
|
|
21
|
-
summary: { type: 'string' },
|
|
22
|
-
metrics: { type: 'object', additionalProperties: true },
|
|
23
|
-
failures: { type: 'array', items: { type: 'string' } },
|
|
24
|
-
},
|
|
25
|
-
}
|
|
26
|
-
|
|
27
|
-
const DIMS = [
|
|
28
|
-
{
|
|
29
|
-
key: 'build', agentType: 'coder',
|
|
30
|
-
prompt: `From the repo root, verify the CLI builds cleanly. Run: \`cd ${CLI} && npm run build\` (this runs tsc). Set ok=true ONLY if the build exits 0 with zero type errors. Put the type-error count in metrics.errors and the first few error lines in failures[]. dimension="build". Do NOT modify any files — read/run only.`,
|
|
31
|
-
},
|
|
32
|
-
{
|
|
33
|
-
key: 'unit-tests', agentType: 'tester',
|
|
34
|
-
prompt: `From the repo root, run the CLI automated test suite. In ${CLI}, read package.json "scripts" to find the test command (likely "vitest run" / "npm test"). Run the FULL suite non-interactively (e.g. \`cd ${CLI} && npx vitest run --reporter=dot\` or the package's test script). Put total/passed/failed/skipped in metrics and list notable failing test files in failures[]. ok=true ONLY if failed=0. If the suite is too large to finish in a reasonable time, run as much as you can, set metrics.truncated=true, and report the counts you got. dimension="unit-tests". Do NOT modify tests or source to make anything pass.`,
|
|
35
|
-
},
|
|
36
|
-
{
|
|
37
|
-
key: 'cli-smoke', agentType: 'tester',
|
|
38
|
-
prompt: `From the repo root, smoke-test the built CLI runtime. Ensure ${CLI} is built (if dist/ is missing, run \`npm run build\` there first). Find the entry from ${CLI}/package.json "bin", then run three commands via node and confirm each exits cleanly with sane output: (1) the version flag, (2) --help, (3) \`doctor\`. Record per-command ok in metrics (e.g. metrics.version, metrics.help, metrics.doctor) and put any crash/stack output in failures[]. ok=true if all three run without crashing. dimension="cli-smoke". Do NOT modify files.`,
|
|
39
|
-
},
|
|
40
|
-
{
|
|
41
|
-
key: 'plugin-contracts', agentType: 'tester',
|
|
42
|
-
prompt: `From the repo root, run EVERY plugin smoke contract: for each script matching the glob plugins/*/scripts/smoke.sh, run it with bash and read its trailing "N passed, M failed" line and exit code. Put metrics.totalPlugins, metrics.passing, metrics.failing. List each failing plugin as "<plugin>: M failed" in failures[]. ok=true ONLY if every plugin smoke exits 0. dimension="plugin-contracts". Do NOT modify files — read/run only.`,
|
|
43
|
-
},
|
|
44
|
-
]
|
|
45
|
-
|
|
46
|
-
phase('Test')
|
|
47
|
-
const active = DIMS.filter((d) => !(skipTests && d.key === 'unit-tests'))
|
|
48
|
-
const results = (await parallel(
|
|
49
|
-
active.map((d) => () =>
|
|
50
|
-
agent(d.prompt, { label: `test:${d.key}`, phase: 'Test', schema: DIM_SCHEMA, agentType: d.agentType })
|
|
51
|
-
)
|
|
52
|
-
)).filter(Boolean)
|
|
53
|
-
|
|
54
|
-
phase('Report')
|
|
55
|
-
const failed = results.filter((r) => !r.ok)
|
|
56
|
-
const summary = {
|
|
57
|
-
green: failed.length === 0,
|
|
58
|
-
passed: results.length - failed.length,
|
|
59
|
-
total: results.length,
|
|
60
|
-
skippedTests: skipTests,
|
|
61
|
-
dimensions: results.map((r) => ({ dimension: r.dimension, ok: r.ok, summary: r.summary, metrics: r.metrics || {} })),
|
|
62
|
-
failures: failed.flatMap((r) => (r.failures || []).map((f) => `[${r.dimension}] ${f}`)),
|
|
63
|
-
}
|
|
64
|
-
log(`Full system test: ${summary.passed}/${summary.total} dimensions green${summary.green ? ' — ALL PASS' : ''}`)
|
|
65
|
-
return summary
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'full-system-test',
|
|
3
|
+
description: 'Full system test — CLI build + test suite + runtime smoke + all plugin smoke contracts, run in parallel, with a synthesized pass/fail report',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Test', detail: 'parallel: build, unit tests, CLI runtime smoke, plugin contracts' },
|
|
6
|
+
{ title: 'Report', detail: 'synthesize a single green/red verdict' },
|
|
7
|
+
],
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
// args (optional): { skipTests?: boolean } — skip the (slow) vitest suite, run the rest
|
|
11
|
+
const skipTests = !!(args && args.skipTests)
|
|
12
|
+
|
|
13
|
+
const CLI = 'v3/@claude-flow/cli'
|
|
14
|
+
|
|
15
|
+
const DIM_SCHEMA = {
|
|
16
|
+
type: 'object', additionalProperties: false,
|
|
17
|
+
required: ['dimension', 'ok', 'summary'],
|
|
18
|
+
properties: {
|
|
19
|
+
dimension: { type: 'string' },
|
|
20
|
+
ok: { type: 'boolean' },
|
|
21
|
+
summary: { type: 'string' },
|
|
22
|
+
metrics: { type: 'object', additionalProperties: true },
|
|
23
|
+
failures: { type: 'array', items: { type: 'string' } },
|
|
24
|
+
},
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
const DIMS = [
|
|
28
|
+
{
|
|
29
|
+
key: 'build', agentType: 'coder',
|
|
30
|
+
prompt: `From the repo root, verify the CLI builds cleanly. Run: \`cd ${CLI} && npm run build\` (this runs tsc). Set ok=true ONLY if the build exits 0 with zero type errors. Put the type-error count in metrics.errors and the first few error lines in failures[]. dimension="build". Do NOT modify any files — read/run only.`,
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
key: 'unit-tests', agentType: 'tester',
|
|
34
|
+
prompt: `From the repo root, run the CLI automated test suite. In ${CLI}, read package.json "scripts" to find the test command (likely "vitest run" / "npm test"). Run the FULL suite non-interactively (e.g. \`cd ${CLI} && npx vitest run --reporter=dot\` or the package's test script). Put total/passed/failed/skipped in metrics and list notable failing test files in failures[]. ok=true ONLY if failed=0. If the suite is too large to finish in a reasonable time, run as much as you can, set metrics.truncated=true, and report the counts you got. dimension="unit-tests". Do NOT modify tests or source to make anything pass.`,
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
key: 'cli-smoke', agentType: 'tester',
|
|
38
|
+
prompt: `From the repo root, smoke-test the built CLI runtime. Ensure ${CLI} is built (if dist/ is missing, run \`npm run build\` there first). Find the entry from ${CLI}/package.json "bin", then run three commands via node and confirm each exits cleanly with sane output: (1) the version flag, (2) --help, (3) \`doctor\`. Record per-command ok in metrics (e.g. metrics.version, metrics.help, metrics.doctor) and put any crash/stack output in failures[]. ok=true if all three run without crashing. dimension="cli-smoke". Do NOT modify files.`,
|
|
39
|
+
},
|
|
40
|
+
{
|
|
41
|
+
key: 'plugin-contracts', agentType: 'tester',
|
|
42
|
+
prompt: `From the repo root, run EVERY plugin smoke contract: for each script matching the glob plugins/*/scripts/smoke.sh, run it with bash and read its trailing "N passed, M failed" line and exit code. Put metrics.totalPlugins, metrics.passing, metrics.failing. List each failing plugin as "<plugin>: M failed" in failures[]. ok=true ONLY if every plugin smoke exits 0. dimension="plugin-contracts". Do NOT modify files — read/run only.`,
|
|
43
|
+
},
|
|
44
|
+
]
|
|
45
|
+
|
|
46
|
+
phase('Test')
|
|
47
|
+
const active = DIMS.filter((d) => !(skipTests && d.key === 'unit-tests'))
|
|
48
|
+
const results = (await parallel(
|
|
49
|
+
active.map((d) => () =>
|
|
50
|
+
agent(d.prompt, { label: `test:${d.key}`, phase: 'Test', schema: DIM_SCHEMA, agentType: d.agentType })
|
|
51
|
+
)
|
|
52
|
+
)).filter(Boolean)
|
|
53
|
+
|
|
54
|
+
phase('Report')
|
|
55
|
+
const failed = results.filter((r) => !r.ok)
|
|
56
|
+
const summary = {
|
|
57
|
+
green: failed.length === 0,
|
|
58
|
+
passed: results.length - failed.length,
|
|
59
|
+
total: results.length,
|
|
60
|
+
skippedTests: skipTests,
|
|
61
|
+
dimensions: results.map((r) => ({ dimension: r.dimension, ok: r.ok, summary: r.summary, metrics: r.metrics || {} })),
|
|
62
|
+
failures: failed.flatMap((r) => (r.failures || []).map((f) => `[${r.dimension}] ${f}`)),
|
|
63
|
+
}
|
|
64
|
+
log(`Full system test: ${summary.passed}/${summary.total} dimensions green${summary.green ? ' — ALL PASS' : ''}`)
|
|
65
|
+
return summary
|
|
@@ -1,121 +1,121 @@
|
|
|
1
|
-
export const meta = {
|
|
2
|
-
name: 'intelligence-system-hardening',
|
|
3
|
-
description: 'Implement audit fixes, build a real benchmark harness, optimize, validate, and rewrite perf docs with measured numbers',
|
|
4
|
-
phases: [
|
|
5
|
-
{ title: 'Implement', detail: 'parallel fixes — distinct files, no conflicts' },
|
|
6
|
-
{ title: 'Validate', detail: 'build + tests; repair if broken' },
|
|
7
|
-
{ title: 'Benchmark', detail: 'real measurement harness -> JSON numbers' },
|
|
8
|
-
{ title: 'Optimize', detail: 'tune HNSW params, re-measure before/after' },
|
|
9
|
-
{ title: 'Docs', detail: 'rewrite README/CLAUDE.md perf claims with measured values' },
|
|
10
|
-
],
|
|
11
|
-
}
|
|
12
|
-
|
|
13
|
-
const REPO = '/Users/cohen/Projects/ruflo'
|
|
14
|
-
const CLI = `${REPO}/v3/@claude-flow/cli`
|
|
15
|
-
const RNG = 'an ' + 'RNG' + ' call (pseudo-random fabrication)'
|
|
16
|
-
|
|
17
|
-
const FIX_SCHEMA = {
|
|
18
|
-
type: 'object', additionalProperties: false,
|
|
19
|
-
required: ['issue', 'applied', 'summary', 'files'],
|
|
20
|
-
properties: {
|
|
21
|
-
issue: { type: 'string' }, applied: { type: 'boolean' },
|
|
22
|
-
summary: { type: 'string' }, files: { type: 'array', items: { type: 'string' } },
|
|
23
|
-
risk: { type: 'string' },
|
|
24
|
-
},
|
|
25
|
-
}
|
|
26
|
-
const BENCH_SCHEMA = {
|
|
27
|
-
type: 'object', additionalProperties: true,
|
|
28
|
-
required: ['ran', 'results', 'harnessPath', 'notes'],
|
|
29
|
-
properties: {
|
|
30
|
-
ran: { type: 'boolean' }, harnessPath: { type: 'string' },
|
|
31
|
-
results: { type: 'object', additionalProperties: true },
|
|
32
|
-
notes: { type: 'string' },
|
|
33
|
-
},
|
|
34
|
-
}
|
|
35
|
-
const VALIDATE_SCHEMA = {
|
|
36
|
-
type: 'object', additionalProperties: false,
|
|
37
|
-
required: ['buildOk', 'testsOk', 'summary'],
|
|
38
|
-
properties: {
|
|
39
|
-
buildOk: { type: 'boolean' }, testsOk: { type: 'boolean' },
|
|
40
|
-
summary: { type: 'string' }, failures: { type: 'array', items: { type: 'string' } },
|
|
41
|
-
},
|
|
42
|
-
}
|
|
43
|
-
|
|
44
|
-
phase('Implement')
|
|
45
|
-
const FIXES = [
|
|
46
|
-
{
|
|
47
|
-
key: 'reward-inversion', label: 'fix:reward-inversion',
|
|
48
|
-
prompt: `Repo ${REPO}. CRITICAL BUG (audit finding #1, follow-up to #2222). In ${CLI}/src/commands/route.ts the \`route feedback\` command: a negative reward passed the documented way (\`-r -1.0\` or \`--reward -1.0\`) is parsed as +1.00 because the CLI flag parser strips the leading '-' from negative numeric values. Only \`--reward=-1.0\` (equals form) preserves the sign. So a user giving NEGATIVE feedback actively REINFORCES the bad agent.
|
|
49
|
-
Investigate the flag-parsing path (route.ts reward flag def ~line 399, value read ~line 419; and the shared CLI arg parser route uses). Fix so \`-r -1.0\`, \`--reward -1.0\`, and \`--reward=-1.0\` ALL yield reward = -1.0. Prefer the most localized correct fix; if the bug is in the shared parser, fix it there but verify other negative-number flags still work. Add a regression test (extend ${CLI}/__tests__/bug-cluster-2219-2226.test.ts or new) asserting parsed reward sign for all three syntaxes. Build (cd ${CLI} && npm run build) must stay clean. Report via schema.`,
|
|
50
|
-
},
|
|
51
|
-
{
|
|
52
|
-
key: 'flash-fabrication', label: 'fix:flash-fabrication',
|
|
53
|
-
prompt: `Repo ${REPO}. AUDIT FINDING #2: ${REPO}/v3/@claude-flow/swarm/src/attention-coordinator.ts line 972 fabricates a fake metric — it sets performanceStats.flashSpeedup to a value computed from ${RNG}: roughly "2.49 plus rng times 4.98", and line 973 hardcodes memoryReduction = 0.75. Reporting a made-up number as real telemetry is a credibility liability. The SAME pattern exists in ${REPO}/v3/@claude-flow/integration/src/attention-coordinator.ts — fix BOTH copies.
|
|
54
|
-
Replace the pseudo-random fabrication with an honest value: either (a) actually invoke the FlashAttention kernel's own benchmark()/measured path to get a real speedup if cheaply available, or (b) if no measurement is wired, set flashSpeedup to a sentinel meaning "unmeasured" (0 or null) and update any consumer/label so it never advertises a made-up 2.49x-7.47x. Do NOT invent a number. Update the doc-comment lines claiming "2.49x-7.47x speedup" in those files to "approximate sparse attention; speedup unverified — see docs/reviews/intelligence-system-audit-2026-05-29.md". Keep builds clean. Report via schema.`,
|
|
55
|
-
},
|
|
56
|
-
{
|
|
57
|
-
key: 'embedding-observability', label: 'fix:embedding-observability',
|
|
58
|
-
prompt: `Repo ${REPO}. AUDIT FINDING #3: in ${CLI}/src/memory/memory-initializer.ts, generateEmbedding() falls back to MOCK/hash embeddings when transformers.js/sharp fails to load, but the returned object still reports model: "Xenova/all-MiniLM-L6-v2" — so an operator cannot tell mock output (inverted semantics) from real ONNX output.
|
|
59
|
-
Add an explicit \`backend: 'onnx' | 'mock'\` field to the generateEmbedding return value, set truthfully by which path produced the vector. Surface it where the model name is reported — at minimum the memory_bridge_status MCP tool and any "embedding: all-MiniLM-L6-v2 (384-dim)" status string should also state backend (e.g. "...384-dim, backend=mock"). Do not change the embedding math. Add/extend a test asserting the field is 'mock' when the real model is unavailable. Keep builds clean. Report via schema.`,
|
|
60
|
-
},
|
|
61
|
-
{
|
|
62
|
-
key: 'mcp-learning', label: 'fix:mcp-learning',
|
|
63
|
-
prompt: `Repo ${REPO}. AUDIT FINDINGS #4 & #5 in ${CLI}/src/mcp-tools/hooks-tools.ts:
|
|
64
|
-
(A) trajectory-end (~line 2474-2493) feeds the EWC consolidator a SYNTHETIC gradient built from a sine wave over the index (an array of 384 values like sin(i*0.01)*(steps/10)) instead of the trajectory's real embedding-derived gradient. Replace it with a gradient derived from the actual recorded trajectory embeddings/outcome (mirror the library DISTILL path), or if real embeddings aren't available there, pass the real available signal or SKIP the EWC update rather than feeding sine-wave noise.
|
|
65
|
-
(B) hooks_intelligence_learn (~line 2920) is named "force learning cycle" but only reads/echoes stats. Either make it actually trigger a real learning/consolidation cycle (call the real distill/consolidate path), or rename/redescribe it truthfully so it doesn't claim to learn.
|
|
66
|
-
Make minimal correct changes. Keep build clean (cd ${CLI} && npm run build). Add a smoke assertion if practical. Report via schema. You are the ONLY agent editing hooks-tools.ts — own it.`,
|
|
67
|
-
},
|
|
68
|
-
]
|
|
69
|
-
const fixes = (await parallel(
|
|
70
|
-
FIXES.map((f) => () => agent(f.prompt, { label: f.label, phase: 'Implement', schema: FIX_SCHEMA, agentType: 'coder' }))
|
|
71
|
-
)).filter(Boolean)
|
|
72
|
-
log(`Implement: ${fixes.filter((f) => f.applied).length}/${FIXES.length} fixes applied`)
|
|
73
|
-
|
|
74
|
-
phase('Validate')
|
|
75
|
-
const validation = await agent(
|
|
76
|
-
`Repo ${REPO}. Validate the working tree after parallel fixes to: route.ts, attention-coordinator.ts (swarm + integration), memory-initializer.ts, hooks-tools.ts.
|
|
77
|
-
1. cd ${CLI} && npm run build — must be clean (tsc). If the fixes introduced type errors, FIX them minimally and rebuild until clean.
|
|
78
|
-
2. If attention-coordinator changed and the swarm package has a build script: cd ${REPO}/v3/@claude-flow/swarm && npm run build (skip if no build script).
|
|
79
|
-
3. Run targeted tests: cd ${CLI} && npx vitest run __tests__/bug-cluster-2219-2226.test.ts __tests__/statusline-cost-display.test.ts plus any new tests the fixes added.
|
|
80
|
-
Report buildOk/testsOk and failures via schema. Do NOT weaken or delete tests to pass — fix the code.`,
|
|
81
|
-
{ label: 'validate:build+test', phase: 'Validate', schema: VALIDATE_SCHEMA, agentType: 'coder' }
|
|
82
|
-
)
|
|
83
|
-
log(`Validate: build=${validation?.buildOk} tests=${validation?.testsOk}`)
|
|
84
|
-
|
|
85
|
-
phase('Benchmark')
|
|
86
|
-
const bench = await agent(
|
|
87
|
-
`Repo ${REPO}. Build a REAL reusable benchmark harness at ${REPO}/scripts/benchmark-intelligence.mjs (clean, documented, exit 0, safe to re-run) and RUN it to produce measured numbers against the built ${CLI}/dist exports on THIS machine:
|
|
88
|
-
- HNSW search vs in-process brute-force cosine baseline at N = 1000, 5000, 20000, and 50000 if feasible: per-query ms + speedup ratio + recall@10.
|
|
89
|
-
- Int8 quantization: measured compression ratio + reconstruction cosine.
|
|
90
|
-
- RaBitQ: memory compression ratio; retrieval speed only if a populated index is feasible else "not measured".
|
|
91
|
-
- SONA WASM adapt latency (ms/call, warmed).
|
|
92
|
-
- MoE: confirm the gate learns (probability shift after rewards).
|
|
93
|
-
- Embedding backend actually in use (onnx vs mock) — honest.
|
|
94
|
-
Every value MUST come from a run — never hardcode or guess; mark unmeasurable items null with a reason. Emit numbers in the schema results object and print a markdown table to stdout.`,
|
|
95
|
-
{ label: 'benchmark:harness', phase: 'Benchmark', schema: BENCH_SCHEMA, agentType: 'perf-analyzer' }
|
|
96
|
-
)
|
|
97
|
-
log(`Benchmark: ran=${bench?.ran} -> ${bench?.harnessPath || 'no harness'}`)
|
|
98
|
-
|
|
99
|
-
phase('Optimize')
|
|
100
|
-
const optimize = await agent(
|
|
101
|
-
`Repo ${REPO}. HNSW search underperforms (audit ~1.48x peak, slower than brute force below N~5k). Attempt a GENUINE optimization, then RE-MEASURE with ${bench?.harnessPath || REPO + '/scripts/benchmark-intelligence.mjs'} and report before/after HONESTLY.
|
|
102
|
-
Levers (only what the code exposes): HNSW ef_construction / M / ef_search in the build/search path (${CLI}/src/memory + @ruvector/core config); the brute-force LIMIT 1000 fallback cap; ensuring the index is used above the crossover N.
|
|
103
|
-
Rules: (1) measure before AND after with the same harness; (2) if a change does NOT improve measured numbers, REVERT it and say so; (3) be honest — if HNSW only wins at large N (expected for ANN), report that rather than forcing a number. Report before/after and which changes you kept. Keep builds clean.`,
|
|
104
|
-
{ label: 'optimize:hnsw', phase: 'Optimize', schema: BENCH_SCHEMA, agentType: 'perf-analyzer' }
|
|
105
|
-
)
|
|
106
|
-
log(`Optimize: ran=${optimize?.ran}`)
|
|
107
|
-
|
|
108
|
-
phase('Docs')
|
|
109
|
-
const measured = JSON.stringify({ benchmark: bench?.results ?? null, optimized: optimize?.results ?? null })
|
|
110
|
-
const docs = await agent(
|
|
111
|
-
`Repo ${REPO}. Rewrite performance claims across docs using the MEASURED numbers below (NOT old hardcoded multipliers). Measured JSON: ${measured}
|
|
112
|
-
Revise ONLY the perf/capability claims in:
|
|
113
|
-
- ${REPO}/README.md — "150x-12,500x", "2.49x-7.47x", "75x", "32x", "3.92x", SONA "<0.05ms" → measured values or honest qualifiers ("approximate", "at N>=20k", "unverified" where no benchmark exists).
|
|
114
|
-
- ${REPO}/CLAUDE.md, ${REPO}/v3/CLAUDE.md, ${CLI}/CLAUDE.md — the "V3 Performance Targets" / "Intelligence System" tables.
|
|
115
|
-
- Add a one-line pointer in each perf table to docs/reviews/intelligence-system-audit-2026-05-29.md and scripts/benchmark-intelligence.mjs as source of truth.
|
|
116
|
-
Rules: every number must trace to the measured JSON or be marked "unverified/target". Keep CONFIRMED real numbers (Int8 ratio, RaBitQ memory ratio, SONA adapt ms, MoE converges). Mark HNSW with its real measured speedup + "ANN wins at large N" caveat. Remove/qualify the Flash Attention 2.49-7.47x claim. Report files changed and before->after for each headline number via schema.`,
|
|
117
|
-
{ label: 'docs:rewrite', phase: 'Docs', schema: FIX_SCHEMA, agentType: 'coder' }
|
|
118
|
-
)
|
|
119
|
-
log(`Docs: applied=${docs?.applied} files=${(docs?.files || []).length}`)
|
|
120
|
-
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'intelligence-system-hardening',
|
|
3
|
+
description: 'Implement audit fixes, build a real benchmark harness, optimize, validate, and rewrite perf docs with measured numbers',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Implement', detail: 'parallel fixes — distinct files, no conflicts' },
|
|
6
|
+
{ title: 'Validate', detail: 'build + tests; repair if broken' },
|
|
7
|
+
{ title: 'Benchmark', detail: 'real measurement harness -> JSON numbers' },
|
|
8
|
+
{ title: 'Optimize', detail: 'tune HNSW params, re-measure before/after' },
|
|
9
|
+
{ title: 'Docs', detail: 'rewrite README/CLAUDE.md perf claims with measured values' },
|
|
10
|
+
],
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
const REPO = '/Users/cohen/Projects/ruflo'
|
|
14
|
+
const CLI = `${REPO}/v3/@claude-flow/cli`
|
|
15
|
+
const RNG = 'an ' + 'RNG' + ' call (pseudo-random fabrication)'
|
|
16
|
+
|
|
17
|
+
const FIX_SCHEMA = {
|
|
18
|
+
type: 'object', additionalProperties: false,
|
|
19
|
+
required: ['issue', 'applied', 'summary', 'files'],
|
|
20
|
+
properties: {
|
|
21
|
+
issue: { type: 'string' }, applied: { type: 'boolean' },
|
|
22
|
+
summary: { type: 'string' }, files: { type: 'array', items: { type: 'string' } },
|
|
23
|
+
risk: { type: 'string' },
|
|
24
|
+
},
|
|
25
|
+
}
|
|
26
|
+
const BENCH_SCHEMA = {
|
|
27
|
+
type: 'object', additionalProperties: true,
|
|
28
|
+
required: ['ran', 'results', 'harnessPath', 'notes'],
|
|
29
|
+
properties: {
|
|
30
|
+
ran: { type: 'boolean' }, harnessPath: { type: 'string' },
|
|
31
|
+
results: { type: 'object', additionalProperties: true },
|
|
32
|
+
notes: { type: 'string' },
|
|
33
|
+
},
|
|
34
|
+
}
|
|
35
|
+
const VALIDATE_SCHEMA = {
|
|
36
|
+
type: 'object', additionalProperties: false,
|
|
37
|
+
required: ['buildOk', 'testsOk', 'summary'],
|
|
38
|
+
properties: {
|
|
39
|
+
buildOk: { type: 'boolean' }, testsOk: { type: 'boolean' },
|
|
40
|
+
summary: { type: 'string' }, failures: { type: 'array', items: { type: 'string' } },
|
|
41
|
+
},
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
phase('Implement')
|
|
45
|
+
const FIXES = [
|
|
46
|
+
{
|
|
47
|
+
key: 'reward-inversion', label: 'fix:reward-inversion',
|
|
48
|
+
prompt: `Repo ${REPO}. CRITICAL BUG (audit finding #1, follow-up to #2222). In ${CLI}/src/commands/route.ts the \`route feedback\` command: a negative reward passed the documented way (\`-r -1.0\` or \`--reward -1.0\`) is parsed as +1.00 because the CLI flag parser strips the leading '-' from negative numeric values. Only \`--reward=-1.0\` (equals form) preserves the sign. So a user giving NEGATIVE feedback actively REINFORCES the bad agent.
|
|
49
|
+
Investigate the flag-parsing path (route.ts reward flag def ~line 399, value read ~line 419; and the shared CLI arg parser route uses). Fix so \`-r -1.0\`, \`--reward -1.0\`, and \`--reward=-1.0\` ALL yield reward = -1.0. Prefer the most localized correct fix; if the bug is in the shared parser, fix it there but verify other negative-number flags still work. Add a regression test (extend ${CLI}/__tests__/bug-cluster-2219-2226.test.ts or new) asserting parsed reward sign for all three syntaxes. Build (cd ${CLI} && npm run build) must stay clean. Report via schema.`,
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
key: 'flash-fabrication', label: 'fix:flash-fabrication',
|
|
53
|
+
prompt: `Repo ${REPO}. AUDIT FINDING #2: ${REPO}/v3/@claude-flow/swarm/src/attention-coordinator.ts line 972 fabricates a fake metric — it sets performanceStats.flashSpeedup to a value computed from ${RNG}: roughly "2.49 plus rng times 4.98", and line 973 hardcodes memoryReduction = 0.75. Reporting a made-up number as real telemetry is a credibility liability. The SAME pattern exists in ${REPO}/v3/@claude-flow/integration/src/attention-coordinator.ts — fix BOTH copies.
|
|
54
|
+
Replace the pseudo-random fabrication with an honest value: either (a) actually invoke the FlashAttention kernel's own benchmark()/measured path to get a real speedup if cheaply available, or (b) if no measurement is wired, set flashSpeedup to a sentinel meaning "unmeasured" (0 or null) and update any consumer/label so it never advertises a made-up 2.49x-7.47x. Do NOT invent a number. Update the doc-comment lines claiming "2.49x-7.47x speedup" in those files to "approximate sparse attention; speedup unverified — see docs/reviews/intelligence-system-audit-2026-05-29.md". Keep builds clean. Report via schema.`,
|
|
55
|
+
},
|
|
56
|
+
{
|
|
57
|
+
key: 'embedding-observability', label: 'fix:embedding-observability',
|
|
58
|
+
prompt: `Repo ${REPO}. AUDIT FINDING #3: in ${CLI}/src/memory/memory-initializer.ts, generateEmbedding() falls back to MOCK/hash embeddings when transformers.js/sharp fails to load, but the returned object still reports model: "Xenova/all-MiniLM-L6-v2" — so an operator cannot tell mock output (inverted semantics) from real ONNX output.
|
|
59
|
+
Add an explicit \`backend: 'onnx' | 'mock'\` field to the generateEmbedding return value, set truthfully by which path produced the vector. Surface it where the model name is reported — at minimum the memory_bridge_status MCP tool and any "embedding: all-MiniLM-L6-v2 (384-dim)" status string should also state backend (e.g. "...384-dim, backend=mock"). Do not change the embedding math. Add/extend a test asserting the field is 'mock' when the real model is unavailable. Keep builds clean. Report via schema.`,
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
key: 'mcp-learning', label: 'fix:mcp-learning',
|
|
63
|
+
prompt: `Repo ${REPO}. AUDIT FINDINGS #4 & #5 in ${CLI}/src/mcp-tools/hooks-tools.ts:
|
|
64
|
+
(A) trajectory-end (~line 2474-2493) feeds the EWC consolidator a SYNTHETIC gradient built from a sine wave over the index (an array of 384 values like sin(i*0.01)*(steps/10)) instead of the trajectory's real embedding-derived gradient. Replace it with a gradient derived from the actual recorded trajectory embeddings/outcome (mirror the library DISTILL path), or if real embeddings aren't available there, pass the real available signal or SKIP the EWC update rather than feeding sine-wave noise.
|
|
65
|
+
(B) hooks_intelligence_learn (~line 2920) is named "force learning cycle" but only reads/echoes stats. Either make it actually trigger a real learning/consolidation cycle (call the real distill/consolidate path), or rename/redescribe it truthfully so it doesn't claim to learn.
|
|
66
|
+
Make minimal correct changes. Keep build clean (cd ${CLI} && npm run build). Add a smoke assertion if practical. Report via schema. You are the ONLY agent editing hooks-tools.ts — own it.`,
|
|
67
|
+
},
|
|
68
|
+
]
|
|
69
|
+
const fixes = (await parallel(
|
|
70
|
+
FIXES.map((f) => () => agent(f.prompt, { label: f.label, phase: 'Implement', schema: FIX_SCHEMA, agentType: 'coder' }))
|
|
71
|
+
)).filter(Boolean)
|
|
72
|
+
log(`Implement: ${fixes.filter((f) => f.applied).length}/${FIXES.length} fixes applied`)
|
|
73
|
+
|
|
74
|
+
phase('Validate')
|
|
75
|
+
const validation = await agent(
|
|
76
|
+
`Repo ${REPO}. Validate the working tree after parallel fixes to: route.ts, attention-coordinator.ts (swarm + integration), memory-initializer.ts, hooks-tools.ts.
|
|
77
|
+
1. cd ${CLI} && npm run build — must be clean (tsc). If the fixes introduced type errors, FIX them minimally and rebuild until clean.
|
|
78
|
+
2. If attention-coordinator changed and the swarm package has a build script: cd ${REPO}/v3/@claude-flow/swarm && npm run build (skip if no build script).
|
|
79
|
+
3. Run targeted tests: cd ${CLI} && npx vitest run __tests__/bug-cluster-2219-2226.test.ts __tests__/statusline-cost-display.test.ts plus any new tests the fixes added.
|
|
80
|
+
Report buildOk/testsOk and failures via schema. Do NOT weaken or delete tests to pass — fix the code.`,
|
|
81
|
+
{ label: 'validate:build+test', phase: 'Validate', schema: VALIDATE_SCHEMA, agentType: 'coder' }
|
|
82
|
+
)
|
|
83
|
+
log(`Validate: build=${validation?.buildOk} tests=${validation?.testsOk}`)
|
|
84
|
+
|
|
85
|
+
phase('Benchmark')
|
|
86
|
+
const bench = await agent(
|
|
87
|
+
`Repo ${REPO}. Build a REAL reusable benchmark harness at ${REPO}/scripts/benchmark-intelligence.mjs (clean, documented, exit 0, safe to re-run) and RUN it to produce measured numbers against the built ${CLI}/dist exports on THIS machine:
|
|
88
|
+
- HNSW search vs in-process brute-force cosine baseline at N = 1000, 5000, 20000, and 50000 if feasible: per-query ms + speedup ratio + recall@10.
|
|
89
|
+
- Int8 quantization: measured compression ratio + reconstruction cosine.
|
|
90
|
+
- RaBitQ: memory compression ratio; retrieval speed only if a populated index is feasible else "not measured".
|
|
91
|
+
- SONA WASM adapt latency (ms/call, warmed).
|
|
92
|
+
- MoE: confirm the gate learns (probability shift after rewards).
|
|
93
|
+
- Embedding backend actually in use (onnx vs mock) — honest.
|
|
94
|
+
Every value MUST come from a run — never hardcode or guess; mark unmeasurable items null with a reason. Emit numbers in the schema results object and print a markdown table to stdout.`,
|
|
95
|
+
{ label: 'benchmark:harness', phase: 'Benchmark', schema: BENCH_SCHEMA, agentType: 'perf-analyzer' }
|
|
96
|
+
)
|
|
97
|
+
log(`Benchmark: ran=${bench?.ran} -> ${bench?.harnessPath || 'no harness'}`)
|
|
98
|
+
|
|
99
|
+
phase('Optimize')
|
|
100
|
+
const optimize = await agent(
|
|
101
|
+
`Repo ${REPO}. HNSW search underperforms (audit ~1.48x peak, slower than brute force below N~5k). Attempt a GENUINE optimization, then RE-MEASURE with ${bench?.harnessPath || REPO + '/scripts/benchmark-intelligence.mjs'} and report before/after HONESTLY.
|
|
102
|
+
Levers (only what the code exposes): HNSW ef_construction / M / ef_search in the build/search path (${CLI}/src/memory + @ruvector/core config); the brute-force LIMIT 1000 fallback cap; ensuring the index is used above the crossover N.
|
|
103
|
+
Rules: (1) measure before AND after with the same harness; (2) if a change does NOT improve measured numbers, REVERT it and say so; (3) be honest — if HNSW only wins at large N (expected for ANN), report that rather than forcing a number. Report before/after and which changes you kept. Keep builds clean.`,
|
|
104
|
+
{ label: 'optimize:hnsw', phase: 'Optimize', schema: BENCH_SCHEMA, agentType: 'perf-analyzer' }
|
|
105
|
+
)
|
|
106
|
+
log(`Optimize: ran=${optimize?.ran}`)
|
|
107
|
+
|
|
108
|
+
phase('Docs')
|
|
109
|
+
const measured = JSON.stringify({ benchmark: bench?.results ?? null, optimized: optimize?.results ?? null })
|
|
110
|
+
const docs = await agent(
|
|
111
|
+
`Repo ${REPO}. Rewrite performance claims across docs using the MEASURED numbers below (NOT old hardcoded multipliers). Measured JSON: ${measured}
|
|
112
|
+
Revise ONLY the perf/capability claims in:
|
|
113
|
+
- ${REPO}/README.md — "150x-12,500x", "2.49x-7.47x", "75x", "32x", "3.92x", SONA "<0.05ms" → measured values or honest qualifiers ("approximate", "at N>=20k", "unverified" where no benchmark exists).
|
|
114
|
+
- ${REPO}/CLAUDE.md, ${REPO}/v3/CLAUDE.md, ${CLI}/CLAUDE.md — the "V3 Performance Targets" / "Intelligence System" tables.
|
|
115
|
+
- Add a one-line pointer in each perf table to docs/reviews/intelligence-system-audit-2026-05-29.md and scripts/benchmark-intelligence.mjs as source of truth.
|
|
116
|
+
Rules: every number must trace to the measured JSON or be marked "unverified/target". Keep CONFIRMED real numbers (Int8 ratio, RaBitQ memory ratio, SONA adapt ms, MoE converges). Mark HNSW with its real measured speedup + "ANN wins at large N" caveat. Remove/qualify the Flash Attention 2.49-7.47x claim. Report files changed and before->after for each headline number via schema.`,
|
|
117
|
+
{ label: 'docs:rewrite', phase: 'Docs', schema: FIX_SCHEMA, agentType: 'coder' }
|
|
118
|
+
)
|
|
119
|
+
log(`Docs: applied=${docs?.applied} files=${(docs?.files || []).length}`)
|
|
120
|
+
|
|
121
121
|
return { fixes, validation, benchmark: bench, optimize, docs }
|
|
@@ -1,91 +1,91 @@
|
|
|
1
|
-
export const meta = {
|
|
2
|
-
name: 'plugin-contract-audit',
|
|
3
|
-
description: 'Run every ruflo plugin smoke contract, fan diagnosis agents out over the failures, and report a punch list',
|
|
4
|
-
phases: [
|
|
5
|
-
{ title: 'Sweep', detail: 'run all plugins/*/scripts/smoke.sh, collect pass/fail' },
|
|
6
|
-
{ title: 'Diagnose', detail: 'one agent per failing plugin — root cause + minimal fix' },
|
|
7
|
-
{ title: 'Report', detail: 'assemble the audit summary' },
|
|
8
|
-
],
|
|
9
|
-
}
|
|
10
|
-
|
|
11
|
-
// args (all optional):
|
|
12
|
-
// string → only audit plugins whose name contains this substring
|
|
13
|
-
// { filter?: string, → same substring filter
|
|
14
|
-
// diagnose?: boolean } → set false to skip the Diagnose phase (sweep only)
|
|
15
|
-
const opts = typeof args === 'string' ? { filter: args } : (args || {})
|
|
16
|
-
const FILTER = opts.filter || ''
|
|
17
|
-
const DIAGNOSE = opts.diagnose !== false
|
|
18
|
-
|
|
19
|
-
const SWEEP_SCHEMA = {
|
|
20
|
-
type: 'object', additionalProperties: false,
|
|
21
|
-
required: ['results'],
|
|
22
|
-
properties: {
|
|
23
|
-
results: {
|
|
24
|
-
type: 'array',
|
|
25
|
-
items: {
|
|
26
|
-
type: 'object', additionalProperties: false,
|
|
27
|
-
required: ['plugin', 'passed', 'failed'],
|
|
28
|
-
properties: {
|
|
29
|
-
plugin: { type: 'string' },
|
|
30
|
-
passed: { type: 'integer' },
|
|
31
|
-
failed: { type: 'integer' },
|
|
32
|
-
exitCode: { type: 'integer' },
|
|
33
|
-
failingChecks: { type: 'array', items: { type: 'string' } },
|
|
34
|
-
},
|
|
35
|
-
},
|
|
36
|
-
},
|
|
37
|
-
notes: { type: 'string' },
|
|
38
|
-
},
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
const DIAGNOSIS_SCHEMA = {
|
|
42
|
-
type: 'object', additionalProperties: false,
|
|
43
|
-
required: ['plugin', 'rootCause', 'proposedFix', 'confident'],
|
|
44
|
-
properties: {
|
|
45
|
-
plugin: { type: 'string' },
|
|
46
|
-
rootCause: { type: 'string' },
|
|
47
|
-
proposedFix: { type: 'string' },
|
|
48
|
-
files: { type: 'array', items: { type: 'string' } },
|
|
49
|
-
confident: { type: 'boolean' },
|
|
50
|
-
},
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
phase('Sweep')
|
|
54
|
-
const filterClause = FILTER
|
|
55
|
-
? `Only audit plugins whose directory name contains "${FILTER}". `
|
|
56
|
-
: ''
|
|
57
|
-
const sweep = await agent(
|
|
58
|
-
`From the repo root, audit every ruflo plugin's smoke contract. ${filterClause}For each script matching the glob plugins/*/scripts/smoke.sh, run it with bash and capture its output and exit code. Each smoke script prints a trailing "N passed, M failed" line.
|
|
59
|
-
For every plugin report: plugin (the directory name under plugins/), passed (integer), failed (integer), exitCode (integer), and failingChecks (the "→ ..." lines that printed FAIL, verbatim, empty array if none).
|
|
60
|
-
Do NOT modify any files — this is read/run only. Return every audited plugin via the schema, not just the failing ones.`,
|
|
61
|
-
{ label: 'sweep:all-smokes', phase: 'Sweep', schema: SWEEP_SCHEMA, agentType: 'tester' }
|
|
62
|
-
)
|
|
63
|
-
|
|
64
|
-
const results = (sweep?.results || []).filter((r) => !FILTER || r.plugin.includes(FILTER))
|
|
65
|
-
const failures = results.filter((r) => r.failed > 0 || (r.exitCode && r.exitCode !== 0))
|
|
66
|
-
log(`Sweep: ${results.length} plugins audited, ${failures.length} failing`)
|
|
67
|
-
|
|
68
|
-
let diagnoses = []
|
|
69
|
-
if (DIAGNOSE && failures.length) {
|
|
70
|
-
phase('Diagnose')
|
|
71
|
-
diagnoses = (await parallel(
|
|
72
|
-
failures.map((f) => () =>
|
|
73
|
-
agent(
|
|
74
|
-
`Plugin "${f.plugin}" fails its smoke contract (plugins/${f.plugin}/scripts/smoke.sh): ${f.failed} check(s) failed. Failing checks:\n${(f.failingChecks || []).join('\n') || '(not captured — re-run the smoke script to see them)'}\n\nRead plugins/${f.plugin}/scripts/smoke.sh and the plugin files it inspects (plugin.json, README.md, skills, agents, commands, docs/adrs). Determine the ROOT CAUSE of each failing check and propose a MINIMAL fix. Distinguish a stale assertion in smoke.sh (the contract drifted from reality) from a genuine plugin defect. Do NOT edit anything — report only, via the schema, with confident=true only if the root cause is unambiguous.`,
|
|
75
|
-
{ label: `diagnose:${f.plugin}`, phase: 'Diagnose', schema: DIAGNOSIS_SCHEMA, agentType: 'code-analyzer' }
|
|
76
|
-
)
|
|
77
|
-
)
|
|
78
|
-
)).filter(Boolean)
|
|
79
|
-
log(`Diagnose: ${diagnoses.length}/${failures.length} diagnosed`)
|
|
80
|
-
}
|
|
81
|
-
|
|
82
|
-
phase('Report')
|
|
83
|
-
const summary = {
|
|
84
|
-
audited: results.length,
|
|
85
|
-
passing: results.length - failures.length,
|
|
86
|
-
failing: failures.length,
|
|
87
|
-
failingPlugins: failures.map((f) => ({ plugin: f.plugin, failed: f.failed })),
|
|
88
|
-
diagnoses,
|
|
89
|
-
}
|
|
90
|
-
log(`Report: ${summary.passing}/${summary.audited} plugins pass their contract`)
|
|
91
|
-
return summary
|
|
1
|
+
export const meta = {
|
|
2
|
+
name: 'plugin-contract-audit',
|
|
3
|
+
description: 'Run every ruflo plugin smoke contract, fan diagnosis agents out over the failures, and report a punch list',
|
|
4
|
+
phases: [
|
|
5
|
+
{ title: 'Sweep', detail: 'run all plugins/*/scripts/smoke.sh, collect pass/fail' },
|
|
6
|
+
{ title: 'Diagnose', detail: 'one agent per failing plugin — root cause + minimal fix' },
|
|
7
|
+
{ title: 'Report', detail: 'assemble the audit summary' },
|
|
8
|
+
],
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
// args (all optional):
|
|
12
|
+
// string → only audit plugins whose name contains this substring
|
|
13
|
+
// { filter?: string, → same substring filter
|
|
14
|
+
// diagnose?: boolean } → set false to skip the Diagnose phase (sweep only)
|
|
15
|
+
const opts = typeof args === 'string' ? { filter: args } : (args || {})
|
|
16
|
+
const FILTER = opts.filter || ''
|
|
17
|
+
const DIAGNOSE = opts.diagnose !== false
|
|
18
|
+
|
|
19
|
+
const SWEEP_SCHEMA = {
|
|
20
|
+
type: 'object', additionalProperties: false,
|
|
21
|
+
required: ['results'],
|
|
22
|
+
properties: {
|
|
23
|
+
results: {
|
|
24
|
+
type: 'array',
|
|
25
|
+
items: {
|
|
26
|
+
type: 'object', additionalProperties: false,
|
|
27
|
+
required: ['plugin', 'passed', 'failed'],
|
|
28
|
+
properties: {
|
|
29
|
+
plugin: { type: 'string' },
|
|
30
|
+
passed: { type: 'integer' },
|
|
31
|
+
failed: { type: 'integer' },
|
|
32
|
+
exitCode: { type: 'integer' },
|
|
33
|
+
failingChecks: { type: 'array', items: { type: 'string' } },
|
|
34
|
+
},
|
|
35
|
+
},
|
|
36
|
+
},
|
|
37
|
+
notes: { type: 'string' },
|
|
38
|
+
},
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
const DIAGNOSIS_SCHEMA = {
|
|
42
|
+
type: 'object', additionalProperties: false,
|
|
43
|
+
required: ['plugin', 'rootCause', 'proposedFix', 'confident'],
|
|
44
|
+
properties: {
|
|
45
|
+
plugin: { type: 'string' },
|
|
46
|
+
rootCause: { type: 'string' },
|
|
47
|
+
proposedFix: { type: 'string' },
|
|
48
|
+
files: { type: 'array', items: { type: 'string' } },
|
|
49
|
+
confident: { type: 'boolean' },
|
|
50
|
+
},
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
phase('Sweep')
|
|
54
|
+
const filterClause = FILTER
|
|
55
|
+
? `Only audit plugins whose directory name contains "${FILTER}". `
|
|
56
|
+
: ''
|
|
57
|
+
const sweep = await agent(
|
|
58
|
+
`From the repo root, audit every ruflo plugin's smoke contract. ${filterClause}For each script matching the glob plugins/*/scripts/smoke.sh, run it with bash and capture its output and exit code. Each smoke script prints a trailing "N passed, M failed" line.
|
|
59
|
+
For every plugin report: plugin (the directory name under plugins/), passed (integer), failed (integer), exitCode (integer), and failingChecks (the "→ ..." lines that printed FAIL, verbatim, empty array if none).
|
|
60
|
+
Do NOT modify any files — this is read/run only. Return every audited plugin via the schema, not just the failing ones.`,
|
|
61
|
+
{ label: 'sweep:all-smokes', phase: 'Sweep', schema: SWEEP_SCHEMA, agentType: 'tester' }
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
const results = (sweep?.results || []).filter((r) => !FILTER || r.plugin.includes(FILTER))
|
|
65
|
+
const failures = results.filter((r) => r.failed > 0 || (r.exitCode && r.exitCode !== 0))
|
|
66
|
+
log(`Sweep: ${results.length} plugins audited, ${failures.length} failing`)
|
|
67
|
+
|
|
68
|
+
let diagnoses = []
|
|
69
|
+
if (DIAGNOSE && failures.length) {
|
|
70
|
+
phase('Diagnose')
|
|
71
|
+
diagnoses = (await parallel(
|
|
72
|
+
failures.map((f) => () =>
|
|
73
|
+
agent(
|
|
74
|
+
`Plugin "${f.plugin}" fails its smoke contract (plugins/${f.plugin}/scripts/smoke.sh): ${f.failed} check(s) failed. Failing checks:\n${(f.failingChecks || []).join('\n') || '(not captured — re-run the smoke script to see them)'}\n\nRead plugins/${f.plugin}/scripts/smoke.sh and the plugin files it inspects (plugin.json, README.md, skills, agents, commands, docs/adrs). Determine the ROOT CAUSE of each failing check and propose a MINIMAL fix. Distinguish a stale assertion in smoke.sh (the contract drifted from reality) from a genuine plugin defect. Do NOT edit anything — report only, via the schema, with confident=true only if the root cause is unambiguous.`,
|
|
75
|
+
{ label: `diagnose:${f.plugin}`, phase: 'Diagnose', schema: DIAGNOSIS_SCHEMA, agentType: 'code-analyzer' }
|
|
76
|
+
)
|
|
77
|
+
)
|
|
78
|
+
)).filter(Boolean)
|
|
79
|
+
log(`Diagnose: ${diagnoses.length}/${failures.length} diagnosed`)
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
phase('Report')
|
|
83
|
+
const summary = {
|
|
84
|
+
audited: results.length,
|
|
85
|
+
passing: results.length - failures.length,
|
|
86
|
+
failing: failures.length,
|
|
87
|
+
failingPlugins: failures.map((f) => ({ plugin: f.plugin, failed: f.failed })),
|
|
88
|
+
diagnoses,
|
|
89
|
+
}
|
|
90
|
+
log(`Report: ${summary.passing}/${summary.audited} plugins pass their contract`)
|
|
91
|
+
return summary
|