@claude-flow/cli 3.32.9 → 3.32.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/.proven-config-version +1 -0
- package/.claude/agents/analysis/analyze-code-quality.md +178 -178
- package/.claude/agents/analysis/code-analyzer.md +209 -209
- package/.claude/agents/analysis/code-review/analyze-code-quality.md +178 -178
- package/.claude/agents/architecture/arch-system-design.md +156 -156
- package/.claude/agents/architecture/system-design/arch-system-design.md +154 -154
- package/.claude/agents/browser/browser-agent.yaml +182 -182
- package/.claude/agents/consensus/byzantine-coordinator.md +62 -62
- package/.claude/agents/consensus/crdt-synchronizer.md +996 -996
- package/.claude/agents/consensus/gossip-coordinator.md +62 -62
- package/.claude/agents/consensus/performance-benchmarker.md +850 -850
- package/.claude/agents/consensus/quorum-manager.md +822 -822
- package/.claude/agents/consensus/raft-manager.md +62 -62
- package/.claude/agents/consensus/security-manager.md +621 -621
- package/.claude/agents/core/planner.md +374 -374
- package/.claude/agents/custom/test-long-runner.md +44 -44
- package/.claude/agents/data/data-ml-model.md +444 -444
- package/.claude/agents/data/ml/data-ml-model.md +192 -192
- package/.claude/agents/development/backend/dev-backend-api.md +141 -141
- package/.claude/agents/development/dev-backend-api.md +344 -344
- package/.claude/agents/devops/ci-cd/ops-cicd-github.md +163 -163
- package/.claude/agents/devops/ops-cicd-github.md +164 -164
- package/.claude/agents/documentation/api-docs/docs-api-openapi.md +173 -173
- package/.claude/agents/documentation/docs-api-openapi.md +354 -354
- package/.claude/agents/flow-nexus/app-store.md +87 -87
- package/.claude/agents/flow-nexus/authentication.md +68 -68
- package/.claude/agents/flow-nexus/challenges.md +80 -80
- package/.claude/agents/flow-nexus/neural-network.md +87 -87
- package/.claude/agents/flow-nexus/payments.md +82 -82
- package/.claude/agents/flow-nexus/sandbox.md +75 -75
- package/.claude/agents/flow-nexus/swarm.md +75 -75
- package/.claude/agents/flow-nexus/user-tools.md +95 -95
- package/.claude/agents/flow-nexus/workflow.md +83 -83
- package/.claude/agents/github/code-review-swarm.md +377 -377
- package/.claude/agents/github/github-modes.md +172 -172
- package/.claude/agents/github/issue-tracker.md +575 -575
- package/.claude/agents/github/multi-repo-swarm.md +552 -552
- package/.claude/agents/github/pr-manager.md +437 -437
- package/.claude/agents/github/project-board-sync.md +508 -508
- package/.claude/agents/github/release-manager.md +604 -604
- package/.claude/agents/github/release-swarm.md +582 -582
- package/.claude/agents/github/repo-architect.md +397 -397
- package/.claude/agents/github/swarm-issue.md +572 -572
- package/.claude/agents/github/swarm-pr.md +427 -427
- package/.claude/agents/github/sync-coordinator.md +451 -451
- package/.claude/agents/github/workflow-automation.md +902 -902
- package/.claude/agents/goal/agent.md +815 -815
- package/.claude/agents/optimization/benchmark-suite.md +664 -664
- package/.claude/agents/optimization/load-balancer.md +430 -430
- package/.claude/agents/optimization/performance-monitor.md +671 -671
- package/.claude/agents/optimization/resource-allocator.md +673 -673
- package/.claude/agents/optimization/topology-optimizer.md +807 -807
- package/.claude/agents/payments/agentic-payments.md +126 -126
- package/.claude/agents/sona/sona-learning-optimizer.md +74 -74
- package/.claude/agents/sparc/architecture.md +698 -698
- package/.claude/agents/sparc/pseudocode.md +519 -519
- package/.claude/agents/sparc/refinement.md +801 -801
- package/.claude/agents/sparc/specification.md +477 -477
- package/.claude/agents/specialized/mobile/spec-mobile-react-native.md +224 -224
- package/.claude/agents/specialized/spec-mobile-react-native.md +226 -226
- package/.claude/agents/sublinear/consensus-coordinator.md +337 -337
- package/.claude/agents/sublinear/matrix-optimizer.md +184 -184
- package/.claude/agents/sublinear/pagerank-analyzer.md +298 -298
- package/.claude/agents/sublinear/performance-optimizer.md +367 -367
- package/.claude/agents/sublinear/trading-predictor.md +245 -245
- package/.claude/agents/swarm/adaptive-coordinator.md +1126 -1126
- package/.claude/agents/swarm/hierarchical-coordinator.md +709 -709
- package/.claude/agents/swarm/mesh-coordinator.md +962 -962
- package/.claude/agents/templates/automation-smart-agent.md +204 -204
- package/.claude/agents/templates/base-template-generator.md +289 -289
- package/.claude/agents/templates/coordinator-swarm-init.md +89 -89
- package/.claude/agents/templates/github-pr-manager.md +176 -176
- package/.claude/agents/templates/implementer-sparc-coder.md +258 -258
- package/.claude/agents/templates/memory-coordinator.md +186 -186
- package/.claude/agents/templates/orchestrator-task.md +138 -138
- package/.claude/agents/templates/performance-analyzer.md +198 -198
- package/.claude/agents/templates/sparc-coordinator.md +513 -513
- package/.claude/agents/testing/production-validator.md +394 -394
- package/.claude/agents/testing/tdd-london-swarm.md +243 -243
- package/.claude/agents/v3/aidefence-guardian.md +282 -282
- package/.claude/agents/v3/claims-authorizer.md +208 -208
- package/.claude/agents/v3/collective-intelligence-coordinator.md +993 -993
- package/.claude/agents/v3/ddd-domain-expert.md +220 -220
- package/.claude/agents/v3/injection-analyst.md +236 -236
- package/.claude/agents/v3/performance-engineer.md +1233 -1233
- package/.claude/agents/v3/pii-detector.md +151 -151
- package/.claude/agents/v3/reasoningbank-learner.md +213 -213
- package/.claude/agents/v3/security-architect-aidefence.md +410 -410
- package/.claude/agents/v3/security-architect.md +867 -867
- package/.claude/agents/v3/swarm-memory-manager.md +157 -157
- package/.claude/agents/v3/v3-integration-architect.md +205 -205
- package/.claude/commands/agents/README.md +50 -50
- package/.claude/commands/agents/agent-capabilities.md +140 -140
- package/.claude/commands/agents/agent-coordination.md +28 -28
- package/.claude/commands/agents/agent-spawning.md +28 -28
- package/.claude/commands/agents/agent-types.md +216 -216
- package/.claude/commands/agents/health.md +139 -139
- package/.claude/commands/agents/list.md +100 -100
- package/.claude/commands/agents/logs.md +130 -130
- package/.claude/commands/agents/metrics.md +122 -122
- package/.claude/commands/agents/pool.md +127 -127
- package/.claude/commands/agents/spawn.md +140 -140
- package/.claude/commands/agents/status.md +115 -115
- package/.claude/commands/agents/stop.md +102 -102
- package/.claude/commands/analysis/COMMAND_COMPLIANCE_REPORT.md +53 -53
- package/.claude/commands/analysis/README.md +9 -9
- package/.claude/commands/analysis/bottleneck-detect.md +162 -162
- package/.claude/commands/analysis/performance-bottlenecks.md +58 -58
- package/.claude/commands/analysis/performance-report.md +25 -25
- package/.claude/commands/analysis/token-efficiency.md +44 -44
- package/.claude/commands/analysis/token-usage.md +25 -25
- package/.claude/commands/automation/README.md +9 -9
- package/.claude/commands/automation/auto-agent.md +122 -122
- package/.claude/commands/automation/self-healing.md +105 -105
- package/.claude/commands/automation/session-memory.md +89 -89
- package/.claude/commands/automation/smart-agents.md +72 -72
- package/.claude/commands/automation/smart-spawn.md +25 -25
- package/.claude/commands/automation/workflow-select.md +25 -25
- package/.claude/commands/claude-flow-help.md +103 -103
- package/.claude/commands/claude-flow-memory.md +107 -107
- package/.claude/commands/claude-flow-swarm.md +205 -205
- package/.claude/commands/coordination/README.md +9 -9
- package/.claude/commands/coordination/agent-spawn.md +25 -25
- package/.claude/commands/coordination/init.md +44 -44
- package/.claude/commands/coordination/orchestrate.md +43 -43
- package/.claude/commands/coordination/spawn.md +45 -45
- package/.claude/commands/coordination/swarm-init.md +85 -85
- package/.claude/commands/coordination/task-orchestrate.md +25 -25
- package/.claude/commands/github/README.md +11 -11
- package/.claude/commands/github/code-review-swarm.md +513 -513
- package/.claude/commands/github/code-review.md +25 -25
- package/.claude/commands/github/github-modes.md +146 -146
- package/.claude/commands/github/github-swarm.md +121 -121
- package/.claude/commands/github/issue-tracker.md +291 -291
- package/.claude/commands/github/issue-triage.md +25 -25
- package/.claude/commands/github/multi-repo-swarm.md +518 -518
- package/.claude/commands/github/pr-enhance.md +26 -26
- package/.claude/commands/github/pr-manager.md +169 -169
- package/.claude/commands/github/project-board-sync.md +470 -470
- package/.claude/commands/github/release-manager.md +339 -339
- package/.claude/commands/github/release-swarm.md +543 -543
- package/.claude/commands/github/repo-analyze.md +25 -25
- package/.claude/commands/github/repo-architect.md +366 -366
- package/.claude/commands/github/swarm-issue.md +484 -484
- package/.claude/commands/github/swarm-pr.md +287 -287
- package/.claude/commands/github/sync-coordinator.md +302 -302
- package/.claude/commands/github/workflow-automation.md +441 -441
- package/.claude/commands/hive-mind/README.md +17 -17
- package/.claude/commands/hive-mind/hive-mind-consensus.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-init.md +18 -18
- package/.claude/commands/hive-mind/hive-mind-memory.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-metrics.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-resume.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-sessions.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-spawn.md +21 -21
- package/.claude/commands/hive-mind/hive-mind-status.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-stop.md +8 -8
- package/.claude/commands/hive-mind/hive-mind-wizard.md +8 -8
- package/.claude/commands/hive-mind/hive-mind.md +27 -27
- package/.claude/commands/hooks/README.md +11 -11
- package/.claude/commands/hooks/overview.md +57 -57
- package/.claude/commands/hooks/post-edit.md +117 -117
- package/.claude/commands/hooks/post-task.md +112 -112
- package/.claude/commands/hooks/pre-edit.md +113 -113
- package/.claude/commands/hooks/pre-task.md +111 -111
- package/.claude/commands/hooks/session-end.md +118 -118
- package/.claude/commands/hooks/setup.md +102 -102
- package/.claude/commands/memory/README.md +9 -9
- package/.claude/commands/memory/memory-persist.md +25 -25
- package/.claude/commands/memory/memory-search.md +25 -25
- package/.claude/commands/memory/memory-usage.md +25 -25
- package/.claude/commands/memory/neural.md +47 -47
- package/.claude/commands/monitoring/README.md +9 -9
- package/.claude/commands/monitoring/agent-metrics.md +25 -25
- package/.claude/commands/monitoring/agents.md +44 -44
- package/.claude/commands/monitoring/real-time-view.md +25 -25
- package/.claude/commands/monitoring/status.md +46 -46
- package/.claude/commands/monitoring/swarm-monitor.md +25 -25
- package/.claude/commands/optimization/README.md +9 -9
- package/.claude/commands/optimization/auto-topology.md +61 -61
- package/.claude/commands/optimization/cache-manage.md +25 -25
- package/.claude/commands/optimization/parallel-execute.md +25 -25
- package/.claude/commands/optimization/parallel-execution.md +49 -49
- package/.claude/commands/optimization/topology-optimize.md +25 -25
- package/.claude/commands/pair/README.md +260 -260
- package/.claude/commands/pair/commands.md +545 -545
- package/.claude/commands/pair/config.md +509 -509
- package/.claude/commands/pair/examples.md +511 -511
- package/.claude/commands/pair/modes.md +347 -347
- package/.claude/commands/pair/session.md +406 -406
- package/.claude/commands/pair/start.md +208 -208
- package/.claude/commands/sparc/analyzer.md +51 -51
- package/.claude/commands/sparc/architect.md +53 -53
- package/.claude/commands/sparc/ask.md +97 -97
- package/.claude/commands/sparc/batch-executor.md +54 -54
- package/.claude/commands/sparc/code.md +89 -89
- package/.claude/commands/sparc/coder.md +54 -54
- package/.claude/commands/sparc/debug.md +83 -83
- package/.claude/commands/sparc/debugger.md +54 -54
- package/.claude/commands/sparc/designer.md +53 -53
- package/.claude/commands/sparc/devops.md +109 -109
- package/.claude/commands/sparc/docs-writer.md +80 -80
- package/.claude/commands/sparc/documenter.md +54 -54
- package/.claude/commands/sparc/innovator.md +54 -54
- package/.claude/commands/sparc/integration.md +83 -83
- package/.claude/commands/sparc/mcp.md +117 -117
- package/.claude/commands/sparc/memory-manager.md +54 -54
- package/.claude/commands/sparc/optimizer.md +54 -54
- package/.claude/commands/sparc/orchestrator.md +131 -131
- package/.claude/commands/sparc/post-deployment-monitoring-mode.md +83 -83
- package/.claude/commands/sparc/refinement-optimization-mode.md +83 -83
- package/.claude/commands/sparc/researcher.md +54 -54
- package/.claude/commands/sparc/reviewer.md +54 -54
- package/.claude/commands/sparc/security-review.md +80 -80
- package/.claude/commands/sparc/sparc-modes.md +174 -174
- package/.claude/commands/sparc/sparc.md +111 -111
- package/.claude/commands/sparc/spec-pseudocode.md +80 -80
- package/.claude/commands/sparc/supabase-admin.md +348 -348
- package/.claude/commands/sparc/swarm-coordinator.md +54 -54
- package/.claude/commands/sparc/tdd.md +54 -54
- package/.claude/commands/sparc/tester.md +54 -54
- package/.claude/commands/sparc/tutorial.md +79 -79
- package/.claude/commands/sparc/workflow-manager.md +54 -54
- package/.claude/commands/sparc.md +166 -166
- package/.claude/commands/stream-chain/pipeline.md +120 -120
- package/.claude/commands/stream-chain/run.md +69 -69
- package/.claude/commands/swarm/README.md +15 -15
- package/.claude/commands/swarm/analysis.md +95 -95
- package/.claude/commands/swarm/development.md +96 -96
- package/.claude/commands/swarm/examples.md +168 -168
- package/.claude/commands/swarm/maintenance.md +102 -102
- package/.claude/commands/swarm/optimization.md +117 -117
- package/.claude/commands/swarm/research.md +136 -136
- package/.claude/commands/swarm/swarm-analysis.md +8 -8
- package/.claude/commands/swarm/swarm-background.md +8 -8
- package/.claude/commands/swarm/swarm-init.md +19 -19
- package/.claude/commands/swarm/swarm-modes.md +8 -8
- package/.claude/commands/swarm/swarm-monitor.md +8 -8
- package/.claude/commands/swarm/swarm-spawn.md +19 -19
- package/.claude/commands/swarm/swarm-status.md +8 -8
- package/.claude/commands/swarm/swarm-strategies.md +8 -8
- package/.claude/commands/swarm/swarm.md +87 -87
- package/.claude/commands/swarm/testing.md +131 -131
- package/.claude/commands/training/README.md +9 -9
- package/.claude/commands/training/model-update.md +25 -25
- package/.claude/commands/training/neural-patterns.md +107 -107
- package/.claude/commands/training/neural-train.md +75 -75
- package/.claude/commands/training/pattern-learn.md +25 -25
- package/.claude/commands/training/specialization.md +62 -62
- package/.claude/commands/truth/start.md +142 -142
- package/.claude/commands/verify/check.md +49 -49
- package/.claude/commands/verify/start.md +127 -127
- package/.claude/commands/workflows/README.md +9 -9
- package/.claude/commands/workflows/development.md +77 -77
- package/.claude/commands/workflows/research.md +62 -62
- package/.claude/commands/workflows/workflow-create.md +25 -25
- package/.claude/commands/workflows/workflow-execute.md +25 -25
- package/.claude/commands/workflows/workflow-export.md +25 -25
- package/.claude/eval/human-relevance-frozen-v1.json +17 -17
- package/.claude/evolve-proof/generation-0.json +211 -211
- package/.claude/evolve-proof/real-generation-0.json +406 -406
- package/.claude/evolve-proof/real-generation-1.json +406 -406
- package/.claude/helpers/.helpers-version +1 -1
- package/.claude/helpers/README.md +96 -96
- package/.claude/helpers/adr-compliance.sh +186 -186
- package/.claude/helpers/auto-commit.sh +178 -178
- package/.claude/helpers/auto-memory-hook.mjs +0 -0
- package/.claude/helpers/checkpoint-manager.sh +251 -251
- package/.claude/helpers/daemon-manager.sh +252 -252
- package/.claude/helpers/ddd-tracker.sh +144 -144
- package/.claude/helpers/github-safe.js +156 -156
- package/.claude/helpers/github-setup.sh +45 -45
- package/.claude/helpers/guidance-hook.sh +13 -13
- package/.claude/helpers/guidance-hooks.sh +102 -102
- package/.claude/helpers/health-monitor.sh +108 -108
- package/.claude/helpers/helpers.manifest.json +2 -2
- package/.claude/helpers/hook-handler.cjs +0 -0
- package/.claude/helpers/intelligence.cjs +0 -0
- package/.claude/helpers/learning-hooks.sh +329 -329
- package/.claude/helpers/learning-optimizer.sh +127 -127
- package/.claude/helpers/learning-service.mjs +1144 -1144
- package/.claude/helpers/memory.js +83 -83
- package/.claude/helpers/metrics-db.mjs +503 -503
- package/.claude/helpers/pattern-consolidator.sh +86 -86
- package/.claude/helpers/perf-worker.sh +160 -160
- package/.claude/helpers/post-commit +16 -16
- package/.claude/helpers/pre-commit +26 -26
- package/.claude/helpers/quick-start.sh +19 -19
- package/.claude/helpers/router.js +105 -105
- package/.claude/helpers/security-scanner.sh +127 -127
- package/.claude/helpers/session.js +157 -157
- package/.claude/helpers/setup-mcp.sh +18 -18
- package/.claude/helpers/standard-checkpoint-hooks.sh +189 -189
- package/.claude/helpers/statusline-hook.sh +21 -21
- package/.claude/helpers/statusline.cjs +0 -0
- package/.claude/helpers/statusline.js +340 -340
- package/.claude/helpers/swarm-comms.sh +353 -353
- package/.claude/helpers/swarm-hooks.sh +761 -761
- package/.claude/helpers/swarm-monitor.sh +210 -210
- package/.claude/helpers/sync-v3-metrics.sh +245 -245
- package/.claude/helpers/update-v3-progress.sh +165 -165
- package/.claude/helpers/v3-quick-status.sh +57 -57
- package/.claude/helpers/v3.sh +110 -110
- package/.claude/helpers/validate-v3-config.sh +215 -215
- package/.claude/helpers/worker-manager.sh +170 -170
- package/.claude/proven-config.json +42 -0
- package/.claude/proven-config.manifest.json +37 -37
- package/.claude/proven-config.signed.json +41 -41
- package/.claude/settings.json +182 -182
- package/.claude/skills/agentdb-advanced/SKILL.md +550 -550
- package/.claude/skills/agentdb-learning/SKILL.md +545 -545
- package/.claude/skills/agentdb-memory-patterns/SKILL.md +339 -339
- package/.claude/skills/agentdb-optimization/SKILL.md +509 -509
- package/.claude/skills/agentdb-vector-search/SKILL.md +339 -339
- package/.claude/skills/browser/SKILL.md +204 -204
- package/.claude/skills/dual-mode/README.md +71 -71
- package/.claude/skills/dual-mode/dual-collect.md +103 -103
- package/.claude/skills/dual-mode/dual-coordinate.md +85 -85
- package/.claude/skills/dual-mode/dual-spawn.md +81 -81
- package/.claude/skills/flow-nexus-neural/SKILL.md +727 -727
- package/.claude/skills/flow-nexus-platform/SKILL.md +1154 -1154
- package/.claude/skills/flow-nexus-swarm/SKILL.md +604 -604
- package/.claude/skills/github-code-review/SKILL.md +1125 -1125
- package/.claude/skills/github-multi-repo/SKILL.md +862 -862
- package/.claude/skills/github-project-management/SKILL.md +1262 -1262
- package/.claude/skills/github-release-management/SKILL.md +1064 -1064
- package/.claude/skills/github-workflow-automation/SKILL.md +1047 -1047
- package/.claude/skills/hooks-automation/SKILL.md +1201 -1201
- package/.claude/skills/pair-programming/SKILL.md +1202 -1202
- package/.claude/skills/reasoningbank-agentdb/SKILL.md +446 -446
- package/.claude/skills/reasoningbank-intelligence/SKILL.md +201 -201
- package/.claude/skills/skill-builder/SKILL.md +910 -910
- package/.claude/skills/sparc-methodology/SKILL.md +1106 -1106
- package/.claude/skills/stream-chain/SKILL.md +560 -560
- package/.claude/skills/swarm-advanced/SKILL.md +970 -970
- package/.claude/skills/swarm-orchestration/SKILL.md +179 -179
- package/.claude/skills/v3-cli-modernization/SKILL.md +871 -871
- package/.claude/skills/v3-core-implementation/SKILL.md +796 -796
- package/.claude/skills/v3-ddd-architecture/SKILL.md +441 -441
- package/.claude/skills/v3-integration-deep/SKILL.md +240 -240
- package/.claude/skills/v3-mcp-optimization/SKILL.md +776 -776
- package/.claude/skills/v3-memory-unification/SKILL.md +173 -173
- package/.claude/skills/v3-performance-optimization/SKILL.md +389 -389
- package/.claude/skills/v3-security-overhaul/SKILL.md +81 -81
- package/.claude/skills/v3-swarm-coordination/SKILL.md +339 -339
- package/.claude/skills/verification-quality/SKILL.md +691 -691
- package/README.md +419 -419
- package/bin/cli.js +314 -314
- package/bin/mcp-server.js +224 -224
- package/bin/preinstall.cjs +2 -2
- package/catalog-manifest.json +2 -2
- package/dist/src/autopilot-state.js +24 -7
- package/dist/src/benchmarks/gaia-critic.js +24 -24
- package/dist/src/business-pods/bbs-budget-tracker.js +53 -53
- package/dist/src/commands/completions.js +409 -409
- package/dist/src/commands/daemon.js +44 -44
- package/dist/src/commands/embeddings.js +26 -26
- package/dist/src/commands/hive-mind.js +97 -97
- package/dist/src/commands/hooks.js +31 -10
- package/dist/src/commands/init.js +202 -34
- package/dist/src/commands/memory.js +12 -1
- package/dist/src/commands/ruvector/backup.js +23 -23
- package/dist/src/commands/ruvector/benchmark.js +31 -31
- package/dist/src/commands/ruvector/import.js +14 -14
- package/dist/src/commands/ruvector/init.js +115 -115
- package/dist/src/commands/ruvector/migrate.js +99 -99
- package/dist/src/commands/ruvector/optimize.js +51 -51
- package/dist/src/commands/ruvector/setup.js +624 -624
- package/dist/src/commands/ruvector/status.js +38 -38
- package/dist/src/config/proven-config.js +2 -2
- package/dist/src/funnel/disclosure.js +13 -2
- package/dist/src/funnel/messages.d.ts +12 -10
- package/dist/src/funnel/messages.js +83 -11
- package/dist/src/init/claudemd-generator.js +231 -231
- package/dist/src/init/executor.js +453 -453
- package/dist/src/init/helper-signing.js +2 -2
- package/dist/src/init/helpers-generator.js +751 -751
- package/dist/src/init/statusline-generator.js +24 -24
- package/dist/src/mcp-tools/agentdb-tools.js +15 -15
- package/dist/src/mcp-tools/browser-intent-tools.js +19 -19
- package/dist/src/mcp-tools/browser-tools.js +8 -0
- package/dist/src/mcp-tools/hooks-tools.js +21 -0
- package/dist/src/mcp-tools/memory-tools.js +4 -3
- package/dist/src/memory/graph-edge-writer.js +22 -22
- package/dist/src/memory/memory-bridge.js +248 -158
- package/dist/src/memory/memory-initializer.js +407 -407
- package/dist/src/memory/rabitq-index.js +5 -5
- package/dist/src/parser.js +25 -9
- package/dist/src/proxy/verify.js +2 -2
- package/dist/src/runtime/headless.js +28 -28
- package/dist/src/services/distill-tuning.js +7 -7
- package/dist/src/services/headless-worker-executor.js +84 -84
- package/dist/src/services/memory-distillation.js +4 -4
- package/dist/src/services/worker-daemon.js +7 -4
- package/dist/src/transfer/deploy-seraphine.js +23 -23
- package/package.json +137 -137
- package/plugins/ruflo-metaharness/.claude-plugin/plugin.json +32 -32
- package/plugins/ruflo-metaharness/README.md +72 -72
- package/plugins/ruflo-metaharness/agents/metaharness-architect.md +58 -58
- package/plugins/ruflo-metaharness/commands/ruflo-metaharness.md +48 -48
- package/plugins/ruflo-metaharness/scripts/_darwin.mjs +210 -210
- package/plugins/ruflo-metaharness/scripts/_harness.mjs +330 -330
- package/plugins/ruflo-metaharness/scripts/_invoke.mjs +231 -231
- package/plugins/ruflo-metaharness/scripts/_redblue.mjs +143 -143
- package/plugins/ruflo-metaharness/scripts/_similarity.mjs +161 -161
- package/plugins/ruflo-metaharness/scripts/_spike-similarity.mjs +223 -223
- package/plugins/ruflo-metaharness/scripts/audit-list.mjs +158 -158
- package/plugins/ruflo-metaharness/scripts/audit-trend.mjs +272 -272
- package/plugins/ruflo-metaharness/scripts/bench-parse-mcp-scan.mjs +146 -146
- package/plugins/ruflo-metaharness/scripts/bench-recordpair-overhead.mjs +186 -186
- package/plugins/ruflo-metaharness/scripts/bench-similarity.mjs +177 -177
- package/plugins/ruflo-metaharness/scripts/bench.mjs +95 -95
- package/plugins/ruflo-metaharness/scripts/drift-from-history.mjs +363 -363
- package/plugins/ruflo-metaharness/scripts/evolve.mjs +404 -404
- package/plugins/ruflo-metaharness/scripts/genome.mjs +80 -80
- package/plugins/ruflo-metaharness/scripts/gepa.mjs +153 -153
- package/plugins/ruflo-metaharness/scripts/learn.mjs +127 -127
- package/plugins/ruflo-metaharness/scripts/mcp-scan.mjs +111 -111
- package/plugins/ruflo-metaharness/scripts/mint.mjs +126 -126
- package/plugins/ruflo-metaharness/scripts/oia-audit.mjs +228 -228
- package/plugins/ruflo-metaharness/scripts/redblue.mjs +286 -286
- package/plugins/ruflo-metaharness/scripts/router-parallel-analyze.mjs +250 -250
- package/plugins/ruflo-metaharness/scripts/score.mjs +92 -92
- package/plugins/ruflo-metaharness/scripts/security-bench.mjs +174 -174
- package/plugins/ruflo-metaharness/scripts/similarity.mjs +158 -158
- package/plugins/ruflo-metaharness/scripts/smoke.sh +2356 -2356
- package/plugins/ruflo-metaharness/scripts/test-graceful-degradation.mjs +165 -165
- package/plugins/ruflo-metaharness/scripts/test-mcp-tools.mjs +472 -472
- package/plugins/ruflo-metaharness/scripts/test-parallel-pipeline.mjs +204 -204
- package/plugins/ruflo-metaharness/scripts/test-pipeline-roundtrip.mjs +586 -586
- package/plugins/ruflo-metaharness/scripts/test-similarity.mjs +334 -334
- package/plugins/ruflo-metaharness/scripts/test-with-openrouter.mjs +229 -229
- package/plugins/ruflo-metaharness/scripts/threat-model.mjs +59 -59
- package/plugins/ruflo-metaharness/skills/harness-bench/SKILL.md +64 -64
- package/plugins/ruflo-metaharness/skills/harness-drift-from-history/SKILL.md +65 -65
- package/plugins/ruflo-metaharness/skills/harness-evolve/SKILL.md +131 -131
- package/plugins/ruflo-metaharness/skills/harness-genome/SKILL.md +54 -54
- package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +65 -65
- package/plugins/ruflo-metaharness/skills/harness-learn/SKILL.md +65 -65
- package/plugins/ruflo-metaharness/skills/harness-mcp-scan/SKILL.md +49 -49
- package/plugins/ruflo-metaharness/skills/harness-mint/SKILL.md +72 -72
- package/plugins/ruflo-metaharness/skills/harness-oia-audit/SKILL.md +79 -79
- package/plugins/ruflo-metaharness/skills/harness-score/SKILL.md +66 -66
- package/plugins/ruflo-metaharness/skills/harness-security-bench/SKILL.md +101 -101
- package/plugins/ruflo-metaharness/skills/harness-similarity/SKILL.md +67 -67
- package/plugins/ruflo-metaharness/skills/harness-threat-model/SKILL.md +41 -41
- package/scripts/postinstall.cjs +153 -153
|
@@ -1,250 +1,250 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
// router-parallel-analyze.mjs — ADR-150 Phase 2 SelfEvolvingRouter
|
|
3
|
-
// promotion gate.
|
|
4
|
-
//
|
|
5
|
-
// Reads paired routing decisions (Thompson bandit pick + SelfEvolvingRouter
|
|
6
|
-
// pick, both with measured outcomes) from a JSONL trajectory file and
|
|
7
|
-
// computes the THREE PROMOTION CRITERIA from ADR-150 review-round-1:
|
|
8
|
-
//
|
|
9
|
-
// (a) qualityScore improvement > 2% (relative)
|
|
10
|
-
// (b) usdPerDecision increase < 1% (relative)
|
|
11
|
-
// (c) p95 routing-decision latency increase < 5% (relative)
|
|
12
|
-
//
|
|
13
|
-
// ALL THREE must hold for promotion. AND, not OR — the OR form let
|
|
14
|
-
// quality gains mask cost regressions, exactly the failure mode ADR-149's
|
|
15
|
-
// Pareto framing was built to prevent.
|
|
16
|
-
//
|
|
17
|
-
// INPUT FORMAT (.swarm/router-parallel.jsonl, one JSON record per line):
|
|
18
|
-
// {
|
|
19
|
-
// "ts": "<iso>",
|
|
20
|
-
// "task": { ... },
|
|
21
|
-
// "bandit": { "pick": "<modelId>", "predictedQuality": 0.84, "predictedCostUsd": 0.003 },
|
|
22
|
-
// "ser": { "pick": "<modelId>", "predictedQuality": 0.87, "predictedCostUsd": 0.004 },
|
|
23
|
-
// "outcome": { "actualModel": "<modelId>", "actualQuality": 0.91, "actualUsd": 0.0035, "actualLatencyMs": 1240 }
|
|
24
|
-
// }
|
|
25
|
-
//
|
|
26
|
-
// USAGE
|
|
27
|
-
// node scripts/router-parallel-analyze.mjs --input .swarm/router-parallel.jsonl
|
|
28
|
-
// node scripts/router-parallel-analyze.mjs --input <file> --format json
|
|
29
|
-
// node scripts/router-parallel-analyze.mjs --input <file> --strict # exit 1 if NOT promotable
|
|
30
|
-
//
|
|
31
|
-
// EXIT CODES
|
|
32
|
-
// 0 analysis complete; in --strict mode → all three criteria passed
|
|
33
|
-
// 1 --strict requested AND at least one criterion failed (NOT promotable)
|
|
34
|
-
// 2 config error or input file missing/malformed
|
|
35
|
-
|
|
36
|
-
import { readFileSync, existsSync } from 'node:fs';
|
|
37
|
-
|
|
38
|
-
const ARGS = (() => {
|
|
39
|
-
const a = { input: '.swarm/router-parallel.jsonl', format: 'table', strict: false };
|
|
40
|
-
for (let i = 2; i < process.argv.length; i++) {
|
|
41
|
-
const v = process.argv[i];
|
|
42
|
-
if (v === '--input') a.input = process.argv[++i];
|
|
43
|
-
else if (v === '--format') a.format = process.argv[++i];
|
|
44
|
-
else if (v === '--strict') a.strict = true;
|
|
45
|
-
}
|
|
46
|
-
return a;
|
|
47
|
-
})();
|
|
48
|
-
|
|
49
|
-
function median(sorted) {
|
|
50
|
-
if (sorted.length === 0) return 0;
|
|
51
|
-
const n = sorted.length;
|
|
52
|
-
return n % 2 === 0 ? (sorted[n / 2 - 1] + sorted[n / 2]) / 2 : sorted[(n - 1) / 2];
|
|
53
|
-
}
|
|
54
|
-
|
|
55
|
-
function pctile(sorted, q) {
|
|
56
|
-
if (sorted.length === 0) return 0;
|
|
57
|
-
const idx = Math.floor(q * (sorted.length - 1));
|
|
58
|
-
return sorted[idx];
|
|
59
|
-
}
|
|
60
|
-
|
|
61
|
-
function mean(arr) {
|
|
62
|
-
if (arr.length === 0) return 0;
|
|
63
|
-
return arr.reduce((s, x) => s + x, 0) / arr.length;
|
|
64
|
-
}
|
|
65
|
-
|
|
66
|
-
function main() {
|
|
67
|
-
if (!existsSync(ARGS.input)) {
|
|
68
|
-
if (ARGS.format === 'json') {
|
|
69
|
-
console.log(JSON.stringify({
|
|
70
|
-
ok: true,
|
|
71
|
-
sufficient: false,
|
|
72
|
-
reason: 'input-file-not-found',
|
|
73
|
-
input: ARGS.input,
|
|
74
|
-
hint: 'Enable parallel logging (CLAUDE_FLOW_ROUTER_PARALLEL_LOG=1) and run a workload first.',
|
|
75
|
-
}, null, 2));
|
|
76
|
-
} else {
|
|
77
|
-
console.log(`# router-parallel-analyze`);
|
|
78
|
-
console.log('');
|
|
79
|
-
console.log(`_Input file not found: ${ARGS.input}_`);
|
|
80
|
-
console.log('');
|
|
81
|
-
console.log('Enable parallel logging by setting `CLAUDE_FLOW_ROUTER_PARALLEL_LOG=1`');
|
|
82
|
-
console.log('and running a real workload, then re-run this analyzer.');
|
|
83
|
-
}
|
|
84
|
-
process.exit(ARGS.strict ? 1 : 0);
|
|
85
|
-
}
|
|
86
|
-
|
|
87
|
-
let rows = [];
|
|
88
|
-
try {
|
|
89
|
-
rows = readFileSync(ARGS.input, 'utf-8')
|
|
90
|
-
.split('\n')
|
|
91
|
-
.filter((l) => l.trim())
|
|
92
|
-
.map((l) => JSON.parse(l));
|
|
93
|
-
} catch (e) {
|
|
94
|
-
console.error(`router-parallel-analyze: cannot parse ${ARGS.input}: ${e.message}`);
|
|
95
|
-
process.exit(2);
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
// Filter to records with both predictions + outcome present.
|
|
99
|
-
const usable = rows.filter((r) => r.bandit && r.ser && r.outcome);
|
|
100
|
-
|
|
101
|
-
if (usable.length < 30) {
|
|
102
|
-
const payload = {
|
|
103
|
-
ok: true,
|
|
104
|
-
sufficient: false,
|
|
105
|
-
reason: `n=${usable.length} < 30 (insufficient sample for AND-gate evaluation)`,
|
|
106
|
-
hint: 'Continue collecting parallel-routing data; statistical significance needs ≥30 paired decisions per arm.',
|
|
107
|
-
sampleSize: usable.length,
|
|
108
|
-
generatedAt: new Date().toISOString(),
|
|
109
|
-
};
|
|
110
|
-
if (ARGS.format === 'json') console.log(JSON.stringify(payload, null, 2));
|
|
111
|
-
else {
|
|
112
|
-
console.log(`# router-parallel-analyze`);
|
|
113
|
-
console.log('');
|
|
114
|
-
console.log(`Sample size: **${usable.length}** paired decisions`);
|
|
115
|
-
console.log('');
|
|
116
|
-
console.log(`_Insufficient — need ≥30 for the 3-criteria AND-gate._`);
|
|
117
|
-
}
|
|
118
|
-
process.exit(ARGS.strict ? 1 : 0);
|
|
119
|
-
}
|
|
120
|
-
|
|
121
|
-
// For each record, attribute the outcome to BOTH arms when they
|
|
122
|
-
// agreed on the model pick; when they disagreed, the actual outcome
|
|
123
|
-
// belongs only to the bandit (which was the executing arm). The SER
|
|
124
|
-
// arm gets its PREDICTED metrics for the counterfactual.
|
|
125
|
-
const banditOutcomes = usable.map((r) => ({
|
|
126
|
-
qualityActual: r.outcome.actualQuality,
|
|
127
|
-
usdActual: r.outcome.actualUsd,
|
|
128
|
-
latencyActual: r.outcome.actualLatencyMs,
|
|
129
|
-
}));
|
|
130
|
-
|
|
131
|
-
// SER counterfactual: use ser.predictedQuality / predictedCostUsd
|
|
132
|
-
// when picks disagreed; use actual when they agreed. Latency: when
|
|
133
|
-
// they agreed, both saw the same actual latency; when they
|
|
134
|
-
// disagreed, the SER pick's latency is unknown — fall back to the
|
|
135
|
-
// mean of agreed-decision latencies for that model.
|
|
136
|
-
const agreedLatencyByModel = {};
|
|
137
|
-
for (const r of usable) {
|
|
138
|
-
if (r.bandit.pick === r.ser.pick) {
|
|
139
|
-
const m = r.outcome.actualModel || r.bandit.pick;
|
|
140
|
-
(agreedLatencyByModel[m] = agreedLatencyByModel[m] || []).push(r.outcome.actualLatencyMs);
|
|
141
|
-
}
|
|
142
|
-
}
|
|
143
|
-
const modelMeanLatency = {};
|
|
144
|
-
for (const [m, arr] of Object.entries(agreedLatencyByModel)) {
|
|
145
|
-
modelMeanLatency[m] = mean(arr);
|
|
146
|
-
}
|
|
147
|
-
|
|
148
|
-
const serOutcomes = usable.map((r) => {
|
|
149
|
-
const agreed = r.bandit.pick === r.ser.pick;
|
|
150
|
-
return {
|
|
151
|
-
qualityActual: agreed ? r.outcome.actualQuality : r.ser.predictedQuality,
|
|
152
|
-
usdActual: agreed ? r.outcome.actualUsd : r.ser.predictedCostUsd,
|
|
153
|
-
latencyActual: agreed
|
|
154
|
-
? r.outcome.actualLatencyMs
|
|
155
|
-
: (modelMeanLatency[r.ser.pick] ?? r.outcome.actualLatencyMs),
|
|
156
|
-
};
|
|
157
|
-
});
|
|
158
|
-
|
|
159
|
-
// Compute the three criteria.
|
|
160
|
-
const banditQuality = mean(banditOutcomes.map((o) => o.qualityActual));
|
|
161
|
-
const serQuality = mean(serOutcomes.map((o) => o.qualityActual));
|
|
162
|
-
const qualityImprovementPct =
|
|
163
|
-
banditQuality > 0 ? ((serQuality - banditQuality) / banditQuality) * 100 : 0;
|
|
164
|
-
|
|
165
|
-
const banditUsd = mean(banditOutcomes.map((o) => o.usdActual));
|
|
166
|
-
const serUsd = mean(serOutcomes.map((o) => o.usdActual));
|
|
167
|
-
const usdIncreasePct = banditUsd > 0 ? ((serUsd - banditUsd) / banditUsd) * 100 : 0;
|
|
168
|
-
|
|
169
|
-
const banditLatencyP95 = pctile(banditOutcomes.map((o) => o.latencyActual).sort((a, b) => a - b), 0.95);
|
|
170
|
-
const serLatencyP95 = pctile(serOutcomes.map((o) => o.latencyActual).sort((a, b) => a - b), 0.95);
|
|
171
|
-
const latencyIncreasePct =
|
|
172
|
-
banditLatencyP95 > 0 ? ((serLatencyP95 - banditLatencyP95) / banditLatencyP95) * 100 : 0;
|
|
173
|
-
|
|
174
|
-
const passes = {
|
|
175
|
-
quality: qualityImprovementPct > 2,
|
|
176
|
-
cost: usdIncreasePct < 1,
|
|
177
|
-
latency: latencyIncreasePct < 5,
|
|
178
|
-
};
|
|
179
|
-
const allPass = passes.quality && passes.cost && passes.latency;
|
|
180
|
-
|
|
181
|
-
// Disagreement rate (informational; not part of the gate).
|
|
182
|
-
const disagreements = usable.filter((r) => r.bandit.pick !== r.ser.pick).length;
|
|
183
|
-
const disagreementPct = (disagreements / usable.length) * 100;
|
|
184
|
-
|
|
185
|
-
const payload = {
|
|
186
|
-
ok: true,
|
|
187
|
-
sufficient: true,
|
|
188
|
-
sampleSize: usable.length,
|
|
189
|
-
disagreement: {
|
|
190
|
-
count: disagreements,
|
|
191
|
-
pct: Math.round(disagreementPct * 100) / 100,
|
|
192
|
-
},
|
|
193
|
-
bandit: {
|
|
194
|
-
meanQuality: Math.round(banditQuality * 10000) / 10000,
|
|
195
|
-
meanUsd: Math.round(banditUsd * 1e6) / 1e6,
|
|
196
|
-
p95LatencyMs: Math.round(banditLatencyP95),
|
|
197
|
-
},
|
|
198
|
-
ser: {
|
|
199
|
-
meanQuality: Math.round(serQuality * 10000) / 10000,
|
|
200
|
-
meanUsd: Math.round(serUsd * 1e6) / 1e6,
|
|
201
|
-
p95LatencyMs: Math.round(serLatencyP95),
|
|
202
|
-
},
|
|
203
|
-
criteria: {
|
|
204
|
-
qualityImprovementPct: Math.round(qualityImprovementPct * 100) / 100,
|
|
205
|
-
qualityThresholdPct: 2,
|
|
206
|
-
qualityPasses: passes.quality,
|
|
207
|
-
usdIncreasePct: Math.round(usdIncreasePct * 100) / 100,
|
|
208
|
-
usdThresholdPct: 1,
|
|
209
|
-
costPasses: passes.cost,
|
|
210
|
-
latencyIncreasePct: Math.round(latencyIncreasePct * 100) / 100,
|
|
211
|
-
latencyThresholdPct: 5,
|
|
212
|
-
latencyPasses: passes.latency,
|
|
213
|
-
},
|
|
214
|
-
verdict: {
|
|
215
|
-
promotable: allPass,
|
|
216
|
-
reason: allPass
|
|
217
|
-
? 'All three criteria met (quality > 2% AND cost < 1% AND latency < 5%)'
|
|
218
|
-
: `Blocked by: ${[
|
|
219
|
-
!passes.quality && `quality only +${(qualityImprovementPct).toFixed(2)}% (need >2%)`,
|
|
220
|
-
!passes.cost && `cost +${(usdIncreasePct).toFixed(2)}% (need <1%)`,
|
|
221
|
-
!passes.latency && `latency +${(latencyIncreasePct).toFixed(2)}% (need <5%)`,
|
|
222
|
-
].filter(Boolean).join(', ')}`,
|
|
223
|
-
},
|
|
224
|
-
generatedAt: new Date().toISOString(),
|
|
225
|
-
};
|
|
226
|
-
|
|
227
|
-
if (ARGS.format === 'json') {
|
|
228
|
-
console.log(JSON.stringify(payload, null, 2));
|
|
229
|
-
} else {
|
|
230
|
-
console.log('# router-parallel-analyze (ADR-150 SelfEvolvingRouter promotion gate)');
|
|
231
|
-
console.log('');
|
|
232
|
-
console.log(`Sample size: **${payload.sampleSize}** paired decisions`);
|
|
233
|
-
console.log(`Disagreement rate: ${payload.disagreement.pct}% (${payload.disagreement.count} / ${payload.sampleSize})`);
|
|
234
|
-
console.log('');
|
|
235
|
-
console.log(`| Metric | Bandit | SER | Delta | Threshold | Passes |`);
|
|
236
|
-
console.log(`|---|---:|---:|---:|---:|:---:|`);
|
|
237
|
-
console.log(`| Mean quality | ${payload.bandit.meanQuality} | ${payload.ser.meanQuality} | ${payload.criteria.qualityImprovementPct >= 0 ? '+' : ''}${payload.criteria.qualityImprovementPct}% | >+2% | ${passes.quality ? '✓' : '⚠'} |`);
|
|
238
|
-
console.log(`| Mean $/decision | $${payload.bandit.meanUsd.toFixed(6)} | $${payload.ser.meanUsd.toFixed(6)} | ${payload.criteria.usdIncreasePct >= 0 ? '+' : ''}${payload.criteria.usdIncreasePct}% | <+1% | ${passes.cost ? '✓' : '⚠'} |`);
|
|
239
|
-
console.log(`| p95 latency | ${payload.bandit.p95LatencyMs}ms | ${payload.ser.p95LatencyMs}ms | ${payload.criteria.latencyIncreasePct >= 0 ? '+' : ''}${payload.criteria.latencyIncreasePct}% | <+5% | ${passes.latency ? '✓' : '⚠'} |`);
|
|
240
|
-
console.log('');
|
|
241
|
-
console.log(`**Verdict**: ${allPass ? '✓ PROMOTABLE' : '⚠ NOT promotable'}`);
|
|
242
|
-
console.log('');
|
|
243
|
-
console.log(`_${payload.verdict.reason}_`);
|
|
244
|
-
}
|
|
245
|
-
|
|
246
|
-
if (ARGS.strict && !allPass) process.exit(1);
|
|
247
|
-
process.exit(0);
|
|
248
|
-
}
|
|
249
|
-
|
|
250
|
-
main();
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// router-parallel-analyze.mjs — ADR-150 Phase 2 SelfEvolvingRouter
|
|
3
|
+
// promotion gate.
|
|
4
|
+
//
|
|
5
|
+
// Reads paired routing decisions (Thompson bandit pick + SelfEvolvingRouter
|
|
6
|
+
// pick, both with measured outcomes) from a JSONL trajectory file and
|
|
7
|
+
// computes the THREE PROMOTION CRITERIA from ADR-150 review-round-1:
|
|
8
|
+
//
|
|
9
|
+
// (a) qualityScore improvement > 2% (relative)
|
|
10
|
+
// (b) usdPerDecision increase < 1% (relative)
|
|
11
|
+
// (c) p95 routing-decision latency increase < 5% (relative)
|
|
12
|
+
//
|
|
13
|
+
// ALL THREE must hold for promotion. AND, not OR — the OR form let
|
|
14
|
+
// quality gains mask cost regressions, exactly the failure mode ADR-149's
|
|
15
|
+
// Pareto framing was built to prevent.
|
|
16
|
+
//
|
|
17
|
+
// INPUT FORMAT (.swarm/router-parallel.jsonl, one JSON record per line):
|
|
18
|
+
// {
|
|
19
|
+
// "ts": "<iso>",
|
|
20
|
+
// "task": { ... },
|
|
21
|
+
// "bandit": { "pick": "<modelId>", "predictedQuality": 0.84, "predictedCostUsd": 0.003 },
|
|
22
|
+
// "ser": { "pick": "<modelId>", "predictedQuality": 0.87, "predictedCostUsd": 0.004 },
|
|
23
|
+
// "outcome": { "actualModel": "<modelId>", "actualQuality": 0.91, "actualUsd": 0.0035, "actualLatencyMs": 1240 }
|
|
24
|
+
// }
|
|
25
|
+
//
|
|
26
|
+
// USAGE
|
|
27
|
+
// node scripts/router-parallel-analyze.mjs --input .swarm/router-parallel.jsonl
|
|
28
|
+
// node scripts/router-parallel-analyze.mjs --input <file> --format json
|
|
29
|
+
// node scripts/router-parallel-analyze.mjs --input <file> --strict # exit 1 if NOT promotable
|
|
30
|
+
//
|
|
31
|
+
// EXIT CODES
|
|
32
|
+
// 0 analysis complete; in --strict mode → all three criteria passed
|
|
33
|
+
// 1 --strict requested AND at least one criterion failed (NOT promotable)
|
|
34
|
+
// 2 config error or input file missing/malformed
|
|
35
|
+
|
|
36
|
+
import { readFileSync, existsSync } from 'node:fs';
|
|
37
|
+
|
|
38
|
+
const ARGS = (() => {
|
|
39
|
+
const a = { input: '.swarm/router-parallel.jsonl', format: 'table', strict: false };
|
|
40
|
+
for (let i = 2; i < process.argv.length; i++) {
|
|
41
|
+
const v = process.argv[i];
|
|
42
|
+
if (v === '--input') a.input = process.argv[++i];
|
|
43
|
+
else if (v === '--format') a.format = process.argv[++i];
|
|
44
|
+
else if (v === '--strict') a.strict = true;
|
|
45
|
+
}
|
|
46
|
+
return a;
|
|
47
|
+
})();
|
|
48
|
+
|
|
49
|
+
function median(sorted) {
|
|
50
|
+
if (sorted.length === 0) return 0;
|
|
51
|
+
const n = sorted.length;
|
|
52
|
+
return n % 2 === 0 ? (sorted[n / 2 - 1] + sorted[n / 2]) / 2 : sorted[(n - 1) / 2];
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function pctile(sorted, q) {
|
|
56
|
+
if (sorted.length === 0) return 0;
|
|
57
|
+
const idx = Math.floor(q * (sorted.length - 1));
|
|
58
|
+
return sorted[idx];
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function mean(arr) {
|
|
62
|
+
if (arr.length === 0) return 0;
|
|
63
|
+
return arr.reduce((s, x) => s + x, 0) / arr.length;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function main() {
|
|
67
|
+
if (!existsSync(ARGS.input)) {
|
|
68
|
+
if (ARGS.format === 'json') {
|
|
69
|
+
console.log(JSON.stringify({
|
|
70
|
+
ok: true,
|
|
71
|
+
sufficient: false,
|
|
72
|
+
reason: 'input-file-not-found',
|
|
73
|
+
input: ARGS.input,
|
|
74
|
+
hint: 'Enable parallel logging (CLAUDE_FLOW_ROUTER_PARALLEL_LOG=1) and run a workload first.',
|
|
75
|
+
}, null, 2));
|
|
76
|
+
} else {
|
|
77
|
+
console.log(`# router-parallel-analyze`);
|
|
78
|
+
console.log('');
|
|
79
|
+
console.log(`_Input file not found: ${ARGS.input}_`);
|
|
80
|
+
console.log('');
|
|
81
|
+
console.log('Enable parallel logging by setting `CLAUDE_FLOW_ROUTER_PARALLEL_LOG=1`');
|
|
82
|
+
console.log('and running a real workload, then re-run this analyzer.');
|
|
83
|
+
}
|
|
84
|
+
process.exit(ARGS.strict ? 1 : 0);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
let rows = [];
|
|
88
|
+
try {
|
|
89
|
+
rows = readFileSync(ARGS.input, 'utf-8')
|
|
90
|
+
.split('\n')
|
|
91
|
+
.filter((l) => l.trim())
|
|
92
|
+
.map((l) => JSON.parse(l));
|
|
93
|
+
} catch (e) {
|
|
94
|
+
console.error(`router-parallel-analyze: cannot parse ${ARGS.input}: ${e.message}`);
|
|
95
|
+
process.exit(2);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
// Filter to records with both predictions + outcome present.
|
|
99
|
+
const usable = rows.filter((r) => r.bandit && r.ser && r.outcome);
|
|
100
|
+
|
|
101
|
+
if (usable.length < 30) {
|
|
102
|
+
const payload = {
|
|
103
|
+
ok: true,
|
|
104
|
+
sufficient: false,
|
|
105
|
+
reason: `n=${usable.length} < 30 (insufficient sample for AND-gate evaluation)`,
|
|
106
|
+
hint: 'Continue collecting parallel-routing data; statistical significance needs ≥30 paired decisions per arm.',
|
|
107
|
+
sampleSize: usable.length,
|
|
108
|
+
generatedAt: new Date().toISOString(),
|
|
109
|
+
};
|
|
110
|
+
if (ARGS.format === 'json') console.log(JSON.stringify(payload, null, 2));
|
|
111
|
+
else {
|
|
112
|
+
console.log(`# router-parallel-analyze`);
|
|
113
|
+
console.log('');
|
|
114
|
+
console.log(`Sample size: **${usable.length}** paired decisions`);
|
|
115
|
+
console.log('');
|
|
116
|
+
console.log(`_Insufficient — need ≥30 for the 3-criteria AND-gate._`);
|
|
117
|
+
}
|
|
118
|
+
process.exit(ARGS.strict ? 1 : 0);
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
// For each record, attribute the outcome to BOTH arms when they
|
|
122
|
+
// agreed on the model pick; when they disagreed, the actual outcome
|
|
123
|
+
// belongs only to the bandit (which was the executing arm). The SER
|
|
124
|
+
// arm gets its PREDICTED metrics for the counterfactual.
|
|
125
|
+
const banditOutcomes = usable.map((r) => ({
|
|
126
|
+
qualityActual: r.outcome.actualQuality,
|
|
127
|
+
usdActual: r.outcome.actualUsd,
|
|
128
|
+
latencyActual: r.outcome.actualLatencyMs,
|
|
129
|
+
}));
|
|
130
|
+
|
|
131
|
+
// SER counterfactual: use ser.predictedQuality / predictedCostUsd
|
|
132
|
+
// when picks disagreed; use actual when they agreed. Latency: when
|
|
133
|
+
// they agreed, both saw the same actual latency; when they
|
|
134
|
+
// disagreed, the SER pick's latency is unknown — fall back to the
|
|
135
|
+
// mean of agreed-decision latencies for that model.
|
|
136
|
+
const agreedLatencyByModel = {};
|
|
137
|
+
for (const r of usable) {
|
|
138
|
+
if (r.bandit.pick === r.ser.pick) {
|
|
139
|
+
const m = r.outcome.actualModel || r.bandit.pick;
|
|
140
|
+
(agreedLatencyByModel[m] = agreedLatencyByModel[m] || []).push(r.outcome.actualLatencyMs);
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
const modelMeanLatency = {};
|
|
144
|
+
for (const [m, arr] of Object.entries(agreedLatencyByModel)) {
|
|
145
|
+
modelMeanLatency[m] = mean(arr);
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const serOutcomes = usable.map((r) => {
|
|
149
|
+
const agreed = r.bandit.pick === r.ser.pick;
|
|
150
|
+
return {
|
|
151
|
+
qualityActual: agreed ? r.outcome.actualQuality : r.ser.predictedQuality,
|
|
152
|
+
usdActual: agreed ? r.outcome.actualUsd : r.ser.predictedCostUsd,
|
|
153
|
+
latencyActual: agreed
|
|
154
|
+
? r.outcome.actualLatencyMs
|
|
155
|
+
: (modelMeanLatency[r.ser.pick] ?? r.outcome.actualLatencyMs),
|
|
156
|
+
};
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
// Compute the three criteria.
|
|
160
|
+
const banditQuality = mean(banditOutcomes.map((o) => o.qualityActual));
|
|
161
|
+
const serQuality = mean(serOutcomes.map((o) => o.qualityActual));
|
|
162
|
+
const qualityImprovementPct =
|
|
163
|
+
banditQuality > 0 ? ((serQuality - banditQuality) / banditQuality) * 100 : 0;
|
|
164
|
+
|
|
165
|
+
const banditUsd = mean(banditOutcomes.map((o) => o.usdActual));
|
|
166
|
+
const serUsd = mean(serOutcomes.map((o) => o.usdActual));
|
|
167
|
+
const usdIncreasePct = banditUsd > 0 ? ((serUsd - banditUsd) / banditUsd) * 100 : 0;
|
|
168
|
+
|
|
169
|
+
const banditLatencyP95 = pctile(banditOutcomes.map((o) => o.latencyActual).sort((a, b) => a - b), 0.95);
|
|
170
|
+
const serLatencyP95 = pctile(serOutcomes.map((o) => o.latencyActual).sort((a, b) => a - b), 0.95);
|
|
171
|
+
const latencyIncreasePct =
|
|
172
|
+
banditLatencyP95 > 0 ? ((serLatencyP95 - banditLatencyP95) / banditLatencyP95) * 100 : 0;
|
|
173
|
+
|
|
174
|
+
const passes = {
|
|
175
|
+
quality: qualityImprovementPct > 2,
|
|
176
|
+
cost: usdIncreasePct < 1,
|
|
177
|
+
latency: latencyIncreasePct < 5,
|
|
178
|
+
};
|
|
179
|
+
const allPass = passes.quality && passes.cost && passes.latency;
|
|
180
|
+
|
|
181
|
+
// Disagreement rate (informational; not part of the gate).
|
|
182
|
+
const disagreements = usable.filter((r) => r.bandit.pick !== r.ser.pick).length;
|
|
183
|
+
const disagreementPct = (disagreements / usable.length) * 100;
|
|
184
|
+
|
|
185
|
+
const payload = {
|
|
186
|
+
ok: true,
|
|
187
|
+
sufficient: true,
|
|
188
|
+
sampleSize: usable.length,
|
|
189
|
+
disagreement: {
|
|
190
|
+
count: disagreements,
|
|
191
|
+
pct: Math.round(disagreementPct * 100) / 100,
|
|
192
|
+
},
|
|
193
|
+
bandit: {
|
|
194
|
+
meanQuality: Math.round(banditQuality * 10000) / 10000,
|
|
195
|
+
meanUsd: Math.round(banditUsd * 1e6) / 1e6,
|
|
196
|
+
p95LatencyMs: Math.round(banditLatencyP95),
|
|
197
|
+
},
|
|
198
|
+
ser: {
|
|
199
|
+
meanQuality: Math.round(serQuality * 10000) / 10000,
|
|
200
|
+
meanUsd: Math.round(serUsd * 1e6) / 1e6,
|
|
201
|
+
p95LatencyMs: Math.round(serLatencyP95),
|
|
202
|
+
},
|
|
203
|
+
criteria: {
|
|
204
|
+
qualityImprovementPct: Math.round(qualityImprovementPct * 100) / 100,
|
|
205
|
+
qualityThresholdPct: 2,
|
|
206
|
+
qualityPasses: passes.quality,
|
|
207
|
+
usdIncreasePct: Math.round(usdIncreasePct * 100) / 100,
|
|
208
|
+
usdThresholdPct: 1,
|
|
209
|
+
costPasses: passes.cost,
|
|
210
|
+
latencyIncreasePct: Math.round(latencyIncreasePct * 100) / 100,
|
|
211
|
+
latencyThresholdPct: 5,
|
|
212
|
+
latencyPasses: passes.latency,
|
|
213
|
+
},
|
|
214
|
+
verdict: {
|
|
215
|
+
promotable: allPass,
|
|
216
|
+
reason: allPass
|
|
217
|
+
? 'All three criteria met (quality > 2% AND cost < 1% AND latency < 5%)'
|
|
218
|
+
: `Blocked by: ${[
|
|
219
|
+
!passes.quality && `quality only +${(qualityImprovementPct).toFixed(2)}% (need >2%)`,
|
|
220
|
+
!passes.cost && `cost +${(usdIncreasePct).toFixed(2)}% (need <1%)`,
|
|
221
|
+
!passes.latency && `latency +${(latencyIncreasePct).toFixed(2)}% (need <5%)`,
|
|
222
|
+
].filter(Boolean).join(', ')}`,
|
|
223
|
+
},
|
|
224
|
+
generatedAt: new Date().toISOString(),
|
|
225
|
+
};
|
|
226
|
+
|
|
227
|
+
if (ARGS.format === 'json') {
|
|
228
|
+
console.log(JSON.stringify(payload, null, 2));
|
|
229
|
+
} else {
|
|
230
|
+
console.log('# router-parallel-analyze (ADR-150 SelfEvolvingRouter promotion gate)');
|
|
231
|
+
console.log('');
|
|
232
|
+
console.log(`Sample size: **${payload.sampleSize}** paired decisions`);
|
|
233
|
+
console.log(`Disagreement rate: ${payload.disagreement.pct}% (${payload.disagreement.count} / ${payload.sampleSize})`);
|
|
234
|
+
console.log('');
|
|
235
|
+
console.log(`| Metric | Bandit | SER | Delta | Threshold | Passes |`);
|
|
236
|
+
console.log(`|---|---:|---:|---:|---:|:---:|`);
|
|
237
|
+
console.log(`| Mean quality | ${payload.bandit.meanQuality} | ${payload.ser.meanQuality} | ${payload.criteria.qualityImprovementPct >= 0 ? '+' : ''}${payload.criteria.qualityImprovementPct}% | >+2% | ${passes.quality ? '✓' : '⚠'} |`);
|
|
238
|
+
console.log(`| Mean $/decision | $${payload.bandit.meanUsd.toFixed(6)} | $${payload.ser.meanUsd.toFixed(6)} | ${payload.criteria.usdIncreasePct >= 0 ? '+' : ''}${payload.criteria.usdIncreasePct}% | <+1% | ${passes.cost ? '✓' : '⚠'} |`);
|
|
239
|
+
console.log(`| p95 latency | ${payload.bandit.p95LatencyMs}ms | ${payload.ser.p95LatencyMs}ms | ${payload.criteria.latencyIncreasePct >= 0 ? '+' : ''}${payload.criteria.latencyIncreasePct}% | <+5% | ${passes.latency ? '✓' : '⚠'} |`);
|
|
240
|
+
console.log('');
|
|
241
|
+
console.log(`**Verdict**: ${allPass ? '✓ PROMOTABLE' : '⚠ NOT promotable'}`);
|
|
242
|
+
console.log('');
|
|
243
|
+
console.log(`_${payload.verdict.reason}_`);
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
if (ARGS.strict && !allPass) process.exit(1);
|
|
247
|
+
process.exit(0);
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
main();
|