ruvnet-brain 4.5.3 → 4.5.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/install.mjs +148 -31
- package/config/model-router/catalog.template.json +126 -52
- package/config/model-router/policy.default.mjs +94 -75
- package/config/model-router/qualification-contract.json +124 -0
- package/config/model-router/routing-eval-cases.json +275 -0
- package/config/model-router/routing-policy.template.json +76 -0
- package/config/model-router/weekly-analyst-instruction.md +60 -0
- package/data/model-catalog.json +44 -49
- package/package.json +3 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/hooks/codex-hooks.json +40 -3
- package/plugin/hooks/hook-contracts.json +218 -19
- package/plugin/hooks/hooks.json +51 -2
- package/plugin/scripts/agentdb-recall.mjs +101 -30
- package/plugin/scripts/codex-hook-adapter.mjs +18 -9
- package/plugin/scripts/continuity-hook-policy.mjs +9 -0
- package/plugin/scripts/continuity-journal.mjs +33 -33
- package/plugin/scripts/ground-ruvnet.sh +5 -5
- package/plugin/scripts/hook-shim.mjs +4 -30
- package/plugin/scripts/project-capture-queue.mjs +333 -0
- package/plugin/scripts/project-progression-contract.mjs +1 -1
- package/plugin/scripts/project-progression-hook.mjs +3 -3
- package/plugin/scripts/project-progression-producer.mjs +53 -28
- package/plugin/scripts/project-progression-session-start.mjs +44 -4
- package/plugin/scripts/project-progression-store.mjs +14 -0
- package/plugin/scripts/project-transition-hook.mjs +204 -0
- package/plugin/scripts/session-snapshot-hook.mjs +44 -272
- package/plugin/scripts/session-start-budget.mjs +2 -2
- package/plugin/scripts/turn-outcome-capture.mjs +125 -47
- package/plugin/scripts/turn-transport-journal.mjs +106 -0
- package/scripts/codex-hook-trust-reconcile.mjs +247 -0
- package/scripts/codex-routed.sh +3 -36
- package/scripts/goldie-weekly.sh +8 -64
- package/scripts/metaharness-router.mjs +7 -1
- package/scripts/model-analyst-sandbox.mjs +54 -0
- package/scripts/model-currency-evidence.mjs +139 -0
- package/scripts/model-currency.mjs +230 -0
- package/scripts/model-native-catalog.mjs +111 -0
- package/scripts/model-native-qualification.mjs +251 -0
- package/scripts/model-router-agent-hook.mjs +136 -0
- package/scripts/model-router-dispatch.mjs +161 -0
- package/scripts/model-router-engine.mjs +155 -104
- package/scripts/model-routing-eval.mjs +108 -0
- package/scripts/model-routing-gateway.mjs +420 -0
- package/scripts/model-routing-launchers.mjs +174 -0
- package/scripts/model-routing-policy-promotion.mjs +203 -0
- package/scripts/model-weekly-analyst.mjs +299 -0
- package/scripts/model-weekly-assessment.mjs +91 -0
- package/scripts/model-weekly-cycle.mjs +183 -0
- package/scripts/model-weekly-qualification.mjs +362 -0
- package/scripts/native-subscription-usage.mjs +57 -0
- package/scripts/release-qualification-contract.mjs +54 -0
- package/scripts/security-guidance-codex-compat.mjs +142 -0
- package/scripts/user-model-prompt-hook.mjs +69 -0
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
{
|
|
2
|
+
"schemaVersion": 1,
|
|
3
|
+
"reviewedAt": "2026-10-04T13:32:24.704091Z",
|
|
4
|
+
"maxAgeMs": 604800000,
|
|
5
|
+
"routes": {
|
|
6
|
+
"codex": {
|
|
7
|
+
"fast": {
|
|
8
|
+
"model": "gpt-6-luna",
|
|
9
|
+
"effort": "low"
|
|
10
|
+
},
|
|
11
|
+
"medium": {
|
|
12
|
+
"model": "gpt-6.1-sol",
|
|
13
|
+
"effort": "medium"
|
|
14
|
+
},
|
|
15
|
+
"hard": {
|
|
16
|
+
"model": "gpt-6-astra",
|
|
17
|
+
"effort": "high"
|
|
18
|
+
},
|
|
19
|
+
"substantial": {
|
|
20
|
+
"model": "gpt-6.1-sol",
|
|
21
|
+
"effort": "high"
|
|
22
|
+
},
|
|
23
|
+
"exceptional": {
|
|
24
|
+
"model": "gpt-6-astra",
|
|
25
|
+
"effort": "xhigh",
|
|
26
|
+
"requiresNamedReason": true
|
|
27
|
+
}
|
|
28
|
+
},
|
|
29
|
+
"claude-code": {
|
|
30
|
+
"fast": {
|
|
31
|
+
"model": "claude-sonnet-5-5",
|
|
32
|
+
"effort": "low"
|
|
33
|
+
},
|
|
34
|
+
"medium": {
|
|
35
|
+
"model": "claude-sonnet-5-5",
|
|
36
|
+
"effort": "medium"
|
|
37
|
+
},
|
|
38
|
+
"hard": {
|
|
39
|
+
"model": "claude-opus-5-5",
|
|
40
|
+
"effort": "high"
|
|
41
|
+
},
|
|
42
|
+
"codingEffort": "high"
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
"sources": [
|
|
46
|
+
{
|
|
47
|
+
"url": "https://artificialanalysis.ai/models/releases/comparisons/gpt-6-1-sol-vs-claude-sonnet-5-5",
|
|
48
|
+
"benchmark": "Intelligence Index v4.3.2; Terminal-Bench 4.0",
|
|
49
|
+
"checkedAt": "2026-10-04T13:32:24.704295Z"
|
|
50
|
+
},
|
|
51
|
+
{
|
|
52
|
+
"url": "https://artificialanalysis.ai/models/releases/comparisons/gpt-6-luna-vs-gpt-6-astra",
|
|
53
|
+
"benchmark": "Intelligence Index v4.3.2; Terminal-Bench 4.0",
|
|
54
|
+
"checkedAt": "2026-10-04T13:32:24.704301Z"
|
|
55
|
+
}
|
|
56
|
+
],
|
|
57
|
+
"qualification": {
|
|
58
|
+
"basis": "live provider metadata, five native subscription smoke launches, independent effort comparisons, user allocation constraint",
|
|
59
|
+
"limits": [
|
|
60
|
+
"Not project-specific optimality proof",
|
|
61
|
+
"Luna fast limited to noncoding",
|
|
62
|
+
"No automatic newmodel entitlement claim",
|
|
63
|
+
"Parent conversation model remains host-controlled",
|
|
64
|
+
"Substantial/exceptional allocation reflects user correctness preference, not newly measured optimality",
|
|
65
|
+
"Free-text task classes are heuristic; caller taskFacts can express uncertainty and scope",
|
|
66
|
+
"No assumed completion-speed or subscription-quota multipliers",
|
|
67
|
+
"Native agent hook updatedInput rewrite has not passed host acceptance proof"
|
|
68
|
+
]
|
|
69
|
+
},
|
|
70
|
+
"policyRevisionAt": "2026-10-04T13:50:05.851549+00:00",
|
|
71
|
+
"objectivePriority": [
|
|
72
|
+
"correctness",
|
|
73
|
+
"subscription-allowance",
|
|
74
|
+
"completion-time"
|
|
75
|
+
]
|
|
76
|
+
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Updated: 2026-10-04 14:16:00 EDT | Version 1.0.4
|
|
2
|
+
Created: 2026-10-04 09:56:00 EDT
|
|
3
|
+
|
|
4
|
+
# Weekly model-routing analyst mandate
|
|
5
|
+
|
|
6
|
+
Act as Stuart's model-routing analyst for software development. Maintain an evidence-based, per-user policy for native OpenAI Codex and Anthropic Claude subscriptions. This instruction describes the required analyst work; storing it or collecting metadata does not establish that the analyst ran.
|
|
7
|
+
|
|
8
|
+
## When this assessment runs
|
|
9
|
+
|
|
10
|
+
Stuart's October 4 clarification governs the schedule: check weekly for newly released OpenAI or Anthropic text-capable models. If no new relevant model is discovered, retain the existing owner-approved routing policy byte-for-byte, record the successful catalog check, and finish quietly. Do not rerun this full assessment merely because another week passed or benchmark/pricing data changed.
|
|
11
|
+
|
|
12
|
+
The first successful catalog check establishes the release baseline while retaining the policy Stuart already approved. That is a baseline receipt, not proof that a semantic assessment ran. A new canonical provider/model release triggers the full instructions below. Failed discovery is unknown, never "no change." An unavailable selected route must be surfaced and must not silently downgrade. Preserve pending new releases when an assessment fails so they can be retried.
|
|
13
|
+
|
|
14
|
+
Allow up to 15 minutes for a triggered assessment, aiming to finish within five minutes. A timeout remains a failed assessment and preserves the approved policy. Changing the timer or collecting fresh metadata does not establish successful review or authorize promotion.
|
|
15
|
+
|
|
16
|
+
## Objective and authority
|
|
17
|
+
|
|
18
|
+
Prioritize correctness, completeness and sound architectural judgment; then efficient included subscription allowance use; then time to a verified result including planning, handoffs, implementation, repairs and review. Treat the providers' allowances separately. Never infer included usage from API prices, credit rates, message counts or token counts. Do not weaken capability when the task needs stronger reasoning. Surface capacity constraints and defer optional work instead.
|
|
19
|
+
|
|
20
|
+
Use native subscription authentication. Never enable paid API fallback, extra credits, or a new subscription. Never introduce API billing, purchase credits, upgrade plans or enable overages. Before a Codex analyst launch verify ordinary included usage is available; block when exhausted or unavailable. Authentication and this check do not reserve allowance or guarantee existing credits cannot be consumed after concurrent usage exhausts it. Disclose that limit without claiming a hard spending cap. Do not change existing billing controls. Comparative inference needs standing authorization and an explicit allowance budget. The routine evidence collector must not launch unbudgeted experiments.
|
|
21
|
+
|
|
22
|
+
## Baseline and research
|
|
23
|
+
|
|
24
|
+
On the first run inspect installed tools, supported configuration, exact models and efforts actually available through the subscriptions, accessible usage/reset information, existing routing rules, user overrides and evaluation history. Never expose credentials. On subsequent runs compare with the prior report, refresh changeable facts and retain valid historical evidence. Report inaccessible dashboards or providers without guessing entitlement or allowance.
|
|
25
|
+
|
|
26
|
+
Distinguish public announcements, API availability, native selectable subscription models and independently observed execution. Check official OpenAI and Anthropic sources for exact identifiers, releases, retirements, client effort support, speed modes, usage rules, coding/review capabilities and context handling.
|
|
27
|
+
|
|
28
|
+
Check independent primary evaluations relevant to architecture, repository understanding, implementation, debugging, long-running coding and review. Start with Artificial Analysis model AND coding-agent evaluations (https://artificialanalysis.ai/), VulcanBench reports and methodology (https://vulcanbench.com/), Terminal-Bench (https://www.tbench.ai/) and SWE-bench (https://www.swebench.com/). Include other relevant reproducible independent evaluations. Trace aggregators to the original evaluator. Trace repeated claims to original experiments; repetitions are not independent evidence. Personal reports are supplementary. Record source URL, evaluation date, benchmark version, exact model and effort, harness, task count, success rate, uncertainty where available, runtime, token use and reported cost basis. Compare efforts within each model; investigate whether more effort reduces total work, retries, tokens or time. Low/medium effort is not automatically more economical. Report conflicting evidence and its workloads. Do not combine incompatible evaluations into a ranking. General intelligence scores do not prove architecture/review superiority; maximum-effort results do not prove medium-effort behavior. Cite direct sources and separate measurements, vendor claims and recommendations.
|
|
29
|
+
|
|
30
|
+
## Evaluate dispatch independently of worker quality
|
|
31
|
+
|
|
32
|
+
The entry model must route reliably and enforce its handoff. Do not select a cheap dispatcher without evidence. Maintain representative routing cases: obvious mechanical tasks, deceptively short hard tasks, ambiguous requirements, architectural decisions, security-sensitive changes, migrations, cross-system bugs and difficult reviews. Measure dangerous under-routing, unnecessary escalation, handoff failures, routing latency and whole-workflow usage. Confidence claims from the dispatcher are insufficient.
|
|
33
|
+
|
|
34
|
+
Prefer deterministic dispatch for clear rules. Uncertain classifications go upward or receive a stronger assessment before implementation. Preserve the original request and relevant evidence; a weak summary must not be the worker's only context. Verify actual execution model and effort. Requested arguments, messages and configuration edits are not proof. State whether each supported route changes the parent, starts a child, or launches another native client.
|
|
35
|
+
|
|
36
|
+
## Recommendations and escalation
|
|
37
|
+
|
|
38
|
+
Produce OpenAI-only, Anthropic-only and supported combined policies. Every route needs exact identifier, supported effort or documented equivalent, speed mode, entry conditions, escalation triggers and confidence. Do not invent equivalent effort semantics across providers.
|
|
39
|
+
|
|
40
|
+
Cover entry/dispatch, mechanical work, routine settled-design implementation, substantial development, consequential architecture/scope/ambiguity and cross-system reasoning, difficult debugging, hardest implementation, substantive final review and exceptional escalation. Include exact model, effort, speed mode, fallback, escalation and separately chosen reviewer for each role. The reviewer checks requirements, implementation, tests and unresolved risks, and requires repairs; approval cannot replace execution evidence.
|
|
41
|
+
|
|
42
|
+
Reassess this current OpenAI hypothesis: Astra high for consequential planning and review; Sol 6.1 high for demanding implementation under a clear design; Sol 6.1 medium for routine implementation; Astra retains implementation when essential judgment remains tightly coupled; Luna low only for explicit mechanical transformations with complete cheap verification. Treat Luna low dispatch, Sol 6.1 high implementation, Sol xhigh difficult implementation and Astra high architecture/review as hypotheses, not permanent rules. Apply equally independent reasoning to Anthropic. Verify the identifiers/settings before recommending replacements. The strongest suitable model handles consequential judgment; do not require failures on weaker models first.
|
|
43
|
+
|
|
44
|
+
Reassess when scope, assumptions, boundaries or verification change. Missing information requires evidence; tooling/environment failure requires repair; difficult implementation reasoning may warrant more effort or capability; architectural uncertainty warrants strongest suitable architectural reasoning promptly. Avoid rigid retry ladders. Final review cannot guarantee recovery from bad early assumptions. Require appropriate executable checks and runtime evidence. Cross-provider review needs expected benefit; different providers do not guarantee independent errors.
|
|
45
|
+
|
|
46
|
+
## Evaluation and application
|
|
47
|
+
|
|
48
|
+
Save allowance through clear scope, relevant context, preserved decisions, fewer redundant investigations, suitable checks and fewer repairs. Measure the whole accepted-task path rather than decode speed. Standard delivery is preferred unless acceleration's documented benefit justifies its allowance use. Use existing local telemetry first. Account for dispatcher work, context transfer, all children, retries, review and repairs. Record actual telemetry where available; mark attribution uncertain under concurrent activity, resets, rounding or other sessions. Keep separate provider budgets and any shared/model-specific limits. Verify usage multipliers each week. Never promise zero quality loss or guaranteed savings.
|
|
49
|
+
|
|
50
|
+
Newer models are not automatically better. Use credible evaluations and task outcomes first. Uncertain options stay experimental. Local comparisons need explicit success criteria, a budget and an independent quality judge. Evaluate correctness, review findings, repairs, completion time and attributable allowance where available.
|
|
51
|
+
|
|
52
|
+
Research and update proposals automatically. Maintain a last-known-good policy and dated candidate. Require a demonstrated quality floor for every role, availability, routing, actual handoff and quality checks. Avoid exhaustive evaluations that consume a large fraction of allowance. Automatically promote only evidence-qualified, supported changes within the router's existing authorized update mechanism and standing authorization; preserve recovery. Preserve user overrides, validate settings, retain previous versions and distinguish applied from recommended. If evidence is insufficient, retain the established route with an uncertainty label. Never mark metadata collection as completed semantic review or reset a policy review date merely because HTTP fetches succeeded.
|
|
53
|
+
|
|
54
|
+
## Weekly output and triggering
|
|
55
|
+
|
|
56
|
+
Save an initial full report and concise dated change reports, sources and a versioned policy proposal in the authorized per-user output directory. Include changes; original third-party charts with dates/links; clearly labelled recreated model-and-effort charts with separate quality-versus-cost and quality-versus-completion-time views; API cost versus measured subscription usage; recommended VS Code entry model and reasons; OpenAI, Anthropic and combined routing diagrams/tables; exact identifiers/efforts/speed/fallback; escalation/review rules; confidence, evidence gaps and policy changes. Include a machine-readable candidate compatible with the existing router. Mark unmeasured configurations missing, never invent scores. Preserve source snapshots and policy reasoning. Preserve the prior policy for comparison and recovery.
|
|
57
|
+
|
|
58
|
+
Keep unchanged findings quiet. Notify for actionable improvement, retirement, availability change, regression, access failure or a required decision. A schedule needs actual run receipts. Active-session weekly catch-up is not a guarantee of execution while clients are closed. Never claim a recommendation is implemented, a model accessible or every prompt enforced without checking the actual supported runtime path.
|
|
59
|
+
|
|
60
|
+
The native release check also refreshes account-visible model metadata without inference. For newly discovered models only, semantic analysis and independent qualification share one fifteen-minute deadline. A source-reviewed fixed role suite compares the incumbent and candidate, and a separate approved hard reviewer grades anonymized outputs. Automatic application requires standing user authorization, actual native configured-turn evidence, passing quality checks and an unchanged prior policy. Requested settings are not backend identity proof. Incomplete qualification keeps its bound proposal for retry without repeating the completed analysis; terminal rejection retains the approved route.
|
package/data/model-catalog.json
CHANGED
|
@@ -1,20 +1,20 @@
|
|
|
1
1
|
{
|
|
2
2
|
"_meta": {
|
|
3
|
-
"purpose": "
|
|
3
|
+
"purpose": "Verified provider tier ladders and task-fit recommendations. Frontier is the hard-work target; it is not the routine default. API prices are comparison metadata, not subscription charges.",
|
|
4
4
|
"generated": "2026-07-15",
|
|
5
5
|
"schema_version": 1,
|
|
6
6
|
"sources": {
|
|
7
|
-
"prices": "OpenRouter /api/v1/models live catalog, pulled 2026-
|
|
8
|
-
"rankings": "
|
|
9
|
-
"provenance_rule": "rUv ADR-206: vendor-reported scores are optimistic and harness-confounded
|
|
10
|
-
"benchmark_lag": "The canonical hard coding benchmarks (SWE-bench Verified standardized harness, LiveCodeBench, Aider polyglot) were ALL months stale on 2026-07-15 and carry NONE of these models. The '88.6% / 95% SWE-bench' figures in the press are vendor self-scaffold scores, not the standardized harness
|
|
7
|
+
"prices": "OpenRouter /api/v1/models live catalog, pulled 2026-10-04 (in/out USD per Mtok).",
|
|
8
|
+
"rankings": "Anthropic/OpenAI official role guidance checked 2026-10-04 for refreshed entries. Historical ranks for other providers unchanged and not revalidated in this two-provider refresh.",
|
|
9
|
+
"provenance_rule": "rUv ADR-206: vendor-reported scores are optimistic and harness-confounded \u2014 trust independent (AA/Arena) numbers, treat vendor self-scaffold SWE-bench/LiveCodeBench figures as noisy features, never as truth.",
|
|
10
|
+
"benchmark_lag": "The canonical hard coding benchmarks (SWE-bench Verified standardized harness, LiveCodeBench, Aider polyglot) were ALL months stale on 2026-07-15 and carry NONE of these models. The '88.6% / 95% SWE-bench' figures in the press are vendor self-scaffold scores, not the standardized harness \u2014 excluded here.",
|
|
11
11
|
"release_refs": "OpenAI GPT-5.6 GA 2026-07-09 (openai.com/index/gpt-5-6); Anthropic Fable 5 2026-06-09 (anthropic.com/news/claude-fable-5-mythos-5); Google Gemini 3.1 Pro 2026-02-19; xAI/SpaceXAI Grok 4.5 2026-07-08."
|
|
12
12
|
},
|
|
13
|
-
"caveat": "
|
|
14
|
-
"effort_note": "
|
|
13
|
+
"caveat": "Refreshed Anthropic/OpenAI tiers use official provider guidance and owner preferences, not newly measured independent rankings. Other provider claims remain historical and were not reassessed in this refresh.",
|
|
14
|
+
"effort_note": "Start ordinary work at low/medium, hard work at high; xhigh/max only when representative evals justify latency and cost. Per-model supports and API/host defaults differ; see dated October 4 refresh."
|
|
15
15
|
},
|
|
16
16
|
"default_provider": "anthropic",
|
|
17
|
-
"default_provider_reason": "This is a Claude Code plugin, so DEVELOPMENT genuinely runs on Anthropic
|
|
17
|
+
"default_provider_reason": "This is a Claude Code plugin, so DEVELOPMENT genuinely runs on Anthropic \u2014 that is a detected fact, not an arbitrary house preference. Set your PRODUCTION house in the console (or $RUVNET_PROVIDER) if your app runs on a different provider; it is never assumed silently.",
|
|
18
18
|
"providers": {
|
|
19
19
|
"anthropic": {
|
|
20
20
|
"label": "Claude (Anthropic)",
|
|
@@ -24,29 +24,27 @@
|
|
|
24
24
|
"CLAUDECODE"
|
|
25
25
|
],
|
|
26
26
|
"frontier": {
|
|
27
|
-
"model": "claude-
|
|
28
|
-
"in":
|
|
29
|
-
"out":
|
|
30
|
-
"
|
|
31
|
-
"
|
|
32
|
-
"source": "independent (AA + Arena)"
|
|
27
|
+
"model": "anthropic/claude-opus-5.5",
|
|
28
|
+
"in": 4,
|
|
29
|
+
"out": 20,
|
|
30
|
+
"rank": "Hard work; Fable 5.1 is a bounded exceptional escalation, not the ordinary baseline",
|
|
31
|
+
"source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
|
|
33
32
|
},
|
|
34
33
|
"mid": {
|
|
35
|
-
"model": "claude-sonnet-5",
|
|
34
|
+
"model": "anthropic/claude-sonnet-5.5",
|
|
36
35
|
"in": 2,
|
|
37
36
|
"out": 10,
|
|
38
|
-
"
|
|
39
|
-
"
|
|
40
|
-
"source": "independent (AA)"
|
|
37
|
+
"rank": "Routine work at medium effort",
|
|
38
|
+
"source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
|
|
41
39
|
},
|
|
42
40
|
"cheap": {
|
|
43
|
-
"model": "claude-
|
|
44
|
-
"in":
|
|
45
|
-
"out":
|
|
46
|
-
"
|
|
47
|
-
"
|
|
48
|
-
|
|
49
|
-
|
|
41
|
+
"model": "anthropic/claude-sonnet-5.5",
|
|
42
|
+
"in": 2,
|
|
43
|
+
"out": 10,
|
|
44
|
+
"rank": "Fast work at low effort; Haiku excluded by owner preference, not claimed discontinued",
|
|
45
|
+
"source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
|
|
46
|
+
},
|
|
47
|
+
"note": "API/OpenRouter IDs differ from native Claude IDs: claude-opus-5-5 and claude-sonnet-5-5. Same Sonnet model serves two effort roles. No measured latency/quality guarantee."
|
|
50
48
|
},
|
|
51
49
|
"openai": {
|
|
52
50
|
"label": "ChatGPT (OpenAI)",
|
|
@@ -54,35 +52,32 @@
|
|
|
54
52
|
"OPENAI_API_KEY"
|
|
55
53
|
],
|
|
56
54
|
"frontier": {
|
|
57
|
-
"model": "openai/gpt-
|
|
58
|
-
"in":
|
|
59
|
-
"out":
|
|
60
|
-
"
|
|
61
|
-
"
|
|
62
|
-
"source": "independent (AA)"
|
|
55
|
+
"model": "openai/gpt-6-astra",
|
|
56
|
+
"in": 10,
|
|
57
|
+
"out": 50,
|
|
58
|
+
"rank": "Bounded difficult reasoning and critical independent review",
|
|
59
|
+
"source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
|
|
63
60
|
},
|
|
64
61
|
"mid": {
|
|
65
|
-
"model": "openai/gpt-
|
|
62
|
+
"model": "openai/gpt-6.1-sol",
|
|
66
63
|
"in": 2,
|
|
67
|
-
"out":
|
|
68
|
-
"
|
|
69
|
-
"
|
|
70
|
-
"source": "independent (AA)"
|
|
64
|
+
"out": 10,
|
|
65
|
+
"rank": "Default ordinary coding, research and debugging",
|
|
66
|
+
"source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
|
|
71
67
|
},
|
|
72
68
|
"cheap": {
|
|
73
|
-
"model": "openai/gpt-
|
|
74
|
-
"in": 0.
|
|
75
|
-
"out":
|
|
76
|
-
"
|
|
77
|
-
"
|
|
78
|
-
"source": "independent (AA)"
|
|
69
|
+
"model": "openai/gpt-6-luna",
|
|
70
|
+
"in": 0.1,
|
|
71
|
+
"out": 0.5,
|
|
72
|
+
"rank": "Focused high-volume tasks and small well-specified coding changes",
|
|
73
|
+
"source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
|
|
79
74
|
}
|
|
80
75
|
},
|
|
81
76
|
"codex": {
|
|
82
77
|
"label": "Codex (OpenAI)",
|
|
83
78
|
"detect_env": [],
|
|
84
79
|
"aliasOf": "openai",
|
|
85
|
-
"note": "
|
|
80
|
+
"note": "Alias of current OpenAI tier ladder; native availability checked through fresh Codex account catalog. Updating metadata does not switch active conversations."
|
|
86
81
|
},
|
|
87
82
|
"google": {
|
|
88
83
|
"label": "Gemini (Google)",
|
|
@@ -95,15 +90,15 @@
|
|
|
95
90
|
"in": 2,
|
|
96
91
|
"out": 12,
|
|
97
92
|
"released": "2026-02-19",
|
|
98
|
-
"rank": "AA Intelligence 46
|
|
99
|
-
"source": "independent (AA)
|
|
93
|
+
"rank": "AA Intelligence 46 \u2014 BELOW the Anthropic/OpenAI/xAI frontier cluster (54\u201360); still preview-named. Honest: Gemini is not frontier-competitive on independent indices right now.",
|
|
94
|
+
"source": "independent (AA) \u2014 vendor's GPQA-D 94.3% / ARC-AGI-2 77.1% are Google's own, uncorroborated"
|
|
100
95
|
},
|
|
101
96
|
"mid": {
|
|
102
97
|
"model": "google/gemini-3.5-flash",
|
|
103
98
|
"in": 1.5,
|
|
104
99
|
"out": 9,
|
|
105
100
|
"released": "2026",
|
|
106
|
-
"rank": "AA 50
|
|
101
|
+
"rank": "AA 50 \u2014 out-scores Google's own 3.1 Pro on this index (real anomaly, not a typo)",
|
|
107
102
|
"source": "independent (AA)"
|
|
108
103
|
},
|
|
109
104
|
"cheap": {
|
|
@@ -121,13 +116,13 @@
|
|
|
121
116
|
"XAI_API_KEY",
|
|
122
117
|
"GROK_API_KEY"
|
|
123
118
|
],
|
|
124
|
-
"note": "xAI's live lineup has no distinct budget SKU below Grok 4.3 as of 2026-07-15 (the 'grok-4.1-fast' I first wrote does not exist in the live catalog
|
|
119
|
+
"note": "xAI's live lineup has no distinct budget SKU below Grok 4.3 as of 2026-07-15 (the 'grok-4.1-fast' I first wrote does not exist in the live catalog \u2014 the verify gate caught it). A Grok shop's cheap tasks route to Grok 4.3 or a cross-provider value pick.",
|
|
125
120
|
"frontier": {
|
|
126
121
|
"model": "x-ai/grok-4.5",
|
|
127
122
|
"in": 2,
|
|
128
123
|
"out": 6,
|
|
129
124
|
"released": "2026-07-08",
|
|
130
|
-
"rank": "AA Intelligence 54 (#8) at ~1/3 Opus 4.7's blended price
|
|
125
|
+
"rank": "AA Intelligence 54 (#8) at ~1/3 Opus 4.7's blended price \u2014 the frontier VALUE standout; 'Opus-class, faster, cheaper' (Musk)",
|
|
131
126
|
"source": "independent (AA)"
|
|
132
127
|
},
|
|
133
128
|
"mid": {
|
|
@@ -135,7 +130,7 @@
|
|
|
135
130
|
"in": 1.25,
|
|
136
131
|
"out": 2.5,
|
|
137
132
|
"released": "2026-04-30",
|
|
138
|
-
"rank": "AA 38
|
|
133
|
+
"rank": "AA 38 \u2014 also xAI's cheapest verified stable tier",
|
|
139
134
|
"source": "independent (AA)"
|
|
140
135
|
}
|
|
141
136
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ruvnet-brain",
|
|
3
|
-
"version": "4.5.
|
|
3
|
+
"version": "4.5.5",
|
|
4
4
|
"description": "One-command installer for RuvNet Brain \u2014 a portable, source-grounded brain over rUv's RuvNet building blocks, delivered as a Claude Code plugin so Claude uses the stack instead of fighting it.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -40,6 +40,8 @@
|
|
|
40
40
|
"test:all": "npm run test:unit && npm run test:mesh && npm run test:mutation && npm run test:regression && npm run test:integration && npm test",
|
|
41
41
|
"metaharness:receipts": "node scripts/metaharness-receipts.mjs",
|
|
42
42
|
"route:cheap": "node scripts/route-cheap.mjs",
|
|
43
|
+
"routing:eval": "node scripts/model-routing-eval.mjs",
|
|
44
|
+
"hooks:security-guidance:compat": "node scripts/security-guidance-codex-compat.mjs",
|
|
43
45
|
"route:receipt": "node scripts/dispatch-receipt.mjs",
|
|
44
46
|
"trismart": "node scripts/trismart.mjs",
|
|
45
47
|
"metaharness:fix": "node scripts/fix-metaharness-memretrieve.mjs --apply",
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ruvnet-brain",
|
|
3
3
|
"description": "RuvNet brain transplant for Claude Code — grounds every RuvNet decision in real source across 77 rUv repositories, prefers Ruflo / RuVector-RVF / AgentDB over training-prior defaults (pgvector, Pinecone, hand-rolled cosine), and can pull in any RuvNet repo on demand. Ships a UserPromptSubmit retrieve-and-inject grounding hook and a PreToolUse write gate that refuses ungrounded rUv-product code until search_ruvnet has been consulted (ADR-0012 / ADR-067).",
|
|
4
|
-
"version": "4.5.
|
|
4
|
+
"version": "4.5.5",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Stuart Kerr"
|
|
7
7
|
},
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"description": "RuvNet Brain continuity plane for Codex. Broad legacy automatic gates remain retired. Registrations here are MEASURED, not assumed: a probe hook was registered on all twelve event names the installed codex-cli 0.154.0 binary declares and a real `codex exec` run was observed (2026-09-11). SessionStart, UserPromptSubmit and SessionEnd FIRED and carry handlers. Stop keeps the pre-existing, project-scoped continuation gate; session-snapshot was added at Stop on 2026-09-29 to record each turn's outcome (turn-outcome-capture.mjs)
|
|
2
|
+
"description": "RuvNet Brain continuity plane for Codex. Broad legacy automatic gates remain retired. Registrations here are MEASURED, not assumed: a probe hook was registered on all twelve event names the installed codex-cli 0.154.0 binary declares and a real `codex exec` run was observed (2026-09-11). SessionStart, UserPromptSubmit and SessionEnd FIRED and carry handlers. Stop keeps the pre-existing, project-scoped continuation gate; session-snapshot was added at Stop on 2026-09-29 to record each turn's outcome (turn-outcome-capture.mjs) \u2014 Codex SessionEnd carries no last_assistant_message, so Stop is the only boundary that can. Its delivery rests on codex-cli 0.158.0's declared stop.command.input schema and the other Codex Stop handlers; it has NOT been live-observed and fails open. PreCompact is DECLARED ABSENT here: the binary declares the event but the probe never observed it, and a capture registered on an unobserved event would look symmetrical while capturing nothing. All handlers are bounded and fail open. Amended 2026-09-11: UserPromptSubmit also runs ground-ruvnet (grounding injection, ADR-040 \u00a7Amendment). Amended 2026-09-12: the 2026-09-11 PreToolUse/PostToolUse \"not observed\" note was measured with a prompt (`codex exec \"reply OK\"`) that never invoked a tool, so neither event had anything to fire on \u2014 that was an untested path, not a failing one. Re-probed with prompts that actually call a tool: a real apply_patch write fired PreToolUse/PostToolUse with tool_name \"apply_patch\", and a real MCP call to this repo's own search_ruvnet server fired both with tool_name \"mcp__ruvnet_brain__search_ruvnet\". Both are now registered below: decision-gate's write route (matcher includes apply_patch, Codex's raw write-tool name) and grounding-stamp (unchanged matcher already recognizes the mcp__..__search_ruvnet shape). The bash route (exec_command) remains unregistered \u2014 today's measurement covered a write and an MCP call, not exec_command, and extending on that evidence would be the same unproven leap this note replaces. Also added 2026-09-12: grounding-turn-mark (UserPromptSubmit) and grounding-turn-gate (Stop), the \"answered without searching\" pair \u2014 a prompt-level grounding directive is advisory, so this records whether it fired and forces continuation at Stop if no search_ruvnet call was recorded since (reusing grounding-stamp's own stamp evidence). Both are registered on Codex identically to Claude: the Stop-block contract (hookSpecificOutput.additionalContext) already has a proven Codex translation via codex-hook-adapter.mjs's Stop branch (see tests/unit/codex-lifecycle-hooks.test.mjs), the same path continuation-gate already uses. Capacity-aware parallel-work guidance is also registered at UserPromptSubmit: it advises the coordinator to launch only real independent workers within sampled resource headroom and the live runtime/tool cap; it never spawns workers or claims execution. Automatic normalized project observations are captured at prompt, pre-tool, post-tool and child completion; Claude also observes PostToolUseFailure. Pending and degraded memory are disclosed; this does not establish synchronous ADR-073 conformance or native Grok recall. SessionStart has an 8-second measured replay/restore envelope.",
|
|
3
3
|
"hooks": {
|
|
4
4
|
"SessionStart": [
|
|
5
5
|
{
|
|
@@ -7,8 +7,8 @@
|
|
|
7
7
|
"hooks": [
|
|
8
8
|
{
|
|
9
9
|
"type": "command",
|
|
10
|
-
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\"
|
|
11
|
-
"timeout":
|
|
10
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 7500 session-start",
|
|
11
|
+
"timeout": 8
|
|
12
12
|
}
|
|
13
13
|
]
|
|
14
14
|
}
|
|
@@ -36,6 +36,16 @@
|
|
|
36
36
|
}
|
|
37
37
|
],
|
|
38
38
|
"PreToolUse": [
|
|
39
|
+
{
|
|
40
|
+
"matcher": "*",
|
|
41
|
+
"hooks": [
|
|
42
|
+
{
|
|
43
|
+
"type": "command",
|
|
44
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 9000 session-snapshot PreToolUse",
|
|
45
|
+
"timeout": 10
|
|
46
|
+
}
|
|
47
|
+
]
|
|
48
|
+
},
|
|
39
49
|
{
|
|
40
50
|
"matcher": "^(Write|Edit|MultiEdit|NotebookEdit|apply_patch)$",
|
|
41
51
|
"hooks": [
|
|
@@ -48,6 +58,16 @@
|
|
|
48
58
|
}
|
|
49
59
|
],
|
|
50
60
|
"PostToolUse": [
|
|
61
|
+
{
|
|
62
|
+
"matcher": "*",
|
|
63
|
+
"hooks": [
|
|
64
|
+
{
|
|
65
|
+
"type": "command",
|
|
66
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 9000 session-snapshot PostToolUse",
|
|
67
|
+
"timeout": 10
|
|
68
|
+
}
|
|
69
|
+
]
|
|
70
|
+
},
|
|
51
71
|
{
|
|
52
72
|
"matcher": "^(?:.*__)?search_ruvnet$",
|
|
53
73
|
"hooks": [
|
|
@@ -63,6 +83,11 @@
|
|
|
63
83
|
{
|
|
64
84
|
"matcher": "*",
|
|
65
85
|
"hooks": [
|
|
86
|
+
{
|
|
87
|
+
"type": "command",
|
|
88
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 9000 session-snapshot UserPromptSubmit",
|
|
89
|
+
"timeout": 10
|
|
90
|
+
},
|
|
66
91
|
{
|
|
67
92
|
"type": "command",
|
|
68
93
|
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 9000 unprompted-speech UserPromptSubmit",
|
|
@@ -97,6 +122,18 @@
|
|
|
97
122
|
}
|
|
98
123
|
]
|
|
99
124
|
}
|
|
125
|
+
],
|
|
126
|
+
"SubagentStop": [
|
|
127
|
+
{
|
|
128
|
+
"matcher": "*",
|
|
129
|
+
"hooks": [
|
|
130
|
+
{
|
|
131
|
+
"type": "command",
|
|
132
|
+
"command": "node -e \"const f=require('node:fs'),o=require('node:os'),p=require('node:path'),c=require('node:child_process'),d=process.env.CODEX_HOME||p.join(o.homedir(),'.codex'),b=process.env.RUVNET_BRAIN_HOME||p.join(p.dirname(d),'.cache','ruvnet-brain'),w=p.join(b,'codex-hook.mjs');let s;try{s=f.statSync(w)}catch{}if(!s?.isFile())process.exit(0);const r=c.spawnSync(process.execPath,[w,...process.argv.slice(2)],{stdio:['inherit','pipe','pipe'],encoding:'utf8',env:process.env,timeout:Number(process.argv[1]),killSignal:'SIGKILL'});if(r.status===0||r.status===2){if(r.stdout)process.stdout.write(r.stdout);if(r.stderr)process.stderr.write(r.stderr)}process.exit(r.status===2?2:0)\" 9000 session-snapshot SubagentStop",
|
|
133
|
+
"timeout": 10
|
|
134
|
+
}
|
|
135
|
+
]
|
|
136
|
+
}
|
|
100
137
|
]
|
|
101
138
|
}
|
|
102
139
|
}
|