ruvnet-brain 4.5.4 → 4.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (37) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +144 -27
  3. package/config/model-router/catalog.template.json +126 -52
  4. package/config/model-router/policy.default.mjs +94 -75
  5. package/config/model-router/qualification-contract.json +124 -0
  6. package/config/model-router/routing-eval-cases.json +275 -0
  7. package/config/model-router/routing-policy.template.json +76 -0
  8. package/config/model-router/weekly-analyst-instruction.md +60 -0
  9. package/data/model-catalog.json +44 -49
  10. package/package.json +3 -1
  11. package/plugin/.claude-plugin/plugin.json +1 -1
  12. package/plugin/.codex-plugin/plugin.json +1 -1
  13. package/plugin/scripts/codex-hook-adapter.mjs +18 -9
  14. package/scripts/codex-hook-trust-reconcile.mjs +247 -0
  15. package/scripts/codex-routed.sh +3 -36
  16. package/scripts/goldie-weekly.sh +8 -64
  17. package/scripts/metaharness-router.mjs +7 -1
  18. package/scripts/model-analyst-sandbox.mjs +54 -0
  19. package/scripts/model-currency-evidence.mjs +139 -0
  20. package/scripts/model-currency.mjs +230 -0
  21. package/scripts/model-native-catalog.mjs +111 -0
  22. package/scripts/model-native-qualification.mjs +251 -0
  23. package/scripts/model-router-agent-hook.mjs +136 -0
  24. package/scripts/model-router-dispatch.mjs +161 -0
  25. package/scripts/model-router-engine.mjs +155 -104
  26. package/scripts/model-routing-eval.mjs +108 -0
  27. package/scripts/model-routing-gateway.mjs +420 -0
  28. package/scripts/model-routing-launchers.mjs +174 -0
  29. package/scripts/model-routing-policy-promotion.mjs +203 -0
  30. package/scripts/model-weekly-analyst.mjs +299 -0
  31. package/scripts/model-weekly-assessment.mjs +91 -0
  32. package/scripts/model-weekly-cycle.mjs +183 -0
  33. package/scripts/model-weekly-qualification.mjs +362 -0
  34. package/scripts/native-subscription-usage.mjs +57 -0
  35. package/scripts/release-qualification-contract.mjs +41 -0
  36. package/scripts/security-guidance-codex-compat.mjs +142 -0
  37. package/scripts/user-model-prompt-hook.mjs +69 -0
@@ -0,0 +1,76 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "reviewedAt": "2026-10-04T13:32:24.704091Z",
4
+ "maxAgeMs": 604800000,
5
+ "routes": {
6
+ "codex": {
7
+ "fast": {
8
+ "model": "gpt-6-luna",
9
+ "effort": "low"
10
+ },
11
+ "medium": {
12
+ "model": "gpt-6.1-sol",
13
+ "effort": "medium"
14
+ },
15
+ "hard": {
16
+ "model": "gpt-6-astra",
17
+ "effort": "high"
18
+ },
19
+ "substantial": {
20
+ "model": "gpt-6.1-sol",
21
+ "effort": "high"
22
+ },
23
+ "exceptional": {
24
+ "model": "gpt-6-astra",
25
+ "effort": "xhigh",
26
+ "requiresNamedReason": true
27
+ }
28
+ },
29
+ "claude-code": {
30
+ "fast": {
31
+ "model": "claude-sonnet-5-5",
32
+ "effort": "low"
33
+ },
34
+ "medium": {
35
+ "model": "claude-sonnet-5-5",
36
+ "effort": "medium"
37
+ },
38
+ "hard": {
39
+ "model": "claude-opus-5-5",
40
+ "effort": "high"
41
+ },
42
+ "codingEffort": "high"
43
+ }
44
+ },
45
+ "sources": [
46
+ {
47
+ "url": "https://artificialanalysis.ai/models/releases/comparisons/gpt-6-1-sol-vs-claude-sonnet-5-5",
48
+ "benchmark": "Intelligence Index v4.3.2; Terminal-Bench 4.0",
49
+ "checkedAt": "2026-10-04T13:32:24.704295Z"
50
+ },
51
+ {
52
+ "url": "https://artificialanalysis.ai/models/releases/comparisons/gpt-6-luna-vs-gpt-6-astra",
53
+ "benchmark": "Intelligence Index v4.3.2; Terminal-Bench 4.0",
54
+ "checkedAt": "2026-10-04T13:32:24.704301Z"
55
+ }
56
+ ],
57
+ "qualification": {
58
+ "basis": "live provider metadata, five native subscription smoke launches, independent effort comparisons, user allocation constraint",
59
+ "limits": [
60
+ "Not project-specific optimality proof",
61
+ "Luna fast limited to noncoding",
62
+ "No automatic newmodel entitlement claim",
63
+ "Parent conversation model remains host-controlled",
64
+ "Substantial/exceptional allocation reflects user correctness preference, not newly measured optimality",
65
+ "Free-text task classes are heuristic; caller taskFacts can express uncertainty and scope",
66
+ "No assumed completion-speed or subscription-quota multipliers",
67
+ "Native agent hook updatedInput rewrite has not passed host acceptance proof"
68
+ ]
69
+ },
70
+ "policyRevisionAt": "2026-10-04T13:50:05.851549+00:00",
71
+ "objectivePriority": [
72
+ "correctness",
73
+ "subscription-allowance",
74
+ "completion-time"
75
+ ]
76
+ }
@@ -0,0 +1,60 @@
1
+ Updated: 2026-10-04 14:16:00 EDT | Version 1.0.4
2
+ Created: 2026-10-04 09:56:00 EDT
3
+
4
+ # Weekly model-routing analyst mandate
5
+
6
+ Act as Stuart's model-routing analyst for software development. Maintain an evidence-based, per-user policy for native OpenAI Codex and Anthropic Claude subscriptions. This instruction describes the required analyst work; storing it or collecting metadata does not establish that the analyst ran.
7
+
8
+ ## When this assessment runs
9
+
10
+ Stuart's October 4 clarification governs the schedule: check weekly for newly released OpenAI or Anthropic text-capable models. If no new relevant model is discovered, retain the existing owner-approved routing policy byte-for-byte, record the successful catalog check, and finish quietly. Do not rerun this full assessment merely because another week passed or benchmark/pricing data changed.
11
+
12
+ The first successful catalog check establishes the release baseline while retaining the policy Stuart already approved. That is a baseline receipt, not proof that a semantic assessment ran. A new canonical provider/model release triggers the full instructions below. Failed discovery is unknown, never "no change." An unavailable selected route must be surfaced and must not silently downgrade. Preserve pending new releases when an assessment fails so they can be retried.
13
+
14
+ Allow up to 15 minutes for a triggered assessment, aiming to finish within five minutes. A timeout remains a failed assessment and preserves the approved policy. Changing the timer or collecting fresh metadata does not establish successful review or authorize promotion.
15
+
16
+ ## Objective and authority
17
+
18
+ Prioritize correctness, completeness and sound architectural judgment; then efficient included subscription allowance use; then time to a verified result including planning, handoffs, implementation, repairs and review. Treat the providers' allowances separately. Never infer included usage from API prices, credit rates, message counts or token counts. Do not weaken capability when the task needs stronger reasoning. Surface capacity constraints and defer optional work instead.
19
+
20
+ Use native subscription authentication. Never enable paid API fallback, extra credits, or a new subscription. Never introduce API billing, purchase credits, upgrade plans or enable overages. Before a Codex analyst launch verify ordinary included usage is available; block when exhausted or unavailable. Authentication and this check do not reserve allowance or guarantee existing credits cannot be consumed after concurrent usage exhausts it. Disclose that limit without claiming a hard spending cap. Do not change existing billing controls. Comparative inference needs standing authorization and an explicit allowance budget. The routine evidence collector must not launch unbudgeted experiments.
21
+
22
+ ## Baseline and research
23
+
24
+ On the first run inspect installed tools, supported configuration, exact models and efforts actually available through the subscriptions, accessible usage/reset information, existing routing rules, user overrides and evaluation history. Never expose credentials. On subsequent runs compare with the prior report, refresh changeable facts and retain valid historical evidence. Report inaccessible dashboards or providers without guessing entitlement or allowance.
25
+
26
+ Distinguish public announcements, API availability, native selectable subscription models and independently observed execution. Check official OpenAI and Anthropic sources for exact identifiers, releases, retirements, client effort support, speed modes, usage rules, coding/review capabilities and context handling.
27
+
28
+ Check independent primary evaluations relevant to architecture, repository understanding, implementation, debugging, long-running coding and review. Start with Artificial Analysis model AND coding-agent evaluations (https://artificialanalysis.ai/), VulcanBench reports and methodology (https://vulcanbench.com/), Terminal-Bench (https://www.tbench.ai/) and SWE-bench (https://www.swebench.com/). Include other relevant reproducible independent evaluations. Trace aggregators to the original evaluator. Trace repeated claims to original experiments; repetitions are not independent evidence. Personal reports are supplementary. Record source URL, evaluation date, benchmark version, exact model and effort, harness, task count, success rate, uncertainty where available, runtime, token use and reported cost basis. Compare efforts within each model; investigate whether more effort reduces total work, retries, tokens or time. Low/medium effort is not automatically more economical. Report conflicting evidence and its workloads. Do not combine incompatible evaluations into a ranking. General intelligence scores do not prove architecture/review superiority; maximum-effort results do not prove medium-effort behavior. Cite direct sources and separate measurements, vendor claims and recommendations.
29
+
30
+ ## Evaluate dispatch independently of worker quality
31
+
32
+ The entry model must route reliably and enforce its handoff. Do not select a cheap dispatcher without evidence. Maintain representative routing cases: obvious mechanical tasks, deceptively short hard tasks, ambiguous requirements, architectural decisions, security-sensitive changes, migrations, cross-system bugs and difficult reviews. Measure dangerous under-routing, unnecessary escalation, handoff failures, routing latency and whole-workflow usage. Confidence claims from the dispatcher are insufficient.
33
+
34
+ Prefer deterministic dispatch for clear rules. Uncertain classifications go upward or receive a stronger assessment before implementation. Preserve the original request and relevant evidence; a weak summary must not be the worker's only context. Verify actual execution model and effort. Requested arguments, messages and configuration edits are not proof. State whether each supported route changes the parent, starts a child, or launches another native client.
35
+
36
+ ## Recommendations and escalation
37
+
38
+ Produce OpenAI-only, Anthropic-only and supported combined policies. Every route needs exact identifier, supported effort or documented equivalent, speed mode, entry conditions, escalation triggers and confidence. Do not invent equivalent effort semantics across providers.
39
+
40
+ Cover entry/dispatch, mechanical work, routine settled-design implementation, substantial development, consequential architecture/scope/ambiguity and cross-system reasoning, difficult debugging, hardest implementation, substantive final review and exceptional escalation. Include exact model, effort, speed mode, fallback, escalation and separately chosen reviewer for each role. The reviewer checks requirements, implementation, tests and unresolved risks, and requires repairs; approval cannot replace execution evidence.
41
+
42
+ Reassess this current OpenAI hypothesis: Astra high for consequential planning and review; Sol 6.1 high for demanding implementation under a clear design; Sol 6.1 medium for routine implementation; Astra retains implementation when essential judgment remains tightly coupled; Luna low only for explicit mechanical transformations with complete cheap verification. Treat Luna low dispatch, Sol 6.1 high implementation, Sol xhigh difficult implementation and Astra high architecture/review as hypotheses, not permanent rules. Apply equally independent reasoning to Anthropic. Verify the identifiers/settings before recommending replacements. The strongest suitable model handles consequential judgment; do not require failures on weaker models first.
43
+
44
+ Reassess when scope, assumptions, boundaries or verification change. Missing information requires evidence; tooling/environment failure requires repair; difficult implementation reasoning may warrant more effort or capability; architectural uncertainty warrants strongest suitable architectural reasoning promptly. Avoid rigid retry ladders. Final review cannot guarantee recovery from bad early assumptions. Require appropriate executable checks and runtime evidence. Cross-provider review needs expected benefit; different providers do not guarantee independent errors.
45
+
46
+ ## Evaluation and application
47
+
48
+ Save allowance through clear scope, relevant context, preserved decisions, fewer redundant investigations, suitable checks and fewer repairs. Measure the whole accepted-task path rather than decode speed. Standard delivery is preferred unless acceleration's documented benefit justifies its allowance use. Use existing local telemetry first. Account for dispatcher work, context transfer, all children, retries, review and repairs. Record actual telemetry where available; mark attribution uncertain under concurrent activity, resets, rounding or other sessions. Keep separate provider budgets and any shared/model-specific limits. Verify usage multipliers each week. Never promise zero quality loss or guaranteed savings.
49
+
50
+ Newer models are not automatically better. Use credible evaluations and task outcomes first. Uncertain options stay experimental. Local comparisons need explicit success criteria, a budget and an independent quality judge. Evaluate correctness, review findings, repairs, completion time and attributable allowance where available.
51
+
52
+ Research and update proposals automatically. Maintain a last-known-good policy and dated candidate. Require a demonstrated quality floor for every role, availability, routing, actual handoff and quality checks. Avoid exhaustive evaluations that consume a large fraction of allowance. Automatically promote only evidence-qualified, supported changes within the router's existing authorized update mechanism and standing authorization; preserve recovery. Preserve user overrides, validate settings, retain previous versions and distinguish applied from recommended. If evidence is insufficient, retain the established route with an uncertainty label. Never mark metadata collection as completed semantic review or reset a policy review date merely because HTTP fetches succeeded.
53
+
54
+ ## Weekly output and triggering
55
+
56
+ Save an initial full report and concise dated change reports, sources and a versioned policy proposal in the authorized per-user output directory. Include changes; original third-party charts with dates/links; clearly labelled recreated model-and-effort charts with separate quality-versus-cost and quality-versus-completion-time views; API cost versus measured subscription usage; recommended VS Code entry model and reasons; OpenAI, Anthropic and combined routing diagrams/tables; exact identifiers/efforts/speed/fallback; escalation/review rules; confidence, evidence gaps and policy changes. Include a machine-readable candidate compatible with the existing router. Mark unmeasured configurations missing, never invent scores. Preserve source snapshots and policy reasoning. Preserve the prior policy for comparison and recovery.
57
+
58
+ Keep unchanged findings quiet. Notify for actionable improvement, retirement, availability change, regression, access failure or a required decision. A schedule needs actual run receipts. Active-session weekly catch-up is not a guarantee of execution while clients are closed. Never claim a recommendation is implemented, a model accessible or every prompt enforced without checking the actual supported runtime path.
59
+
60
+ The native release check also refreshes account-visible model metadata without inference. For newly discovered models only, semantic analysis and independent qualification share one fifteen-minute deadline. A source-reviewed fixed role suite compares the incumbent and candidate, and a separate approved hard reviewer grades anonymized outputs. Automatic application requires standing user authorization, actual native configured-turn evidence, passing quality checks and an unchanged prior policy. Requested settings are not backend identity proof. Incomplete qualification keeps its bound proposal for retry without repeating the completed analysis; terminal rejection retains the approved route.
@@ -1,20 +1,20 @@
1
1
  {
2
2
  "_meta": {
3
- "purpose": "Per-provider (house) tier ladders. The FRONTIER — the escalation target AND the savings baseline — is personalized to the user's own house: a Claude shop's frontier is Fable 5, a ChatGPT shop's is GPT-5.6 Sol, a Codex shop's is Sol, a Gemini shop's is Gemini 3.1 Pro, a Grok shop's is Grok 4.5. Modeled on ruflo ADR-148 (assets/model-router/openrouter-alts.json), extended with a provider-house axis rUv's registry does not carry.",
3
+ "purpose": "Verified provider tier ladders and task-fit recommendations. Frontier is the hard-work target; it is not the routine default. API prices are comparison metadata, not subscription charges.",
4
4
  "generated": "2026-07-15",
5
5
  "schema_version": 1,
6
6
  "sources": {
7
- "prices": "OpenRouter /api/v1/models live catalog, pulled 2026-09-19 (in/out USD per Mtok).",
8
- "rankings": "Artificial Analysis Intelligence Index (artificialanalysis.ai) + Arena/LMArena (arena.ai) — the ONLY independent evaluators carrying current-generation models as of 2026-07-15; each figure cross-verified twice.",
9
- "provenance_rule": "rUv ADR-206: vendor-reported scores are optimistic and harness-confounded — trust independent (AA/Arena) numbers, treat vendor self-scaffold SWE-bench/LiveCodeBench figures as noisy features, never as truth.",
10
- "benchmark_lag": "The canonical hard coding benchmarks (SWE-bench Verified standardized harness, LiveCodeBench, Aider polyglot) were ALL months stale on 2026-07-15 and carry NONE of these models. The '88.6% / 95% SWE-bench' figures in the press are vendor self-scaffold scores, not the standardized harness — excluded here.",
7
+ "prices": "OpenRouter /api/v1/models live catalog, pulled 2026-10-04 (in/out USD per Mtok).",
8
+ "rankings": "Anthropic/OpenAI official role guidance checked 2026-10-04 for refreshed entries. Historical ranks for other providers unchanged and not revalidated in this two-provider refresh.",
9
+ "provenance_rule": "rUv ADR-206: vendor-reported scores are optimistic and harness-confounded \u2014 trust independent (AA/Arena) numbers, treat vendor self-scaffold SWE-bench/LiveCodeBench figures as noisy features, never as truth.",
10
+ "benchmark_lag": "The canonical hard coding benchmarks (SWE-bench Verified standardized harness, LiveCodeBench, Aider polyglot) were ALL months stale on 2026-07-15 and carry NONE of these models. The '88.6% / 95% SWE-bench' figures in the press are vendor self-scaffold scores, not the standardized harness \u2014 excluded here.",
11
11
  "release_refs": "OpenAI GPT-5.6 GA 2026-07-09 (openai.com/index/gpt-5-6); Anthropic Fable 5 2026-06-09 (anthropic.com/news/claude-fable-5-mythos-5); Google Gemini 3.1 Pro 2026-02-19; xAI/SpaceXAI Grok 4.5 2026-07-08."
12
12
  },
13
- "caveat": "Prices are live-verified; tier placements are sensible, independent-benchmark-grounded starters, NOT this user's measured results. Override per-installation via $RUVNET_MODEL_CATALOG or the console. The automated always-current path is rUv's ADR-206 (BenchPress predictor + hourly OpenRouter watcher) — see scripts/refresh-model-catalog.mjs for the live re-pull.",
14
- "effort_note": "Frontier effort defaults to 'xhigh' (the max reasoning a shop reaches for on the hardest tasks); this is a principled default, not a per-effort measurement."
13
+ "caveat": "Refreshed Anthropic/OpenAI tiers use official provider guidance and owner preferences, not newly measured independent rankings. Other provider claims remain historical and were not reassessed in this refresh.",
14
+ "effort_note": "Start ordinary work at low/medium, hard work at high; xhigh/max only when representative evals justify latency and cost. Per-model supports and API/host defaults differ; see dated October 4 refresh."
15
15
  },
16
16
  "default_provider": "anthropic",
17
- "default_provider_reason": "This is a Claude Code plugin, so DEVELOPMENT genuinely runs on Anthropic — that is a detected fact, not an arbitrary house preference. Set your PRODUCTION house in the console (or $RUVNET_PROVIDER) if your app runs on a different provider; it is never assumed silently.",
17
+ "default_provider_reason": "This is a Claude Code plugin, so DEVELOPMENT genuinely runs on Anthropic \u2014 that is a detected fact, not an arbitrary house preference. Set your PRODUCTION house in the console (or $RUVNET_PROVIDER) if your app runs on a different provider; it is never assumed silently.",
18
18
  "providers": {
19
19
  "anthropic": {
20
20
  "label": "Claude (Anthropic)",
@@ -24,29 +24,27 @@
24
24
  "CLAUDECODE"
25
25
  ],
26
26
  "frontier": {
27
- "model": "claude-fable-5",
28
- "in": 10,
29
- "out": 50,
30
- "released": "2026-06-09",
31
- "rank": "AA Intelligence #1 (60) · Arena #1 (1508 Elo)",
32
- "source": "independent (AA + Arena)"
27
+ "model": "anthropic/claude-opus-5.5",
28
+ "in": 4,
29
+ "out": 20,
30
+ "rank": "Hard work; Fable 5.1 is a bounded exceptional escalation, not the ordinary baseline",
31
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
33
32
  },
34
33
  "mid": {
35
- "model": "claude-sonnet-5",
34
+ "model": "anthropic/claude-sonnet-5.5",
36
35
  "in": 2,
37
36
  "out": 10,
38
- "released": "2026-06-30",
39
- "rank": "AA 53; intro price $2/$10 → $3/$15 after 2026-08-31",
40
- "source": "independent (AA)"
37
+ "rank": "Routine work at medium effort",
38
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
41
39
  },
42
40
  "cheap": {
43
- "model": "claude-haiku-4.5",
44
- "in": 1,
45
- "out": 5,
46
- "released": "2025-10-15",
47
- "rank": "cheap tier (AA index low; cost-cascade prefers a cross-provider value pick when an OpenRouter key is present)",
48
- "source": "catalog"
49
- }
41
+ "model": "anthropic/claude-sonnet-5.5",
42
+ "in": 2,
43
+ "out": 10,
44
+ "rank": "Fast work at low effort; Haiku excluded by owner preference, not claimed discontinued",
45
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
46
+ },
47
+ "note": "API/OpenRouter IDs differ from native Claude IDs: claude-opus-5-5 and claude-sonnet-5-5. Same Sonnet model serves two effort roles. No measured latency/quality guarantee."
50
48
  },
51
49
  "openai": {
52
50
  "label": "ChatGPT (OpenAI)",
@@ -54,35 +52,32 @@
54
52
  "OPENAI_API_KEY"
55
53
  ],
56
54
  "frontier": {
57
- "model": "openai/gpt-5.6-sol",
58
- "in": 2,
59
- "out": 10,
60
- "released": "2026-07-09",
61
- "rank": "AA Intelligence #2 (59) · AA Coding Index leader (80) · Terminal-Bench 2.1 SOTA · ~1/3 Fable 5's cost/task",
62
- "source": "independent (AA)"
55
+ "model": "openai/gpt-6-astra",
56
+ "in": 10,
57
+ "out": 50,
58
+ "rank": "Bounded difficult reasoning and critical independent review",
59
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
63
60
  },
64
61
  "mid": {
65
- "model": "openai/gpt-5.6-terra",
62
+ "model": "openai/gpt-6.1-sol",
66
63
  "in": 2,
67
- "out": 12,
68
- "released": "2026-07-09",
69
- "rank": "AA Coding 77.4 — beats prev-gen flagship GPT-5.5 (76.4) at lower live catalog pricing",
70
- "source": "independent (AA)"
64
+ "out": 10,
65
+ "rank": "Default ordinary coding, research and debugging",
66
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
71
67
  },
72
68
  "cheap": {
73
- "model": "openai/gpt-5.6-luna",
74
- "in": 0.2,
75
- "out": 1.2,
76
- "released": "2026-07-09",
77
- "rank": "AA 51 · Coding 74.6 — beats most last-gen mid-tier",
78
- "source": "independent (AA)"
69
+ "model": "openai/gpt-6-luna",
70
+ "in": 0.1,
71
+ "out": 0.5,
72
+ "rank": "Focused high-volume tasks and small well-specified coding changes",
73
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
79
74
  }
80
75
  },
81
76
  "codex": {
82
77
  "label": "Codex (OpenAI)",
83
78
  "detect_env": [],
84
79
  "aliasOf": "openai",
85
- "note": "Codex folded into the GPT-5.6 Sol/Terra/Luna tiers on 2026-07-09 — there is NO separate '-codex' SKU this generation; gpt-5.3-codex (2026-02-05) was the last dedicated one and is superseded. A Codex shop's frontier IS Sol."
80
+ "note": "Alias of current OpenAI tier ladder; native availability checked through fresh Codex account catalog. Updating metadata does not switch active conversations."
86
81
  },
87
82
  "google": {
88
83
  "label": "Gemini (Google)",
@@ -95,15 +90,15 @@
95
90
  "in": 2,
96
91
  "out": 12,
97
92
  "released": "2026-02-19",
98
- "rank": "AA Intelligence 46 — BELOW the Anthropic/OpenAI/xAI frontier cluster (54–60); still preview-named. Honest: Gemini is not frontier-competitive on independent indices right now.",
99
- "source": "independent (AA) — vendor's GPQA-D 94.3% / ARC-AGI-2 77.1% are Google's own, uncorroborated"
93
+ "rank": "AA Intelligence 46 \u2014 BELOW the Anthropic/OpenAI/xAI frontier cluster (54\u201360); still preview-named. Honest: Gemini is not frontier-competitive on independent indices right now.",
94
+ "source": "independent (AA) \u2014 vendor's GPQA-D 94.3% / ARC-AGI-2 77.1% are Google's own, uncorroborated"
100
95
  },
101
96
  "mid": {
102
97
  "model": "google/gemini-3.5-flash",
103
98
  "in": 1.5,
104
99
  "out": 9,
105
100
  "released": "2026",
106
- "rank": "AA 50 — out-scores Google's own 3.1 Pro on this index (real anomaly, not a typo)",
101
+ "rank": "AA 50 \u2014 out-scores Google's own 3.1 Pro on this index (real anomaly, not a typo)",
107
102
  "source": "independent (AA)"
108
103
  },
109
104
  "cheap": {
@@ -121,13 +116,13 @@
121
116
  "XAI_API_KEY",
122
117
  "GROK_API_KEY"
123
118
  ],
124
- "note": "xAI's live lineup has no distinct budget SKU below Grok 4.3 as of 2026-07-15 (the 'grok-4.1-fast' I first wrote does not exist in the live catalog — the verify gate caught it). A Grok shop's cheap tasks route to Grok 4.3 or a cross-provider value pick.",
119
+ "note": "xAI's live lineup has no distinct budget SKU below Grok 4.3 as of 2026-07-15 (the 'grok-4.1-fast' I first wrote does not exist in the live catalog \u2014 the verify gate caught it). A Grok shop's cheap tasks route to Grok 4.3 or a cross-provider value pick.",
125
120
  "frontier": {
126
121
  "model": "x-ai/grok-4.5",
127
122
  "in": 2,
128
123
  "out": 6,
129
124
  "released": "2026-07-08",
130
- "rank": "AA Intelligence 54 (#8) at ~1/3 Opus 4.7's blended price — the frontier VALUE standout; 'Opus-class, faster, cheaper' (Musk)",
125
+ "rank": "AA Intelligence 54 (#8) at ~1/3 Opus 4.7's blended price \u2014 the frontier VALUE standout; 'Opus-class, faster, cheaper' (Musk)",
131
126
  "source": "independent (AA)"
132
127
  },
133
128
  "mid": {
@@ -135,7 +130,7 @@
135
130
  "in": 1.25,
136
131
  "out": 2.5,
137
132
  "released": "2026-04-30",
138
- "rank": "AA 38 — also xAI's cheapest verified stable tier",
133
+ "rank": "AA 38 \u2014 also xAI's cheapest verified stable tier",
139
134
  "source": "independent (AA)"
140
135
  }
141
136
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ruvnet-brain",
3
- "version": "4.5.4",
3
+ "version": "4.5.5",
4
4
  "description": "One-command installer for RuvNet Brain \u2014 a portable, source-grounded brain over rUv's RuvNet building blocks, delivered as a Claude Code plugin so Claude uses the stack instead of fighting it.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -40,6 +40,8 @@
40
40
  "test:all": "npm run test:unit && npm run test:mesh && npm run test:mutation && npm run test:regression && npm run test:integration && npm test",
41
41
  "metaharness:receipts": "node scripts/metaharness-receipts.mjs",
42
42
  "route:cheap": "node scripts/route-cheap.mjs",
43
+ "routing:eval": "node scripts/model-routing-eval.mjs",
44
+ "hooks:security-guidance:compat": "node scripts/security-guidance-codex-compat.mjs",
43
45
  "route:receipt": "node scripts/dispatch-receipt.mjs",
44
46
  "trismart": "node scripts/trismart.mjs",
45
47
  "metaharness:fix": "node scripts/fix-metaharness-memretrieve.mjs --apply",
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "ruvnet-brain",
3
3
  "description": "RuvNet brain transplant for Claude Code — grounds every RuvNet decision in real source across 77 rUv repositories, prefers Ruflo / RuVector-RVF / AgentDB over training-prior defaults (pgvector, Pinecone, hand-rolled cosine), and can pull in any RuvNet repo on demand. Ships a UserPromptSubmit retrieve-and-inject grounding hook and a PreToolUse write gate that refuses ungrounded rUv-product code until search_ruvnet has been consulted (ADR-0012 / ADR-067).",
4
- "version": "4.5.4",
4
+ "version": "4.5.5",
5
5
  "author": {
6
6
  "name": "Stuart Kerr"
7
7
  },
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ruvnet-brain",
3
- "version": "4.5.4",
3
+ "version": "4.5.5",
4
4
  "description": "Source-grounded RuvNet knowledge, lifecycle enforcement, and learning for Codex.",
5
5
  "author": {
6
6
  "name": "Stuart Kerr"
@@ -216,21 +216,20 @@ function validPostToolUseOutput(value) {
216
216
  if (!value || typeof value !== 'object' || Array.isArray(value)) return false;
217
217
  const topLevelKeys = new Set([
218
218
  'continue', 'stopReason', 'suppressOutput', 'systemMessage',
219
- 'terminalSequence', 'decision', 'reason', 'hookSpecificOutput',
219
+ 'decision', 'reason', 'hookSpecificOutput',
220
220
  ]);
221
221
  if (Object.keys(value).some((key) => !topLevelKeys.has(key))) return false;
222
222
  if (value.continue !== undefined && typeof value.continue !== 'boolean') return false;
223
223
  if (value.stopReason !== undefined && typeof value.stopReason !== 'string') return false;
224
224
  if (value.suppressOutput !== undefined && typeof value.suppressOutput !== 'boolean') return false;
225
225
  if (value.systemMessage !== undefined && typeof value.systemMessage !== 'string') return false;
226
- if (value.terminalSequence !== undefined && typeof value.terminalSequence !== 'string') return false;
227
226
  if (value.decision !== undefined && value.decision !== 'block') return false;
228
227
  if (value.decision === 'block' && (typeof value.reason !== 'string' || !value.reason.trim())) return false;
229
228
  if (value.hookSpecificOutput !== undefined) {
230
229
  const specific = value.hookSpecificOutput;
231
230
  if (!specific || typeof specific !== 'object' || Array.isArray(specific)) return false;
232
231
  const specificKeys = new Set([
233
- 'hookEventName', 'additionalContext', 'updatedToolOutput', 'updatedMCPToolOutput',
232
+ 'hookEventName', 'additionalContext',
234
233
  ]);
235
234
  if (Object.keys(specific).some((key) => !specificKeys.has(key))) return false;
236
235
  if (specific.hookEventName !== 'PostToolUse') return false;
@@ -268,13 +267,23 @@ if (!parsed) {
268
267
  process.exit(0);
269
268
  }
270
269
 
271
- // A body can emit syntactically valid JSON that is still invalid for Codex's event-specific wire
272
- // schema (wrong event name, unsupported fields, or bad field types). Preserve the advisory as text
273
- // inside the known-good PostToolUse envelope instead of forwarding a payload the host rejects.
270
+ // Codex 0.160.0 post-tool-use.command.output (extracted from the installed binary on
271
+ // 2026-10-04) rejects terminalSequence, updatedToolOutput, and telemetry fields. The
272
+ // host additionally rejects updatedMCPToolOutput semantically after schema parsing. Keep
273
+ // supported control fields even when a shared body includes incompatible metadata: wrapping
274
+ // the whole output as context alone would silently discard a security block.
274
275
  if (event === 'PostToolUse' && !validPostToolUseOutput(parsed)) {
275
- process.stdout.write(JSON.stringify({
276
- hookSpecificOutput: { hookEventName: event, additionalContext: stdout.trim() },
277
- }));
276
+ const normalized = {};
277
+ for (const key of ['continue', 'stopReason', 'suppressOutput', 'systemMessage', 'reason']) {
278
+ const type = ['continue', 'suppressOutput'].includes(key) ? 'boolean' : 'string';
279
+ if (typeof parsed[key] === type) normalized[key] = parsed[key];
280
+ }
281
+ if (parsed.decision === 'block') {
282
+ normalized.decision = 'block';
283
+ if (!normalized.reason?.trim()) normalized.reason = stdout.trim();
284
+ }
285
+ normalized.hookSpecificOutput = { hookEventName: event, additionalContext: stdout.trim() };
286
+ process.stdout.write(JSON.stringify(normalized));
278
287
  process.exit(0);
279
288
  }
280
289