ruvnet-brain 4.5.4 → 4.5.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/README.md +2 -2
  2. package/bin/install.mjs +162 -27
  3. package/config/model-router/catalog.template.json +126 -52
  4. package/config/model-router/claude-terminal-mod/.claude-plugin/plugin.json +1 -0
  5. package/config/model-router/claude-terminal-mod/README.md +22 -0
  6. package/config/model-router/claude-terminal-mod/hooks/hooks.json +1 -0
  7. package/config/model-router/claude-terminal-mod/hooks/policy.default.mjs +2 -0
  8. package/config/model-router/claude-terminal-mod/hooks/register.js +66 -0
  9. package/config/model-router/claude-terminal-mod/hooks/routing.js +55 -0
  10. package/config/model-router/claude-terminal-mod/hooks/runtime.js +2 -0
  11. package/config/model-router/claude-terminal-mod/tests/native.test.ts +86 -0
  12. package/config/model-router/policy.default.mjs +97 -75
  13. package/config/model-router/qualification-contract.json +124 -0
  14. package/config/model-router/routing-eval-cases.json +275 -0
  15. package/config/model-router/routing-policy.template.json +76 -0
  16. package/config/model-router/weekly-analyst-instruction.md +60 -0
  17. package/data/model-catalog.json +44 -49
  18. package/package.json +6 -3
  19. package/plugin/.claude-plugin/plugin.json +1 -1
  20. package/plugin/.codex-plugin/plugin.json +1 -1
  21. package/plugin/mcp/managed-cli-interface.mjs +9 -4
  22. package/plugin/scripts/capability-claim-evidence.mjs +9 -2
  23. package/plugin/scripts/codex-hook-adapter.mjs +18 -9
  24. package/plugin/scripts/project-capture-queue.mjs +7 -1
  25. package/plugin/scripts/project-progression-hook.mjs +16 -7
  26. package/plugin/scripts/project-progression-outbox.mjs +96 -15
  27. package/plugin/scripts/project-progression-producer.mjs +12 -1
  28. package/plugin/scripts/project-progression-store.mjs +53 -1
  29. package/plugin/scripts/project-transition-hook.mjs +19 -8
  30. package/plugin/scripts/session-snapshot-hook.mjs +2 -1
  31. package/scripts/claude-terminal-mod.mjs +89 -0
  32. package/scripts/codex-hook-trust-reconcile.mjs +247 -0
  33. package/scripts/codex-routed.sh +3 -36
  34. package/scripts/goldie-weekly.sh +8 -64
  35. package/scripts/metaharness-router.mjs +7 -1
  36. package/scripts/model-analyst-sandbox.mjs +54 -0
  37. package/scripts/model-currency-evidence.mjs +139 -0
  38. package/scripts/model-currency.mjs +230 -0
  39. package/scripts/model-native-catalog.mjs +111 -0
  40. package/scripts/model-native-qualification.mjs +251 -0
  41. package/scripts/model-router-agent-hook.mjs +136 -0
  42. package/scripts/model-router-dispatch.mjs +161 -0
  43. package/scripts/model-router-engine.mjs +155 -104
  44. package/scripts/model-routing-eval.mjs +108 -0
  45. package/scripts/model-routing-gateway.mjs +453 -0
  46. package/scripts/model-routing-launchers.mjs +209 -0
  47. package/scripts/model-routing-policy-promotion.mjs +203 -0
  48. package/scripts/model-terminal-gateway.mjs +310 -0
  49. package/scripts/model-terminal-launchers.mjs +300 -0
  50. package/scripts/model-weekly-analyst.mjs +299 -0
  51. package/scripts/model-weekly-assessment.mjs +91 -0
  52. package/scripts/model-weekly-cycle.mjs +183 -0
  53. package/scripts/model-weekly-qualification.mjs +362 -0
  54. package/scripts/native-subscription-usage.mjs +57 -0
  55. package/scripts/release-qualification-contract.mjs +50 -1
  56. package/scripts/release-qualification.mjs +9 -6
  57. package/scripts/security-guidance-codex-compat.mjs +142 -0
  58. package/scripts/user-model-prompt-hook.mjs +69 -0
@@ -0,0 +1,275 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "createdAt": "2026-10-04",
4
+ "labelBasis": "User allocation mandate: mechanical noncoding work, ordinary work, substantial implementation, bounded consequential reasoning, and explicitly justified exceptional reasoning. Labels assess the requested work, not the classifier's keywords.",
5
+ "harness": "codex",
6
+ "cases": [
7
+ {
8
+ "id": "mechanical-sort", "group": "mechanical",
9
+ "request": "Put these names in alphabetical order: Mina, Theo, Ana, Jules.",
10
+ "risk": "low",
11
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
12
+ "rationale": "The complete input and ordering rule are supplied; no code or judgment is needed."
13
+ },
14
+ {
15
+ "id": "mechanical-format", "group": "mechanical",
16
+ "request": "Format these three dates as YYYY-MM-DD: October 1 2026, October 2 2026, October 3 2026.",
17
+ "risk": "low",
18
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
19
+ "rationale": "A supplied finite list and explicit transformation make this mechanical."
20
+ },
21
+ {
22
+ "id": "mechanical-extract", "group": "mechanical",
23
+ "request": "Extract the three invoice identifiers from this paragraph: INV-101 arrived, INV-102 was paid, and INV-103 was voided.",
24
+ "risk": "low",
25
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
26
+ "rationale": "Copying explicit identifiers requires no domain inference."
27
+ },
28
+ {
29
+ "id": "mechanical-translate", "group": "mechanical",
30
+ "request": "Translate 'The meeting starts at noon' into Spanish.",
31
+ "risk": "low",
32
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
33
+ "rationale": "A short everyday sentence is a bounded noncoding transformation."
34
+ },
35
+ {
36
+ "id": "mechanical-typo", "group": "mechanical",
37
+ "request": "Fix only the typo in 'The pakage arrived today'.",
38
+ "risk": "low",
39
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
40
+ "rationale": "The requested edit is narrow and reversible."
41
+ },
42
+ {
43
+ "id": "mechanical-table", "group": "mechanical",
44
+ "request": "Turn this list into a two-column Markdown table: red=3, blue=5, green=2.",
45
+ "risk": "low",
46
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
47
+ "rationale": "Only presentation changes; the values are already given."
48
+ },
49
+ {
50
+ "id": "mechanical-summary", "group": "mechanical",
51
+ "request": "Summarize this supplied changelog into three bullets without adding recommendations: fixed typo, corrected link, updated title.",
52
+ "risk": "low",
53
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
54
+ "rationale": "A closed-input summary needs no external investigation."
55
+ },
56
+ {
57
+ "id": "ordinary-code-extraction", "group": "ordinary",
58
+ "request": "Implement an extraction function that reads invoice IDs and add a test for an empty input.",
59
+ "risk": "moderate",
60
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
61
+ "rationale": "Despite extraction vocabulary, this asks for ordinary code and validation."
62
+ },
63
+ {
64
+ "id": "ordinary-file-inspection", "group": "ordinary",
65
+ "request": "Read the two module files and explain which one owns timeout configuration.",
66
+ "risk": "low",
67
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
68
+ "rationale": "Routine source inspection does not justify frontier reasoning."
69
+ },
70
+ {
71
+ "id": "ordinary-environment-path", "group": "environment-repair",
72
+ "request": "The CLI is missing from PATH after opening a new terminal. Inspect the shell startup files and repair the user-level path.",
73
+ "taskFacts": { "taskType": "coding", "scope": "routine", "uncertainty": "environment" },
74
+ "risk": "moderate",
75
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
76
+ "rationale": "An environment problem alone does not establish difficult reasoning."
77
+ },
78
+ {
79
+ "id": "ordinary-missing-information", "group": "missing-information",
80
+ "request": "The test fails, but I forgot to include the error. Find the test command and collect the failure before deciding what to change.",
81
+ "taskFacts": { "taskType": "coding", "uncertainty": "missing-information" },
82
+ "risk": "moderate",
83
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
84
+ "rationale": "Gathering absent evidence is ordinary debugging, not automatically architecture uncertainty."
85
+ },
86
+ {
87
+ "id": "ordinary-ambiguous-speed", "group": "ambiguity",
88
+ "request": "Make it faster. First ask which page and measurement I mean; do not redesign anything yet.",
89
+ "taskFacts": { "uncertainty": "missing-information" },
90
+ "risk": "low",
91
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
92
+ "rationale": "The immediate authorized work is clarification, without evidence of a difficult system problem."
93
+ },
94
+ {
95
+ "id": "ordinary-permission-repair", "group": "environment-repair",
96
+ "request": "A local fixture cannot write its own temporary directory. Check ownership and repair only that disposable fixture.",
97
+ "taskFacts": { "scope": "routine", "uncertainty": "environment" },
98
+ "risk": "moderate",
99
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
100
+ "rationale": "A scoped filesystem repair should not be treated as an exceptional security review."
101
+ },
102
+ {
103
+ "id": "ordinary-architecture-doc", "group": "ordinary",
104
+ "request": "Inspect architecture documentation and tell me where the cache path is documented.",
105
+ "risk": "low",
106
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
107
+ "rationale": "Reading an architecture document is not making an architectural decision."
108
+ },
109
+ {
110
+ "id": "ordinary-routine-review", "group": "ordinary",
111
+ "request": "Review a one-line typo correction in the README and confirm the replacement spelling.",
112
+ "taskFacts": { "taskType": "review", "scope": "routine" },
113
+ "risk": "low",
114
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
115
+ "rationale": "A routine review label must not automatically trigger final substantive review allocation."
116
+ },
117
+ {
118
+ "id": "substantial-invitations", "group": "substantial-implementation",
119
+ "request": "Add team invitations with storage, expiration, acceptance endpoints, permissions, UI states, and integration coverage.",
120
+ "risk": "moderate",
121
+ "expected": { "minimumClass": "substantial", "maximumClass": "substantial", "minimumRole": "substantial-implementation" },
122
+ "rationale": "Several coordinated implementation surfaces warrant a stronger ordinary coding effort."
123
+ },
124
+ {
125
+ "id": "substantial-facts", "group": "substantial-implementation",
126
+ "request": "Implement the accepted invitation design across the backend, client, and integration fixtures.",
127
+ "taskFacts": { "taskType": "coding", "scope": "substantial", "uncertainty": "none" },
128
+ "risk": "moderate",
129
+ "expected": { "minimumClass": "substantial", "maximumClass": "substantial", "minimumRole": "substantial-implementation" },
130
+ "rationale": "Substantial implementation with a settled design needs high-effort ordinary execution, not a new architecture escalation."
131
+ },
132
+ {
133
+ "id": "substantial-parser-adoption", "group": "substantial-implementation",
134
+ "request": "Replace the old parser in every importer, preserve compatibility, and cover each import format with integration fixtures.",
135
+ "risk": "moderate",
136
+ "expected": { "minimumClass": "substantial", "maximumClass": "substantial", "minimumRole": "substantial-implementation" },
137
+ "rationale": "Broad coordinated implementation is substantial even when the request is short."
138
+ },
139
+ {
140
+ "id": "substantial-observability", "group": "substantial-implementation",
141
+ "request": "Implement the already approved tracing design across the scheduler, worker, API, and dashboard with tests.",
142
+ "taskFacts": { "scope": "substantial" },
143
+ "risk": "moderate",
144
+ "expected": { "minimumClass": "substantial", "maximumClass": "substantial", "minimumRole": "substantial-implementation" },
145
+ "rationale": "Partial scope facts supplement a clearly broad implementation task."
146
+ },
147
+ {
148
+ "id": "hard-double-charge", "group": "deceptively-short-hard",
149
+ "request": "It charged the same card twice after failover. Find the smallest safe fix.",
150
+ "risk": "critical",
151
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
152
+ "rationale": "Financial correctness across failover requires understanding coupled failure behavior; brevity does not make it routine."
153
+ },
154
+ {
155
+ "id": "hard-cross-tenant", "group": "deceptively-short-hard",
156
+ "request": "One tenant can see another tenant's invoices. Explain the cause and the safe repair.",
157
+ "risk": "critical",
158
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
159
+ "rationale": "A cross-tenant data exposure is a consequential security reasoning task."
160
+ },
161
+ {
162
+ "id": "hard-migration", "group": "migration",
163
+ "request": "Plan moving live billing records to a new schema without losing payments or breaking rollback while both versions serve traffic.",
164
+ "risk": "critical",
165
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
166
+ "rationale": "A live money-data migration combines consequential planning, rollback, and concurrent-version invariants."
167
+ },
168
+ {
169
+ "id": "hard-storage-tradeoff", "group": "architecture",
170
+ "request": "Choose between one durable writer and replicated writers for this service; explain what happens when a region vanishes during an acknowledgment.",
171
+ "risk": "high",
172
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
173
+ "rationale": "This is an unresolved durability and deployment tradeoff rather than routine code work."
174
+ },
175
+ {
176
+ "id": "hard-cross-system-bug", "group": "cross-system-bug",
177
+ "request": "Jobs vanish only when the scheduler retries during database failover and the worker reconnects. Trace the interaction before fixing it.",
178
+ "risk": "high",
179
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
180
+ "rationale": "The reproduction depends on coupled uncertainty across three systems and failure recovery."
181
+ },
182
+ {
183
+ "id": "hard-ambiguous-architecture", "group": "ambiguity",
184
+ "request": "The architecture tradeoff is unresolved: should acknowledgments precede durable replication? Establish the invariant before choosing.",
185
+ "risk": "high",
186
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
187
+ "rationale": "Architecture uncertainty with a durability consequence warrants bounded difficult reasoning."
188
+ },
189
+ {
190
+ "id": "hard-difficult-review", "group": "difficult-review",
191
+ "request": "Give the final substantive review of the lease-fencing patch, including stale owners, restart races, and failure recovery.",
192
+ "risk": "high",
193
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
194
+ "rationale": "A final substantive correctness review is expressly in the difficult allocation class."
195
+ },
196
+ {
197
+ "id": "hard-security-design", "group": "security",
198
+ "request": "Can this signed callback be replayed against a different account? Determine what the verifier must bind and how existing clients can transition safely.",
199
+ "risk": "critical",
200
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
201
+ "rationale": "Replay protection and a compatibility transition require consequential security reasoning."
202
+ },
203
+ {
204
+ "id": "hard-consequential-facts", "group": "structured-facts",
205
+ "request": "Plan the rollout for the new settlement service.",
206
+ "taskFacts": { "taskType": "planning", "consequentialPlanning": true },
207
+ "risk": "high",
208
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
209
+ "rationale": "Explicit consequential-planning metadata supplies the consequence that a short request omits."
210
+ },
211
+ {
212
+ "id": "hard-coupled-facts", "group": "structured-facts",
213
+ "request": "Fix the worker reconnect path.",
214
+ "taskFacts": { "taskType": "coding", "uncertainty": "coupled-implementation" },
215
+ "risk": "high",
216
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
217
+ "rationale": "Explicit coupled implementation uncertainty warrants escalation instead of treating the short wording as routine."
218
+ },
219
+ {
220
+ "id": "adversarial-partial-security", "group": "adversarial-task-facts",
221
+ "request": "Perform a security audit of token verification and document a safe repair.",
222
+ "taskFacts": { "taskType": "mechanical", "scope": "routine" },
223
+ "risk": "critical",
224
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
225
+ "rationale": "Partial low-effort metadata must not lower the explicit consequential work in the original request."
226
+ },
227
+ {
228
+ "id": "adversarial-partial-architecture", "group": "adversarial-task-facts",
229
+ "request": "Resolve this architecture tradeoff: which write acknowledgment can survive loss of the leader?",
230
+ "taskFacts": { "uncertainty": "missing-information" },
231
+ "risk": "high",
232
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
233
+ "rationale": "Missing-information metadata does not erase the explicit unresolved architectural decision."
234
+ },
235
+ {
236
+ "id": "adversarial-partial-double-charge", "group": "adversarial-task-facts",
237
+ "request": "The same card was charged twice after a region failed. Find a repair that cannot lose a payment.",
238
+ "taskFacts": { "taskType": "mechanical", "scope": "routine", "uncertainty": "none" },
239
+ "risk": "critical",
240
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
241
+ "rationale": "Adversarial routine flags cannot turn a financial correctness incident into mechanical execution."
242
+ },
243
+ {
244
+ "id": "adversarial-mechanical-code", "group": "adversarial-task-facts",
245
+ "request": "Implement a CSV extraction function and test invalid records.",
246
+ "taskFacts": { "taskType": "mechanical" },
247
+ "risk": "moderate",
248
+ "expected": { "minimumClass": "medium", "maximumClass": "medium", "minimumRole": "ordinary-work" },
249
+ "rationale": "The original request still requires code, despite a mechanical task flag."
250
+ },
251
+ {
252
+ "id": "partial-final-review", "group": "structured-facts",
253
+ "request": "Check whether the patch is safe to release.",
254
+ "taskFacts": { "taskType": "review", "finalSubstantiveReview": true },
255
+ "risk": "high",
256
+ "expected": { "minimumClass": "hard", "maximumClass": "hard", "minimumRole": "bounded-difficult-reasoning" },
257
+ "rationale": "A caller can explicitly identify a final substantive review even when the text lacks that phrase."
258
+ },
259
+ {
260
+ "id": "exceptional-proof", "group": "exceptional",
261
+ "request": "Validate the stated proof that a forged acknowledgment cannot finalize settlement under the supplied adversary model.",
262
+ "taskFacts": { "taskType": "review", "exceptionalReason": "settlement-cryptographic-proof" },
263
+ "risk": "critical",
264
+ "expected": { "minimumClass": "exceptional", "maximumClass": "exceptional", "minimumRole": "exceptional-reasoning" },
265
+ "rationale": "A named bounded exceptional justification is supplied for unusually demanding proof review."
266
+ },
267
+ {
268
+ "id": "ordinary-misleading-term", "group": "ordinary",
269
+ "request": "Summarize the headings in this supplied document titled Security Audit; do not assess security or recommend changes.",
270
+ "risk": "low",
271
+ "expected": { "minimumClass": "fast", "maximumClass": "fast", "minimumRole": "mechanical-work" },
272
+ "rationale": "A document title is data; copying its headings is not conducting an audit."
273
+ }
274
+ ]
275
+ }
@@ -0,0 +1,76 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "reviewedAt": "2026-10-04T13:32:24.704091Z",
4
+ "maxAgeMs": 604800000,
5
+ "routes": {
6
+ "codex": {
7
+ "fast": {
8
+ "model": "gpt-6-luna",
9
+ "effort": "low"
10
+ },
11
+ "medium": {
12
+ "model": "gpt-6.1-sol",
13
+ "effort": "medium"
14
+ },
15
+ "hard": {
16
+ "model": "gpt-6-astra",
17
+ "effort": "high"
18
+ },
19
+ "substantial": {
20
+ "model": "gpt-6.1-sol",
21
+ "effort": "high"
22
+ },
23
+ "exceptional": {
24
+ "model": "gpt-6-astra",
25
+ "effort": "xhigh",
26
+ "requiresNamedReason": true
27
+ }
28
+ },
29
+ "claude-code": {
30
+ "fast": {
31
+ "model": "claude-sonnet-5-5",
32
+ "effort": "low"
33
+ },
34
+ "medium": {
35
+ "model": "claude-sonnet-5-5",
36
+ "effort": "medium"
37
+ },
38
+ "hard": {
39
+ "model": "claude-opus-5-5",
40
+ "effort": "high"
41
+ },
42
+ "codingEffort": "high"
43
+ }
44
+ },
45
+ "sources": [
46
+ {
47
+ "url": "https://artificialanalysis.ai/models/releases/comparisons/gpt-6-1-sol-vs-claude-sonnet-5-5",
48
+ "benchmark": "Intelligence Index v4.3.2; Terminal-Bench 4.0",
49
+ "checkedAt": "2026-10-04T13:32:24.704295Z"
50
+ },
51
+ {
52
+ "url": "https://artificialanalysis.ai/models/releases/comparisons/gpt-6-luna-vs-gpt-6-astra",
53
+ "benchmark": "Intelligence Index v4.3.2; Terminal-Bench 4.0",
54
+ "checkedAt": "2026-10-04T13:32:24.704301Z"
55
+ }
56
+ ],
57
+ "qualification": {
58
+ "basis": "live provider metadata, five native subscription smoke launches, independent effort comparisons, user allocation constraint",
59
+ "limits": [
60
+ "Not project-specific optimality proof",
61
+ "Luna fast limited to noncoding",
62
+ "No automatic newmodel entitlement claim",
63
+ "Parent conversation model remains host-controlled",
64
+ "Substantial/exceptional allocation reflects user correctness preference, not newly measured optimality",
65
+ "Free-text task classes are heuristic; caller taskFacts can express uncertainty and scope",
66
+ "No assumed completion-speed or subscription-quota multipliers",
67
+ "Native agent hook updatedInput rewrite has not passed host acceptance proof"
68
+ ]
69
+ },
70
+ "policyRevisionAt": "2026-10-04T13:50:05.851549+00:00",
71
+ "objectivePriority": [
72
+ "correctness",
73
+ "subscription-allowance",
74
+ "completion-time"
75
+ ]
76
+ }
@@ -0,0 +1,60 @@
1
+ Updated: 2026-10-04 14:16:00 EDT | Version 1.0.4
2
+ Created: 2026-10-04 09:56:00 EDT
3
+
4
+ # Weekly model-routing analyst mandate
5
+
6
+ Act as Stuart's model-routing analyst for software development. Maintain an evidence-based, per-user policy for native OpenAI Codex and Anthropic Claude subscriptions. This instruction describes the required analyst work; storing it or collecting metadata does not establish that the analyst ran.
7
+
8
+ ## When this assessment runs
9
+
10
+ Stuart's October 4 clarification governs the schedule: check weekly for newly released OpenAI or Anthropic text-capable models. If no new relevant model is discovered, retain the existing owner-approved routing policy byte-for-byte, record the successful catalog check, and finish quietly. Do not rerun this full assessment merely because another week passed or benchmark/pricing data changed.
11
+
12
+ The first successful catalog check establishes the release baseline while retaining the policy Stuart already approved. That is a baseline receipt, not proof that a semantic assessment ran. A new canonical provider/model release triggers the full instructions below. Failed discovery is unknown, never "no change." An unavailable selected route must be surfaced and must not silently downgrade. Preserve pending new releases when an assessment fails so they can be retried.
13
+
14
+ Allow up to 15 minutes for a triggered assessment, aiming to finish within five minutes. A timeout remains a failed assessment and preserves the approved policy. Changing the timer or collecting fresh metadata does not establish successful review or authorize promotion.
15
+
16
+ ## Objective and authority
17
+
18
+ Prioritize correctness, completeness and sound architectural judgment; then efficient included subscription allowance use; then time to a verified result including planning, handoffs, implementation, repairs and review. Treat the providers' allowances separately. Never infer included usage from API prices, credit rates, message counts or token counts. Do not weaken capability when the task needs stronger reasoning. Surface capacity constraints and defer optional work instead.
19
+
20
+ Use native subscription authentication. Never enable paid API fallback, extra credits, or a new subscription. Never introduce API billing, purchase credits, upgrade plans or enable overages. Before a Codex analyst launch verify ordinary included usage is available; block when exhausted or unavailable. Authentication and this check do not reserve allowance or guarantee existing credits cannot be consumed after concurrent usage exhausts it. Disclose that limit without claiming a hard spending cap. Do not change existing billing controls. Comparative inference needs standing authorization and an explicit allowance budget. The routine evidence collector must not launch unbudgeted experiments.
21
+
22
+ ## Baseline and research
23
+
24
+ On the first run inspect installed tools, supported configuration, exact models and efforts actually available through the subscriptions, accessible usage/reset information, existing routing rules, user overrides and evaluation history. Never expose credentials. On subsequent runs compare with the prior report, refresh changeable facts and retain valid historical evidence. Report inaccessible dashboards or providers without guessing entitlement or allowance.
25
+
26
+ Distinguish public announcements, API availability, native selectable subscription models and independently observed execution. Check official OpenAI and Anthropic sources for exact identifiers, releases, retirements, client effort support, speed modes, usage rules, coding/review capabilities and context handling.
27
+
28
+ Check independent primary evaluations relevant to architecture, repository understanding, implementation, debugging, long-running coding and review. Start with Artificial Analysis model AND coding-agent evaluations (https://artificialanalysis.ai/), VulcanBench reports and methodology (https://vulcanbench.com/), Terminal-Bench (https://www.tbench.ai/) and SWE-bench (https://www.swebench.com/). Include other relevant reproducible independent evaluations. Trace aggregators to the original evaluator. Trace repeated claims to original experiments; repetitions are not independent evidence. Personal reports are supplementary. Record source URL, evaluation date, benchmark version, exact model and effort, harness, task count, success rate, uncertainty where available, runtime, token use and reported cost basis. Compare efforts within each model; investigate whether more effort reduces total work, retries, tokens or time. Low/medium effort is not automatically more economical. Report conflicting evidence and its workloads. Do not combine incompatible evaluations into a ranking. General intelligence scores do not prove architecture/review superiority; maximum-effort results do not prove medium-effort behavior. Cite direct sources and separate measurements, vendor claims and recommendations.
29
+
30
+ ## Evaluate dispatch independently of worker quality
31
+
32
+ The entry model must route reliably and enforce its handoff. Do not select a cheap dispatcher without evidence. Maintain representative routing cases: obvious mechanical tasks, deceptively short hard tasks, ambiguous requirements, architectural decisions, security-sensitive changes, migrations, cross-system bugs and difficult reviews. Measure dangerous under-routing, unnecessary escalation, handoff failures, routing latency and whole-workflow usage. Confidence claims from the dispatcher are insufficient.
33
+
34
+ Prefer deterministic dispatch for clear rules. Uncertain classifications go upward or receive a stronger assessment before implementation. Preserve the original request and relevant evidence; a weak summary must not be the worker's only context. Verify actual execution model and effort. Requested arguments, messages and configuration edits are not proof. State whether each supported route changes the parent, starts a child, or launches another native client.
35
+
36
+ ## Recommendations and escalation
37
+
38
+ Produce OpenAI-only, Anthropic-only and supported combined policies. Every route needs exact identifier, supported effort or documented equivalent, speed mode, entry conditions, escalation triggers and confidence. Do not invent equivalent effort semantics across providers.
39
+
40
+ Cover entry/dispatch, mechanical work, routine settled-design implementation, substantial development, consequential architecture/scope/ambiguity and cross-system reasoning, difficult debugging, hardest implementation, substantive final review and exceptional escalation. Include exact model, effort, speed mode, fallback, escalation and separately chosen reviewer for each role. The reviewer checks requirements, implementation, tests and unresolved risks, and requires repairs; approval cannot replace execution evidence.
41
+
42
+ Reassess this current OpenAI hypothesis: Astra high for consequential planning and review; Sol 6.1 high for demanding implementation under a clear design; Sol 6.1 medium for routine implementation; Astra retains implementation when essential judgment remains tightly coupled; Luna low only for explicit mechanical transformations with complete cheap verification. Treat Luna low dispatch, Sol 6.1 high implementation, Sol xhigh difficult implementation and Astra high architecture/review as hypotheses, not permanent rules. Apply equally independent reasoning to Anthropic. Verify the identifiers/settings before recommending replacements. The strongest suitable model handles consequential judgment; do not require failures on weaker models first.
43
+
44
+ Reassess when scope, assumptions, boundaries or verification change. Missing information requires evidence; tooling/environment failure requires repair; difficult implementation reasoning may warrant more effort or capability; architectural uncertainty warrants strongest suitable architectural reasoning promptly. Avoid rigid retry ladders. Final review cannot guarantee recovery from bad early assumptions. Require appropriate executable checks and runtime evidence. Cross-provider review needs expected benefit; different providers do not guarantee independent errors.
45
+
46
+ ## Evaluation and application
47
+
48
+ Save allowance through clear scope, relevant context, preserved decisions, fewer redundant investigations, suitable checks and fewer repairs. Measure the whole accepted-task path rather than decode speed. Standard delivery is preferred unless acceleration's documented benefit justifies its allowance use. Use existing local telemetry first. Account for dispatcher work, context transfer, all children, retries, review and repairs. Record actual telemetry where available; mark attribution uncertain under concurrent activity, resets, rounding or other sessions. Keep separate provider budgets and any shared/model-specific limits. Verify usage multipliers each week. Never promise zero quality loss or guaranteed savings.
49
+
50
+ Newer models are not automatically better. Use credible evaluations and task outcomes first. Uncertain options stay experimental. Local comparisons need explicit success criteria, a budget and an independent quality judge. Evaluate correctness, review findings, repairs, completion time and attributable allowance where available.
51
+
52
+ Research and update proposals automatically. Maintain a last-known-good policy and dated candidate. Require a demonstrated quality floor for every role, availability, routing, actual handoff and quality checks. Avoid exhaustive evaluations that consume a large fraction of allowance. Automatically promote only evidence-qualified, supported changes within the router's existing authorized update mechanism and standing authorization; preserve recovery. Preserve user overrides, validate settings, retain previous versions and distinguish applied from recommended. If evidence is insufficient, retain the established route with an uncertainty label. Never mark metadata collection as completed semantic review or reset a policy review date merely because HTTP fetches succeeded.
53
+
54
+ ## Weekly output and triggering
55
+
56
+ Save an initial full report and concise dated change reports, sources and a versioned policy proposal in the authorized per-user output directory. Include changes; original third-party charts with dates/links; clearly labelled recreated model-and-effort charts with separate quality-versus-cost and quality-versus-completion-time views; API cost versus measured subscription usage; recommended VS Code entry model and reasons; OpenAI, Anthropic and combined routing diagrams/tables; exact identifiers/efforts/speed/fallback; escalation/review rules; confidence, evidence gaps and policy changes. Include a machine-readable candidate compatible with the existing router. Mark unmeasured configurations missing, never invent scores. Preserve source snapshots and policy reasoning. Preserve the prior policy for comparison and recovery.
57
+
58
+ Keep unchanged findings quiet. Notify for actionable improvement, retirement, availability change, regression, access failure or a required decision. A schedule needs actual run receipts. Active-session weekly catch-up is not a guarantee of execution while clients are closed. Never claim a recommendation is implemented, a model accessible or every prompt enforced without checking the actual supported runtime path.
59
+
60
+ The native release check also refreshes account-visible model metadata without inference. For newly discovered models only, semantic analysis and independent qualification share one fifteen-minute deadline. A source-reviewed fixed role suite compares the incumbent and candidate, and a separate approved hard reviewer grades anonymized outputs. Automatic application requires standing user authorization, actual native configured-turn evidence, passing quality checks and an unchanged prior policy. Requested settings are not backend identity proof. Incomplete qualification keeps its bound proposal for retry without repeating the completed analysis; terminal rejection retains the approved route.
@@ -1,20 +1,20 @@
1
1
  {
2
2
  "_meta": {
3
- "purpose": "Per-provider (house) tier ladders. The FRONTIER — the escalation target AND the savings baseline — is personalized to the user's own house: a Claude shop's frontier is Fable 5, a ChatGPT shop's is GPT-5.6 Sol, a Codex shop's is Sol, a Gemini shop's is Gemini 3.1 Pro, a Grok shop's is Grok 4.5. Modeled on ruflo ADR-148 (assets/model-router/openrouter-alts.json), extended with a provider-house axis rUv's registry does not carry.",
3
+ "purpose": "Verified provider tier ladders and task-fit recommendations. Frontier is the hard-work target; it is not the routine default. API prices are comparison metadata, not subscription charges.",
4
4
  "generated": "2026-07-15",
5
5
  "schema_version": 1,
6
6
  "sources": {
7
- "prices": "OpenRouter /api/v1/models live catalog, pulled 2026-09-19 (in/out USD per Mtok).",
8
- "rankings": "Artificial Analysis Intelligence Index (artificialanalysis.ai) + Arena/LMArena (arena.ai) — the ONLY independent evaluators carrying current-generation models as of 2026-07-15; each figure cross-verified twice.",
9
- "provenance_rule": "rUv ADR-206: vendor-reported scores are optimistic and harness-confounded — trust independent (AA/Arena) numbers, treat vendor self-scaffold SWE-bench/LiveCodeBench figures as noisy features, never as truth.",
10
- "benchmark_lag": "The canonical hard coding benchmarks (SWE-bench Verified standardized harness, LiveCodeBench, Aider polyglot) were ALL months stale on 2026-07-15 and carry NONE of these models. The '88.6% / 95% SWE-bench' figures in the press are vendor self-scaffold scores, not the standardized harness — excluded here.",
7
+ "prices": "OpenRouter /api/v1/models live catalog, pulled 2026-10-04 (in/out USD per Mtok).",
8
+ "rankings": "Anthropic/OpenAI official role guidance checked 2026-10-04 for refreshed entries. Historical ranks for other providers unchanged and not revalidated in this two-provider refresh.",
9
+ "provenance_rule": "rUv ADR-206: vendor-reported scores are optimistic and harness-confounded \u2014 trust independent (AA/Arena) numbers, treat vendor self-scaffold SWE-bench/LiveCodeBench figures as noisy features, never as truth.",
10
+ "benchmark_lag": "The canonical hard coding benchmarks (SWE-bench Verified standardized harness, LiveCodeBench, Aider polyglot) were ALL months stale on 2026-07-15 and carry NONE of these models. The '88.6% / 95% SWE-bench' figures in the press are vendor self-scaffold scores, not the standardized harness \u2014 excluded here.",
11
11
  "release_refs": "OpenAI GPT-5.6 GA 2026-07-09 (openai.com/index/gpt-5-6); Anthropic Fable 5 2026-06-09 (anthropic.com/news/claude-fable-5-mythos-5); Google Gemini 3.1 Pro 2026-02-19; xAI/SpaceXAI Grok 4.5 2026-07-08."
12
12
  },
13
- "caveat": "Prices are live-verified; tier placements are sensible, independent-benchmark-grounded starters, NOT this user's measured results. Override per-installation via $RUVNET_MODEL_CATALOG or the console. The automated always-current path is rUv's ADR-206 (BenchPress predictor + hourly OpenRouter watcher) — see scripts/refresh-model-catalog.mjs for the live re-pull.",
14
- "effort_note": "Frontier effort defaults to 'xhigh' (the max reasoning a shop reaches for on the hardest tasks); this is a principled default, not a per-effort measurement."
13
+ "caveat": "Refreshed Anthropic/OpenAI tiers use official provider guidance and owner preferences, not newly measured independent rankings. Other provider claims remain historical and were not reassessed in this refresh.",
14
+ "effort_note": "Start ordinary work at low/medium, hard work at high; xhigh/max only when representative evals justify latency and cost. Per-model supports and API/host defaults differ; see dated October 4 refresh."
15
15
  },
16
16
  "default_provider": "anthropic",
17
- "default_provider_reason": "This is a Claude Code plugin, so DEVELOPMENT genuinely runs on Anthropic — that is a detected fact, not an arbitrary house preference. Set your PRODUCTION house in the console (or $RUVNET_PROVIDER) if your app runs on a different provider; it is never assumed silently.",
17
+ "default_provider_reason": "This is a Claude Code plugin, so DEVELOPMENT genuinely runs on Anthropic \u2014 that is a detected fact, not an arbitrary house preference. Set your PRODUCTION house in the console (or $RUVNET_PROVIDER) if your app runs on a different provider; it is never assumed silently.",
18
18
  "providers": {
19
19
  "anthropic": {
20
20
  "label": "Claude (Anthropic)",
@@ -24,29 +24,27 @@
24
24
  "CLAUDECODE"
25
25
  ],
26
26
  "frontier": {
27
- "model": "claude-fable-5",
28
- "in": 10,
29
- "out": 50,
30
- "released": "2026-06-09",
31
- "rank": "AA Intelligence #1 (60) · Arena #1 (1508 Elo)",
32
- "source": "independent (AA + Arena)"
27
+ "model": "anthropic/claude-opus-5.5",
28
+ "in": 4,
29
+ "out": 20,
30
+ "rank": "Hard work; Fable 5.1 is a bounded exceptional escalation, not the ordinary baseline",
31
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
33
32
  },
34
33
  "mid": {
35
- "model": "claude-sonnet-5",
34
+ "model": "anthropic/claude-sonnet-5.5",
36
35
  "in": 2,
37
36
  "out": 10,
38
- "released": "2026-06-30",
39
- "rank": "AA 53; intro price $2/$10 → $3/$15 after 2026-08-31",
40
- "source": "independent (AA)"
37
+ "rank": "Routine work at medium effort",
38
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
41
39
  },
42
40
  "cheap": {
43
- "model": "claude-haiku-4.5",
44
- "in": 1,
45
- "out": 5,
46
- "released": "2025-10-15",
47
- "rank": "cheap tier (AA index low; cost-cascade prefers a cross-provider value pick when an OpenRouter key is present)",
48
- "source": "catalog"
49
- }
41
+ "model": "anthropic/claude-sonnet-5.5",
42
+ "in": 2,
43
+ "out": 10,
44
+ "rank": "Fast work at low effort; Haiku excluded by owner preference, not claimed discontinued",
45
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
46
+ },
47
+ "note": "API/OpenRouter IDs differ from native Claude IDs: claude-opus-5-5 and claude-sonnet-5-5. Same Sonnet model serves two effort roles. No measured latency/quality guarantee."
50
48
  },
51
49
  "openai": {
52
50
  "label": "ChatGPT (OpenAI)",
@@ -54,35 +52,32 @@
54
52
  "OPENAI_API_KEY"
55
53
  ],
56
54
  "frontier": {
57
- "model": "openai/gpt-5.6-sol",
58
- "in": 2,
59
- "out": 10,
60
- "released": "2026-07-09",
61
- "rank": "AA Intelligence #2 (59) · AA Coding Index leader (80) · Terminal-Bench 2.1 SOTA · ~1/3 Fable 5's cost/task",
62
- "source": "independent (AA)"
55
+ "model": "openai/gpt-6-astra",
56
+ "in": 10,
57
+ "out": 50,
58
+ "rank": "Bounded difficult reasoning and critical independent review",
59
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
63
60
  },
64
61
  "mid": {
65
- "model": "openai/gpt-5.6-terra",
62
+ "model": "openai/gpt-6.1-sol",
66
63
  "in": 2,
67
- "out": 12,
68
- "released": "2026-07-09",
69
- "rank": "AA Coding 77.4 — beats prev-gen flagship GPT-5.5 (76.4) at lower live catalog pricing",
70
- "source": "independent (AA)"
64
+ "out": 10,
65
+ "rank": "Default ordinary coding, research and debugging",
66
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
71
67
  },
72
68
  "cheap": {
73
- "model": "openai/gpt-5.6-luna",
74
- "in": 0.2,
75
- "out": 1.2,
76
- "released": "2026-07-09",
77
- "rank": "AA 51 · Coding 74.6 — beats most last-gen mid-tier",
78
- "source": "independent (AA)"
69
+ "model": "openai/gpt-6-luna",
70
+ "in": 0.1,
71
+ "out": 0.5,
72
+ "rank": "Focused high-volume tasks and small well-specified coding changes",
73
+ "source": "2026-10-04 provider documentation + live OpenRouter catalog; project task-fit recommendation, not measured independent rank"
79
74
  }
80
75
  },
81
76
  "codex": {
82
77
  "label": "Codex (OpenAI)",
83
78
  "detect_env": [],
84
79
  "aliasOf": "openai",
85
- "note": "Codex folded into the GPT-5.6 Sol/Terra/Luna tiers on 2026-07-09 — there is NO separate '-codex' SKU this generation; gpt-5.3-codex (2026-02-05) was the last dedicated one and is superseded. A Codex shop's frontier IS Sol."
80
+ "note": "Alias of current OpenAI tier ladder; native availability checked through fresh Codex account catalog. Updating metadata does not switch active conversations."
86
81
  },
87
82
  "google": {
88
83
  "label": "Gemini (Google)",
@@ -95,15 +90,15 @@
95
90
  "in": 2,
96
91
  "out": 12,
97
92
  "released": "2026-02-19",
98
- "rank": "AA Intelligence 46 — BELOW the Anthropic/OpenAI/xAI frontier cluster (54–60); still preview-named. Honest: Gemini is not frontier-competitive on independent indices right now.",
99
- "source": "independent (AA) — vendor's GPQA-D 94.3% / ARC-AGI-2 77.1% are Google's own, uncorroborated"
93
+ "rank": "AA Intelligence 46 \u2014 BELOW the Anthropic/OpenAI/xAI frontier cluster (54\u201360); still preview-named. Honest: Gemini is not frontier-competitive on independent indices right now.",
94
+ "source": "independent (AA) \u2014 vendor's GPQA-D 94.3% / ARC-AGI-2 77.1% are Google's own, uncorroborated"
100
95
  },
101
96
  "mid": {
102
97
  "model": "google/gemini-3.5-flash",
103
98
  "in": 1.5,
104
99
  "out": 9,
105
100
  "released": "2026",
106
- "rank": "AA 50 — out-scores Google's own 3.1 Pro on this index (real anomaly, not a typo)",
101
+ "rank": "AA 50 \u2014 out-scores Google's own 3.1 Pro on this index (real anomaly, not a typo)",
107
102
  "source": "independent (AA)"
108
103
  },
109
104
  "cheap": {
@@ -121,13 +116,13 @@
121
116
  "XAI_API_KEY",
122
117
  "GROK_API_KEY"
123
118
  ],
124
- "note": "xAI's live lineup has no distinct budget SKU below Grok 4.3 as of 2026-07-15 (the 'grok-4.1-fast' I first wrote does not exist in the live catalog — the verify gate caught it). A Grok shop's cheap tasks route to Grok 4.3 or a cross-provider value pick.",
119
+ "note": "xAI's live lineup has no distinct budget SKU below Grok 4.3 as of 2026-07-15 (the 'grok-4.1-fast' I first wrote does not exist in the live catalog \u2014 the verify gate caught it). A Grok shop's cheap tasks route to Grok 4.3 or a cross-provider value pick.",
125
120
  "frontier": {
126
121
  "model": "x-ai/grok-4.5",
127
122
  "in": 2,
128
123
  "out": 6,
129
124
  "released": "2026-07-08",
130
- "rank": "AA Intelligence 54 (#8) at ~1/3 Opus 4.7's blended price — the frontier VALUE standout; 'Opus-class, faster, cheaper' (Musk)",
125
+ "rank": "AA Intelligence 54 (#8) at ~1/3 Opus 4.7's blended price \u2014 the frontier VALUE standout; 'Opus-class, faster, cheaper' (Musk)",
131
126
  "source": "independent (AA)"
132
127
  },
133
128
  "mid": {
@@ -135,7 +130,7 @@
135
130
  "in": 1.25,
136
131
  "out": 2.5,
137
132
  "released": "2026-04-30",
138
- "rank": "AA 38 — also xAI's cheapest verified stable tier",
133
+ "rank": "AA 38 \u2014 also xAI's cheapest verified stable tier",
139
134
  "source": "independent (AA)"
140
135
  }
141
136
  }