ruvnet-brain 4.0.1 → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (185) hide show
  1. package/.claude-plugin/marketplace.json +1 -0
  2. package/README.md +4 -4
  3. package/bin/install.mjs +100 -5
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  25. package/package.json +8 -22
  26. package/plugin/.claude-plugin/marketplace.json +1 -0
  27. package/plugin/.claude-plugin/plugin.json +2 -3
  28. package/plugin/.codex-plugin/plugin.json +1 -1
  29. package/plugin/commands/brain-console.md +2 -2
  30. package/plugin/commands/configure.md +3 -2
  31. package/plugin/commands/rvbc.md +4 -3
  32. package/plugin/commands/rvcb.md +2 -2
  33. package/plugin/hooks/hooks.json +1 -2
  34. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  35. package/plugin/mcp/server.mjs +21 -0
  36. package/plugin/scripts/detach.mjs +14 -0
  37. package/plugin/scripts/first-session-worker.mjs +38 -0
  38. package/plugin/scripts/ground-ruvnet.sh +16 -6
  39. package/plugin/scripts/hook-shim.mjs +7 -7
  40. package/plugin/scripts/learn-capture.sh +22 -3
  41. package/plugin/scripts/learn-flush.mjs +21 -4
  42. package/plugin/scripts/runtime-preferences.mjs +269 -0
  43. package/plugin/scripts/session-start-core.mjs +477 -0
  44. package/plugin/scripts/session-start.sh +3 -858
  45. package/plugin/skills/brain-console/SKILL.md +4 -2
  46. package/plugin/skills/release-proof/SKILL.md +81 -0
  47. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  48. package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
  49. package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
  50. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
  51. package/plugin/skills/rvbc/SKILL.md +9 -6
  52. package/scripts/adr-backfill.mjs +107 -0
  53. package/scripts/advocacy-outcomes.mjs +808 -0
  54. package/scripts/agentdb-context.mjs +216 -0
  55. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  56. package/scripts/ascii-drift.mjs +236 -0
  57. package/scripts/behavioral-l1-l4.mjs +210 -0
  58. package/scripts/brain-capability-check.mjs +72 -0
  59. package/scripts/brain-grade-groundtruth.mjs +100 -0
  60. package/scripts/brain-latency-50.mjs +227 -0
  61. package/scripts/brain-novice-50.mjs +189 -0
  62. package/scripts/brain-stamp.mjs +94 -0
  63. package/scripts/brain-state.mjs +212 -0
  64. package/scripts/build-bundle.mjs +522 -0
  65. package/scripts/build-concepts.mjs +132 -0
  66. package/scripts/build-l2.mjs +71 -0
  67. package/scripts/build-primer.mjs +73 -0
  68. package/scripts/build-symbols.mjs +68 -0
  69. package/scripts/calibrate-router.mjs +97 -0
  70. package/scripts/capability-audit.mjs +321 -0
  71. package/scripts/capability-registry.mjs +876 -0
  72. package/scripts/check-indexation.mjs +108 -0
  73. package/scripts/check-legibility.mjs +189 -0
  74. package/scripts/ci/build-fixture-kb.mjs +67 -0
  75. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  76. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  77. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  78. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  79. package/scripts/ci/stranger-scenario.mjs +228 -0
  80. package/scripts/ci/stranger-timeout.mjs +25 -0
  81. package/scripts/ci-verdict.mjs +29 -0
  82. package/scripts/claims-verify.mjs +710 -0
  83. package/scripts/clear-claude-tmp.sh +31 -0
  84. package/scripts/console-engine.mjs +434 -0
  85. package/scripts/console-engine.test.mjs +125 -0
  86. package/scripts/corpus-qa.mjs +250 -0
  87. package/scripts/correction-detect-embed.mjs +346 -0
  88. package/scripts/correction-detect-measure.mjs +270 -0
  89. package/scripts/correction-detect.mjs +686 -0
  90. package/scripts/count-chunks.mjs +54 -0
  91. package/scripts/described-questions.json +30 -0
  92. package/scripts/design-grade.mjs +58 -0
  93. package/scripts/dev-plugin-link.sh +105 -0
  94. package/scripts/distill-project.mjs +200 -0
  95. package/scripts/doc-currency.mjs +801 -0
  96. package/scripts/eval-brain.mjs +244 -0
  97. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  98. package/scripts/full-hints.mjs +87 -0
  99. package/scripts/gate.sh +39 -0
  100. package/scripts/gates.mjs +146 -0
  101. package/scripts/gen-console-images.mjs +54 -0
  102. package/scripts/gen-images.mjs +47 -0
  103. package/scripts/git-clone-refresh.mjs +52 -0
  104. package/scripts/git-hooks/pre-push +126 -0
  105. package/scripts/goal-match.mjs +398 -0
  106. package/scripts/goldie-research.mjs +223 -0
  107. package/scripts/goldie-weekly.sh +67 -0
  108. package/scripts/health-repair.mjs +250 -0
  109. package/scripts/helix-scenario-questions.json +10 -0
  110. package/scripts/ingest-gists.mjs +230 -0
  111. package/scripts/ingest-meeting.mjs +115 -0
  112. package/scripts/ingest-repo.mjs +79 -0
  113. package/scripts/install-npx-witness.sh +49 -0
  114. package/scripts/issue-fix.mjs +639 -0
  115. package/scripts/issue-watch.mjs +276 -0
  116. package/scripts/issue4-close-note.md +31 -0
  117. package/scripts/key-canary.mjs +91 -0
  118. package/scripts/latency-to-surface.mjs +233 -0
  119. package/scripts/learning-enable.mjs +380 -0
  120. package/scripts/learning-replay.mjs +1570 -0
  121. package/scripts/learnings.mjs +62 -0
  122. package/scripts/lesson-gate.mjs +680 -0
  123. package/scripts/lesson-lifecycle.mjs +449 -0
  124. package/scripts/lesson-promote.mjs +262 -0
  125. package/scripts/lesson-ratify.mjs +98 -0
  126. package/scripts/lesson-seed.mjs +252 -0
  127. package/scripts/lesson-store.mjs +447 -0
  128. package/scripts/loop-checkpoint.mjs +86 -0
  129. package/scripts/memdb-health.sh +14 -0
  130. package/scripts/memory-doctor.mjs +271 -0
  131. package/scripts/model-catalog.mjs +79 -0
  132. package/scripts/nightly-controller.mjs +66 -0
  133. package/scripts/nightly-gists.sh +72 -0
  134. package/scripts/nightly-wrapper.sh +180 -0
  135. package/scripts/notify.sh +12 -0
  136. package/scripts/npx-witness.sh +56 -0
  137. package/scripts/onboarding-console.mjs +2749 -0
  138. package/scripts/private-fence.mjs +69 -0
  139. package/scripts/proactivity-metrics.mjs +118 -0
  140. package/scripts/proof-questions.json +56 -0
  141. package/scripts/prove.mjs +95 -0
  142. package/scripts/proxy/claude-proxied.sh +57 -0
  143. package/scripts/proxy/proxy-revert.sh +59 -0
  144. package/scripts/proxy/proxy-up.sh +60 -0
  145. package/scripts/proxy/proxy-verify.mjs +142 -0
  146. package/scripts/published-surface-probe.mjs +241 -0
  147. package/scripts/qe/card-lane-gate.mjs +162 -0
  148. package/scripts/qe/session-start-gate.mjs +229 -0
  149. package/scripts/qe/ux-suite.mjs +323 -0
  150. package/scripts/reconcile-project.mjs +0 -0
  151. package/scripts/record-lesson.mjs +113 -0
  152. package/scripts/refresh-model-catalog.mjs +99 -0
  153. package/scripts/release-proof.mjs +9 -0
  154. package/scripts/release-vector.mjs +281 -0
  155. package/scripts/release.mjs +395 -0
  156. package/scripts/remedy-registry.mjs +247 -0
  157. package/scripts/rerank-cap-eval.mjs +265 -0
  158. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  159. package/scripts/route-cheap.mjs +20 -15
  160. package/scripts/router-utilization.mjs +182 -0
  161. package/scripts/routing-flywheel.mjs +596 -0
  162. package/scripts/rvf-generation.mjs +104 -0
  163. package/scripts/rvf-index-audit.mjs +138 -0
  164. package/scripts/self-update.mjs +508 -0
  165. package/scripts/selfcheck.mjs +7 -1
  166. package/scripts/sign-bundle.mjs +69 -0
  167. package/scripts/signal-watch.mjs +171 -0
  168. package/scripts/stack-sync.mjs +469 -0
  169. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  170. package/scripts/stamp-sweep.mjs +144 -0
  171. package/scripts/status-honesty.mjs +102 -0
  172. package/scripts/sync-version.mjs +217 -0
  173. package/scripts/token-report.mjs +102 -0
  174. package/scripts/top100-benchmark.mjs +479 -0
  175. package/scripts/top100-corpus.mjs +112 -0
  176. package/scripts/top100-semantic-assertions.mjs +449 -0
  177. package/scripts/update-apply.mjs +9 -0
  178. package/scripts/upgrade-notice.mjs +14 -0
  179. package/scripts/verify-bundle.mjs +51 -0
  180. package/scripts/verify-channels.mjs +184 -0
  181. package/scripts/verify-model-catalog.mjs +104 -0
  182. package/scripts/verify-nightly-close-issue4.sh +31 -0
  183. package/scripts/version.mjs +40 -0
  184. package/scripts/wired-check.mjs +864 -0
  185. package/plugin/scripts/finalize-token-meter.mjs +0 -25
@@ -0,0 +1,398 @@
1
+ // goal-match.mjs — L4 ANTICIPATORY: infer the GOAL, name the capability that serves it, and
2
+ // otherwise SAY NOTHING.
3
+ //
4
+ // PURE. No I/O, no network, no filesystem, no process.exit — same discipline as console-engine.mjs,
5
+ // and for the same reason: this file only DECIDES. It is a total function of (prompt, capabilities),
6
+ // so it is testable by table and can never, by construction, change the machine.
7
+ //
8
+ // ── THE CONSTRAINT THAT IS THE ENTIRE DESIGN ────────────────────────────────────────────────────
9
+ //
10
+ // ADR-027 (the constraint that keeps it honest) and ADR-028 (anti-goals) say the same thing twice,
11
+ // because it is the failure mode that kills this feature:
12
+ //
13
+ // "This is goal-aware capability matching, NOT evangelism. Recommending a tool to someone whose
14
+ // problem it does not fit is the same failure in the opposite direction, and it is the FASTER
15
+ // way to destroy trust, because it is indistinguishable from salesmanship."
16
+ //
17
+ // ADR-028's metric table makes it numeric and non-negotiable: false-alarm rate target is **0**,
18
+ // annotated "one false alarm costs more trust than ten true ones earn." Recall's target is 0.80 —
19
+ // deliberately lower. **The asymmetry is the specification.** This module is therefore built to
20
+ // return [] and treats every match as something it must earn. If you are ever choosing between a
21
+ // miss and a false positive here, take the miss; that choice is already made, in the ADR, on
22
+ // purpose.
23
+ //
24
+ // ── WHY KEYWORD MATCHING ALONE WOULD SHIP A LIAR ────────────────────────────────────────────────
25
+ //
26
+ // Look at the actual vocabulary of the eleven capabilities in capability-registry.auditAll():
27
+ // memory, hooks, routing, sessions, context, patterns, gates, cache, nightly, learning. **Every
28
+ // single one of those words is a homonym for something in ordinary software work**, and the other
29
+ // meaning is far more common in a developer's prompt:
30
+ //
31
+ // "fix the memory leak" → C heap, not AgentDB
32
+ // "my useEffect hook fires 2x" → React, not ruflo hooks
33
+ // "set up routing for /admin" → a router, not cheap-model routing
34
+ // "the session cookie expires" → HTTP, not a Claude session
35
+ // "our nightly build failed" → CI, not the KB refresh
36
+ // "the model isn't learning" → gradient descent, not workflow learning
37
+ //
38
+ // A bag-of-words detector fires on all six and is wrong all six times. That is not a hypothetical:
39
+ // this project already shipped exactly this bug in a different costume — a detector that read a
40
+ // CLI's human-readable table and announced "26 hooks off" while the learner held 457 trajectories.
41
+ // It matched a surface pattern and reported the match as a fact.
42
+ //
43
+ // So a goal here requires TWO INDEPENDENT KEYS, and one alone is never enough:
44
+ //
45
+ // 1. INTENT — the prompt describes the problem the capability actually solves.
46
+ // 2. SUBJECT — the thing being discussed is the user's AI/agent workflow, not their application.
47
+ //
48
+ // plus a VETO list that only ever contains disambiguators for words this file genuinely uses. A
49
+ // veto for a word we never match on would be superstition, not engineering.
50
+ //
51
+ // The two-key rule is what makes "fix the memory leak in my C++ parser" silent: INTENT plausibly
52
+ // matches, SUBJECT does not match at all, and the veto catches it a second time. Belt and braces,
53
+ // because the cost of being wrong here is measured in trust rather than in a stack trace.
54
+
55
+ /**
56
+ * SUBJECT — evidence that the conversation is about the user's AI assistant and how it works,
57
+ * rather than about code the user is writing.
58
+ *
59
+ * This is the load-bearing half of the two-key rule. Every entry names the ASSISTANT or the
60
+ * ASSISTANT'S WORKFLOW explicitly. Nothing here can be satisfied by a prompt about a web app, which
61
+ * is the property that produces silence on the negative table.
62
+ */
63
+ const SUBJECT = Object.freeze([
64
+ /\bclaude\b/,
65
+ /\b(my|the|this) (ai|assistant|agent|llm|copilot)\b/,
66
+ /\b(coding|ai) (agent|assistant)\b/,
67
+ /\bcursor\b/,
68
+ /\bruflo\b/, /\bruvnet\b/, /\bruvector\b/, /\bagentdb\b/, /\bmcp\b/,
69
+ /\bclaude[ .-]?md\b/,
70
+ /\b(every|each|new|another) (chat|conversation|session)\b/,
71
+ /\bacross (sessions|chats|conversations|projects)\b/,
72
+ /\bcontext window\b/,
73
+ /\bcompact(s|ed|ion|ing)?\b/,
74
+ /\b(my|our) (workflow|setup|harness|stack|tooling)\b/,
75
+ /\bit keeps\b/, // "it keeps forgetting" — the assistant, in the user's own voice
76
+ /\bit forgets\b/,
77
+ /\bit never\b/,
78
+ /\bit doesn'?t\b/,
79
+ ]);
80
+
81
+ /**
82
+ * GLOBAL_VETO — the prompt is about building or operating SOFTWARE, so our whole vocabulary is
83
+ * being used in its other sense.
84
+ *
85
+ * DELIBERATELY OVER-BROAD. Some of these ("deploy", "production") could appear in a prompt that was
86
+ * genuinely about the user's AI workflow, and vetoing it costs us a true positive. That trade is
87
+ * made knowingly and in one direction only, per ADR-028: recall 0.80, false alarms 0. A miss is
88
+ * invisible. A false alarm reads as salesmanship, and you only get to do that once.
89
+ */
90
+ const GLOBAL_VETO = Object.freeze([
91
+ // — the assistant's words used as an application's words —
92
+ /\bmemory leak\b/, /\bheap\b/, /\bmalloc\b/, /\bvalgrind\b/, /\bgarbage collect/, /\boom\b/, /\bram\b/,
93
+ /\buse(effect|state|context|memo|callback|ref)\b/, /\breact hook/, /\bcustom hook/, /\blifecycle hook/,
94
+ /\b(react|next|vue|express|api) rout/, /\brouter\b/, /\b\/api\//, /\bendpoint\b/, /\bmiddleware\b/,
95
+ /\bsession (cookie|token|id|storage)\b/, /\bjwt\b/, /\bexpress-session\b/, /\bcookie\b/,
96
+ // Auth owns the word "session" at least as strongly as we do. These began as a veto private to
97
+ // the losing-work goal, and a test proved that scoping WRONG: suppressing one goal simply handed
98
+ // "forgets the login state and starts over from scratch" to a different goal, which recommended
99
+ // the learning capabilities instead. A prompt about authentication is application work outright,
100
+ // so the disambiguation belongs here, once, globally — not per-goal, where it silences one
101
+ // claimant and leaves ten others holding the same bad match.
102
+ /\blogin\b/, /\bauth(entication)?\b/, /\bsign ?(in|out)\b/, /\blogged (in|out)\b/,
103
+ /\bnightly (build|release)\b/, /\bci (pipeline|gate|job)\b/, /\bgithub actions\b/, /\bquality gate\b/,
104
+ /\bcache-control\b/, /\bhttp cache\b/, /\bcdn\b/, /\bredis\b/, /\bmemcached\b/,
105
+ /\bnpm outdated\b/, /\bdependabot\b/, /\bdependenc(y|ies)\b/,
106
+ /\bregex(p)? pattern\b/, /\bdesign pattern\b/,
107
+ // — machine learning, which owns "learn", "train", "model" and "pattern" outright —
108
+ /\btraining (loss|data|set)\b/, /\boverfit/, /\bepochs?\b/, /\bgradient\b/, /\bhyperparameter/,
109
+ /\bpytorch\b/, /\btensorflow\b/, /\bneural net/, /\bdataset\b/, /\bfine-?tun/,
110
+ // — building AGAINST an AI API is app work, not workflow work; "claude" appears in both —
111
+ /\bsdk\b/, /\bapi key\b/, /\brate limit/, /\b429\b/, /\bmy app\b/, /\bproduction\b/,
112
+ /\bend users\b/, /\bcustomers\b/, /\bdeploy(ing|ment)?\b/,
113
+ ]);
114
+
115
+ /**
116
+ * GOALS — the closed taxonomy.
117
+ *
118
+ * EVERY goal here is reverse-engineered from a real capability's own `whatItBuysYou` string in
119
+ * capability-registry.auditAll(). None was invented from a sense of what would be nice to detect.
120
+ * That direction of derivation is the point: a goal no shipped capability serves is a goal whose
121
+ * only possible outcome is a recommendation we cannot fulfil, which is the salesmanship failure
122
+ * arriving by a side door.
123
+ *
124
+ * `serves` holds capability KEYS, not labels — labels are prose for humans and drift; keys are the
125
+ * registry's identity.
126
+ */
127
+ export const GOALS = Object.freeze([
128
+ {
129
+ id: 'reteaching-every-project',
130
+ because: 'You are re-teaching the same rule in project after project',
131
+ serves: ['cross-project-lessons'],
132
+ intent: [
133
+ /\bre-?(explain|teach|state|specify)/,
134
+ /\b(every|each|another|a new) (new )?(project|repo|repository|codebase)\b/,
135
+ /\bsame (rule|standard|convention|instruction|correction|preference|guideline)/,
136
+ /\bkeep (telling|reminding|explaining)/,
137
+ /\b(over and over|again and again)\b/,
138
+ /\bproject by project\b/,
139
+ ],
140
+ },
141
+ {
142
+ id: 'corrections-not-obeyed',
143
+ because: 'You are correcting the same behaviour more than once',
144
+ serves: ['lessons-in-force'],
145
+ intent: [
146
+ /\bignor(es|ing|ed) (my|the|these)\b/,
147
+ /\balready (told|asked|corrected)\b/,
148
+ /\bsame mistake\b/,
149
+ /\bkeeps? (doing|making|repeating)\b/,
150
+ /\b(won'?t|doesn'?t|does not) (follow|listen|respect|obey)\b/,
151
+ /\bcorrected (it|this|that) (again|twice|three times|\d+ times)\b/,
152
+ ],
153
+ },
154
+ {
155
+ id: 'losing-work-between-sessions',
156
+ because: 'Work you established in one session is not surviving into the next',
157
+ serves: ['session-capture'],
158
+ intent: [
159
+ /\bforget(s|ting)?\b/,
160
+ /\bdoesn'?t remember\b/,
161
+ /\blos(e|es|ing|t) (the |all )?(context|thread|history|everything)\b/,
162
+ /\bstart(s|ing)? (over|from scratch)\b/,
163
+ /\bafter (a |the )?(compact|restart)/,
164
+ /\bwhen the (session|conversation) ends\b/,
165
+ ],
166
+ },
167
+ {
168
+ id: 'spend-too-high',
169
+ because: 'You are paying top-tier model prices for work that does not need them',
170
+ serves: ['cheap-model-routing'],
171
+ intent: [
172
+ /\b(bill|costs?|spend(ing)?|expensive|pricey)\b/,
173
+ /\btoken (spend|usage|burn)\b/,
174
+ /\bcheaper model\b/,
175
+ /\bsave money\b/,
176
+ /\bburning (through )?(credits|tokens|cash)\b/,
177
+ /\bhow much (am i|i'?m) (paying|spending)\b/,
178
+ ],
179
+ },
180
+ {
181
+ id: 'notes-that-teach-nothing',
182
+ because: 'You have accumulated notes and history that are never actually reused',
183
+ serves: ['memory-distillation'],
184
+ intent: [
185
+ /\bdistill/,
186
+ /\breusable patterns?\b/,
187
+ /\bnever (recalls?|reuses?|surfaces?)\b/,
188
+ /\b(notes|memories|decisions) (are )?(just )?(sitting|piling|pile)/,
189
+ /\bdoesn'?t (use|recall|remember) (my|the|past|previous|old)\b/,
190
+ /\bpast (sessions|work|decisions|notes)\b/,
191
+ ],
192
+ },
193
+ {
194
+ id: 'resolving-the-same-problem',
195
+ because: 'You are solving the same problem from scratch instead of building on what worked',
196
+ serves: ['learning-hooks', 'workflow-pattern-learning'],
197
+ intent: [
198
+ /\bsolv(e|es|ed|ing) the same\b/,
199
+ /\bfrom scratch (every|each)\b/,
200
+ /\b(doesn'?t|does not|never) (learn|improve|get better)\b/,
201
+ /\bsame (problem|approach|bug|issue) (again|every)\b/,
202
+ /\breinvent/,
203
+ /\bwhat (worked|we did) last time\b/,
204
+ ],
205
+ },
206
+ {
207
+ id: 'catching-bad-writes-in-review',
208
+ because: 'You are catching in review what should have been refused at write time',
209
+ serves: ['write-gates'],
210
+ intent: [
211
+ /\bcatch(ing)? (it|them|this|that) in review\b/,
212
+ /\b(stop|prevent|block) (it|claude|the agent|the ai) (from )?writ/,
213
+ /\bkeeps? writing\b/,
214
+ /\bguard ?rails?\b/,
215
+ /\benforce (a|my|our|the) (rule|standard|convention|policy)\b/,
216
+ /\bshould have (been )?(refused|blocked|stopped)\b/,
217
+ ],
218
+ },
219
+ {
220
+ id: 'needs-an-external-service',
221
+ because: 'You want your AI to reach a service it currently cannot see',
222
+ serves: ['mcp-servers'],
223
+ intent: [
224
+ /\bconnect (claude|it|the ai|the agent) to\b/,
225
+ /\bhook (it|claude) up to\b/,
226
+ /\b(access|read|see) my (notion|gmail|email|calendar|drive|slack|linear|jira|figma|notes)\b/,
227
+ /\bcan'?t (see|reach|access) (my|our)\b/,
228
+ /\bgive (it|claude) access to\b/,
229
+ ],
230
+ },
231
+ {
232
+ id: 'knowledge-going-stale',
233
+ because: 'What your AI knows about your tools is drifting out of date',
234
+ serves: ['nightly-refresh'],
235
+ intent: [
236
+ /\b(stale|outdated|out of date)\b/,
237
+ /\bdoesn'?t know about the (new|latest)\b/,
238
+ /\bold version of\b/,
239
+ /\bknowledge ?base\b/,
240
+ /\bkeeps? (citing|using) the old\b/,
241
+ ],
242
+ },
243
+ {
244
+ id: 'tuning-the-harness-itself',
245
+ because: 'You are trying to work out which set of rules actually performs better',
246
+ serves: ['harness-evolution'],
247
+ intent: [
248
+ /\bimprove (my|the) (prompt|rules|instructions|harness|setup)\b/,
249
+ /\bwhich (prompt|rule|policy|version) (works|performs) better\b/,
250
+ /\ba\/b test/,
251
+ /\btune (my|the) (rules|instructions|prompt)\b/,
252
+ /\bmeasure (which|whether) .{0,20}(prompt|rule)/,
253
+ ],
254
+ },
255
+ ]);
256
+
257
+ /**
258
+ * The floor, and the reason it sits where it does.
259
+ *
260
+ * BASE is deliberately BELOW the floor. That single fact encodes the rule that matters: one intent
261
+ * cue plus one subject cue scores 0.55 and is therefore SILENT. A goal must be CORROBORATED — a
262
+ * second intent cue or a second subject cue — before this module will say anything at all.
263
+ *
264
+ * One cue is a coincidence. Two is a statement.
265
+ */
266
+ export const CONFIDENCE_FLOOR = 0.6;
267
+ const BASE = 0.55;
268
+ const PER_EXTRA_CUE = 0.12;
269
+ const MAX_EXTRA_CUES = 2;
270
+
271
+ /**
272
+ * An 'unknown' capability is discounted, and this is the honesty rule from capability-registry's own
273
+ * header made arithmetic: "'unknown' is a first-class state, and it outranks 'off' every single time
274
+ * a probe could not run."
275
+ *
276
+ * We do not KNOW an unknown capability is dormant. Surfacing one is a guess about the machine on top
277
+ * of a guess about the goal, so it must clear a higher bar of evidence — 0.85 pushes the minimum
278
+ * corroborated score (0.67) back below the floor, meaning an unknown capability needs strictly more
279
+ * than an 'off' one before it may be named. The `why` string never calls it "off" either.
280
+ */
281
+ const UNKNOWN_DISCOUNT = 0.85;
282
+
283
+ /**
284
+ * At most two. The nudge principle ("correct, clear, confident, proactive, deferential, never
285
+ * pushy") does not survive a bulleted list of five things the user should turn on — that reads as a
286
+ * pitch no matter how each line is worded. Anticipation that dumps its whole inventory is evangelism
287
+ * with better targeting.
288
+ */
289
+ const MAX_RESULTS = 2;
290
+
291
+ const hits = (text, patterns) => patterns.reduce((n, re) => (re.test(text) ? n + 1 : n), 0);
292
+ const any = (text, patterns) => patterns.some((re) => re.test(text));
293
+
294
+ /**
295
+ * Which goals the prompt supports, with a score. Exported for testing and introspection: a scorer
296
+ * that cannot be examined in isolation is a scorer whose threshold nobody can defend.
297
+ *
298
+ * @param {string} promptText
299
+ * @returns {Array<{goal: object, confidence: number, intentHits: number, subjectHits: number}>}
300
+ */
301
+ export function classifyGoals(promptText) {
302
+ if (typeof promptText !== 'string' || !promptText.trim()) return [];
303
+ const text = promptText.toLowerCase();
304
+
305
+ // Veto first, and veto globally. If the prompt is about software rather than about the assistant,
306
+ // nothing below can rescue it and nothing below should get the chance to try.
307
+ if (any(text, GLOBAL_VETO)) return [];
308
+
309
+ const subjectHits = hits(text, SUBJECT);
310
+ if (subjectHits === 0) return []; // key 2 absent ⇒ every goal fails, no exceptions
311
+
312
+ const out = [];
313
+ for (const goal of GOALS) {
314
+ const intentHits = hits(text, goal.intent);
315
+ if (intentHits === 0) continue; // key 1 absent
316
+
317
+ const extra = Math.min(intentHits - 1, MAX_EXTRA_CUES) + Math.min(subjectHits - 1, MAX_EXTRA_CUES);
318
+ const confidence = Math.min(0.95, BASE + extra * PER_EXTRA_CUE);
319
+ out.push({ goal, confidence: +confidence.toFixed(4), intentHits, subjectHits });
320
+ }
321
+ return out.sort((a, b) => b.confidence - a.confidence);
322
+ }
323
+
324
+ /**
325
+ * matchGoal — the L4 surface.
326
+ *
327
+ * Given what the user says they are trying to do, name the capability that serves THAT goal, and
328
+ * only if it is not already serving them. Returns [] far more often than not; that is the feature.
329
+ *
330
+ * Three independent conditions must ALL hold before a single row comes back:
331
+ * 1. the prompt clears the two-key test and the vetoes (classifyGoals),
332
+ * 2. a capability the goal actually names is present in the audit, and is off or unknown,
333
+ * 3. the resulting confidence clears CONFIDENCE_FLOOR after any state discount.
334
+ *
335
+ * @param {string} promptText what the user said they are trying to do
336
+ * @param {Array} capabilities rows from capability-registry.auditAll()
337
+ * @returns {Array<{capability: object, why: string, confidence: number, goal: string}>}
338
+ */
339
+ export function matchGoal(promptText, capabilities) {
340
+ if (!Array.isArray(capabilities) || capabilities.length === 0) return [];
341
+
342
+ const scored = classifyGoals(promptText);
343
+ if (scored.length === 0) return [];
344
+
345
+ const byKey = new Map();
346
+ for (const c of capabilities) if (c && typeof c.key === 'string') byKey.set(c.key, c);
347
+
348
+ // Best row per capability. Two goals can legitimately point at the same capability; the user is
349
+ // owed one sentence about it, from whichever goal explains it best — not the same suggestion twice
350
+ // wearing different rationales.
351
+ const best = new Map();
352
+
353
+ for (const { goal, confidence } of scored) {
354
+ for (const key of goal.serves) {
355
+ const cap = byKey.get(key);
356
+ if (!cap) continue;
357
+
358
+ // Never advocate for something already working. A recommendation to switch on what is already
359
+ // on is not merely useless — it proves to the reader that we did not look, and every other
360
+ // claim we make is downgraded accordingly.
361
+ const state = cap.state;
362
+ if (state !== 'off' && state !== 'unknown') continue;
363
+
364
+ const adjusted = +(confidence * (state === 'unknown' ? UNKNOWN_DISCOUNT : 1)).toFixed(4);
365
+ if (adjusted < CONFIDENCE_FLOOR) continue;
366
+
367
+ const prior = best.get(key);
368
+ if (prior && prior.confidence >= adjusted) continue;
369
+ best.set(key, { capability: cap, why: explain(goal, cap), confidence: adjusted, goal: goal.id });
370
+ }
371
+ }
372
+
373
+ return [...best.values()]
374
+ .sort((a, b) => b.confidence - a.confidence)
375
+ .slice(0, MAX_RESULTS);
376
+ }
377
+
378
+ /**
379
+ * The sentence the user reads.
380
+ *
381
+ * Every clause is DERIVED — the goal's own `because`, the registry's own `label`, the registry's own
382
+ * `whatItBuysYou`, and the registry's own `evidence` string verbatim. Nothing here is written to be
383
+ * persuasive, because the moment this function starts generating copy it starts generating claims,
384
+ * and a claim about the machine that did not come from a probe is the thing this repo exists to
385
+ * refuse.
386
+ *
387
+ * The state clause is the specific guard: an 'unknown' capability is described as unreadable, never
388
+ * as off. Reporting "off" for something we failed to measure is the exact defect that shipped as
389
+ * "26 hooks off" against a learner holding 457 trajectories.
390
+ */
391
+ function explain(goal, cap) {
392
+ const stateClause = cap.state === 'unknown'
393
+ ? 'whether this is already switched on could not be read on this machine'
394
+ : 'it is switched off';
395
+ const evidence = cap.evidence ? ` — ${cap.evidence}` : '';
396
+ return `${goal.because}. ${cap.label} is the part of your stack that serves that: `
397
+ + `${cap.whatItBuysYou} Right now ${stateClause}${evidence}.`;
398
+ }
@@ -0,0 +1,223 @@
1
+ #!/usr/bin/env node
2
+ // goldie-research.mjs — Goldie's DETERMINISTIC core: keep the model-router catalog verifiably fresh.
3
+ //
4
+ // Stuart's standing mandate (2026-07-12): the router must never run on stale beliefs about the model
5
+ // landscape. Goldie runs WEEKLY (scripts/goldie-weekly.sh via launchd) and answers, with live data:
6
+ // • what do the router's candidate models ACTUALLY cost right now (OpenRouter /models, public API)?
7
+ // • did any price drift >20% since the catalog was last verified (ruflo ADR-149's own re-measure
8
+ // trigger — that threshold is rUv's, not invented here)?
9
+ // • which Codex tiers exist on this machine right now (~/.codex/models_cache.json, fetched live
10
+ // by Codex itself)?
11
+ // • which cheap, tool-capable OpenRouter models exist that we DON'T track (the "worth a look" radar)?
12
+ //
13
+ // It updates ONLY prices + verified-stamps in ~/.claude/model-router/catalog.json. It NEVER adds an
14
+ // execution path (harness/subscription fields) — the catalog's own rule: an engine that "chooses" a
15
+ // model it can't run is worse than one that doesn't know it exists. New-model adoption and bucket
16
+ // taxonomy are JUDGMENT calls: goldie-weekly.sh layers a headless-Claude research pass on top, and
17
+ // its output lands in the same brief as a PROPOSAL for a human/session to apply.
18
+ //
19
+ // Output: ~/.claude/model-router/goldie/YYYY-MM-DD.md (the brief) + updated catalog.json.
20
+ // Exit 0 = brief written; exit 1 = could not produce a brief (wrapper alerts loudly — no silent death).
21
+
22
+ import fs from 'node:fs';
23
+ import path from 'node:path';
24
+ import os from 'node:os';
25
+
26
+ const CONFIG_DIR = path.join(os.homedir(), '.claude', 'model-router');
27
+ const CATALOG = path.join(CONFIG_DIR, 'catalog.json');
28
+ const GOLDIE_DIR = path.join(CONFIG_DIR, 'goldie');
29
+ const DECISIONS = path.join(os.homedir(), '.claude', 'metaharness', 'routing-decisions.jsonl');
30
+ const TODAY = new Date().toISOString().slice(0, 10);
31
+ const DRIFT_THRESHOLD = 0.20; // ADR-149 open question #3: re-measure when pricing moves >20%
32
+
33
+ async function fetchOpenRouterModels() {
34
+ const res = await fetch('https://openrouter.ai/api/v1/models', { signal: AbortSignal.timeout(20000) });
35
+ if (!res.ok) throw new Error(`OpenRouter /models HTTP ${res.status}`);
36
+ const j = await res.json();
37
+ if (!Array.isArray(j.data) || !j.data.length) throw new Error('OpenRouter /models returned no data');
38
+ return j.data;
39
+ }
40
+
41
+ // OpenRouter pricing is $/token as strings; catalog speaks $/MTok.
42
+ const perMTok = (v) => (v == null || isNaN(+v) ? null : +(+v * 1e6).toFixed(4));
43
+
44
+ function refreshCatalogPrices(catalog, orModels) {
45
+ const byId = new Map(orModels.map((m) => [m.id, m]));
46
+ const changes = [];
47
+ for (const c of catalog.candidates) {
48
+ if (c.provider !== 'openrouter') continue;
49
+ const live = byId.get(c.id);
50
+ if (!live) { changes.push({ id: c.id, note: 'NOT FOUND on OpenRouter anymore — investigate before next route' }); continue; }
51
+ const fresh = { in: perMTok(live.pricing?.prompt), out: perMTok(live.pricing?.completion) };
52
+ if (fresh.in == null || fresh.out == null) { changes.push({ id: c.id, note: 'listed but pricing unparsable — left as-is' }); continue; }
53
+ const old = c.costPerMTok;
54
+ if (old && typeof old.out === 'number') {
55
+ const drift = Math.max(Math.abs(fresh.in - old.in) / old.in, Math.abs(fresh.out - old.out) / old.out);
56
+ if (drift > DRIFT_THRESHOLD) changes.push({ id: c.id, note: `PRICE DRIFT ${(drift * 100).toFixed(0)}%: in $${old.in}->$${fresh.in}, out $${old.out}->$${fresh.out} /MTok (>${DRIFT_THRESHOLD * 100}% — ADR-149 says re-measure quality/cost now)` });
57
+ else if (fresh.in !== old.in || fresh.out !== old.out) changes.push({ id: c.id, note: `price updated: in $${old.in}->$${fresh.in}, out $${old.out}->$${fresh.out} /MTok` });
58
+ }
59
+ c.costPerMTok = fresh;
60
+ c.verified = `${TODAY} OpenRouter API (goldie)`;
61
+ }
62
+ catalog.updated = TODAY;
63
+ return changes;
64
+ }
65
+
66
+ // The radar: cheap, tool-capable models we don't track. Tool support matters because agentic work
67
+ // (the router's whole domain) is useless without it. Deterministic shortlist only — adoption is a
68
+ // judgment call for the brief's reader / the judgment layer.
69
+ function radar(catalog, orModels) {
70
+ const tracked = new Set(catalog.candidates.map((c) => c.id));
71
+ return orModels
72
+ .filter((m) => !tracked.has(m.id))
73
+ .filter((m) => (m.supported_parameters || []).includes('tools'))
74
+ .map((m) => ({ id: m.id, in: perMTok(m.pricing?.prompt), out: perMTok(m.pricing?.completion), ctx: m.context_length }))
75
+ .filter((m) => m.in != null && m.in >= 0 && m.in <= 0.5 && m.out != null && m.out >= 0) // negative = OpenRouter's dynamic-pricing sentinel, not a price
76
+ .sort((a, b) => a.in - b.in)
77
+ .slice(0, 8);
78
+ }
79
+
80
+ function codexTiers() {
81
+ try {
82
+ const cache = JSON.parse(fs.readFileSync(path.join(os.homedir(), '.codex', 'models_cache.json'), 'utf8'));
83
+ const models = cache.models || cache.data || [];
84
+ const names = models.map((m) => m.id || m.slug || m.name).filter(Boolean);
85
+ // `models` (below) is truncated to 20 for the brief's display list; `allModels` is the FULL set,
86
+ // kept separately so the gpt-5.6 reinstate check (reinstateGpt56, below) can never miss a real
87
+ // slug just because it landed past position #20 in the cache.
88
+ return { fetchedAt: cache.fetched_at || cache.fetchedAt || 'unknown', models: names.slice(0, 20), allModels: names };
89
+ } catch { return null; }
90
+ }
91
+
92
+ // GPT-5.6 auto-reinstate (Stuart's mandate, 2026-07-12): gpt-5.6-sol/terra/luna were demoted to
93
+ // landscape-only THAT SAME DAY because their "verified live" stamps were false — two independent
94
+ // reads of models_cache.json showed only gpt-5.5/5.4/5.4-mini/codex-auto-review. The demotion note
95
+ // on each entry says: reinstate ONLY when a fresh read of models_cache.json actually shows the slug.
96
+ // This is that fresh read, automated, every week Goldie runs — no more relying on a human to notice.
97
+ //
98
+ // Inverse direction (any 'codex' harness entry whose gpt-* id vanishes from this week's cache) is
99
+ // FLAG-ONLY, never auto-demote: a cache miss could be rollout timing or a transient cache staleness,
100
+ // not proof the model is gone — the same false-confidence mistake in reverse. Demotion stays a
101
+ // judgment call for a human/Goldie's judgment pass, same as the 2026-07-12 demotion itself was.
102
+ function reinstateGpt56(catalog, codex) {
103
+ const result = { reinstated: [], warnings: [] };
104
+ if (!codex) return result; // no live cache this run (codexTiers() returned null) — skip both checks silently
105
+ const live = new Set(codex.allModels);
106
+ for (const c of catalog.candidates) {
107
+ if (c.id.startsWith('gpt-5.6') && Array.isArray(c.harness) && c.harness.length === 0 && live.has(c.id)) {
108
+ c.harness = ['codex'];
109
+ c.subscription = ['codex'];
110
+ c.verified = `${TODAY} ~/.codex/models_cache.json live (goldie auto-reinstate)`;
111
+ c.note = `auto-reinstated by goldie on ${TODAY}; was demoted for false verification 2026-07-12`;
112
+ result.reinstated.push(c.id);
113
+ }
114
+ if (c.id.startsWith('gpt-') && Array.isArray(c.harness) && c.harness.includes('codex') && !live.has(c.id)) {
115
+ result.warnings.push(c.id);
116
+ }
117
+ }
118
+ return result;
119
+ }
120
+
121
+ // models.env cross-check (pure informational drift-detection between the two model registries —
122
+ // ~/.claude/models.env auto-refreshes provider pins independently of the router's own catalog.json,
123
+ // so the two can silently drift apart; this just surfaces that, it never edits either file).
124
+ function parseModelsEnv() {
125
+ try {
126
+ const lines = fs.readFileSync(path.join(os.homedir(), '.claude', 'models.env'), 'utf8').split('\n');
127
+ const pins = [];
128
+ for (const line of lines) {
129
+ const t = line.trim();
130
+ if (!t || t.startsWith('#')) continue;
131
+ const eq = t.indexOf('=');
132
+ if (eq === -1) continue;
133
+ pins.push({ key: t.slice(0, eq).trim(), value: t.slice(eq + 1).trim() });
134
+ }
135
+ return pins;
136
+ } catch { return null; }
137
+ }
138
+
139
+ // Loose match: models.env pins bare model names (e.g. 'deepseek-v4-flash') while the catalog often
140
+ // prefixes openrouter candidates with a provider ('deepseek/deepseek-v4-flash') — compare both the
141
+ // raw id and the segment after the last '/' in either direction.
142
+ function pinMatchesCatalog(pinValue, candidates) {
143
+ const norm = pinValue.toLowerCase();
144
+ return candidates.some((c) => {
145
+ const cid = c.id.toLowerCase();
146
+ const short = cid.split('/').pop();
147
+ return cid === norm || short === norm;
148
+ });
149
+ }
150
+
151
+ function modelsEnvCrossCheck(catalog) {
152
+ const pins = parseModelsEnv();
153
+ if (!pins) return null;
154
+ return pins.map((p) => ({ ...p, known: pinMatchesCatalog(p.value, catalog.candidates) }));
155
+ }
156
+
157
+ function decisionStats() {
158
+ try {
159
+ const lines = fs.readFileSync(DECISIONS, 'utf8').trim().split('\n').filter(Boolean);
160
+ const byModel = {};
161
+ for (const l of lines) { try { const d = JSON.parse(l); byModel[d.model] = (byModel[d.model] || 0) + 1; } catch { /* skip */ } }
162
+ return { total: lines.length, byModel };
163
+ } catch { return { total: 0, byModel: {} } }
164
+ }
165
+
166
+ async function main() {
167
+ const catalog = JSON.parse(fs.readFileSync(CATALOG, 'utf8'));
168
+ const orModels = await fetchOpenRouterModels();
169
+ const changes = refreshCatalogPrices(catalog, orModels);
170
+ const watch = radar(catalog, orModels);
171
+ const codex = codexTiers();
172
+ const gpt56 = reinstateGpt56(catalog, codex);
173
+ for (const id of gpt56.reinstated) changes.push({ id, note: `AUTO-REINSTATED: harness/subscription set to ['codex'] — slug now confirmed live in models_cache.json (goldie auto-reinstate)` });
174
+ const envCross = modelsEnvCrossCheck(catalog);
175
+ const stats = decisionStats();
176
+
177
+ fs.writeFileSync(CATALOG, JSON.stringify(catalog, null, 2) + '\n');
178
+ fs.mkdirSync(GOLDIE_DIR, { recursive: true });
179
+
180
+ const brief = [
181
+ `# Goldie weekly model-landscape brief — ${TODAY}`,
182
+ '',
183
+ `Live sources: OpenRouter /api/v1/models (${orModels.length} models), ~/.codex/models_cache.json, routing-decisions.jsonl.`,
184
+ '',
185
+ `## Catalog price refresh (${changes.length ? changes.length + ' change(s)' : 'no changes'})`,
186
+ ...(changes.length ? changes.map((c) => `- ${c.id}: ${c.note}`) : ['- all tracked OpenRouter prices unchanged; verified-stamps refreshed to today']),
187
+ '',
188
+ '## Radar: cheap tool-capable models we do NOT track (≤$0.50/MTok in, top 8 by input price)',
189
+ ...(watch.length ? watch.map((m) => `- ${m.id} — in $${m.in}, out $${m.out} /MTok, ctx ${m.ctx}`) : ['- none matched the filter this week']),
190
+ '',
191
+ '## Codex tiers on this machine (live cache)',
192
+ codex ? `- fetched_at: ${codex.fetchedAt}` : '- no ~/.codex/models_cache.json found',
193
+ ...(codex ? codex.models.map((m) => `- ${m}`) : []),
194
+ '',
195
+ `## GPT-5.6 auto-reinstate check (${gpt56.reinstated.length} reinstated this run)`,
196
+ ...(!codex ? ['- skipped: no live ~/.codex/models_cache.json this run'] :
197
+ gpt56.reinstated.length ? gpt56.reinstated.map((id) => `- ${id}: REINSTATED — harness/subscription set to ['codex'], now confirmed live`) :
198
+ ['- none reinstated: no gpt-5.6-* landscape entry appeared in this week\'s live cache']),
199
+ ...(gpt56.warnings.length ? gpt56.warnings.map((id) => `- WARNING: ${id} has harness:['codex'] but is NOT in this week's live models_cache.json — verify manually before trusting it (flag only, not auto-demoted)`) : []),
200
+ '',
201
+ '## models.env cross-check (informational drift-detection between the two model registries; never edits either file)',
202
+ ...(!envCross ? ['- ~/.claude/models.env not found'] : envCross.map((p) => `- ${p.key}=${p.value}${p.known ? '' : ' — NOT recognized in catalog.json (provider the router doesn\'t track, or the two registries have drifted — worth a look)'}`)),
203
+ '',
204
+ `## Router decision log so far (${stats.total} decisions)`,
205
+ ...Object.entries(stats.byModel).map(([m, n]) => `- ${m}: ${n}`),
206
+ '',
207
+ '## Standing questions for the judgment layer (goldie-weekly.sh appends its answers below)',
208
+ '1. How many BUCKETS should prompts be classified into, given this landscape? (current policy: 3-tier placeholder)',
209
+ '2. What is the best model per bucket right now, per the public evals (Artificial Analysis, LMArena, SWE-bench)?',
210
+ '3. Should any radar model be wired up (needs: route-cheap PRICING entry + catalog harness path + a measured quality check)?',
211
+ '',
212
+ ].join('\n');
213
+
214
+ const briefPath = path.join(GOLDIE_DIR, `${TODAY}.md`);
215
+ fs.writeFileSync(briefPath, brief);
216
+ console.log(`[goldie] catalog refreshed (${changes.length} change(s)); brief: ${briefPath}`);
217
+ if (changes.some((c) => c.note.startsWith('PRICE DRIFT') || c.note.startsWith('NOT FOUND'))) {
218
+ console.log('[goldie] ATTENTION: drift or delisting detected — see brief');
219
+ process.exitCode = 0; // informational; the wrapper reads the brief for the push
220
+ }
221
+ }
222
+
223
+ main().catch((e) => { console.error(`goldie-research: ${e.message}`); process.exit(1); });