ruvnet-brain 3.9.134-dev → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. package/.claude-plugin/marketplace.json +14 -0
  2. package/README.md +5 -5
  3. package/bin/install.mjs +382 -36
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/kb/zip-extract.mjs +53 -14
  25. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  26. package/package.json +14 -22
  27. package/plugin/.claude-plugin/marketplace.json +14 -0
  28. package/plugin/.claude-plugin/plugin.json +22 -0
  29. package/plugin/.codex-plugin/plugin.json +21 -0
  30. package/plugin/.mcp.json +8 -0
  31. package/plugin/commands/brain-console.md +16 -0
  32. package/plugin/commands/configure.md +33 -0
  33. package/plugin/commands/rvbc.md +79 -0
  34. package/plugin/commands/rvcb.md +16 -0
  35. package/plugin/commands/whats-new.md +57 -0
  36. package/plugin/hooks/codex-hooks.json +160 -0
  37. package/plugin/hooks/hook-contracts.json +77 -0
  38. package/plugin/hooks/hooks.json +202 -0
  39. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  40. package/plugin/mcp/server.mjs +56 -6
  41. package/plugin/scripts/anticipate.sh +534 -0
  42. package/plugin/scripts/codex-hook-adapter.mjs +96 -0
  43. package/plugin/scripts/continuation-gate.mjs +267 -0
  44. package/plugin/scripts/design-wall.sh +137 -0
  45. package/plugin/scripts/detach.mjs +182 -0
  46. package/plugin/scripts/first-session-worker.mjs +38 -0
  47. package/plugin/scripts/gate-receipt.sh +35 -0
  48. package/plugin/scripts/ground-before-write.sh +199 -0
  49. package/plugin/scripts/ground-ruvnet.sh +517 -0
  50. package/plugin/scripts/grounding-stamp.sh +113 -0
  51. package/plugin/scripts/grounding-substance.mjs +595 -0
  52. package/plugin/scripts/hijack-ruvnet.sh +81 -0
  53. package/plugin/scripts/hook-input.mjs +558 -0
  54. package/plugin/scripts/hook-shim-bash.mjs +55 -0
  55. package/plugin/scripts/hook-shim.mjs +303 -0
  56. package/plugin/scripts/host-update.mjs +58 -0
  57. package/plugin/scripts/kling-preflight.sh +146 -0
  58. package/plugin/scripts/learn-capture.sh +173 -0
  59. package/plugin/scripts/learn-flush.mjs +155 -0
  60. package/plugin/scripts/lesson-hooks.sh +213 -0
  61. package/plugin/scripts/md-stamp.mjs +219 -0
  62. package/plugin/scripts/protect-brain-state.sh +84 -0
  63. package/plugin/scripts/route-dispatch.sh +147 -0
  64. package/plugin/scripts/routing-outcome-capture.mjs +89 -0
  65. package/plugin/scripts/runtime-preferences.mjs +269 -0
  66. package/plugin/scripts/session-start-core.mjs +477 -0
  67. package/plugin/scripts/session-start.sh +13 -0
  68. package/plugin/scripts/signal-watch.mjs +193 -0
  69. package/plugin/scripts/unprompted-runtime.mjs +377 -0
  70. package/plugin/scripts/update-apply.mjs +419 -0
  71. package/plugin/scripts/verify-interface.sh +53 -0
  72. package/plugin/scripts/version-bump-gate.sh +112 -0
  73. package/plugin/skills/brain-build/SKILL.md +123 -0
  74. package/plugin/skills/brain-console/SKILL.md +22 -0
  75. package/plugin/skills/brain-prompt/SKILL.md +83 -0
  76. package/plugin/skills/brain-score/SKILL.md +101 -0
  77. package/plugin/skills/release-proof/SKILL.md +81 -0
  78. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  79. package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
  80. package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
  81. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +121 -0
  82. package/plugin/skills/ruvnet-brain/SKILL.md +234 -0
  83. package/plugin/skills/rvbc/SKILL.md +23 -0
  84. package/plugin/skills/savings/SKILL.md +46 -0
  85. package/plugin/skills/whats-new/SKILL.md +22 -0
  86. package/scripts/adr-backfill.mjs +107 -0
  87. package/scripts/advocacy-outcomes.mjs +808 -0
  88. package/scripts/agentdb-context.mjs +216 -0
  89. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  90. package/scripts/ascii-drift.mjs +236 -0
  91. package/scripts/behavioral-l1-l4.mjs +210 -0
  92. package/scripts/brain-capability-check.mjs +72 -0
  93. package/scripts/brain-grade-groundtruth.mjs +100 -0
  94. package/scripts/brain-latency-50.mjs +227 -0
  95. package/scripts/brain-novice-50.mjs +189 -0
  96. package/scripts/brain-stamp.mjs +94 -0
  97. package/scripts/brain-state.mjs +212 -0
  98. package/scripts/build-bundle.mjs +522 -0
  99. package/scripts/build-concepts.mjs +132 -0
  100. package/scripts/build-l2.mjs +71 -0
  101. package/scripts/build-primer.mjs +73 -0
  102. package/scripts/build-symbols.mjs +68 -0
  103. package/scripts/calibrate-router.mjs +97 -0
  104. package/scripts/capability-audit.mjs +321 -0
  105. package/scripts/capability-registry.mjs +876 -0
  106. package/scripts/check-indexation.mjs +108 -0
  107. package/scripts/check-legibility.mjs +189 -0
  108. package/scripts/ci/build-fixture-kb.mjs +67 -0
  109. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  110. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  111. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  112. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  113. package/scripts/ci/stranger-scenario.mjs +228 -0
  114. package/scripts/ci/stranger-timeout.mjs +25 -0
  115. package/scripts/ci-verdict.mjs +29 -0
  116. package/scripts/claims-verify.mjs +710 -0
  117. package/scripts/clear-claude-tmp.sh +31 -0
  118. package/scripts/console-engine.mjs +434 -0
  119. package/scripts/console-engine.test.mjs +125 -0
  120. package/scripts/corpus-qa.mjs +250 -0
  121. package/scripts/correction-detect-embed.mjs +346 -0
  122. package/scripts/correction-detect-measure.mjs +270 -0
  123. package/scripts/correction-detect.mjs +686 -0
  124. package/scripts/count-chunks.mjs +54 -0
  125. package/scripts/described-questions.json +30 -0
  126. package/scripts/design-grade.mjs +58 -0
  127. package/scripts/dev-plugin-link.sh +105 -0
  128. package/scripts/distill-project.mjs +200 -0
  129. package/scripts/doc-currency.mjs +801 -0
  130. package/scripts/eval-brain.mjs +244 -0
  131. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  132. package/scripts/full-hints.mjs +87 -0
  133. package/scripts/gate.sh +39 -0
  134. package/scripts/gates.mjs +146 -0
  135. package/scripts/gen-console-images.mjs +54 -0
  136. package/scripts/gen-images.mjs +47 -0
  137. package/scripts/git-clone-refresh.mjs +52 -0
  138. package/scripts/git-hooks/pre-push +126 -0
  139. package/scripts/goal-match.mjs +398 -0
  140. package/scripts/goldie-research.mjs +223 -0
  141. package/scripts/goldie-weekly.sh +67 -0
  142. package/scripts/health-repair.mjs +250 -0
  143. package/scripts/helix-scenario-questions.json +10 -0
  144. package/scripts/ingest-gists.mjs +230 -0
  145. package/scripts/ingest-meeting.mjs +115 -0
  146. package/scripts/ingest-repo.mjs +79 -0
  147. package/scripts/install-npx-witness.sh +49 -0
  148. package/scripts/issue-fix.mjs +639 -0
  149. package/scripts/issue-watch.mjs +276 -0
  150. package/scripts/issue4-close-note.md +31 -0
  151. package/scripts/key-canary.mjs +91 -0
  152. package/scripts/latency-to-surface.mjs +233 -0
  153. package/scripts/learning-enable.mjs +380 -0
  154. package/scripts/learning-replay.mjs +1570 -0
  155. package/scripts/learnings.mjs +62 -0
  156. package/scripts/lesson-gate.mjs +680 -0
  157. package/scripts/lesson-lifecycle.mjs +449 -0
  158. package/scripts/lesson-promote.mjs +262 -0
  159. package/scripts/lesson-ratify.mjs +98 -0
  160. package/scripts/lesson-seed.mjs +252 -0
  161. package/scripts/lesson-store.mjs +447 -0
  162. package/scripts/loop-checkpoint.mjs +86 -0
  163. package/scripts/memdb-health.sh +14 -0
  164. package/scripts/memory-doctor.mjs +271 -0
  165. package/scripts/model-catalog.mjs +79 -0
  166. package/scripts/nightly-controller.mjs +66 -0
  167. package/scripts/nightly-gists.sh +72 -0
  168. package/scripts/nightly-wrapper.sh +180 -0
  169. package/scripts/notify.sh +12 -0
  170. package/scripts/npx-witness.sh +56 -0
  171. package/scripts/onboarding-console.mjs +2749 -0
  172. package/scripts/private-fence.mjs +69 -0
  173. package/scripts/proactivity-metrics.mjs +118 -0
  174. package/scripts/proof-questions.json +56 -0
  175. package/scripts/prove.mjs +95 -0
  176. package/scripts/proxy/claude-proxied.sh +57 -0
  177. package/scripts/proxy/proxy-revert.sh +59 -0
  178. package/scripts/proxy/proxy-up.sh +60 -0
  179. package/scripts/proxy/proxy-verify.mjs +142 -0
  180. package/scripts/published-surface-probe.mjs +241 -0
  181. package/scripts/qe/card-lane-gate.mjs +162 -0
  182. package/scripts/qe/session-start-gate.mjs +229 -0
  183. package/scripts/qe/ux-suite.mjs +323 -0
  184. package/scripts/reconcile-project.mjs +0 -0
  185. package/scripts/record-lesson.mjs +113 -0
  186. package/scripts/refresh-model-catalog.mjs +99 -0
  187. package/scripts/release-proof.mjs +9 -0
  188. package/scripts/release-vector.mjs +281 -0
  189. package/scripts/release.mjs +395 -0
  190. package/scripts/remedy-registry.mjs +247 -0
  191. package/scripts/rerank-cap-eval.mjs +265 -0
  192. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  193. package/scripts/route-cheap.mjs +20 -15
  194. package/scripts/router-utilization.mjs +182 -0
  195. package/scripts/routing-flywheel.mjs +596 -0
  196. package/scripts/rvf-generation.mjs +104 -0
  197. package/scripts/rvf-index-audit.mjs +138 -0
  198. package/scripts/self-update.mjs +508 -0
  199. package/scripts/selfcheck.mjs +7 -1
  200. package/scripts/sign-bundle.mjs +69 -0
  201. package/scripts/signal-watch.mjs +171 -0
  202. package/scripts/stack-sync.mjs +469 -0
  203. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  204. package/scripts/stamp-sweep.mjs +144 -0
  205. package/scripts/status-honesty.mjs +102 -0
  206. package/scripts/sync-version.mjs +217 -0
  207. package/scripts/token-report.mjs +102 -0
  208. package/scripts/top100-benchmark.mjs +479 -0
  209. package/scripts/top100-corpus.mjs +112 -0
  210. package/scripts/top100-semantic-assertions.mjs +449 -0
  211. package/scripts/update-apply.mjs +9 -0
  212. package/scripts/upgrade-notice.mjs +14 -0
  213. package/scripts/verify-bundle.mjs +51 -0
  214. package/scripts/verify-channels.mjs +184 -0
  215. package/scripts/verify-model-catalog.mjs +104 -0
  216. package/scripts/verify-nightly-close-issue4.sh +31 -0
  217. package/scripts/version.mjs +40 -0
  218. package/scripts/wired-check.mjs +864 -0
@@ -0,0 +1,229 @@
1
+ #!/usr/bin/env node
2
+ // session-start-gate.mjs — ADR-058 D6's SECOND hard gate: session-start WALL TIME, the first
3
+ // user-felt number in this repo that a build can fail on.
4
+ //
5
+ // WHY THIS EXISTS (the deduction, quoted): an independent grader scored D6 68 and wrote — "the hard
6
+ // gate measures a 0.03–0.22ms in-process function against a 250ms budget (~1000x headroom — it can
7
+ // only catch catastrophic regression classes, by design per its header). Everything the user
8
+ // actually FEELS — heavy-lane query seconds, session-start wall time, install minutes, dead air,
9
+ // refusal clarity — is advisory or unmeasured", and "the gate has trivially never failed in earnest
10
+ // (thresholds set at 1000x measured cost)". Their own cheapest fix was named: promote ONE user-felt
11
+ // number to a hard gate, and give it a budget row in the same governed manifest. This is that.
12
+ //
13
+ // WHAT MAKES IT USER-FELT: this is the wall time of the hook a stranger's Claude Code fires at
14
+ // SessionStart, BEFORE their first prompt is answered. Nobody experiences kb/card-lane.mjs's
15
+ // 0.1158ms. Everybody experiences this.
16
+ //
17
+ // NOTHING HERE HAND-ROLLS A SECOND TIMER. scripts/selfcheck.mjs ALREADY fires the literal registered
18
+ // command through an external process-group watchdog and already returns elapsedMs per firing, and
19
+ // already enforces the declared timeout with TIMEOUT_MARGIN. Writing a private timer beside it would
20
+ // recreate the adjacent-door defect (ADR-055 F16: a gate and its evidence as two different code
21
+ // paths). So this file is a THRESHOLD POLICY over selfcheck's existing measurement — fireHook(),
22
+ // resolveInstalledSurface() and readInstalledRegistrations() are imported, not reproduced.
23
+ //
24
+ // MEASUREMENT METHOD, DELIBERATE — and the OPPOSITE of the card lane's, for a stated reason:
25
+ // card-lane-gate.mjs measures IN-PROCESS because the thing it measures is an in-process function and
26
+ // a subprocess per firing would measure the OS scheduler instead. Here the thing measured IS a
27
+ // subprocess (node → hook-shim → bash → session-start.sh), so subprocess-per-firing is not a
28
+ // concession, it is the only honest method. The four consequences that follow are handled explicitly
29
+ // rather than assumed away, and are restated in kb/card-lane-budget.json's `measurementMethod`:
30
+ // 1. SURFACE — resolveInstalledSurface() prefers a machine's INSTALLED plugin cache over the
31
+ // checkout. On a developer's machine that cache is usually an older release, so a gate that
32
+ // took the default would grade code that is not in this commit. We therefore hand it a fresh
33
+ // EMPTY home, which leaves the checkout as the only candidate. (Verified live 2026-07-28: with
34
+ // the real homedir it selected `installed:` and measured a build 12 versions old.)
35
+ // 2. HOME — a fresh temp dir per run. The hook writes once-per-machine marker files; pointing it
36
+ // at the developer's real HOME would both perturb the measurement and silently consume their
37
+ // real first-run offers.
38
+ // 3. COLD + STEADY — ONE cold fire before the steady-state window. The first-ever fire in a virgin
39
+ // HOME emits once-per-machine offers, but it is still a real user wait: it must finish inside
40
+ // both absoluteFailMs and the hook's declared timeout. The following samples measure the common
41
+ // steady state without allowing a failed cold start to disappear into a percentile.
42
+ // 4. SEQUENTIAL — never concurrent. Concurrency would measure the runner's core count.
43
+ //
44
+ // THE THRESHOLDS ARE NOT HARDCODED HERE. They live in kb/card-lane-budget.json under `sessionStart`,
45
+ // which docs/adr/0058-the-95-contract.md `governs:` — so a silent raise shows up as governed-set
46
+ // drift under `node scripts/doc-currency.mjs --check` rather than being a free edit. And they are
47
+ // set from a measured distribution (n=110, p50 148ms, worst p95 323ms, max 440ms), not from a round
48
+ // number: p95 budget 1000ms is ~3.1x the worst measured p95, sized for a 2-vCPU CI runner. A budget
49
+ // at 1000x measured cost is the exact criticism above; it is not repeated here.
50
+ import fs from 'node:fs';
51
+ import os from 'node:os';
52
+ import path from 'node:path';
53
+ import { fileURLToPath } from 'node:url';
54
+ import { fireHook, resolveInstalledSurface, readInstalledRegistrations } from '../selfcheck.mjs';
55
+
56
+ const HERE = path.dirname(fileURLToPath(import.meta.url));
57
+ export const REPO_ROOT = path.resolve(HERE, '../..');
58
+ export const BUDGET_PATH = path.join(REPO_ROOT, 'kb', 'card-lane-budget.json');
59
+
60
+ /** Read the `sessionStart` block. Same validation shape as card-lane-gate.mjs's loadBudget(). */
61
+ export function loadBudget(budgetPath = BUDGET_PATH) {
62
+ const doc = JSON.parse(fs.readFileSync(budgetPath, 'utf8'));
63
+ const budget = doc.sessionStart;
64
+ if (!budget || typeof budget !== 'object') {
65
+ throw new Error(`card-lane-budget.json: no "sessionStart" block — this gate has no checked-in budget to enforce`);
66
+ }
67
+ for (const key of ['sampleSize', 'p95BudgetMs', 'absoluteFailMs']) {
68
+ if (typeof budget[key] !== 'number' || !(budget[key] > 0)) {
69
+ throw new Error(`card-lane-budget.json sessionStart: "${key}" must be a positive number, got ${JSON.stringify(budget[key])}`);
70
+ }
71
+ }
72
+ return budget;
73
+ }
74
+
75
+ /** Nearest-rank percentile over an ASCENDING-sorted array. p in [0,100]. */
76
+ export function percentile(sortedAsc, p) {
77
+ if (!sortedAsc.length) return null;
78
+ const idx = Math.min(sortedAsc.length - 1, Math.max(0, Math.ceil((p / 100) * sortedAsc.length) - 1));
79
+ return sortedAsc[idx];
80
+ }
81
+
82
+ /**
83
+ * Resolve the SessionStart registration to fire, from the CHECKOUT's plugin tree.
84
+ * `home` is a fresh empty dir on purpose (see note 1 in the header) — it is what forces
85
+ * resolveInstalledSurface() to pick `source: 'checkout'` instead of a stale installed cache.
86
+ */
87
+ export function resolveSessionStart({ repo = REPO_ROOT, home = null } = {}) {
88
+ const emptyHome = home ?? fs.mkdtempSync(path.join(os.tmpdir(), 'ssgate-resolve-'));
89
+ const surface = resolveInstalledSurface({ home: emptyHome, repo });
90
+ if (!surface.ok) throw new Error(`could not resolve a plugin surface to measure: ${surface.reason}`);
91
+ const reg = readInstalledRegistrations(surface.hooksFile).find((r) => r.event === 'SessionStart');
92
+ if (!reg) throw new Error(`no SessionStart registration in ${surface.hooksFile} — there is nothing to measure`);
93
+ return { surface, reg, command: reg.command.replaceAll('${CLAUDE_PLUGIN_ROOT}', surface.root) };
94
+ }
95
+
96
+ /**
97
+ * Fire the SessionStart hook `n` times through selfcheck's watchdog and return each firing's wall
98
+ * time in ms, plus the warm-up's own numbers (reported, never gated — it is a different regime).
99
+ *
100
+ * `fireFn` defaults to the real fireHook and exists as an injectable seam ONLY so
101
+ * tests/unit/session-start-gate.test.mjs can prove the THRESHOLD LOGIC catches a slow hook without
102
+ * mutating the shipped plugin to do it. The shipped hook's own mutant is exercised separately, for
103
+ * real, against the real path — the same split card-lane-gate.mjs uses and for the same reason.
104
+ */
105
+ export async function measureFirings({ n = 30, repo = REPO_ROOT, fireFn = fireHook, resolved = null } = {}) {
106
+ const r = resolved ?? resolveSessionStart({ repo });
107
+ const home = fs.mkdtempSync(path.join(os.tmpdir(), 'ssgate-home-'));
108
+ const brainHome = path.join(home, '.cache', 'ruvnet-brain');
109
+ const stateDir = path.join(home, '.config', 'ruvnet-brain');
110
+ // HOME alone is not isolation on Windows: os.homedir() follows USERPROFILE there, while Git Bash
111
+ // follows HOME. The old gate therefore let hook-shim.mjs read the runner's real spine while the
112
+ // shell body wrote to the fixture home. Keep every authority on one root, exactly as the shipped
113
+ // Windows installer/host tests do.
114
+ const env = {
115
+ HOME: home,
116
+ USERPROFILE: home,
117
+ XDG_CACHE_HOME: path.join(home, '.cache'),
118
+ RUVNET_BRAIN_HOME: brainHome,
119
+ RUVNET_BRAIN_STATE_DIR: stateDir,
120
+ RUVNET_SESSION_TRACE: '1',
121
+ CLAUDE_PLUGIN_ROOT: r.surface.root,
122
+ };
123
+ const timeoutSec = typeof r.reg.timeout === 'number' ? r.reg.timeout : 5;
124
+ const fire = () => fireFn({ command: r.command, event: 'SessionStart', regime: 'valid', timeoutSec, cwd: os.tmpdir(), env });
125
+
126
+ const warmup = await fire(); // separate regime, but its declared-timeout result is still gated
127
+ const samplesMs = [];
128
+ for (let i = 0; i < n; i++) {
129
+ const m = await fire();
130
+ // A firing the watchdog had to kill has no meaningful elapsedMs to average — it is a hang, and a
131
+ // hang must never be smoothed into a percentile. Charge it as the full timeout so it can only
132
+ // ever make the verdict worse, and name it in the verdict below.
133
+ samplesMs.push(m.timedOut ? timeoutSec * 1000 : m.elapsedMs);
134
+ }
135
+ return {
136
+ samplesMs,
137
+ warmupMs: warmup.elapsedMs,
138
+ warmupStdoutBytes: warmup.stdoutBytes,
139
+ warmupTimedOut: Boolean(warmup.timedOut),
140
+ warmupStatus: warmup.status,
141
+ warmupStderr: String(warmup.stderr || '').slice(-1000),
142
+ timeoutSec,
143
+ surface: r.surface,
144
+ home,
145
+ };
146
+ }
147
+
148
+ /**
149
+ * The gate. Returns a verdict object; never throws on a threshold breach (that is a normal result,
150
+ * not an exceptional one) — it throws only if the hook or the manifest could not be resolved at all,
151
+ * which scripts/qe/ux-suite.mjs treats as its own hard failure ("could not measure" is never success).
152
+ */
153
+ export async function runSessionStartGate(opts = {}) {
154
+ const budget = loadBudget(opts.budgetPath);
155
+ const {
156
+ samplesMs, warmupMs, warmupStdoutBytes, warmupTimedOut, warmupStatus, warmupStderr,
157
+ timeoutSec, surface,
158
+ } = await measureFirings({
159
+ n: budget.sampleSize, repo: opts.repo, fireFn: opts.fireFn, resolved: opts.resolved,
160
+ });
161
+ const sorted = [...samplesMs].sort((a, b) => a - b);
162
+ const p50 = percentile(sorted, 50);
163
+ const p95 = percentile(sorted, 95);
164
+ const max = sorted[sorted.length - 1];
165
+
166
+ const reasons = [];
167
+ if (warmupTimedOut) {
168
+ reasons.push(`COLD-START FAIL — the first SessionStart fire exceeded its declared ${timeoutSec}s timeout (${warmupMs.toFixed(0)}ms); first-run latency is user-felt and may not be hidden as an untimed warm-up`);
169
+ }
170
+ if (warmupMs > budget.absoluteFailMs) {
171
+ reasons.push(`COLD-START ABSOLUTE FAIL — the first SessionStart fire took ${warmupMs.toFixed(0)}ms > absoluteFailMs=${budget.absoluteFailMs}ms, even though the command timeout is ${timeoutSec}s; first-run latency may not bypass the absolute limit as an untimed warm-up`);
172
+ }
173
+ if (p95 > budget.absoluteFailMs || max > budget.absoluteFailMs) {
174
+ reasons.push(`ABSOLUTE FAIL — the hook has no margin left inside its own declared ${timeoutSec}s timeout: max=${max.toFixed(0)}ms p95=${p95.toFixed(0)}ms > absoluteFailMs=${budget.absoluteFailMs}ms (= TIMEOUT_MARGIN 0.8 x ${timeoutSec}s, the same wall scripts/selfcheck.mjs already enforces on a stranger's machine)`);
175
+ } else if (p95 > budget.p95BudgetMs) {
176
+ reasons.push(`BUDGET BREACH: p95=${p95.toFixed(0)}ms > p95BudgetMs=${budget.p95BudgetMs}ms over ${budget.sampleSize} real firings of the registered SessionStart command (measured baseline p95 ${budget.measuredBaseline?.worstRunP95Ms ?? '?'}ms)`);
177
+ }
178
+
179
+ return {
180
+ pass: reasons.length === 0,
181
+ n: samplesMs.length,
182
+ p50,
183
+ p95,
184
+ max,
185
+ warmupMs,
186
+ warmupStdoutBytes,
187
+ warmupTimedOut,
188
+ warmupStatus,
189
+ warmupStderr,
190
+ timeoutSec,
191
+ surface,
192
+ budget,
193
+ reasons,
194
+ samplesMs,
195
+ };
196
+ }
197
+
198
+ // ── CLI ─────────────────────────────────────────────────────────────────────────────────────────
199
+ function fmt(ms) { return `${ms.toFixed(0)}ms`; }
200
+
201
+ async function main() {
202
+ console.log('\n session-start wall-time gate (ADR-058 D6 — HARD gate, real registered command, user-felt)\n');
203
+ let result;
204
+ try {
205
+ result = await runSessionStartGate();
206
+ } catch (e) {
207
+ console.error(` ✗ could not run the gate: ${e.message}`);
208
+ process.exit(2);
209
+ }
210
+ console.log(` budget source kb/card-lane-budget.json → sessionStart`);
211
+ console.log(` surface ${result.surface.source} (${result.surface.root})`);
212
+ console.log(` firings 1 cold + ${result.n} steady-state, sequential, fresh isolated HOME`);
213
+ console.log(` cold first fire ${fmt(result.warmupMs)} / ${result.warmupStdoutBytes} bytes (${result.warmupTimedOut ? 'TIMED OUT — HARD FAIL' : 'inside declared timeout'})`);
214
+ console.log(` p50 ${fmt(result.p50)}`);
215
+ console.log(` p95 ${fmt(result.p95)} (budget ${result.budget.p95BudgetMs}ms)`);
216
+ console.log(` max ${fmt(result.max)} (absolute fail ${result.budget.absoluteFailMs}ms = 0.8 x the ${result.timeoutSec}s declared timeout)`);
217
+ console.log('');
218
+ if (result.pass) {
219
+ console.log(' PASS — session start inside budget.\n');
220
+ process.exit(0);
221
+ }
222
+ console.log(' FAIL (hard):');
223
+ for (const r of result.reasons) console.log(` ✗ ${r}`);
224
+ console.log('');
225
+ process.exit(1);
226
+ }
227
+
228
+ const invokedDirectly = process.argv[1] && path.resolve(process.argv[1]) === path.resolve(fileURLToPath(import.meta.url));
229
+ if (invokedDirectly) main();
@@ -0,0 +1,323 @@
1
+ // ux-suite.mjs — the UX-experience QE suite runner (owner request 2026-07-24).
2
+ //
3
+ // Runs the deterministic UX probes, prints a table of MEASURED numbers, writes a machine-readable
4
+ // receipt when UX_QE_EVIDENCE is set, and exits non-zero on any HARD failure.
5
+ // 1. Environment-sensitive timings (server-ready, console/tips paint, command→explanation,
6
+ // dead-air) are HARD user-experience budgets. Platform calibration gives slower hosted runners
7
+ // honest headroom without turning "slow enough for a person to notice" into advisory green.
8
+ // 2. kb/card-lane.mjs's decision lane is MODEL-FREE, ML-FREE keyword overlap with a measured warm
9
+ // baseline of 0.1158ms. Its budget (kb/card-lane-budget.json, p95 <= 250ms / absolute fail
10
+ // >1000ms — ~2,159x / ~8,600x the baseline) has so much headroom that a breach cannot be
11
+ // scheduler jitter — it can only be a correctness regression. THIS is a genuine hard gate: a
12
+ // breach here fails the suite, not warns it. See scripts/qe/card-lane-gate.mjs for the full
13
+ // reasoning and the in-process (no subprocess per firing) measurement method.
14
+ // 3. SESSION-START WALL TIME (added 2026-07-28) is the SAME tier as 2, and is here because tier 2
15
+ // alone was not enough. An independent grader's words: the card-lane gate "measures a
16
+ // 0.03–0.22ms in-process function against a 250ms budget (~1000x headroom — it can only catch
17
+ // catastrophic regression classes)", while "everything the user actually FEELS — heavy-lane
18
+ // query seconds, session-start WALL TIME, install minutes, dead air, refusal clarity — is
19
+ // advisory or unmeasured". Session-start wall time is the first of those promoted out of tier 1:
20
+ // it is the hook a stranger's Claude Code fires before their first prompt is answered, it is
21
+ // already measured by scripts/selfcheck.mjs's external process-group watchdog (no second timer
22
+ // was written), and its budget is set from a measured distribution — p95 1000ms is ~3.1x the
23
+ // worst measured p95 (323ms over n=110), NOT 1000x. See scripts/qe/session-start-gate.mjs.
24
+ //
25
+ // HONESTY (same rules as the product):
26
+ // • Every number is measured on THIS run. Nothing is asserted from memory.
27
+ // • A probe that could not execute is reported "not run" and HARD-fails — silence is not success.
28
+ // • The probes are MODEL-FREE (render + PTY-style timing, plus the in-process card-lane firings).
29
+ // They call no LLM, use no API key, touch no account — the cleanest satisfaction of the owner's
30
+ // "no API keys, run on our account" rule.
31
+ // • aqe orchestration: we OPTIONALLY register this run as an `aqe task` for visibility in
32
+ // `aqe status`, but the MEASUREMENT is a plain deterministic probe, NOT aqe-internal. Verified live
33
+ // 2026-07-24: `aqe domain` supports only list/health (not create), so inventing an "onboarding-ux"
34
+ // domain would be fiction. We do not. If aqe isn't present, the suite runs identically and says so.
35
+ import { spawn, spawnSync } from 'node:child_process';
36
+ import fs from 'node:fs';
37
+ import path from 'node:path';
38
+ import os from 'node:os';
39
+ import { fileURLToPath } from 'node:url';
40
+ import { runCommandProbe } from '../../tests/ux/command-probe.mjs';
41
+ import { runCardLaneGate } from './card-lane-gate.mjs';
42
+ import { runSessionStartGate } from './session-start-gate.mjs';
43
+
44
+ const HERE = path.dirname(fileURLToPath(import.meta.url));
45
+ const RENDER_PROBE = path.resolve(HERE, '../../tests/ux/render-probe.mjs');
46
+ // The child now performs seven acceptance assertions, two real settings writes + reload, one real
47
+ // batch remedy and one real undo in addition to paint timings. Its total wall clock is test-runtime,
48
+ // not user-visible latency; each user action has its own hard 4s assertion inside the probe.
49
+ const RENDER_PROBE_TIMEOUT_MS = 60_000;
50
+
51
+ function stopProcessTree(child) {
52
+ if (!child?.pid) return;
53
+ if (process.platform === 'win32') {
54
+ try { spawnSync('taskkill', ['/pid', String(child.pid), '/T', '/F'], { stdio: 'ignore' }); } catch {}
55
+ return;
56
+ }
57
+ try { process.kill(-child.pid, 'SIGTERM'); } catch {}
58
+ try { process.kill(-child.pid, 'SIGKILL'); } catch {}
59
+ }
60
+
61
+ /**
62
+ * Browser drivers can wedge below JavaScript, so an in-process Promise timeout is not a bound.
63
+ * Run the render probe in its own process group and kill the whole group at the deadline.
64
+ */
65
+ export function runRenderProbeIsolated({
66
+ probeFile = RENDER_PROBE,
67
+ timeoutMs = RENDER_PROBE_TIMEOUT_MS,
68
+ } = {}) {
69
+ return new Promise((resolve) => {
70
+ const child = spawn(process.execPath, [probeFile], {
71
+ cwd: path.resolve(HERE, '../..'),
72
+ env: process.env,
73
+ detached: process.platform !== 'win32',
74
+ stdio: ['ignore', 'pipe', 'pipe'],
75
+ windowsHide: true,
76
+ });
77
+ let stdout = '';
78
+ let stderr = '';
79
+ let settled = false;
80
+ child.stdout.on('data', (chunk) => { stdout += String(chunk); });
81
+ child.stderr.on('data', (chunk) => { stderr += String(chunk); });
82
+
83
+ const finish = (result) => {
84
+ if (settled) return;
85
+ settled = true;
86
+ clearTimeout(timer);
87
+ resolve(result);
88
+ };
89
+ const timer = setTimeout(() => {
90
+ stopProcessTree(child);
91
+ const trace = stderr.trim().split('\n').filter(Boolean).slice(-4).join(' | ');
92
+ finish({
93
+ results: [],
94
+ notes: [`render probe exceeded ${timeoutMs}ms process deadline${trace ? `; last stages: ${trace}` : ''}`],
95
+ });
96
+ }, timeoutMs);
97
+
98
+ child.on('error', (error) => finish({ results: [], notes: [`render probe spawn failed: ${error.message}`] }));
99
+ child.on('close', () => {
100
+ try {
101
+ const parsed = JSON.parse(stdout);
102
+ finish(parsed);
103
+ } catch {
104
+ const trace = stderr.trim().split('\n').filter(Boolean).slice(-4).join(' | ');
105
+ finish({ results: [], notes: [`render probe returned no readable JSON${trace ? `; last stages: ${trace}` : ''}`] });
106
+ }
107
+ });
108
+ });
109
+ }
110
+
111
+ // Darwin values are frozen from the measured 2026-07-24 baseline in docs/qe/ux-first-run.md.
112
+ // Linux and Windows receive bounded hosted-runner startup headroom; the visible-paint and dead-air
113
+ // product promises stay tight. These are release budgets, not performance claims about GitHub's
114
+ // hardware. CI receipts make future recalibration evidence-based rather than guessed.
115
+ export const PLATFORM_BUDGETS = Object.freeze({
116
+ darwin: Object.freeze({
117
+ 'server-ready': 2500,
118
+ 'console time-to-visible': 2500,
119
+ 'tips time-to-visible (hero)': 2000,
120
+ 'tips first-section': 2000,
121
+ commandToExplanationMs: 1500,
122
+ maxDeadAirMs: 3000,
123
+ }),
124
+ linux: Object.freeze({
125
+ 'server-ready': 4000,
126
+ 'console time-to-visible': 3000,
127
+ 'tips time-to-visible (hero)': 2500,
128
+ 'tips first-section': 2500,
129
+ commandToExplanationMs: 2500,
130
+ maxDeadAirMs: 3000,
131
+ }),
132
+ win32: Object.freeze({
133
+ 'server-ready': 6000,
134
+ 'console time-to-visible': 4000,
135
+ 'tips time-to-visible (hero)': 3500,
136
+ 'tips first-section': 3500,
137
+ commandToExplanationMs: 3000,
138
+ maxDeadAirMs: 3000,
139
+ }),
140
+ });
141
+
142
+ export function budgetsForPlatform(platform = process.platform) {
143
+ const budgets = PLATFORM_BUDGETS[platform];
144
+ if (!budgets) throw new Error(`unsupported UX-QE platform: ${platform}`);
145
+ return budgets;
146
+ }
147
+
148
+ export function timingFailure(label, measured, budget) {
149
+ if (measured == null) return `${label}: could not measure`;
150
+ if (measured > budget) return `${label}: ${measured}ms exceeds HARD ${budget}ms budget`;
151
+ return null;
152
+ }
153
+
154
+ function line(label, measured, unit, hardAt) {
155
+ const val = measured == null ? 'NOT RUN' : `${measured}${unit}`;
156
+ let flag = '';
157
+ if (measured == null) flag = ' ✗ could not measure';
158
+ else if (hardAt != null && measured > hardAt) flag = ` ✗ HARD FAIL (>${hardAt}${unit})`;
159
+ else if (hardAt != null) flag = ` ✓ HARD budget ${hardAt}${unit}`;
160
+ return ` ${label.padEnd(30)} ${String(val).padStart(10)}${flag}`;
161
+ }
162
+
163
+ function writeEvidence(receipt) {
164
+ const jsonOutIndex = process.argv.indexOf('--json-out');
165
+ if (jsonOutIndex >= 0 && !process.argv[jsonOutIndex + 1]) {
166
+ throw new Error('--json-out requires a file path');
167
+ }
168
+ const target = jsonOutIndex >= 0
169
+ ? process.argv[jsonOutIndex + 1]
170
+ : process.env.UX_QE_EVIDENCE;
171
+ if (!target) return;
172
+ const resolved = path.resolve(target);
173
+ fs.mkdirSync(path.dirname(resolved), { recursive: true });
174
+ fs.writeFileSync(resolved, `${JSON.stringify(receipt, null, 2)}\n`);
175
+ };
176
+
177
+ function tryRegisterAqeTask() {
178
+ // Best-effort visibility only. Never fails the suite; never bills a model. `submit` enqueues
179
+ // metadata to the Queen Coordinator; `--no-progress` and no `--wait` keep it fire-and-forget, so no
180
+ // model is invoked. Flags grounded live 2026-07-24 against `aqe task submit --help` (type positional,
181
+ // -p/-d/-t/--payload — there is NO --description).
182
+ const payload = JSON.stringify({ probe: 'ruvnet-brain-ux-time-to-visible', model_free: true });
183
+ const r = spawnSync('aqe', ['task', 'submit', 'quality-assessment', '-p', 'p3', '--payload', payload, '--no-progress'], { encoding: 'utf8', timeout: 15000 });
184
+ if (r.error || r.status !== 0) return { registered: false, why: (r.error && r.error.message) || (r.stderr || '').trim().split('\n').filter(Boolean).pop() || `exit ${r.status}` };
185
+ const id = ((r.stdout || '').match(/task[- ]?id[:\s]+(\S+)/i) || [])[1] || 'submitted';
186
+ return { registered: true, id };
187
+ }
188
+
189
+ export async function runUxSuite() {
190
+ console.log('\n RuvNet Brain — UX-experience QE suite (deterministic · model-free · runs on your account)\n');
191
+ const platform = process.platform;
192
+ const budgets = budgetsForPlatform(platform);
193
+ const startedAt = new Date().toISOString();
194
+
195
+ const aqe = tryRegisterAqeTask();
196
+ console.log(aqe.registered
197
+ ? ` aqe: registered task ${aqe.id} for orchestration visibility (measurement is a plain probe)\n`
198
+ : ` aqe: not registered (${aqe.why}) — probes run identically; orchestration visibility only\n`);
199
+
200
+ const hardFailures = [];
201
+
202
+ // ── Probe 1: render time-to-visible ──────────────────────────────────────────────────────────
203
+ console.log(' ── time-to-visible (console + tips) ──');
204
+ const render = await runRenderProbeIsolated();
205
+ for (const r of render.results) {
206
+ console.log(line(r.label, r.ms, 'ms', budgets[r.label]));
207
+ const failure = timingFailure(r.label, r.ms, budgets[r.label]);
208
+ if (failure) hardFailures.push(failure);
209
+ }
210
+ for (const n of render.notes) { console.log(` ! ${n}`); hardFailures.push(`render: ${n}`); }
211
+ console.log('\n ── console control acceptance ──');
212
+ for (const row of render.acceptance || []) {
213
+ console.log(` ${row.pass ? '✓' : '✗'} ${row.label}: ${row.detail}`);
214
+ if (!row.pass) hardFailures.push(`console control acceptance: ${row.label} — ${row.detail}`);
215
+ }
216
+ if (!(render.acceptance || []).length) hardFailures.push('console control acceptance: NOT RUN');
217
+ // Any expected render row missing entirely = not run = hard fail.
218
+ const gotConsole = render.results.some((r) => r.label === 'console time-to-visible' && r.ms != null);
219
+ if (!gotConsole) hardFailures.push('console time-to-visible: NOT RUN');
220
+
221
+ // ── Probe 2/3: command → explanation → "it's live" ──────────────────────────────────────────
222
+ console.log('\n ── command → explanation → completion signal ──');
223
+ const cmd = await runCommandProbe();
224
+ console.log(line('command→explanation', cmd.commandToExplanationMs, 'ms', budgets.commandToExplanationMs));
225
+ console.log(line('command→"it\'s live"', cmd.commandToLiveMs, 'ms', null) + ' (reported, not gated)');
226
+ console.log(line('max dead-air gap', cmd.maxDeadAirMs, 'ms', budgets.maxDeadAirMs));
227
+ console.log(` completion signal present ${cmd.completionSignalPresent ? ' YES ✓' : ' NO ✗ (GAP)'}`);
228
+ if (cmd.liveSignalText) console.log(` signal: "${cmd.liveSignalText}"`);
229
+
230
+ const explanationFailure = timingFailure('command→explanation', cmd.commandToExplanationMs, budgets.commandToExplanationMs);
231
+ if (explanationFailure) hardFailures.push(explanationFailure);
232
+ const deadAirFailure = timingFailure('max dead-air gap', cmd.maxDeadAirMs, budgets.maxDeadAirMs);
233
+ if (deadAirFailure) hardFailures.push(deadAirFailure);
234
+ if (!cmd.completionSignalPresent) hardFailures.push('completion signal MISSING — the "it\'s live, take a look at your page" line never printed');
235
+
236
+ // ── Probe 4: decision-lane latency — HARD GATE, not advisory (ADR-058 D6) ───────────────────
237
+ // Deliberately NOT reusing line()'s warnAt/"(proposed)" formatting above: that phrasing is correct
238
+ // for the advisory timings but would misreport a HARD budget breach as merely "proposed".
239
+ console.log('\n ── decision-lane latency (kb/card-lane.mjs) — HARD GATE, deterministic, model-free ──');
240
+ try {
241
+ const laneResult = await runCardLaneGate();
242
+ const b = laneResult.budget;
243
+ const tag = (ok) => (ok ? '✓' : '✗ HARD FAIL');
244
+ console.log(` ${'card-lane p50'.padEnd(30)} ${laneResult.p50.toFixed(4).padStart(10)}ms (reported, not gated)`);
245
+ console.log(` ${'card-lane p95'.padEnd(30)} ${laneResult.p95.toFixed(4).padStart(10)}ms budget ${b.p95BudgetMs}ms ${tag(laneResult.p95 <= b.p95BudgetMs)}`);
246
+ console.log(` ${'card-lane max'.padEnd(30)} ${laneResult.max.toFixed(4).padStart(10)}ms absolute-fail ${b.absoluteFailMs}ms ${tag(laneResult.max <= b.absoluteFailMs)}`);
247
+ console.log(` firings: ${laneResult.n} in-process (no subprocess per firing — see card-lane-gate.mjs)`);
248
+ if (!laneResult.pass) for (const r of laneResult.reasons) hardFailures.push(`card-lane latency: ${r}`);
249
+ } catch (e) {
250
+ console.log(` ! could not run the card-lane latency gate: ${e.message}`);
251
+ hardFailures.push(`card-lane latency gate: could not run — ${e.message}`);
252
+ }
253
+
254
+ // ── Probe 5: session-start wall time — HARD GATE, the first USER-FELT number (ADR-058 D6) ───
255
+ // Wired exactly like probe 4 above and for the same reason: same tier, same "could not measure is
256
+ // never success" handling, same refusal to reuse line()'s "(proposed)" phrasing, which is correct
257
+ // for an advisory row and would misreport a HARD breach.
258
+ console.log('\n ── session-start wall time (plugin/hooks/hooks.json SessionStart) — HARD GATE, user-felt ──');
259
+ try {
260
+ const ss = await runSessionStartGate();
261
+ const b = ss.budget;
262
+ const tag = (ok) => (ok ? '✓' : '✗ HARD FAIL');
263
+ console.log(` ${'session-start cold first fire'.padEnd(30)} ${ss.warmupMs.toFixed(0).padStart(10)}ms ${ss.warmupTimedOut ? '✗ HARD FAIL (declared timeout exceeded)' : '✓ inside declared timeout'}`);
264
+ console.log(` ${'session-start p50'.padEnd(30)} ${ss.p50.toFixed(0).padStart(10)}ms (reported, not gated)`);
265
+ console.log(` ${'session-start p95'.padEnd(30)} ${ss.p95.toFixed(0).padStart(10)}ms budget ${b.p95BudgetMs}ms ${tag(ss.p95 <= b.p95BudgetMs)}`);
266
+ console.log(` ${'session-start max'.padEnd(30)} ${ss.max.toFixed(0).padStart(10)}ms absolute-fail ${b.absoluteFailMs}ms ${tag(ss.max <= b.absoluteFailMs)}`);
267
+ console.log(` firings: ${ss.n} sequential fires of the REAL registered command via selfcheck.mjs's watchdog, from ${ss.surface.source}`);
268
+ if (ss.warmupStderr) {
269
+ console.log(` cold trace: ${ss.warmupStderr.trim().split('\n').join(' | ')}`);
270
+ }
271
+ if (!ss.pass) for (const r of ss.reasons) hardFailures.push(`session-start wall time: ${r}`);
272
+ } catch (e) {
273
+ console.log(` ! could not run the session-start wall-time gate: ${e.message}`);
274
+ hardFailures.push(`session-start wall-time gate: could not run — ${e.message}`);
275
+ }
276
+
277
+ // ── Not run on this host (stated, never faked) ──────────────────────────────────────────────
278
+ console.log('\n ── execution scope ──');
279
+ console.log(` Platform — ${platform} ${os.arch()} (this process; other OSes execute as separate CI jobs)`);
280
+ console.log(' Codex host — NOT RUN: GitHub-hosted runners do not provide a configured Codex host; this probes the shipped console process directly');
281
+
282
+ // ── Verdict ─────────────────────────────────────────────────────────────────────────────────
283
+ const receipt = {
284
+ schemaVersion: 1,
285
+ suite: 'ruvnet-brain-ux-qe',
286
+ startedAt,
287
+ finishedAt: new Date().toISOString(),
288
+ gitSha: process.env.GITHUB_SHA || null,
289
+ platform,
290
+ arch: os.arch(),
291
+ node: process.version,
292
+ budgetsMs: budgets,
293
+ render,
294
+ command: cmd,
295
+ hardFailures,
296
+ pass: hardFailures.length === 0,
297
+ scope: {
298
+ browser: 'Playwright Chromium, real local console HTTP server',
299
+ command: 'direct shipped console process',
300
+ codexHost: 'not-run',
301
+ },
302
+ };
303
+ writeEvidence(receipt);
304
+
305
+ console.log('\n ── verdict ──');
306
+ if (hardFailures.length === 0) {
307
+ console.log(' PASS — every probe ran and every render, explanation, dead-air, decision-lane, and session-start HARD budget passed.\n');
308
+ return receipt;
309
+ }
310
+ console.log(' FAIL (hard):');
311
+ for (const f of hardFailures) console.log(` ✗ ${f}`);
312
+ console.log('');
313
+ const error = new Error(`UX QE failed with ${hardFailures.length} hard failure(s)`);
314
+ error.receipt = receipt;
315
+ throw error;
316
+ }
317
+
318
+ if (process.argv[1] && path.resolve(process.argv[1]) === fileURLToPath(import.meta.url)) {
319
+ runUxSuite().catch((e) => {
320
+ if (!e.receipt) console.error(' ux-suite crashed:', e.message);
321
+ process.exit(e.receipt ? 1 : 2);
322
+ });
323
+ }
Binary file
@@ -0,0 +1,113 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * record-lesson.mjs — the durable "capture a lesson the RIGHT way" habit.
4
+ *
5
+ * WHY: AgentDB auto-capture records session transcripts (logging), not lessons
6
+ * (learning), and that telemetry drowns real lessons in recall. This records a
7
+ * lesson *structured* (task / tried / worked / critique / outcome) into a dedicated
8
+ * `lessons` signal namespace, refines it via native distill, and proves recall.
9
+ *
10
+ * NATIVE ONLY — shells to `ruflo memory` (store + distill + search). It does NOT
11
+ * reimplement any rUv capability; it enforces the structured-capture discipline
12
+ * that rUv's own `/remember` command recommends (agentdb-memory/commands/remember.md).
13
+ *
14
+ * Usage:
15
+ * node scripts/record-lesson.mjs \
16
+ * --task "..." --tried "..." --worked "..." --critique "..." --outcome success \
17
+ * [--slug short-name] [--dir <projectDir>] [--namespace lessons]
18
+ */
19
+ import { execFileSync } from 'node:child_process';
20
+ import fs from 'node:fs';
21
+ import path from 'node:path';
22
+
23
+ const arg = (name, def = '') => {
24
+ const i = process.argv.indexOf(`--${name}`);
25
+ return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : def;
26
+ };
27
+
28
+ const task = arg('task');
29
+ if (!task) {
30
+ console.error('ERROR: --task is required (what were you trying to do?)');
31
+ process.exit(2);
32
+ }
33
+ const tried = arg('tried');
34
+ const worked = arg('worked');
35
+ const critique = arg('critique');
36
+ const outcome = arg('outcome', 'success');
37
+ const dir = path.resolve(arg('dir', process.cwd()));
38
+ const ns = arg('namespace', 'lessons');
39
+ const slug =
40
+ arg('slug') ||
41
+ task.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').slice(0, 40);
42
+
43
+ const db = path.join(dir, '.swarm', 'memory.db');
44
+ if (!fs.existsSync(db)) {
45
+ console.error(`ERROR: no AgentDB at ${db}\n -> run \`ruflo memory init\` in that project first.`);
46
+ process.exit(2);
47
+ }
48
+
49
+ const key = `lesson-${slug}`;
50
+ const value = [
51
+ `TASK: ${task}`,
52
+ tried ? `TRIED(failed): ${tried}` : null,
53
+ worked ? `WORKED: ${worked}` : null,
54
+ critique ? `CRITIQUE: ${critique}` : null,
55
+ `OUTCOME: ${outcome}`,
56
+ ].filter(Boolean).join(' ');
57
+
58
+ const ruflo = (args) =>
59
+ execFileSync('ruflo', args, { cwd: dir, encoding: 'utf8', timeout: 60000 });
60
+
61
+ console.log(`\nRecording lesson into ${path.basename(dir)}/.swarm/memory.db (namespace: ${ns})`);
62
+ console.log(` key: ${key}`);
63
+
64
+ // 1. STORE (native, signal namespace) — L1 content + L2 embedding
65
+ let stored = false;
66
+ try {
67
+ const out = ruflo(['memory', 'store', '-k', key, '-n', ns, '--value', value]);
68
+ stored = /OK|stored/i.test(out);
69
+ } catch (e) {
70
+ console.error(' store FAILED:', String(e.stdout || e.message).split('\n')[0]);
71
+ process.exit(1);
72
+ }
73
+ console.log(` 1. store -> ${stored ? 'OK' : '?'}`);
74
+
75
+ // 2. REFINE (native) — L3 patterns + L4 episodes
76
+ let batchEpisodes = '?';
77
+ let distillOk = false;
78
+ try {
79
+ const dist = ruflo(['memory', 'distill', 'run']);
80
+ const m = dist.match(/Episodes\s*\|\s*(\d+)/i);
81
+ if (m) batchEpisodes = m[1];
82
+ distillOk = true;
83
+ } catch (e) {
84
+ /* distill is best-effort; the store already succeeded */
85
+ }
86
+ // DERIVED, not asserted (F15): say what actually happened — the old line printed "refined into
87
+ // episodes+patterns" even when distill threw.
88
+ console.log(distillOk
89
+ ? ` 2. distill -> refined into episodes+patterns (batch: ${batchEpisodes})`
90
+ : ' 2. distill -> FAILED (best-effort; the raw lesson is stored, refinement will catch up on a later distill)');
91
+
92
+ // 3. VERIFY recall by the task text (paraphrase-ish), filtered to the namespace
93
+ let recalled = false;
94
+ try {
95
+ const search = ruflo(['memory', 'search', '-q', task, '-n', ns]);
96
+ recalled = search.includes(key.slice(0, 16));
97
+ } catch (e) {
98
+ /* search failure shouldn't fail the record */
99
+ }
100
+ console.log(
101
+ ` 3. recall -> ${
102
+ recalled
103
+ ? `✅ "${task.slice(0, 44)}…" returns ${key}`
104
+ : '⚠️ not the top in-namespace hit (stored fine; ranking improves as signal grows)'
105
+ }`,
106
+ );
107
+
108
+ // DERIVED, not asserted (F15): the closing line reports exactly what was verified, never more. The
109
+ // old line claimed "captured, refined, and recall-verified" even when distill failed and recall
110
+ // didn't return the key — asserted prose over an honest exit code.
111
+ const parts = ['captured', distillOk ? 'refined' : 'NOT refined (distill failed)', recalled ? 'recall-verified' : 'recall NOT verified'];
112
+ console.log(`\nDone. Lesson is ${parts.join(', ')}.\n`);
113
+ process.exit(stored ? 0 : 1);