ruvnet-brain 3.9.134-dev → 4.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (218) hide show
  1. package/.claude-plugin/marketplace.json +14 -0
  2. package/README.md +5 -5
  3. package/bin/install.mjs +382 -36
  4. package/console/CONTRACT.md +172 -0
  5. package/console/activity.js +753 -0
  6. package/console/app.js +4189 -0
  7. package/console/architecture.html +1221 -0
  8. package/console/assets/depth-1.webp +0 -0
  9. package/console/assets/depth-2.webp +0 -0
  10. package/console/assets/depth-3.webp +0 -0
  11. package/console/assets/harness-vs-plain.svg +259 -0
  12. package/console/assets/hero.webp +0 -0
  13. package/console/assets/memory.webp +0 -0
  14. package/console/assets/metaharness.svg +247 -0
  15. package/console/index.html +777 -0
  16. package/console/install-architecture.html +162 -0
  17. package/console/install-mockup.html +543 -0
  18. package/console/style.css +2144 -0
  19. package/console/tips.css +926 -0
  20. package/console/tips.html +858 -0
  21. package/console/tips.js +128 -0
  22. package/docs/RELEASE-NOTES-4.0.md +88 -0
  23. package/kb/model-requirements.mjs +37 -6
  24. package/kb/zip-extract.mjs +53 -14
  25. package/keys/ruvnet-brain-signing.pub.pem +3 -0
  26. package/package.json +14 -22
  27. package/plugin/.claude-plugin/marketplace.json +14 -0
  28. package/plugin/.claude-plugin/plugin.json +22 -0
  29. package/plugin/.codex-plugin/plugin.json +21 -0
  30. package/plugin/.mcp.json +8 -0
  31. package/plugin/commands/brain-console.md +16 -0
  32. package/plugin/commands/configure.md +33 -0
  33. package/plugin/commands/rvbc.md +79 -0
  34. package/plugin/commands/rvcb.md +16 -0
  35. package/plugin/commands/whats-new.md +57 -0
  36. package/plugin/hooks/codex-hooks.json +160 -0
  37. package/plugin/hooks/hook-contracts.json +77 -0
  38. package/plugin/hooks/hooks.json +202 -0
  39. package/plugin/mcp/managed-cli-interface.mjs +47 -4
  40. package/plugin/mcp/server.mjs +56 -6
  41. package/plugin/scripts/anticipate.sh +534 -0
  42. package/plugin/scripts/codex-hook-adapter.mjs +96 -0
  43. package/plugin/scripts/continuation-gate.mjs +267 -0
  44. package/plugin/scripts/design-wall.sh +137 -0
  45. package/plugin/scripts/detach.mjs +182 -0
  46. package/plugin/scripts/first-session-worker.mjs +38 -0
  47. package/plugin/scripts/gate-receipt.sh +35 -0
  48. package/plugin/scripts/ground-before-write.sh +199 -0
  49. package/plugin/scripts/ground-ruvnet.sh +517 -0
  50. package/plugin/scripts/grounding-stamp.sh +113 -0
  51. package/plugin/scripts/grounding-substance.mjs +595 -0
  52. package/plugin/scripts/hijack-ruvnet.sh +81 -0
  53. package/plugin/scripts/hook-input.mjs +558 -0
  54. package/plugin/scripts/hook-shim-bash.mjs +55 -0
  55. package/plugin/scripts/hook-shim.mjs +303 -0
  56. package/plugin/scripts/host-update.mjs +58 -0
  57. package/plugin/scripts/kling-preflight.sh +146 -0
  58. package/plugin/scripts/learn-capture.sh +173 -0
  59. package/plugin/scripts/learn-flush.mjs +155 -0
  60. package/plugin/scripts/lesson-hooks.sh +213 -0
  61. package/plugin/scripts/md-stamp.mjs +219 -0
  62. package/plugin/scripts/protect-brain-state.sh +84 -0
  63. package/plugin/scripts/route-dispatch.sh +147 -0
  64. package/plugin/scripts/routing-outcome-capture.mjs +89 -0
  65. package/plugin/scripts/runtime-preferences.mjs +269 -0
  66. package/plugin/scripts/session-start-core.mjs +477 -0
  67. package/plugin/scripts/session-start.sh +13 -0
  68. package/plugin/scripts/signal-watch.mjs +193 -0
  69. package/plugin/scripts/unprompted-runtime.mjs +377 -0
  70. package/plugin/scripts/update-apply.mjs +419 -0
  71. package/plugin/scripts/verify-interface.sh +53 -0
  72. package/plugin/scripts/version-bump-gate.sh +112 -0
  73. package/plugin/skills/brain-build/SKILL.md +123 -0
  74. package/plugin/skills/brain-console/SKILL.md +22 -0
  75. package/plugin/skills/brain-prompt/SKILL.md +83 -0
  76. package/plugin/skills/brain-score/SKILL.md +101 -0
  77. package/plugin/skills/release-proof/SKILL.md +81 -0
  78. package/plugin/skills/release-proof/agents/openai.yaml +4 -0
  79. package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
  80. package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
  81. package/plugin/skills/ruvnet-brain/PLAYBOOK.md +121 -0
  82. package/plugin/skills/ruvnet-brain/SKILL.md +234 -0
  83. package/plugin/skills/rvbc/SKILL.md +23 -0
  84. package/plugin/skills/savings/SKILL.md +46 -0
  85. package/plugin/skills/whats-new/SKILL.md +22 -0
  86. package/scripts/adr-backfill.mjs +107 -0
  87. package/scripts/advocacy-outcomes.mjs +808 -0
  88. package/scripts/agentdb-context.mjs +216 -0
  89. package/scripts/agentdb-fleet-doctor.mjs +101 -0
  90. package/scripts/ascii-drift.mjs +236 -0
  91. package/scripts/behavioral-l1-l4.mjs +210 -0
  92. package/scripts/brain-capability-check.mjs +72 -0
  93. package/scripts/brain-grade-groundtruth.mjs +100 -0
  94. package/scripts/brain-latency-50.mjs +227 -0
  95. package/scripts/brain-novice-50.mjs +189 -0
  96. package/scripts/brain-stamp.mjs +94 -0
  97. package/scripts/brain-state.mjs +212 -0
  98. package/scripts/build-bundle.mjs +522 -0
  99. package/scripts/build-concepts.mjs +132 -0
  100. package/scripts/build-l2.mjs +71 -0
  101. package/scripts/build-primer.mjs +73 -0
  102. package/scripts/build-symbols.mjs +68 -0
  103. package/scripts/calibrate-router.mjs +97 -0
  104. package/scripts/capability-audit.mjs +321 -0
  105. package/scripts/capability-registry.mjs +876 -0
  106. package/scripts/check-indexation.mjs +108 -0
  107. package/scripts/check-legibility.mjs +189 -0
  108. package/scripts/ci/build-fixture-kb.mjs +67 -0
  109. package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
  110. package/scripts/ci/learning-replay-recorder.mjs +59 -0
  111. package/scripts/ci/mutate-hook-timeout.mjs +70 -0
  112. package/scripts/ci/stranger-fixture-stage.mjs +17 -0
  113. package/scripts/ci/stranger-scenario.mjs +228 -0
  114. package/scripts/ci/stranger-timeout.mjs +25 -0
  115. package/scripts/ci-verdict.mjs +29 -0
  116. package/scripts/claims-verify.mjs +710 -0
  117. package/scripts/clear-claude-tmp.sh +31 -0
  118. package/scripts/console-engine.mjs +434 -0
  119. package/scripts/console-engine.test.mjs +125 -0
  120. package/scripts/corpus-qa.mjs +250 -0
  121. package/scripts/correction-detect-embed.mjs +346 -0
  122. package/scripts/correction-detect-measure.mjs +270 -0
  123. package/scripts/correction-detect.mjs +686 -0
  124. package/scripts/count-chunks.mjs +54 -0
  125. package/scripts/described-questions.json +30 -0
  126. package/scripts/design-grade.mjs +58 -0
  127. package/scripts/dev-plugin-link.sh +105 -0
  128. package/scripts/distill-project.mjs +200 -0
  129. package/scripts/doc-currency.mjs +801 -0
  130. package/scripts/eval-brain.mjs +244 -0
  131. package/scripts/fix-metaharness-memretrieve.mjs +121 -0
  132. package/scripts/full-hints.mjs +87 -0
  133. package/scripts/gate.sh +39 -0
  134. package/scripts/gates.mjs +146 -0
  135. package/scripts/gen-console-images.mjs +54 -0
  136. package/scripts/gen-images.mjs +47 -0
  137. package/scripts/git-clone-refresh.mjs +52 -0
  138. package/scripts/git-hooks/pre-push +126 -0
  139. package/scripts/goal-match.mjs +398 -0
  140. package/scripts/goldie-research.mjs +223 -0
  141. package/scripts/goldie-weekly.sh +67 -0
  142. package/scripts/health-repair.mjs +250 -0
  143. package/scripts/helix-scenario-questions.json +10 -0
  144. package/scripts/ingest-gists.mjs +230 -0
  145. package/scripts/ingest-meeting.mjs +115 -0
  146. package/scripts/ingest-repo.mjs +79 -0
  147. package/scripts/install-npx-witness.sh +49 -0
  148. package/scripts/issue-fix.mjs +639 -0
  149. package/scripts/issue-watch.mjs +276 -0
  150. package/scripts/issue4-close-note.md +31 -0
  151. package/scripts/key-canary.mjs +91 -0
  152. package/scripts/latency-to-surface.mjs +233 -0
  153. package/scripts/learning-enable.mjs +380 -0
  154. package/scripts/learning-replay.mjs +1570 -0
  155. package/scripts/learnings.mjs +62 -0
  156. package/scripts/lesson-gate.mjs +680 -0
  157. package/scripts/lesson-lifecycle.mjs +449 -0
  158. package/scripts/lesson-promote.mjs +262 -0
  159. package/scripts/lesson-ratify.mjs +98 -0
  160. package/scripts/lesson-seed.mjs +252 -0
  161. package/scripts/lesson-store.mjs +447 -0
  162. package/scripts/loop-checkpoint.mjs +86 -0
  163. package/scripts/memdb-health.sh +14 -0
  164. package/scripts/memory-doctor.mjs +271 -0
  165. package/scripts/model-catalog.mjs +79 -0
  166. package/scripts/nightly-controller.mjs +66 -0
  167. package/scripts/nightly-gists.sh +72 -0
  168. package/scripts/nightly-wrapper.sh +180 -0
  169. package/scripts/notify.sh +12 -0
  170. package/scripts/npx-witness.sh +56 -0
  171. package/scripts/onboarding-console.mjs +2749 -0
  172. package/scripts/private-fence.mjs +69 -0
  173. package/scripts/proactivity-metrics.mjs +118 -0
  174. package/scripts/proof-questions.json +56 -0
  175. package/scripts/prove.mjs +95 -0
  176. package/scripts/proxy/claude-proxied.sh +57 -0
  177. package/scripts/proxy/proxy-revert.sh +59 -0
  178. package/scripts/proxy/proxy-up.sh +60 -0
  179. package/scripts/proxy/proxy-verify.mjs +142 -0
  180. package/scripts/published-surface-probe.mjs +241 -0
  181. package/scripts/qe/card-lane-gate.mjs +162 -0
  182. package/scripts/qe/session-start-gate.mjs +229 -0
  183. package/scripts/qe/ux-suite.mjs +323 -0
  184. package/scripts/reconcile-project.mjs +0 -0
  185. package/scripts/record-lesson.mjs +113 -0
  186. package/scripts/refresh-model-catalog.mjs +99 -0
  187. package/scripts/release-proof.mjs +9 -0
  188. package/scripts/release-vector.mjs +281 -0
  189. package/scripts/release.mjs +395 -0
  190. package/scripts/remedy-registry.mjs +247 -0
  191. package/scripts/rerank-cap-eval.mjs +265 -0
  192. package/scripts/rerank-cap-warm-ab.mjs +129 -0
  193. package/scripts/route-cheap.mjs +20 -15
  194. package/scripts/router-utilization.mjs +182 -0
  195. package/scripts/routing-flywheel.mjs +596 -0
  196. package/scripts/rvf-generation.mjs +104 -0
  197. package/scripts/rvf-index-audit.mjs +138 -0
  198. package/scripts/self-update.mjs +508 -0
  199. package/scripts/selfcheck.mjs +7 -1
  200. package/scripts/sign-bundle.mjs +69 -0
  201. package/scripts/signal-watch.mjs +171 -0
  202. package/scripts/stack-sync.mjs +469 -0
  203. package/scripts/stamp-existing-rvf-generations.mjs +53 -0
  204. package/scripts/stamp-sweep.mjs +144 -0
  205. package/scripts/status-honesty.mjs +102 -0
  206. package/scripts/sync-version.mjs +217 -0
  207. package/scripts/token-report.mjs +102 -0
  208. package/scripts/top100-benchmark.mjs +479 -0
  209. package/scripts/top100-corpus.mjs +112 -0
  210. package/scripts/top100-semantic-assertions.mjs +449 -0
  211. package/scripts/update-apply.mjs +9 -0
  212. package/scripts/upgrade-notice.mjs +14 -0
  213. package/scripts/verify-bundle.mjs +51 -0
  214. package/scripts/verify-channels.mjs +184 -0
  215. package/scripts/verify-model-catalog.mjs +104 -0
  216. package/scripts/verify-nightly-close-issue4.sh +31 -0
  217. package/scripts/version.mjs +40 -0
  218. package/scripts/wired-check.mjs +864 -0
@@ -0,0 +1,270 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * correction-detect-measure.mjs — the measurement harness ADR-033 §2 requires and correction-
4
+ * detect.mjs's own header cites, but that never existed as a script: something that walks the REAL
5
+ * transcript corpus, builds the (promptText, precedingAssistantAction) pairs detectCorrection()
6
+ * actually consumes, runs the detector, and reports precision/recall on a held-out split — so the
7
+ * numbers in correction-detect.mjs's header are reproducible, not asserted.
8
+ *
9
+ * WHY A FILE-LEVEL TUNE/HOLDOUT SPLIT, NOT A RANDOM ROW SPLIT. Two rows from the same transcript
10
+ * session are not independent draws — they share the user's phrasing habits for that session, and
11
+ * sometimes repeat the same correction verbatim minutes apart. Splitting by ROW would leak: a
12
+ * heuristic tuned on one half of a duplicated correction would trivially "recall" the other half.
13
+ * Splitting by FILE (a deterministic hash of the transcript filename, not a random seed, so re-runs
14
+ * are reproducible) keeps every row from one session on one side of the line.
15
+ *
16
+ * WHAT THIS DOES NOT DO. It does not hand-label anything — that is a human's job, and doing it
17
+ * mechanically here would be the exact "the fixture cannot falsify its own choice" trap this
18
+ * measurement exists to avoid. `--dump-pool` writes a lexically-loose CANDIDATE POOL (a superset of
19
+ * what the real detector would ever fire on) to a file OUTSIDE this repo by default, for a human to
20
+ * read and label true/false. Real user transcripts can contain secrets, business content, or simply
21
+ * more of a conversation than its owner intends to publish — this script never writes transcript
22
+ * text into the repo, and the default output path is under the OS temp directory for exactly that
23
+ * reason. Committing labelled examples back into tests/unit/correction-detect.test.mjs is a separate,
24
+ * deliberate, human-reviewed step (see that file's "FROM THE REAL CORPUS" entries for the precedent).
25
+ *
26
+ * USAGE
27
+ * node scripts/correction-detect-measure.mjs
28
+ * Reports adjacency-candidate counts and detection counts/rates, split tune vs. holdout, for
29
+ * whichever corpus directory is being read (default: this project's own Claude Code transcript
30
+ * directory, i.e. `~/.claude/projects/<mangled-cwd>`).
31
+ *
32
+ * node scripts/correction-detect-measure.mjs --corpus-dir <path>
33
+ * Point at a different transcript directory (e.g. to reproduce this measurement on someone
34
+ * else's machine, or a different project's history).
35
+ *
36
+ * node scripts/correction-detect-measure.mjs --dump-pool <path> [--split tune|holdout|all]
37
+ * Additionally writes the loose-net candidate pool (JSONL: file, turnIndex, promptText,
38
+ * precedingAssistantAction, split, detectorResult) to <path> for hand-labelling. Defaults to
39
+ * tune+holdout combined; pass --split to isolate one side.
40
+ *
41
+ * BROADENING RESULT, 2026-07-24 (agent-directed-imperative signal, see correction-detect.mjs).
42
+ *
43
+ * Acting on the recall finding below: added one gated signal class — directives whose object is the
44
+ * agent ("I want you to X", "you need to Y"), which the quantifier net structurally could not see —
45
+ * requiring strong rejection valence so a first-time request stays silent. MEASURED effect on the
46
+ * real corpus: holdout firings 3 -> 10 (total 7 -> 19). All 94 detector unit tests stay green, so no
47
+ * blind-rater-certified case regressed.
48
+ *
49
+ * THE TRADE, stated honestly: single-rater inspection of the 10 holdout firings read ~7 genuine
50
+ * corrections and ~3 borderline false positives (directives phrased as questions). That is a ~70%
51
+ * point estimate — HIGHER RECALL, slightly LOWER precision point-estimate than the narrow net's
52
+ * 77.8%. Neither is certifiable: 10 firings still cannot bound precision above 74.1% (needs n>=29),
53
+ * and single-rater labels are direction-finding, not the blind 3-rater majority a floor claim needs.
54
+ * The structural win is the VOLUME: 10 is most of the way to the n>=29 that any future >=90%
55
+ * certification requires — you cannot certify a floor on n=3 at all. Deliberately NOT tuned further
56
+ * against the holdout firings above: tuning on the set you measure with corrupts the only unbiased
57
+ * measurement you have. Further tuning belongs on the tune split, with blind labelling.
58
+ *
59
+ * HAND-LABELLED FINDINGS, 2026-07-24 — N3 IS A RECALL PROBLEM, NOT A PRECISION PROBLEM.
60
+ *
61
+ * The open work item read "raise correction-detect precision 27% -> 90%". A hand-labelling pass over
62
+ * this pool says that framing is wrong, and it is worth writing down before anyone tunes a regex again.
63
+ *
64
+ * PRECISION, holdout firings: 2 of 3 correct. Also uncertifiable — see the certifiability block at
65
+ * the bottom of main(): three firings cannot bound precision above 36.8% no matter what, and >=90%
66
+ * needs n >= 29. No regex change moves that; only more firings do.
67
+ *
68
+ * BASE RATE, 28-row holdout sample of NON-firing candidates: after discarding harness artifacts,
69
+ * 13 of 20 real user turns (65%) were genuine corrections the detector did not catch.
70
+ *
71
+ * RECALL, extrapolated over 156 holdout non-firings: roughly 73 missed against 2 caught, i.e.
72
+ * ABOUT 3%. The detector misses ~97% of the corrections in front of it.
73
+ *
74
+ * So precision was never the binding constraint. And the two problems share ONE fix: broadening the
75
+ * net raises recall AND produces the firing volume that certifying precision requires. Tuning for
76
+ * precision on n=3 does neither, while looking like progress.
77
+ *
78
+ * CAVEAT ON THESE LABELS, stated because it bounds them: they are ONE rater's judgement (mine), not
79
+ * the blind 3-rater majority the earlier 77.8% figure used. Treat them as a direction-finding
80
+ * measurement that reframes the problem, not as a certified precision number. The certified number
81
+ * still requires the volume above.
82
+ *
83
+ * THE NUMBERS THIS PRODUCED ON 2026-07-23 are recorded in correction-detect.mjs's own header
84
+ * (search that file for "MEASURED ON THE REAL CORPUS, 2026-07-23") rather than duplicated here,
85
+ * since a measurement script that also claims to BE the measurement is how numbers rot out of sync
86
+ * with the code they describe.
87
+ */
88
+
89
+ import fs from 'node:fs';
90
+ import os from 'node:os';
91
+ import path from 'node:path';
92
+ import crypto from 'node:crypto';
93
+ import readline from 'node:readline';
94
+ import { detectCorrection, HARNESS_TEMPLATES } from './correction-detect.mjs';
95
+
96
+ const argv = process.argv.slice(2);
97
+ const flag = (name, fallback = null) => {
98
+ const i = argv.indexOf(name);
99
+ return i >= 0 && argv[i + 1] ? argv[i + 1] : fallback;
100
+ };
101
+
102
+ const defaultCorpusDir = path.join(
103
+ os.homedir(), '.claude', 'projects',
104
+ process.cwd().replace(/\//g, '-'),
105
+ );
106
+ const CORPUS_DIR = flag('--corpus-dir', defaultCorpusDir);
107
+ const DUMP_POOL = flag('--dump-pool', null);
108
+ const SPLIT_FILTER = flag('--split', 'all'); // tune | holdout | all
109
+
110
+ /** Deterministic 55/45 tune/holdout split BY TRANSCRIPT FILE, fixed so re-runs are reproducible. */
111
+ function splitOf(fileName) {
112
+ const h = crypto.createHash('md5').update(fileName).digest('hex');
113
+ const n = parseInt(h.slice(0, 8), 16) % 100;
114
+ return n < 55 ? 'tune' : 'holdout';
115
+ }
116
+
117
+ /** Best-effort one-line description of a tool_use block, mirroring how a real hook would summarise it. */
118
+ function summarizeToolUse(block) {
119
+ const name = block.name || 'unknown';
120
+ const input = block.input || {};
121
+ let detail;
122
+ if (name === 'Bash') detail = input.command;
123
+ else if (['Edit', 'Write', 'NotebookEdit', 'Read'].includes(name)) detail = input.file_path;
124
+ else if (['Grep', 'Glob'].includes(name)) detail = input.pattern;
125
+ else if (name === 'Task') detail = input.description || input.prompt;
126
+ else if (name === 'TodoWrite') detail = 'todo update';
127
+ else detail = JSON.stringify(input);
128
+ return { tool: name, summary: String(detail ?? '').slice(0, 200) };
129
+ }
130
+
131
+ /**
132
+ * Walk one transcript, emitting a candidate row for every genuinely-typed user turn (string
133
+ * `message.content`, never a tool_result array) that has SOME preceding assistant turn — a tool
134
+ * action if the assistant's last message used one, otherwise a truncated summary of what it said.
135
+ * A pure-text-then-nothing-since boundary correctly resets this to null, matching Signal 1's actual
136
+ * meaning: "is there something for this utterance to be responding to."
137
+ */
138
+ async function extractFromFile(file) {
139
+ const rows = [];
140
+ let lastAssistantAction = null;
141
+ let turnIndex = 0;
142
+ const rl = readline.createInterface({ input: fs.createReadStream(file, { encoding: 'utf8' }), crlfDelay: Infinity });
143
+ for await (const line of rl) {
144
+ if (!line.trim()) continue;
145
+ let obj;
146
+ try { obj = JSON.parse(line); } catch { continue; }
147
+
148
+ if (obj.type === 'assistant' && obj.message && Array.isArray(obj.message.content)) {
149
+ const toolUses = obj.message.content.filter((b) => b && b.type === 'tool_use');
150
+ if (toolUses.length) {
151
+ lastAssistantAction = summarizeToolUse(toolUses[toolUses.length - 1]);
152
+ } else {
153
+ const text = obj.message.content.filter((b) => b && b.type === 'text' && b.text).map((b) => b.text).join(' ').trim();
154
+ lastAssistantAction = text ? { tool: null, summary: text.slice(0, 200) } : null;
155
+ }
156
+ continue;
157
+ }
158
+
159
+ if (obj.type === 'user' && obj.message && typeof obj.message.content === 'string') {
160
+ turnIndex += 1;
161
+ if (lastAssistantAction && (lastAssistantAction.summary || lastAssistantAction.tool)) {
162
+ rows.push({
163
+ file: path.basename(file), turnIndex, timestamp: obj.timestamp,
164
+ promptText: obj.message.content, precedingAssistantAction: lastAssistantAction,
165
+ });
166
+ }
167
+ }
168
+ }
169
+ return rows;
170
+ }
171
+
172
+ /** A deliberately LOOSE lexical net — a superset of every signal the real detector requires — used
173
+ * only to build a candidate pool small enough for a human to hand-label, never to decide anything. */
174
+ const BROAD_NET = /\b(?:always|never|no longer|no more|constantly|repeatedly|stop\b|don'?t\b|do not\b|quit\b|wrong\b|incorrect\b|instead of|rather than|isn'?t what|why (?:did|didn'?t|are|aren'?t) you|you keep|you always|i (?:told|asked) you|already (?:told|asked|said)|should have|failed to|forgot to|from now on|going forward|in the future|henceforth|next time|that'?s not|not what i (?:asked|wanted|said))\b/i;
175
+
176
+ async function main() {
177
+ let files;
178
+ try {
179
+ files = fs.readdirSync(CORPUS_DIR).filter((f) => f.endsWith('.jsonl')).map((f) => path.join(CORPUS_DIR, f));
180
+ } catch (e) {
181
+ console.error(`Cannot read corpus dir ${CORPUS_DIR}: ${e.message}`);
182
+ console.error('Pass --corpus-dir <path> to point at a real Claude Code transcript directory.');
183
+ process.exit(1);
184
+ }
185
+ console.error(`[measure] ${files.length} transcript file(s) in ${CORPUS_DIR}`);
186
+
187
+ const bySplit = { tune: { total: 0, hits: 0 }, holdout: { total: 0, hits: 0 } };
188
+ const poolRows = [];
189
+ let userTurns = 0;
190
+
191
+ for (const file of files) {
192
+ let rows;
193
+ try { rows = await extractFromFile(file); } catch (e) { console.error(`[measure] skip ${file}: ${e.message}`); continue; }
194
+ const split = splitOf(path.basename(file));
195
+ for (const row of rows) {
196
+ userTurns += 1;
197
+ bySplit[split].total += 1;
198
+ const got = detectCorrection(row.promptText, {
199
+ precedingAssistantAction: row.precedingAssistantAction,
200
+ transcriptPath: row.file, turnIndex: row.turnIndex, timestamp: row.timestamp,
201
+ });
202
+ if (got) bySplit[split].hits += 1;
203
+
204
+ // Harness artifacts are excluded from the LABELLING POOL, not just from detection. The
205
+ // detector already rejects them (correction-detect.mjs HARNESS_TEMPLATES), so they could never
206
+ // fire — but they were still written out for a human to label. MEASURED in a 28-row holdout
207
+ // sample: 8 of them (29%) were <local-command-caveat> blocks, i.e. a third of the labelling
208
+ // effort spent on rows that are not user speech and whose answer is definitionally "no".
209
+ // Labelled examples are the scarcest resource in this problem; spending 29% of them on
210
+ // harness noise is why the pool looked bigger than it usefully was.
211
+ const isArtifact = HARNESS_TEMPLATES.some((re) => re.test(row.promptText));
212
+ if (DUMP_POOL && !isArtifact && (SPLIT_FILTER === 'all' || SPLIT_FILTER === split)
213
+ && row.promptText.length <= 2000 && BROAD_NET.test(row.promptText)) {
214
+ poolRows.push({ ...row, split, detectorResult: got });
215
+ }
216
+ }
217
+ }
218
+
219
+ const total = bySplit.tune.total + bySplit.holdout.total;
220
+ const hits = bySplit.tune.hits + bySplit.holdout.hits;
221
+ console.log(`\nadjacency candidates (Signal 1): ${total} (tune ${bySplit.tune.total} / holdout ${bySplit.holdout.total})`);
222
+ console.log(`detections: ${hits} (tune ${bySplit.tune.hits} / holdout ${bySplit.holdout.hits})`);
223
+ console.log(`rate: ${(100 * hits / total).toFixed(3)}% (tune ${(100 * bySplit.tune.hits / bySplit.tune.total).toFixed(3)}% / holdout ${(100 * bySplit.holdout.hits / bySplit.holdout.total).toFixed(3)}%)`);
224
+ console.log(`\nPrecision and recall require HAND-LABELLING — this script only counts firings.`);
225
+ console.log(`Use --dump-pool <path> to write a labellable candidate pool; the holdout half is the`);
226
+ console.log(`only one whose precision/recall counts as an unbiased measurement.`);
227
+
228
+ // ── CAN THE ≥90% FLOOR EVEN BE CERTIFIED FROM THIS MUCH DATA? ────────────────────────────────────
229
+ // ADR-033 holds lesson auto-extraction behind a ≥90% precision floor, and the open work item read
230
+ // "raise correction-detect precision 27% -> 90%" — which frames it as a TUNING problem. It is not,
231
+ // and stating the arithmetic here is what stops it being mistaken for one again.
232
+ //
233
+ // Precision is estimated from the detections, not from the candidate pool, so the holdout FIRING
234
+ // count is the sample size. With every single detection correct, the exact (Clopper-Pearson) 95%
235
+ // one-sided lower bound is p = alpha^(1/n) — closed form, checkable by hand: for n=3 that is the
236
+ // cube root of 0.05, 36.8%. Clearing 90% needs n >= 29 CONSECUTIVE correct detections, and any
237
+ // error pushes the requirement higher still.
238
+ //
239
+ // So a "77.8% at n=19" measurement cannot certify a 90% floor even in principle — at n=19 a PERFECT
240
+ // 19/19 bounds at only 85.4%. Tuning the regex until the point estimate crosses 0.90 on a sample
241
+ // this small is fitting the sample, not the property, and it would produce exactly the confident
242
+ // wrong number this project keeps catching elsewhere.
243
+ const holdoutHits = bySplit.holdout.hits;
244
+ const lowerBoundIfPerfect = holdoutHits > 0 ? Math.pow(0.05, 1 / holdoutHits) : 0;
245
+ const N_FOR_90 = 29;
246
+ console.log(`\ncertifiability of the ADR-033 >=90% precision floor, from THIS run:`);
247
+ console.log(` holdout detections (the precision sample) : ${holdoutHits}`);
248
+ console.log(` best possible 95% lower bound (all correct): ${(lowerBoundIfPerfect * 100).toFixed(1)}%`);
249
+ if (lowerBoundIfPerfect >= 0.90) {
250
+ console.log(` => the sample is LARGE ENOUGH to certify 90% — hand-label the holdout detections.`);
251
+ } else {
252
+ console.log(` => NOT CERTIFIABLE at any precision: ${holdoutHits} detections cannot bound above`);
253
+ console.log(` ${(lowerBoundIfPerfect * 100).toFixed(1)}%, and >=90% requires n >= ${N_FOR_90} consecutive correct.`);
254
+ console.log(` N3 is blocked on LABELLED VOLUME, not on the detector. Tuning against a sample`);
255
+ console.log(` this small overfits it. The unblock is more transcript corpus (or a broader net`);
256
+ console.log(` that fires more often), then hand-labelling — not another regex pass.`);
257
+ }
258
+
259
+ if (DUMP_POOL) {
260
+ fs.writeFileSync(DUMP_POOL, poolRows.map((r) => JSON.stringify(r)).join('\n') + (poolRows.length ? '\n' : ''));
261
+ console.log(`\nWrote ${poolRows.length} candidate(s) to ${DUMP_POOL} for hand-labelling.`);
262
+ console.log(`This file may contain real transcript text — do not commit it into the repo.`);
263
+ }
264
+ }
265
+
266
+ const invokedDirectly = process.argv[1]
267
+ && path.resolve(process.argv[1]).endsWith(`correction-detect-measure${path.extname(process.argv[1])}`);
268
+ if (invokedDirectly) main();
269
+
270
+ export { extractFromFile, splitOf, BROAD_NET };