ruvnet-brain 4.0.1 → 4.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -0
- package/README.md +4 -4
- package/bin/install.mjs +100 -5
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +8 -22
- package/plugin/.claude-plugin/marketplace.json +1 -0
- package/plugin/.claude-plugin/plugin.json +2 -3
- package/plugin/.codex-plugin/plugin.json +1 -1
- package/plugin/commands/brain-console.md +2 -2
- package/plugin/commands/configure.md +3 -2
- package/plugin/commands/rvbc.md +4 -3
- package/plugin/commands/rvcb.md +2 -2
- package/plugin/hooks/hooks.json +1 -2
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +21 -0
- package/plugin/scripts/detach.mjs +14 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/ground-ruvnet.sh +16 -6
- package/plugin/scripts/hook-shim.mjs +7 -7
- package/plugin/scripts/learn-capture.sh +22 -3
- package/plugin/scripts/learn-flush.mjs +21 -4
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +477 -0
- package/plugin/scripts/session-start.sh +3 -858
- package/plugin/skills/brain-console/SKILL.md +4 -2
- package/plugin/skills/release-proof/SKILL.md +81 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +5 -1
- package/plugin/skills/rvbc/SKILL.md +9 -6
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +522 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +250 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +639 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +271 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +180 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2749 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +395 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +508 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +864 -0
- package/plugin/scripts/finalize-token-meter.mjs +0 -25
|
@@ -0,0 +1,270 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* correction-detect-measure.mjs — the measurement harness ADR-033 §2 requires and correction-
|
|
4
|
+
* detect.mjs's own header cites, but that never existed as a script: something that walks the REAL
|
|
5
|
+
* transcript corpus, builds the (promptText, precedingAssistantAction) pairs detectCorrection()
|
|
6
|
+
* actually consumes, runs the detector, and reports precision/recall on a held-out split — so the
|
|
7
|
+
* numbers in correction-detect.mjs's header are reproducible, not asserted.
|
|
8
|
+
*
|
|
9
|
+
* WHY A FILE-LEVEL TUNE/HOLDOUT SPLIT, NOT A RANDOM ROW SPLIT. Two rows from the same transcript
|
|
10
|
+
* session are not independent draws — they share the user's phrasing habits for that session, and
|
|
11
|
+
* sometimes repeat the same correction verbatim minutes apart. Splitting by ROW would leak: a
|
|
12
|
+
* heuristic tuned on one half of a duplicated correction would trivially "recall" the other half.
|
|
13
|
+
* Splitting by FILE (a deterministic hash of the transcript filename, not a random seed, so re-runs
|
|
14
|
+
* are reproducible) keeps every row from one session on one side of the line.
|
|
15
|
+
*
|
|
16
|
+
* WHAT THIS DOES NOT DO. It does not hand-label anything — that is a human's job, and doing it
|
|
17
|
+
* mechanically here would be the exact "the fixture cannot falsify its own choice" trap this
|
|
18
|
+
* measurement exists to avoid. `--dump-pool` writes a lexically-loose CANDIDATE POOL (a superset of
|
|
19
|
+
* what the real detector would ever fire on) to a file OUTSIDE this repo by default, for a human to
|
|
20
|
+
* read and label true/false. Real user transcripts can contain secrets, business content, or simply
|
|
21
|
+
* more of a conversation than its owner intends to publish — this script never writes transcript
|
|
22
|
+
* text into the repo, and the default output path is under the OS temp directory for exactly that
|
|
23
|
+
* reason. Committing labelled examples back into tests/unit/correction-detect.test.mjs is a separate,
|
|
24
|
+
* deliberate, human-reviewed step (see that file's "FROM THE REAL CORPUS" entries for the precedent).
|
|
25
|
+
*
|
|
26
|
+
* USAGE
|
|
27
|
+
* node scripts/correction-detect-measure.mjs
|
|
28
|
+
* Reports adjacency-candidate counts and detection counts/rates, split tune vs. holdout, for
|
|
29
|
+
* whichever corpus directory is being read (default: this project's own Claude Code transcript
|
|
30
|
+
* directory, i.e. `~/.claude/projects/<mangled-cwd>`).
|
|
31
|
+
*
|
|
32
|
+
* node scripts/correction-detect-measure.mjs --corpus-dir <path>
|
|
33
|
+
* Point at a different transcript directory (e.g. to reproduce this measurement on someone
|
|
34
|
+
* else's machine, or a different project's history).
|
|
35
|
+
*
|
|
36
|
+
* node scripts/correction-detect-measure.mjs --dump-pool <path> [--split tune|holdout|all]
|
|
37
|
+
* Additionally writes the loose-net candidate pool (JSONL: file, turnIndex, promptText,
|
|
38
|
+
* precedingAssistantAction, split, detectorResult) to <path> for hand-labelling. Defaults to
|
|
39
|
+
* tune+holdout combined; pass --split to isolate one side.
|
|
40
|
+
*
|
|
41
|
+
* BROADENING RESULT, 2026-07-24 (agent-directed-imperative signal, see correction-detect.mjs).
|
|
42
|
+
*
|
|
43
|
+
* Acting on the recall finding below: added one gated signal class — directives whose object is the
|
|
44
|
+
* agent ("I want you to X", "you need to Y"), which the quantifier net structurally could not see —
|
|
45
|
+
* requiring strong rejection valence so a first-time request stays silent. MEASURED effect on the
|
|
46
|
+
* real corpus: holdout firings 3 -> 10 (total 7 -> 19). All 94 detector unit tests stay green, so no
|
|
47
|
+
* blind-rater-certified case regressed.
|
|
48
|
+
*
|
|
49
|
+
* THE TRADE, stated honestly: single-rater inspection of the 10 holdout firings read ~7 genuine
|
|
50
|
+
* corrections and ~3 borderline false positives (directives phrased as questions). That is a ~70%
|
|
51
|
+
* point estimate — HIGHER RECALL, slightly LOWER precision point-estimate than the narrow net's
|
|
52
|
+
* 77.8%. Neither is certifiable: 10 firings still cannot bound precision above 74.1% (needs n>=29),
|
|
53
|
+
* and single-rater labels are direction-finding, not the blind 3-rater majority a floor claim needs.
|
|
54
|
+
* The structural win is the VOLUME: 10 is most of the way to the n>=29 that any future >=90%
|
|
55
|
+
* certification requires — you cannot certify a floor on n=3 at all. Deliberately NOT tuned further
|
|
56
|
+
* against the holdout firings above: tuning on the set you measure with corrupts the only unbiased
|
|
57
|
+
* measurement you have. Further tuning belongs on the tune split, with blind labelling.
|
|
58
|
+
*
|
|
59
|
+
* HAND-LABELLED FINDINGS, 2026-07-24 — N3 IS A RECALL PROBLEM, NOT A PRECISION PROBLEM.
|
|
60
|
+
*
|
|
61
|
+
* The open work item read "raise correction-detect precision 27% -> 90%". A hand-labelling pass over
|
|
62
|
+
* this pool says that framing is wrong, and it is worth writing down before anyone tunes a regex again.
|
|
63
|
+
*
|
|
64
|
+
* PRECISION, holdout firings: 2 of 3 correct. Also uncertifiable — see the certifiability block at
|
|
65
|
+
* the bottom of main(): three firings cannot bound precision above 36.8% no matter what, and >=90%
|
|
66
|
+
* needs n >= 29. No regex change moves that; only more firings do.
|
|
67
|
+
*
|
|
68
|
+
* BASE RATE, 28-row holdout sample of NON-firing candidates: after discarding harness artifacts,
|
|
69
|
+
* 13 of 20 real user turns (65%) were genuine corrections the detector did not catch.
|
|
70
|
+
*
|
|
71
|
+
* RECALL, extrapolated over 156 holdout non-firings: roughly 73 missed against 2 caught, i.e.
|
|
72
|
+
* ABOUT 3%. The detector misses ~97% of the corrections in front of it.
|
|
73
|
+
*
|
|
74
|
+
* So precision was never the binding constraint. And the two problems share ONE fix: broadening the
|
|
75
|
+
* net raises recall AND produces the firing volume that certifying precision requires. Tuning for
|
|
76
|
+
* precision on n=3 does neither, while looking like progress.
|
|
77
|
+
*
|
|
78
|
+
* CAVEAT ON THESE LABELS, stated because it bounds them: they are ONE rater's judgement (mine), not
|
|
79
|
+
* the blind 3-rater majority the earlier 77.8% figure used. Treat them as a direction-finding
|
|
80
|
+
* measurement that reframes the problem, not as a certified precision number. The certified number
|
|
81
|
+
* still requires the volume above.
|
|
82
|
+
*
|
|
83
|
+
* THE NUMBERS THIS PRODUCED ON 2026-07-23 are recorded in correction-detect.mjs's own header
|
|
84
|
+
* (search that file for "MEASURED ON THE REAL CORPUS, 2026-07-23") rather than duplicated here,
|
|
85
|
+
* since a measurement script that also claims to BE the measurement is how numbers rot out of sync
|
|
86
|
+
* with the code they describe.
|
|
87
|
+
*/
|
|
88
|
+
|
|
89
|
+
import fs from 'node:fs';
|
|
90
|
+
import os from 'node:os';
|
|
91
|
+
import path from 'node:path';
|
|
92
|
+
import crypto from 'node:crypto';
|
|
93
|
+
import readline from 'node:readline';
|
|
94
|
+
import { detectCorrection, HARNESS_TEMPLATES } from './correction-detect.mjs';
|
|
95
|
+
|
|
96
|
+
const argv = process.argv.slice(2);
|
|
97
|
+
const flag = (name, fallback = null) => {
|
|
98
|
+
const i = argv.indexOf(name);
|
|
99
|
+
return i >= 0 && argv[i + 1] ? argv[i + 1] : fallback;
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
const defaultCorpusDir = path.join(
|
|
103
|
+
os.homedir(), '.claude', 'projects',
|
|
104
|
+
process.cwd().replace(/\//g, '-'),
|
|
105
|
+
);
|
|
106
|
+
const CORPUS_DIR = flag('--corpus-dir', defaultCorpusDir);
|
|
107
|
+
const DUMP_POOL = flag('--dump-pool', null);
|
|
108
|
+
const SPLIT_FILTER = flag('--split', 'all'); // tune | holdout | all
|
|
109
|
+
|
|
110
|
+
/** Deterministic 55/45 tune/holdout split BY TRANSCRIPT FILE, fixed so re-runs are reproducible. */
|
|
111
|
+
function splitOf(fileName) {
|
|
112
|
+
const h = crypto.createHash('md5').update(fileName).digest('hex');
|
|
113
|
+
const n = parseInt(h.slice(0, 8), 16) % 100;
|
|
114
|
+
return n < 55 ? 'tune' : 'holdout';
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** Best-effort one-line description of a tool_use block, mirroring how a real hook would summarise it. */
|
|
118
|
+
function summarizeToolUse(block) {
|
|
119
|
+
const name = block.name || 'unknown';
|
|
120
|
+
const input = block.input || {};
|
|
121
|
+
let detail;
|
|
122
|
+
if (name === 'Bash') detail = input.command;
|
|
123
|
+
else if (['Edit', 'Write', 'NotebookEdit', 'Read'].includes(name)) detail = input.file_path;
|
|
124
|
+
else if (['Grep', 'Glob'].includes(name)) detail = input.pattern;
|
|
125
|
+
else if (name === 'Task') detail = input.description || input.prompt;
|
|
126
|
+
else if (name === 'TodoWrite') detail = 'todo update';
|
|
127
|
+
else detail = JSON.stringify(input);
|
|
128
|
+
return { tool: name, summary: String(detail ?? '').slice(0, 200) };
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Walk one transcript, emitting a candidate row for every genuinely-typed user turn (string
|
|
133
|
+
* `message.content`, never a tool_result array) that has SOME preceding assistant turn — a tool
|
|
134
|
+
* action if the assistant's last message used one, otherwise a truncated summary of what it said.
|
|
135
|
+
* A pure-text-then-nothing-since boundary correctly resets this to null, matching Signal 1's actual
|
|
136
|
+
* meaning: "is there something for this utterance to be responding to."
|
|
137
|
+
*/
|
|
138
|
+
async function extractFromFile(file) {
|
|
139
|
+
const rows = [];
|
|
140
|
+
let lastAssistantAction = null;
|
|
141
|
+
let turnIndex = 0;
|
|
142
|
+
const rl = readline.createInterface({ input: fs.createReadStream(file, { encoding: 'utf8' }), crlfDelay: Infinity });
|
|
143
|
+
for await (const line of rl) {
|
|
144
|
+
if (!line.trim()) continue;
|
|
145
|
+
let obj;
|
|
146
|
+
try { obj = JSON.parse(line); } catch { continue; }
|
|
147
|
+
|
|
148
|
+
if (obj.type === 'assistant' && obj.message && Array.isArray(obj.message.content)) {
|
|
149
|
+
const toolUses = obj.message.content.filter((b) => b && b.type === 'tool_use');
|
|
150
|
+
if (toolUses.length) {
|
|
151
|
+
lastAssistantAction = summarizeToolUse(toolUses[toolUses.length - 1]);
|
|
152
|
+
} else {
|
|
153
|
+
const text = obj.message.content.filter((b) => b && b.type === 'text' && b.text).map((b) => b.text).join(' ').trim();
|
|
154
|
+
lastAssistantAction = text ? { tool: null, summary: text.slice(0, 200) } : null;
|
|
155
|
+
}
|
|
156
|
+
continue;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
if (obj.type === 'user' && obj.message && typeof obj.message.content === 'string') {
|
|
160
|
+
turnIndex += 1;
|
|
161
|
+
if (lastAssistantAction && (lastAssistantAction.summary || lastAssistantAction.tool)) {
|
|
162
|
+
rows.push({
|
|
163
|
+
file: path.basename(file), turnIndex, timestamp: obj.timestamp,
|
|
164
|
+
promptText: obj.message.content, precedingAssistantAction: lastAssistantAction,
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
}
|
|
169
|
+
return rows;
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/** A deliberately LOOSE lexical net — a superset of every signal the real detector requires — used
|
|
173
|
+
* only to build a candidate pool small enough for a human to hand-label, never to decide anything. */
|
|
174
|
+
const BROAD_NET = /\b(?:always|never|no longer|no more|constantly|repeatedly|stop\b|don'?t\b|do not\b|quit\b|wrong\b|incorrect\b|instead of|rather than|isn'?t what|why (?:did|didn'?t|are|aren'?t) you|you keep|you always|i (?:told|asked) you|already (?:told|asked|said)|should have|failed to|forgot to|from now on|going forward|in the future|henceforth|next time|that'?s not|not what i (?:asked|wanted|said))\b/i;
|
|
175
|
+
|
|
176
|
+
async function main() {
|
|
177
|
+
let files;
|
|
178
|
+
try {
|
|
179
|
+
files = fs.readdirSync(CORPUS_DIR).filter((f) => f.endsWith('.jsonl')).map((f) => path.join(CORPUS_DIR, f));
|
|
180
|
+
} catch (e) {
|
|
181
|
+
console.error(`Cannot read corpus dir ${CORPUS_DIR}: ${e.message}`);
|
|
182
|
+
console.error('Pass --corpus-dir <path> to point at a real Claude Code transcript directory.');
|
|
183
|
+
process.exit(1);
|
|
184
|
+
}
|
|
185
|
+
console.error(`[measure] ${files.length} transcript file(s) in ${CORPUS_DIR}`);
|
|
186
|
+
|
|
187
|
+
const bySplit = { tune: { total: 0, hits: 0 }, holdout: { total: 0, hits: 0 } };
|
|
188
|
+
const poolRows = [];
|
|
189
|
+
let userTurns = 0;
|
|
190
|
+
|
|
191
|
+
for (const file of files) {
|
|
192
|
+
let rows;
|
|
193
|
+
try { rows = await extractFromFile(file); } catch (e) { console.error(`[measure] skip ${file}: ${e.message}`); continue; }
|
|
194
|
+
const split = splitOf(path.basename(file));
|
|
195
|
+
for (const row of rows) {
|
|
196
|
+
userTurns += 1;
|
|
197
|
+
bySplit[split].total += 1;
|
|
198
|
+
const got = detectCorrection(row.promptText, {
|
|
199
|
+
precedingAssistantAction: row.precedingAssistantAction,
|
|
200
|
+
transcriptPath: row.file, turnIndex: row.turnIndex, timestamp: row.timestamp,
|
|
201
|
+
});
|
|
202
|
+
if (got) bySplit[split].hits += 1;
|
|
203
|
+
|
|
204
|
+
// Harness artifacts are excluded from the LABELLING POOL, not just from detection. The
|
|
205
|
+
// detector already rejects them (correction-detect.mjs HARNESS_TEMPLATES), so they could never
|
|
206
|
+
// fire — but they were still written out for a human to label. MEASURED in a 28-row holdout
|
|
207
|
+
// sample: 8 of them (29%) were <local-command-caveat> blocks, i.e. a third of the labelling
|
|
208
|
+
// effort spent on rows that are not user speech and whose answer is definitionally "no".
|
|
209
|
+
// Labelled examples are the scarcest resource in this problem; spending 29% of them on
|
|
210
|
+
// harness noise is why the pool looked bigger than it usefully was.
|
|
211
|
+
const isArtifact = HARNESS_TEMPLATES.some((re) => re.test(row.promptText));
|
|
212
|
+
if (DUMP_POOL && !isArtifact && (SPLIT_FILTER === 'all' || SPLIT_FILTER === split)
|
|
213
|
+
&& row.promptText.length <= 2000 && BROAD_NET.test(row.promptText)) {
|
|
214
|
+
poolRows.push({ ...row, split, detectorResult: got });
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
const total = bySplit.tune.total + bySplit.holdout.total;
|
|
220
|
+
const hits = bySplit.tune.hits + bySplit.holdout.hits;
|
|
221
|
+
console.log(`\nadjacency candidates (Signal 1): ${total} (tune ${bySplit.tune.total} / holdout ${bySplit.holdout.total})`);
|
|
222
|
+
console.log(`detections: ${hits} (tune ${bySplit.tune.hits} / holdout ${bySplit.holdout.hits})`);
|
|
223
|
+
console.log(`rate: ${(100 * hits / total).toFixed(3)}% (tune ${(100 * bySplit.tune.hits / bySplit.tune.total).toFixed(3)}% / holdout ${(100 * bySplit.holdout.hits / bySplit.holdout.total).toFixed(3)}%)`);
|
|
224
|
+
console.log(`\nPrecision and recall require HAND-LABELLING — this script only counts firings.`);
|
|
225
|
+
console.log(`Use --dump-pool <path> to write a labellable candidate pool; the holdout half is the`);
|
|
226
|
+
console.log(`only one whose precision/recall counts as an unbiased measurement.`);
|
|
227
|
+
|
|
228
|
+
// ── CAN THE ≥90% FLOOR EVEN BE CERTIFIED FROM THIS MUCH DATA? ────────────────────────────────────
|
|
229
|
+
// ADR-033 holds lesson auto-extraction behind a ≥90% precision floor, and the open work item read
|
|
230
|
+
// "raise correction-detect precision 27% -> 90%" — which frames it as a TUNING problem. It is not,
|
|
231
|
+
// and stating the arithmetic here is what stops it being mistaken for one again.
|
|
232
|
+
//
|
|
233
|
+
// Precision is estimated from the detections, not from the candidate pool, so the holdout FIRING
|
|
234
|
+
// count is the sample size. With every single detection correct, the exact (Clopper-Pearson) 95%
|
|
235
|
+
// one-sided lower bound is p = alpha^(1/n) — closed form, checkable by hand: for n=3 that is the
|
|
236
|
+
// cube root of 0.05, 36.8%. Clearing 90% needs n >= 29 CONSECUTIVE correct detections, and any
|
|
237
|
+
// error pushes the requirement higher still.
|
|
238
|
+
//
|
|
239
|
+
// So a "77.8% at n=19" measurement cannot certify a 90% floor even in principle — at n=19 a PERFECT
|
|
240
|
+
// 19/19 bounds at only 85.4%. Tuning the regex until the point estimate crosses 0.90 on a sample
|
|
241
|
+
// this small is fitting the sample, not the property, and it would produce exactly the confident
|
|
242
|
+
// wrong number this project keeps catching elsewhere.
|
|
243
|
+
const holdoutHits = bySplit.holdout.hits;
|
|
244
|
+
const lowerBoundIfPerfect = holdoutHits > 0 ? Math.pow(0.05, 1 / holdoutHits) : 0;
|
|
245
|
+
const N_FOR_90 = 29;
|
|
246
|
+
console.log(`\ncertifiability of the ADR-033 >=90% precision floor, from THIS run:`);
|
|
247
|
+
console.log(` holdout detections (the precision sample) : ${holdoutHits}`);
|
|
248
|
+
console.log(` best possible 95% lower bound (all correct): ${(lowerBoundIfPerfect * 100).toFixed(1)}%`);
|
|
249
|
+
if (lowerBoundIfPerfect >= 0.90) {
|
|
250
|
+
console.log(` => the sample is LARGE ENOUGH to certify 90% — hand-label the holdout detections.`);
|
|
251
|
+
} else {
|
|
252
|
+
console.log(` => NOT CERTIFIABLE at any precision: ${holdoutHits} detections cannot bound above`);
|
|
253
|
+
console.log(` ${(lowerBoundIfPerfect * 100).toFixed(1)}%, and >=90% requires n >= ${N_FOR_90} consecutive correct.`);
|
|
254
|
+
console.log(` N3 is blocked on LABELLED VOLUME, not on the detector. Tuning against a sample`);
|
|
255
|
+
console.log(` this small overfits it. The unblock is more transcript corpus (or a broader net`);
|
|
256
|
+
console.log(` that fires more often), then hand-labelling — not another regex pass.`);
|
|
257
|
+
}
|
|
258
|
+
|
|
259
|
+
if (DUMP_POOL) {
|
|
260
|
+
fs.writeFileSync(DUMP_POOL, poolRows.map((r) => JSON.stringify(r)).join('\n') + (poolRows.length ? '\n' : ''));
|
|
261
|
+
console.log(`\nWrote ${poolRows.length} candidate(s) to ${DUMP_POOL} for hand-labelling.`);
|
|
262
|
+
console.log(`This file may contain real transcript text — do not commit it into the repo.`);
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const invokedDirectly = process.argv[1]
|
|
267
|
+
&& path.resolve(process.argv[1]).endsWith(`correction-detect-measure${path.extname(process.argv[1])}`);
|
|
268
|
+
if (invokedDirectly) main();
|
|
269
|
+
|
|
270
|
+
export { extractFromFile, splitOf, BROAD_NET };
|