ruvnet-brain 3.9.134-dev → 4.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +14 -0
- package/README.md +5 -5
- package/bin/install.mjs +382 -36
- package/console/CONTRACT.md +172 -0
- package/console/activity.js +753 -0
- package/console/app.js +4189 -0
- package/console/architecture.html +1221 -0
- package/console/assets/depth-1.webp +0 -0
- package/console/assets/depth-2.webp +0 -0
- package/console/assets/depth-3.webp +0 -0
- package/console/assets/harness-vs-plain.svg +259 -0
- package/console/assets/hero.webp +0 -0
- package/console/assets/memory.webp +0 -0
- package/console/assets/metaharness.svg +247 -0
- package/console/index.html +777 -0
- package/console/install-architecture.html +162 -0
- package/console/install-mockup.html +543 -0
- package/console/style.css +2144 -0
- package/console/tips.css +926 -0
- package/console/tips.html +858 -0
- package/console/tips.js +128 -0
- package/docs/RELEASE-NOTES-4.0.md +88 -0
- package/kb/model-requirements.mjs +37 -6
- package/kb/zip-extract.mjs +53 -14
- package/keys/ruvnet-brain-signing.pub.pem +3 -0
- package/package.json +14 -22
- package/plugin/.claude-plugin/marketplace.json +14 -0
- package/plugin/.claude-plugin/plugin.json +22 -0
- package/plugin/.codex-plugin/plugin.json +21 -0
- package/plugin/.mcp.json +8 -0
- package/plugin/commands/brain-console.md +16 -0
- package/plugin/commands/configure.md +33 -0
- package/plugin/commands/rvbc.md +79 -0
- package/plugin/commands/rvcb.md +16 -0
- package/plugin/commands/whats-new.md +57 -0
- package/plugin/hooks/codex-hooks.json +160 -0
- package/plugin/hooks/hook-contracts.json +77 -0
- package/plugin/hooks/hooks.json +202 -0
- package/plugin/mcp/managed-cli-interface.mjs +47 -4
- package/plugin/mcp/server.mjs +56 -6
- package/plugin/scripts/anticipate.sh +534 -0
- package/plugin/scripts/codex-hook-adapter.mjs +96 -0
- package/plugin/scripts/continuation-gate.mjs +267 -0
- package/plugin/scripts/design-wall.sh +137 -0
- package/plugin/scripts/detach.mjs +182 -0
- package/plugin/scripts/first-session-worker.mjs +38 -0
- package/plugin/scripts/gate-receipt.sh +35 -0
- package/plugin/scripts/ground-before-write.sh +199 -0
- package/plugin/scripts/ground-ruvnet.sh +517 -0
- package/plugin/scripts/grounding-stamp.sh +113 -0
- package/plugin/scripts/grounding-substance.mjs +595 -0
- package/plugin/scripts/hijack-ruvnet.sh +81 -0
- package/plugin/scripts/hook-input.mjs +558 -0
- package/plugin/scripts/hook-shim-bash.mjs +55 -0
- package/plugin/scripts/hook-shim.mjs +303 -0
- package/plugin/scripts/host-update.mjs +58 -0
- package/plugin/scripts/kling-preflight.sh +146 -0
- package/plugin/scripts/learn-capture.sh +173 -0
- package/plugin/scripts/learn-flush.mjs +155 -0
- package/plugin/scripts/lesson-hooks.sh +213 -0
- package/plugin/scripts/md-stamp.mjs +219 -0
- package/plugin/scripts/protect-brain-state.sh +84 -0
- package/plugin/scripts/route-dispatch.sh +147 -0
- package/plugin/scripts/routing-outcome-capture.mjs +89 -0
- package/plugin/scripts/runtime-preferences.mjs +269 -0
- package/plugin/scripts/session-start-core.mjs +477 -0
- package/plugin/scripts/session-start.sh +13 -0
- package/plugin/scripts/signal-watch.mjs +193 -0
- package/plugin/scripts/unprompted-runtime.mjs +377 -0
- package/plugin/scripts/update-apply.mjs +419 -0
- package/plugin/scripts/verify-interface.sh +53 -0
- package/plugin/scripts/version-bump-gate.sh +112 -0
- package/plugin/skills/brain-build/SKILL.md +123 -0
- package/plugin/skills/brain-console/SKILL.md +22 -0
- package/plugin/skills/brain-prompt/SKILL.md +83 -0
- package/plugin/skills/brain-score/SKILL.md +101 -0
- package/plugin/skills/release-proof/SKILL.md +81 -0
- package/plugin/skills/release-proof/agents/openai.yaml +4 -0
- package/plugin/skills/release-proof/references/receipt-contract.md +38 -0
- package/plugin/skills/release-proof/scripts/release-proof.mjs +210 -0
- package/plugin/skills/ruvnet-brain/PLAYBOOK.md +121 -0
- package/plugin/skills/ruvnet-brain/SKILL.md +234 -0
- package/plugin/skills/rvbc/SKILL.md +23 -0
- package/plugin/skills/savings/SKILL.md +46 -0
- package/plugin/skills/whats-new/SKILL.md +22 -0
- package/scripts/adr-backfill.mjs +107 -0
- package/scripts/advocacy-outcomes.mjs +808 -0
- package/scripts/agentdb-context.mjs +216 -0
- package/scripts/agentdb-fleet-doctor.mjs +101 -0
- package/scripts/ascii-drift.mjs +236 -0
- package/scripts/behavioral-l1-l4.mjs +210 -0
- package/scripts/brain-capability-check.mjs +72 -0
- package/scripts/brain-grade-groundtruth.mjs +100 -0
- package/scripts/brain-latency-50.mjs +227 -0
- package/scripts/brain-novice-50.mjs +189 -0
- package/scripts/brain-stamp.mjs +94 -0
- package/scripts/brain-state.mjs +212 -0
- package/scripts/build-bundle.mjs +522 -0
- package/scripts/build-concepts.mjs +132 -0
- package/scripts/build-l2.mjs +71 -0
- package/scripts/build-primer.mjs +73 -0
- package/scripts/build-symbols.mjs +68 -0
- package/scripts/calibrate-router.mjs +97 -0
- package/scripts/capability-audit.mjs +321 -0
- package/scripts/capability-registry.mjs +876 -0
- package/scripts/check-indexation.mjs +108 -0
- package/scripts/check-legibility.mjs +189 -0
- package/scripts/ci/build-fixture-kb.mjs +67 -0
- package/scripts/ci/learning-replay-codex-adapter.mjs +62 -0
- package/scripts/ci/learning-replay-recorder.mjs +59 -0
- package/scripts/ci/mutate-hook-timeout.mjs +70 -0
- package/scripts/ci/stranger-fixture-stage.mjs +17 -0
- package/scripts/ci/stranger-scenario.mjs +228 -0
- package/scripts/ci/stranger-timeout.mjs +25 -0
- package/scripts/ci-verdict.mjs +29 -0
- package/scripts/claims-verify.mjs +710 -0
- package/scripts/clear-claude-tmp.sh +31 -0
- package/scripts/console-engine.mjs +434 -0
- package/scripts/console-engine.test.mjs +125 -0
- package/scripts/corpus-qa.mjs +250 -0
- package/scripts/correction-detect-embed.mjs +346 -0
- package/scripts/correction-detect-measure.mjs +270 -0
- package/scripts/correction-detect.mjs +686 -0
- package/scripts/count-chunks.mjs +54 -0
- package/scripts/described-questions.json +30 -0
- package/scripts/design-grade.mjs +58 -0
- package/scripts/dev-plugin-link.sh +105 -0
- package/scripts/distill-project.mjs +200 -0
- package/scripts/doc-currency.mjs +801 -0
- package/scripts/eval-brain.mjs +244 -0
- package/scripts/fix-metaharness-memretrieve.mjs +121 -0
- package/scripts/full-hints.mjs +87 -0
- package/scripts/gate.sh +39 -0
- package/scripts/gates.mjs +146 -0
- package/scripts/gen-console-images.mjs +54 -0
- package/scripts/gen-images.mjs +47 -0
- package/scripts/git-clone-refresh.mjs +52 -0
- package/scripts/git-hooks/pre-push +126 -0
- package/scripts/goal-match.mjs +398 -0
- package/scripts/goldie-research.mjs +223 -0
- package/scripts/goldie-weekly.sh +67 -0
- package/scripts/health-repair.mjs +250 -0
- package/scripts/helix-scenario-questions.json +10 -0
- package/scripts/ingest-gists.mjs +230 -0
- package/scripts/ingest-meeting.mjs +115 -0
- package/scripts/ingest-repo.mjs +79 -0
- package/scripts/install-npx-witness.sh +49 -0
- package/scripts/issue-fix.mjs +639 -0
- package/scripts/issue-watch.mjs +276 -0
- package/scripts/issue4-close-note.md +31 -0
- package/scripts/key-canary.mjs +91 -0
- package/scripts/latency-to-surface.mjs +233 -0
- package/scripts/learning-enable.mjs +380 -0
- package/scripts/learning-replay.mjs +1570 -0
- package/scripts/learnings.mjs +62 -0
- package/scripts/lesson-gate.mjs +680 -0
- package/scripts/lesson-lifecycle.mjs +449 -0
- package/scripts/lesson-promote.mjs +262 -0
- package/scripts/lesson-ratify.mjs +98 -0
- package/scripts/lesson-seed.mjs +252 -0
- package/scripts/lesson-store.mjs +447 -0
- package/scripts/loop-checkpoint.mjs +86 -0
- package/scripts/memdb-health.sh +14 -0
- package/scripts/memory-doctor.mjs +271 -0
- package/scripts/model-catalog.mjs +79 -0
- package/scripts/nightly-controller.mjs +66 -0
- package/scripts/nightly-gists.sh +72 -0
- package/scripts/nightly-wrapper.sh +180 -0
- package/scripts/notify.sh +12 -0
- package/scripts/npx-witness.sh +56 -0
- package/scripts/onboarding-console.mjs +2749 -0
- package/scripts/private-fence.mjs +69 -0
- package/scripts/proactivity-metrics.mjs +118 -0
- package/scripts/proof-questions.json +56 -0
- package/scripts/prove.mjs +95 -0
- package/scripts/proxy/claude-proxied.sh +57 -0
- package/scripts/proxy/proxy-revert.sh +59 -0
- package/scripts/proxy/proxy-up.sh +60 -0
- package/scripts/proxy/proxy-verify.mjs +142 -0
- package/scripts/published-surface-probe.mjs +241 -0
- package/scripts/qe/card-lane-gate.mjs +162 -0
- package/scripts/qe/session-start-gate.mjs +229 -0
- package/scripts/qe/ux-suite.mjs +323 -0
- package/scripts/reconcile-project.mjs +0 -0
- package/scripts/record-lesson.mjs +113 -0
- package/scripts/refresh-model-catalog.mjs +99 -0
- package/scripts/release-proof.mjs +9 -0
- package/scripts/release-vector.mjs +281 -0
- package/scripts/release.mjs +395 -0
- package/scripts/remedy-registry.mjs +247 -0
- package/scripts/rerank-cap-eval.mjs +265 -0
- package/scripts/rerank-cap-warm-ab.mjs +129 -0
- package/scripts/route-cheap.mjs +20 -15
- package/scripts/router-utilization.mjs +182 -0
- package/scripts/routing-flywheel.mjs +596 -0
- package/scripts/rvf-generation.mjs +104 -0
- package/scripts/rvf-index-audit.mjs +138 -0
- package/scripts/self-update.mjs +508 -0
- package/scripts/selfcheck.mjs +7 -1
- package/scripts/sign-bundle.mjs +69 -0
- package/scripts/signal-watch.mjs +171 -0
- package/scripts/stack-sync.mjs +469 -0
- package/scripts/stamp-existing-rvf-generations.mjs +53 -0
- package/scripts/stamp-sweep.mjs +144 -0
- package/scripts/status-honesty.mjs +102 -0
- package/scripts/sync-version.mjs +217 -0
- package/scripts/token-report.mjs +102 -0
- package/scripts/top100-benchmark.mjs +479 -0
- package/scripts/top100-corpus.mjs +112 -0
- package/scripts/top100-semantic-assertions.mjs +449 -0
- package/scripts/update-apply.mjs +9 -0
- package/scripts/upgrade-notice.mjs +14 -0
- package/scripts/verify-bundle.mjs +51 -0
- package/scripts/verify-channels.mjs +184 -0
- package/scripts/verify-model-catalog.mjs +104 -0
- package/scripts/verify-nightly-close-issue4.sh +31 -0
- package/scripts/version.mjs +40 -0
- package/scripts/wired-check.mjs +864 -0
|
@@ -0,0 +1,686 @@
|
|
|
1
|
+
// correction-detect.mjs — decide whether a user utterance is a BEHAVIOURAL CORRECTION.
|
|
2
|
+
//
|
|
3
|
+
// This is the missing beginning of the learning pipeline. ADR-033 measured the gap: 14 lessons in
|
|
4
|
+
// the store, 14 of them transcribed by hand, 0 captured from a live correction. The middle and the
|
|
5
|
+
// end were built (store, trust boundary, gate); nothing ever answered "where does a lesson come
|
|
6
|
+
// from?". This module answers exactly that question and nothing else — it is pure, does no I/O,
|
|
7
|
+
// reads no transcript, writes no store. It is given one utterance and one piece of context, and it
|
|
8
|
+
// returns a candidate or, far more often, null.
|
|
9
|
+
//
|
|
10
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
11
|
+
// WHY THIS IS NOT THE DETECTOR ADR-033 FIRST DESCRIBED
|
|
12
|
+
//
|
|
13
|
+
// ADR-033 §1 specifies a conjunction of four signals. Two independent adversarial reviews took that
|
|
14
|
+
// specification apart before a line was written, and both were right. The corrections they forced
|
|
15
|
+
// are the whole substance of this file:
|
|
16
|
+
//
|
|
17
|
+
// • FABLE'S F1 — the load-bearing finding. Signals 2 ("directed at behaviour") and 4 ("negative
|
|
18
|
+
// valence") were specified as semantic-role and sentiment problems with NO lexical realization
|
|
19
|
+
// given, while LLM detection was simultaneously refused — a spec for a deadlock. Worse, a
|
|
20
|
+
// 15-utterance hand-walk showed the conjunction as written FIRES on the two classes that
|
|
21
|
+
// dominate a coding transcript:
|
|
22
|
+
//
|
|
23
|
+
// "It always crashes when I pass null — fix it." ← a bug report
|
|
24
|
+
// "Make sure the parser never accepts unquoted keys." ← spec-speak
|
|
25
|
+
//
|
|
26
|
+
// Signal 3 was quantifying over PROGRAM EXECUTIONS and over REQUIREMENTS, not over occasions of
|
|
27
|
+
// agent behaviour, and no stated rule could tell the domains apart. With a 1.2% base rate, the
|
|
28
|
+
// allowed false-positive rate is ~0.07% per turn; requirements dialect natively uses always/never,
|
|
29
|
+
// so that FP class alone buries the detector.
|
|
30
|
+
//
|
|
31
|
+
// THE FIX, which is Fable's own (a) and the single most important idea here: Signal 3 does not
|
|
32
|
+
// look for a quantifier. It looks for a quantifier SYNTACTICALLY BOUND TO THE SECOND PERSON —
|
|
33
|
+
// `you always`, `every time you`, `you keep`, `stop <gerund>`, or a quantifier in CLAUSE-INITIAL
|
|
34
|
+
// IMPERATIVE position with nothing but discourse fillers in front of it. "the parser never" has a
|
|
35
|
+
// third-person nominal subject and is refused; "Never just link to the HTML page" has an empty
|
|
36
|
+
// clause prefix and is kept. That one syntactic move is what separates a rule about the agent
|
|
37
|
+
// from a rule about a program, and it is implementable in exactly the lexical terms ADR-033
|
|
38
|
+
// demands.
|
|
39
|
+
//
|
|
40
|
+
// Because that same test IS the honest realization of Signal 2 (is the complaint about something
|
|
41
|
+
// the agent did?), signals 2 and 3 are implemented as ONE predicate and said so out loud. Two
|
|
42
|
+
// names for one test would be a claim of independent evidence we do not have — the precise
|
|
43
|
+
// dishonesty lesson-store.mjs was built to refuse elsewhere.
|
|
44
|
+
//
|
|
45
|
+
// • SOL'S C5 — the quote must come from the authenticated user utterance, not a reconstruction.
|
|
46
|
+
// `plugin/scripts/ground-ruvnet.sh:19` already reads `.prompt` off the UserPromptSubmit payload;
|
|
47
|
+
// that IS the user's words, structurally labelled as theirs by the harness. Hence the signature:
|
|
48
|
+
// `promptText` is the payload, and `context` carries only what the payload cannot know (what the
|
|
49
|
+
// assistant did immediately before). A transcript is never the source of the words.
|
|
50
|
+
//
|
|
51
|
+
// • SOL'S C2 — `makeLesson()` destructures a fixed key list and silently DROPS unknown keys, so
|
|
52
|
+
// every rich evidence field ADR-033 §3 specifies (quote, respondingTo, source, signals) would
|
|
53
|
+
// vanish on the way into the store, leaving the human ratifier with no sentence to read. The one
|
|
54
|
+
// array that passes through whole is `evidence[]`. So everything rides INSIDE evidence[0], and
|
|
55
|
+
// `observed` is populated because `lesson-gate.mjs:75` is what actually prints it.
|
|
56
|
+
//
|
|
57
|
+
// • SOL'S C7 + the meta-risk — hypotheticals, delegated instructions ("tell the subagent to
|
|
58
|
+
// always…"), quoted policy, and this repository's OWN design documents (which are wall-to-wall
|
|
59
|
+
// quantified second-person rules) are all live false-positive classes. They are excluded
|
|
60
|
+
// structurally, before any signal is evaluated.
|
|
61
|
+
//
|
|
62
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
63
|
+
// THE OPERATING POINT, and why silence is the default
|
|
64
|
+
//
|
|
65
|
+
// ADR-033 §2 settles the asymmetry with a structural argument, not a preference: a miss costs one
|
|
66
|
+
// repetition and is SELF-HEALING (the repeat is itself ADR-030's escalation signal). A false
|
|
67
|
+
// positive becomes a candidate, then a ratification prompt, then noise the user learns to skip — and
|
|
68
|
+
// at the end of that road it reaches ADR-031 §4's objective function, where a search pursues it
|
|
69
|
+
// faithfully and at scale. Recall is a convenience. Precision is a safety property.
|
|
70
|
+
//
|
|
71
|
+
// So this module returns null on every doubt, and several genuine corrections are knowingly let go.
|
|
72
|
+
// The accepted misses are enumerated in ACCEPTED_MISSES below rather than left to be rediscovered —
|
|
73
|
+
// under-enumeration being its own recorded failure here (L10).
|
|
74
|
+
//
|
|
75
|
+
// ─────────────────────────────────────────────────────────────────────────────────────────────────
|
|
76
|
+
// MEASURED ON THE REAL CORPUS, 2026-07-23 — a held-out re-measurement, and what it still does NOT prove
|
|
77
|
+
//
|
|
78
|
+
// The 2026-07-22 measurement below (n=4, kept for provenance) was two orders of magnitude short of
|
|
79
|
+
// ADR-033 §2's ≥100-detection floor, so this round built the harness the floor requires: a candidate
|
|
80
|
+
// pool pulled from this project's live transcript corpus (`scripts/correction-detect-measure.mjs`,
|
|
81
|
+
// same corpus ADR-033 measured, 1,328 files at the time of this snapshot and still growing — this is
|
|
82
|
+
// an active project, not a frozen fixture), SPLIT BY TRANSCRIPT FILE into a 55/45 tune/holdout
|
|
83
|
+
// partition BEFORE any hand-labelling — so heuristics were only ever adjusted against the tune half,
|
|
84
|
+
// and the numbers below are the detector's FIRST look at the holdout half. Reproduce with
|
|
85
|
+
// `node scripts/correction-detect-measure.mjs --dump-pool <path> --split holdout` (expect small drift
|
|
86
|
+
// run to run: the corpus is live).
|
|
87
|
+
//
|
|
88
|
+
// adjacency-satisfying candidates (signal 1) 1,338 (554 tune / 784 holdout)
|
|
89
|
+
// hand-labelled via a loose superset lexical net 271 (112 tune / 159 holdout)
|
|
90
|
+
//
|
|
91
|
+
// Five real, load-bearing bugs surfaced by mining the TUNE half (never the holdout — the fifth was
|
|
92
|
+
// self-inflicted, caught by the new tests before shipping, not by the holdout), each fixed and
|
|
93
|
+
// commented at its site:
|
|
94
|
+
// 1. THIRD_PERSON_QUANT was case-sensitive, so its `[A-Z][\w.'-]*` proper-noun catch-all also
|
|
95
|
+
// matched sentence-initial "You" and "I" — silently rejecting this file's OWN canonical example
|
|
96
|
+
// ("You never bump the version.") as third-person. Every shipped positive-table case with that
|
|
97
|
+
// shape only passed because a second sentence happened to mask it.
|
|
98
|
+
// 2. BOUND_SECOND_PERSON required "you" to DIRECTLY precede the quantifier, missing the extremely
|
|
99
|
+
// common copula form "you are/were never…", "you're always…".
|
|
100
|
+
// 3. No pattern existed for "I never/always want/expect/need you to <verb>" — a quantifier bound to
|
|
101
|
+
// the agent's occasions, just phrased as the speaker's expectation rather than second person.
|
|
102
|
+
// 4. `write-code`'s trigger vocabulary had "wrote" but not bare "write/writing/written".
|
|
103
|
+
// 5. The FIRST version of fix #2 required whitespace between "you" and the auxiliary — `you\s+(?:
|
|
104
|
+
// are|'re|…)` — which matches "you are never" but not "you're never" (no space before a
|
|
105
|
+
// contraction's apostrophe). Masked the same way as bug #1: an isolated "You're never…" sentence
|
|
106
|
+
// with no other qualifying sentence in the same utterance is what a new regression test caught,
|
|
107
|
+
// before this ever reached measurement.
|
|
108
|
+
// Two harness-artifact tags seen live in the corpus (`<local-command-caveat>`, `<task-notification>`)
|
|
109
|
+
// were added to HARNESS_TEMPLATES on the same hygiene principle as the existing entries, though
|
|
110
|
+
// neither was independently responsible for a false positive — signal 2+3 already killed them.
|
|
111
|
+
// One change was TRIED AND REJECTED: raising MAX_UTTERANCE_CHARS (to admit longer real corrections
|
|
112
|
+
// that were being length-gated) was tested against the full tune pool at 2000 chars and produced
|
|
113
|
+
// exactly one new detection — a false positive (a one-off "get this working perfectly" demand) — and
|
|
114
|
+
// not one of the four length-gated true positives it was meant to rescue, because each of those was
|
|
115
|
+
// independently blocked by a different signal anyway. Reverted; the 800-char bound stands.
|
|
116
|
+
//
|
|
117
|
+
// RESULT, holdout half only (the number that counts — nothing above was tuned against it):
|
|
118
|
+
//
|
|
119
|
+
// holdout candidates 784
|
|
120
|
+
// DETECTIONS 4 0.510% (was 2 pre-fix, same holdout)
|
|
121
|
+
// hand-labelled TRUE (unambiguous) 2 clickable-link's sibling (scores-out-of-100,
|
|
122
|
+
// already known) + "partial solutions" (ship)
|
|
123
|
+
// hand-labelled BORDERLINE 2 "get the operating guide to the point you'd
|
|
124
|
+
// never repeat this mistake" (finish), and "you're
|
|
125
|
+
// still writing code that fakes results — that's
|
|
126
|
+
// toxic" (write-code) — both defensible, genuinely
|
|
127
|
+
// arguable calls, counted as false positives below
|
|
128
|
+
//
|
|
129
|
+
// PRECISION, holdout, strict 2/4 = 50.0% (only the two unambiguous ones count)
|
|
130
|
+
// PRECISION, holdout, lenient 4/4 = 100% (if both borderline cases are ratified as real)
|
|
131
|
+
// PRECISION, tune (4 detections, all 4 unambiguous — but this is the set the fixes were derived
|
|
132
|
+
// against, so it is not independent evidence; reported for completeness only)
|
|
133
|
+
// 4/4 = 100%
|
|
134
|
+
// PRECISION, combined (tune+holdout, all 8 detections) 6/8 = 75.0% strict, 8/8 = 100% lenient
|
|
135
|
+
//
|
|
136
|
+
// RECALL is the harder number and the one most worth being honest about. Against the 2 unambiguous
|
|
137
|
+
// genuine corrections found by hand-labelling the 159-item holdout pool, recall is 2/2 — but that
|
|
138
|
+
// denominator is too small to mean anything on its own (n=2). Widening the ground truth to every
|
|
139
|
+
// utterance in that same 159 that a human WOULD plausibly ratify as a real standing order — most
|
|
140
|
+
// phrased as an impersonal "it must never / it should always" system-property claim (structurally
|
|
141
|
+
// identical to the bug-report false-positive class Fable's review killed, e.g. "npx/npm/GitHub
|
|
142
|
+
// getting out of sync should never happen" — verified this exact shape also appears as a genuine
|
|
143
|
+
// BUG REPORT elsewhere in the same corpus), or carried only by repeated reproach with no explicit
|
|
144
|
+
// always/never lexeme, or with its trigger vocabulary sitting in a sentence adjacent to — but not
|
|
145
|
+
// inside — the one that actually carries the quantifier (the SAME shape that sank a "always push
|
|
146
|
+
// through them… close the issue and comment" detection in the tune half; deliberately not widened,
|
|
147
|
+
// since re-including neighbour sentences reopens the exact cross-turn vocabulary bug fixed by
|
|
148
|
+
// narrowing to bearing sentences) — puts the denominator closer to 15, of which 2-4 are caught:
|
|
149
|
+
// roughly 15-25%. Ranges are reported because the true denominator is a judgment call, not because
|
|
150
|
+
// any single number flatters the result.
|
|
151
|
+
//
|
|
152
|
+
// WHAT THIS DOES NOT ESTABLISH, stated plainly because the gate is a number and this is not it:
|
|
153
|
+
// • ADR-033 §2 requires ≥90% precision on ≥100 detections. This round's holdout sample is n=4
|
|
154
|
+
// detections (159 hand-labelled candidates, not 100 detections) — still far short of the floor's
|
|
155
|
+
// actual denominator. Precision measured on so few firings swings by a whole detection: the
|
|
156
|
+
// difference between 50% and 100% here is TWO borderline judgment calls out of four total firings.
|
|
157
|
+
// • The classification is the detector author's own — this has NOT been independently graded, same
|
|
158
|
+
// caveat as 2026-07-22.
|
|
159
|
+
// • The residual misses are not random noise; they cluster in named, understood shapes (impersonal
|
|
160
|
+
// system-property phrasing, adjacent-sentence trigger vocabulary, no-lexical-quantifier reproach
|
|
161
|
+
// chains) that were deliberately left unaddressed because closing them lexically reopens the
|
|
162
|
+
// bug-report and spec-language false-positive classes the original adversarial review killed.
|
|
163
|
+
// The honest status: precision on the specific bugs fixed is high (all clean synthetic regression
|
|
164
|
+
// cases), two real latent bugs were found and fixed, but the live-corpus holdout sample is both too
|
|
165
|
+
// small (n=4) and too ambiguous (half its firings are defensible-but-arguable) to claim it clears
|
|
166
|
+
// ADR-033's floor, and the residual recall gap looks structural to a pure-lexical approach on THIS
|
|
167
|
+
// corpus, not a tuning oversight. See `scripts/correction-detect-measure.mjs`'s own header for the
|
|
168
|
+
// full methodology and how to reproduce or extend this measurement.
|
|
169
|
+
//
|
|
170
|
+
// ── 2026-07-22 baseline, kept for provenance ───────────────────────────────────────────────────────
|
|
171
|
+
// Run over this project's 1,299 transcripts, before any of the fixes above:
|
|
172
|
+
// user-role turns 2,768
|
|
173
|
+
// with a preceding agent action 1,451
|
|
174
|
+
// DETECTIONS 4 0.276% of considered turns
|
|
175
|
+
// All four hand-classified as genuine durable behavioural corrections. THREE independently
|
|
176
|
+
// rediscovered standing orders a human had already transcribed by hand into the project memory index
|
|
177
|
+
// — the clickable-link rule, the scores-out-of-100 rule, and "never show me a page you haven't gone
|
|
178
|
+
// through and checked visually" (ADR-033 §5's own worked example). Recall then: ~9% against 45+ known
|
|
179
|
+
// standing orders. That measurement's own three earlier revisions (temporal-scope double-count, the
|
|
180
|
+
// contentless `stop doing that`, and whole-turn trigger inference) remain fixed and commented at their
|
|
181
|
+
// sites; nothing about them changed in this round.
|
|
182
|
+
//
|
|
183
|
+
// WHAT THIS CAN NEVER DO. Every returned candidate is `origin: model-inferred`, `status: candidate`,
|
|
184
|
+
// unconditionally — INCLUDING when the user's words are quoted verbatim, and including when the
|
|
185
|
+
// utterance is a flawless imitation of a correction. ADR-033 §4: `user-stated` means a human
|
|
186
|
+
// asserted this rule, not that a string was found which looks like one. An automatic extractor that
|
|
187
|
+
// could mint `user-stated` would be the injection path of the original adversarial review,
|
|
188
|
+
// industrialised. `confidence` orders the ratification queue and does nothing else — a confidence
|
|
189
|
+
// threshold is just ratification with the human removed and the word "confidence" in front of it.
|
|
190
|
+
|
|
191
|
+
/** Corrections are short. The measured tightened detector used this bound; specs and briefs exceed it. */
|
|
192
|
+
export const MAX_UTTERANCE_CHARS = 800;
|
|
193
|
+
|
|
194
|
+
/** `makeLesson()` refuses a statement under 15 chars ("must say what to DO, specifically"). */
|
|
195
|
+
const MIN_STATEMENT_CHARS = 15;
|
|
196
|
+
|
|
197
|
+
/** Verbatim is the point (ADR-033 §3), but a ratification card must fit on one screen. */
|
|
198
|
+
const MAX_STATEMENT_CHARS = 300;
|
|
199
|
+
const MAX_QUOTE_CHARS = 600;
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Genuine corrections this detector KNOWINGLY drops. Written down because "a satisfyingly round
|
|
203
|
+
* number is evidence of rounding" (L10), and because the next person to widen a rule should have to
|
|
204
|
+
* argue against a named cost rather than discover it.
|
|
205
|
+
*/
|
|
206
|
+
export const ACCEPTED_MISSES = Object.freeze([
|
|
207
|
+
'Recurrence carried only by "again" with no second-person binding — "I asked for a table. This is prose again."',
|
|
208
|
+
'Bare prohibitions with no scope over occasions — "Don\'t do that." (also below the 15-char statement floor)',
|
|
209
|
+
'Purely positive standing orders with no rejection anywhere — "Always give me a table." is indistinguishable from an ordinary instruction or a project convention.',
|
|
210
|
+
'Corrections whose subject matter maps to no trigger in the closed TRIGGERS enum — a lesson that cannot name when it fires is prose, and the store refuses it anyway.',
|
|
211
|
+
'Corrections phrased as a request to write a rule elsewhere — "add to CLAUDE.md that…" — which are authoring tasks, not corrections of the turn.',
|
|
212
|
+
'Corrections that tie across two or more triggers. They fire at more than one moment, and picking one by list order would interrupt at the wrong one.',
|
|
213
|
+
]);
|
|
214
|
+
|
|
215
|
+
// ── HARD EXCLUSIONS ──────────────────────────────────────────────────────────────────────────────
|
|
216
|
+
// Applied before any signal. Each entry killed a real false-positive class, named in the comment.
|
|
217
|
+
|
|
218
|
+
/**
|
|
219
|
+
* Not an utterance at all. The single highest-scoring hit in ADR-033's measurement was one of these.
|
|
220
|
+
*
|
|
221
|
+
* EXPORTED as of 2026-07-24 so the MEASUREMENT harness can apply the same filter when it writes the
|
|
222
|
+
* hand-labelling pool. It could not before, and the consequence was measured rather than guessed: in
|
|
223
|
+
* a 28-row sample of the holdout pool, EIGHT (29%) were `<local-command-caveat>` blocks — not user
|
|
224
|
+
* speech at all. The detector was right to ignore them; the pool handed them to a human to label
|
|
225
|
+
* anyway, burning ~29% of the scarcest resource in this whole problem (labelled examples) on rows
|
|
226
|
+
* whose answer is definitionally "no", and diluting the base rate with them.
|
|
227
|
+
*/
|
|
228
|
+
export const HARNESS_TEMPLATES = [
|
|
229
|
+
/\[Your previous response/i,
|
|
230
|
+
/\[Request interrupted/i,
|
|
231
|
+
/<\/?system-reminder>/i,
|
|
232
|
+
/<\/?(?:command-name|command-message|command-args|local-command-stdout|local-command-stderr|local-command-caveat|task-notification|function_results|function_calls|budget)\b/i,
|
|
233
|
+
/^\s*Caveat:/i,
|
|
234
|
+
/Base directory for this skill:/i,
|
|
235
|
+
/This session is being continued from a previous conversation/i,
|
|
236
|
+
/^\s*#\s*claudeMd\b/im,
|
|
237
|
+
/\[INTELLIGENCE\]/i,
|
|
238
|
+
];
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Pasted content, not speech. Includes markdown structure — this repository's own ADRs and DDDs are
|
|
242
|
+
* dense with quantified second-person rules, so its design documents are a minefield for its own
|
|
243
|
+
* detector (Sol C7, meta-risk). A document is never a correction.
|
|
244
|
+
*/
|
|
245
|
+
const DOCUMENT_MARKERS = [
|
|
246
|
+
/```/, // code fence
|
|
247
|
+
/^\s*(?:\+\+\+|---\s|@@ )/m, // diff
|
|
248
|
+
/^\s*#{1,6}\s+\S/m, // markdown heading
|
|
249
|
+
/^\s*\|.*\|/m, // markdown table
|
|
250
|
+
/^\s*[-*•]\s+\S/m, // bullet list
|
|
251
|
+
/^\s*\d+\.\s+\S/m, // numbered list
|
|
252
|
+
/^\s*\{\s*"/m, // JSON blob
|
|
253
|
+
/^\s+at\s+\S+\s*\(/m, // stack frame
|
|
254
|
+
];
|
|
255
|
+
|
|
256
|
+
/**
|
|
257
|
+
* THE INJECTION KILLER. A real user does not refer to themselves in the third person. Every planted
|
|
258
|
+
* "the user told me to always…" — the exact sentence the original adversarial review used to
|
|
259
|
+
* demonstrate the attack — carries this tell, because it is written ABOUT a user by something that
|
|
260
|
+
* is not one.
|
|
261
|
+
*/
|
|
262
|
+
const USER_THIRD_PERSON = [
|
|
263
|
+
/\bthe user\b/i,
|
|
264
|
+
/\bthe owner\s+(?:said|told|wants|corrected|asked)/i,
|
|
265
|
+
/\buser\s+(?:told|said|corrected|instructed)\s+(?:me|you|us|the model|the assistant)\b/i,
|
|
266
|
+
/\bper the user\b/i,
|
|
267
|
+
/\bas instructed by\b/i,
|
|
268
|
+
];
|
|
269
|
+
|
|
270
|
+
/** Reported policy. Quoting a rule is not issuing one — "CLAUDE.md already says never pin versions". */
|
|
271
|
+
const ATTRIBUTION = [
|
|
272
|
+
/\b(?:says?|said|states?|reads?|specifies|requires|mandates)\b[^.!?]{0,40}?\b(?:always|never)\b/i,
|
|
273
|
+
/\baccording to\b/i,
|
|
274
|
+
/\b(?:CLAUDE\.md|the\s+(?:docs?|readme|adr|rules?|spec|guide|standing order|policy|instructions))\b[^.!?]{0,30}?\b(?:says?|said|states?|tells?)\b/i,
|
|
275
|
+
/\brule\s*\d+\b/i,
|
|
276
|
+
];
|
|
277
|
+
|
|
278
|
+
/** Delegated instruction: the quantified action's subject is a third party, not this agent. */
|
|
279
|
+
const DELEGATION = [
|
|
280
|
+
/\b(?:tell|ask|have|make|instruct|remind|get)\s+(?:the\s+|a\s+|an\s+|my\s+|your\s+)?[\w-]+\s+to\s+(?:always|never|not\b)/i,
|
|
281
|
+
/\b(?:add|write|put|append|record|save)\b[^.!?]{0,50}?\b(?:to|in|into)\b[^.!?]{0,25}?(?:CLAUDE\.md|memory|the rules?|a rule|the store|the lessons?)\b/i,
|
|
282
|
+
];
|
|
283
|
+
|
|
284
|
+
/** Spec dialect. "make sure the X never…" is a requirement about an artifact, not about the agent. */
|
|
285
|
+
const SPEC_FRAME = /\b(?:make sure|ensure|guarantee)\s+(?:that\s+)?(?:the|a|an|it|this|these|those|my|our|your)\b/i;
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Third-person subject immediately governing a quantifier. Belt-and-braces behind the clause test:
|
|
289
|
+
* it closes the conjunct leak where "Make sure the parser never accepts unquoted keys AND always
|
|
290
|
+
* preserves order" splits on `and` and hands the second conjunct a clause with an empty prefix.
|
|
291
|
+
* `you` is excluded from the subject set by lookahead — "You never bump the version" must survive.
|
|
292
|
+
*/
|
|
293
|
+
// FIX (found on the real corpus, 2026-07-23): this regex is deliberately case-SENSITIVE so `[A-Z]
|
|
294
|
+
// [\w.'-]*` only catches genuine capitalized proper nouns ("Vercel always…", "npm never…") and not
|
|
295
|
+
// every capitalized common word — but that same catch-all also swallows "You" and "I" whenever they
|
|
296
|
+
// start a sentence, since a capital letter is a capital letter regardless of which pronoun it opens.
|
|
297
|
+
// The old `(?!(?:you)\b)` guard only excluded LOWERCASE "you", so it did nothing for sentence-initial
|
|
298
|
+
// "You" — meaning "You never bump the version." (this file's own canonical example, cited below as
|
|
299
|
+
// text that "must survive") was silently swallowed as a third-person subject and REJECTED. Confirmed
|
|
300
|
+
// live against the shipped detector before this fix: a single-sentence "You always hand-roll instead
|
|
301
|
+
// of searching for the tool." — which should fire — returned null, and every positive-table case with
|
|
302
|
+
// this shape only passed because a second sentence happened to carry an independent clause-initial
|
|
303
|
+
// quantifier that masked the defect. Excluding "You" and "I" explicitly (both cases, since the match
|
|
304
|
+
// is case-sensitive) closes it without touching `[A-Z]`'s actual job of catching real proper nouns.
|
|
305
|
+
const THIRD_PERSON_QUANT =
|
|
306
|
+
/\b(?!(?:you|You|I)\b)(?:it|they|he|she|we|this|that|these|those|there|[A-Z][\w.'-]*|(?:the|a|an|my|our|its|their|his|her)\s+[\w.'-]+)\s+(?:(?:should|must|shall|will|would|can|could|may|might|does|do|did|is|are|was|were|has|have|had|keeps?|seems?|tends? to)\s+)*(?:always|never)\b/;
|
|
307
|
+
|
|
308
|
+
/** Hedges and hypotheticals — thinking aloud is not instructing (Sol C7). Applied per sentence. */
|
|
309
|
+
const HEDGE =
|
|
310
|
+
/\b(?:maybe|perhaps|possibly|might want|what if|suppose|hypothetically|for example|for instance|e\.?g\.?|imagine|let'?s say|in theory|i wonder|not sure if|thinking out loud|just brainstorming)\b/i;
|
|
311
|
+
|
|
312
|
+
// ── SIGNAL 3 (+2): A QUANTIFIER BOUND TO THE AGENT ───────────────────────────────────────────────
|
|
313
|
+
|
|
314
|
+
/** Explicit second-person subject. The quantified occasions are unambiguously the agent's. */
|
|
315
|
+
const BOUND_SECOND_PERSON = [
|
|
316
|
+
/\byou\s+(?:always|never|constantly|repeatedly|keep|keep on)\b/i,
|
|
317
|
+
/\byou'?(?:ve|\s+have)\s+(?:always|never|repeatedly|constantly)\b/i,
|
|
318
|
+
/\b(?:every|each|any)\s*time\s+you\b/i,
|
|
319
|
+
/\bwhenever\s+you\b/i,
|
|
320
|
+
/\b(?:second|third|fourth|fifth|sixth|\d+(?:st|nd|rd|th))\s+time\s+(?:you|i'?ve|i have)\b/i,
|
|
321
|
+
// A copula/modal between "you" and the quantifier — "you are never", "you're always", "you were
|
|
322
|
+
// never" — is the SAME binding as the bare form above, just with an auxiliary in between. Found on
|
|
323
|
+
// the real corpus (2026-07-23): "You are never, ever, ever supposed to do things from memory" and
|
|
324
|
+
// "You're never supposed to take things from old memory" both missed the bare-form regex because of
|
|
325
|
+
// the copula, even though the subject is unambiguously "you". Mirrors the modal list THIRD_PERSON_
|
|
326
|
+
// QUANT already uses for third-person subjects — this was an asymmetry, not a deliberate choice.
|
|
327
|
+
/\byou(?:\s+(?:are|were|was|do|does|did|have|had|will|would|should|must|shall|can|could|may|might)|'re|'ve)\s+(?:always|never|constantly|repeatedly|still)\b/i,
|
|
328
|
+
// "I never/always want/expect/need/require you to <verb>" quantifies over the AGENT's occasions
|
|
329
|
+
// exactly as much as "you never <verb>" does — it is just phrased as the speaker's expectation
|
|
330
|
+
// rather than a direct second-person claim. Found on the real corpus: "I never, ever, ever, ever,
|
|
331
|
+
// ever expect you to do shit from memory" and "One thing I always, always, always want you to do
|
|
332
|
+
// is..." both carry a clean quantifier bound to "you to <verb>", and neither matched anything above
|
|
333
|
+
// because the quantifier's grammatical subject is "I", not "you". The bound occasions are still the
|
|
334
|
+
// agent's, so this earns the same signal.
|
|
335
|
+
/\bi\s+(?:always|never)(?:[\s,]+(?:always|never|ever))*\s+(?:want(?:ed)?|expect(?:ed)?|need(?:ed)?|require[ds]?|ask(?:ed)?)\s+you\s+to\b/i,
|
|
336
|
+
];
|
|
337
|
+
|
|
338
|
+
/**
|
|
339
|
+
* `stop <gerund>` is inherently imperative-to-you and is in ADR-033 §1's own Signal 3 list. The
|
|
340
|
+
* gerund requirement is load-bearing: it is what separates "Stop giving me scores out of 10" (a
|
|
341
|
+
* rule) from "Stop and clear it" (the tail of a rant at Vercel, Fable's N4).
|
|
342
|
+
*
|
|
343
|
+
* The PRO-VERB exclusion was added after a corpus run: "Stop doing that." fired, and it is a
|
|
344
|
+
* correction — but `doing`/`being`/`having` carry no transferable content, so the resulting lesson
|
|
345
|
+
* is the literal string "Stop doing that", which would interrupt a future gate with a rule nobody
|
|
346
|
+
* can act on. A statement must carry content beyond the correction frame itself, or it is a
|
|
347
|
+
* deictic pointing at a moment that has already passed.
|
|
348
|
+
*/
|
|
349
|
+
const STOP_GERUND = /\bstop\s+(?:\w+\s+){0,2}?(?!(?:doing|being|having)\b)\w+ing\b/i;
|
|
350
|
+
|
|
351
|
+
/**
|
|
352
|
+
* AGENT-DIRECTED IMPERATIVE (added 2026-07-24 for N3 recall — GATED, see the detector).
|
|
353
|
+
*
|
|
354
|
+
* The detector's recall was measured at ~3%: it fired only on utterances carrying an explicit
|
|
355
|
+
* quantifier ("you always/never", "from now on"). Most real corrections are plain directives with no
|
|
356
|
+
* quantifier at all — "I want you to pick it up", "you need to run both suites". This catches the
|
|
357
|
+
* class the quantifier net structurally cannot.
|
|
358
|
+
*
|
|
359
|
+
* It is grammatically AGENT-BOUND by construction — the object of the directive is "you" — which is
|
|
360
|
+
* exactly what keeps it clear of the negative classes that sink a naive broadening. Every false-
|
|
361
|
+
* positive class in the test suite addresses an ARTIFACT, not the agent: "Make sure the parser never
|
|
362
|
+
* accepts…", "The retry policy should always back off", "It always crashes". None of them say "you".
|
|
363
|
+
* So "you need to / you must / I need you to" cannot match them.
|
|
364
|
+
*
|
|
365
|
+
* A directive is NOT a correction on its own, though — "you should add error handling here" is an
|
|
366
|
+
* ordinary first-time request. So this binding is the ONE signal the detector refuses to let stand
|
|
367
|
+
* alone: when it is the only quantifier signal, the detector requires STRONG negative valence
|
|
368
|
+
* (a prohibition, reproach, or rejection — not a bare temporal scope) before it fires. A directive
|
|
369
|
+
* that also REJECTS something is a correction; a directive that merely instructs is not. See the
|
|
370
|
+
* `directive-imperative`-only gate in detectCorrection.
|
|
371
|
+
*/
|
|
372
|
+
const AGENT_DIRECTED_IMPERATIVE = [
|
|
373
|
+
/\bi\s+(?:really\s+|just\s+)?(?:need|want|expect|require)\s+you\s+to\b/i,
|
|
374
|
+
/\byou\s+(?:need|have|ought)\s+to\b/i,
|
|
375
|
+
/\byou\s+(?:must|should)\b/i,
|
|
376
|
+
];
|
|
377
|
+
|
|
378
|
+
/**
|
|
379
|
+
* Temporal scope markers. These quantify over future occasions, but say nothing about WHOSE — so
|
|
380
|
+
* they only count when the utterance also addresses the agent (second person, or a clause-initial
|
|
381
|
+
* prohibition). "From now on, close the issue yourself" counts; "From now on the parser should
|
|
382
|
+
* validate input" does not.
|
|
383
|
+
*/
|
|
384
|
+
const TEMPORAL_SCOPE = /\b(?:from now on|going forward|in (?:the )?future|henceforth|next time)\b/i;
|
|
385
|
+
const ADDRESSES_AGENT = /\byou(?:r|rself)?\b/i;
|
|
386
|
+
|
|
387
|
+
/** Clause boundaries. Splitting on coordinators is why THIRD_PERSON_QUANT exists as a backstop. */
|
|
388
|
+
const CLAUSE_SPLIT = /\s*(?:[,;:—–]|\.\.\.|\band\b|\bbut\b|\bso\b|\bthen\b|\byet\b)\s*/i;
|
|
389
|
+
|
|
390
|
+
/** Discourse material that may precede a genuine imperative without giving it a subject. */
|
|
391
|
+
const FILLERS =
|
|
392
|
+
'(?:please|just|also|again|ok|okay|well|right|look|hey|actually|honestly|seriously|by the way|btw|remember|note|and|but|from now on|going forward|in the future|in future|next time)';
|
|
393
|
+
|
|
394
|
+
/** A quantifier in clause-initial imperative position — nothing but fillers between it and the boundary. */
|
|
395
|
+
const CLAUSE_INITIAL_QUANT = new RegExp(
|
|
396
|
+
`^(?:${FILLERS}[\\s,]+)*(always|never(?!\\s*mind)|no longer|no more|constantly|repeatedly)\\b`,
|
|
397
|
+
'i',
|
|
398
|
+
);
|
|
399
|
+
|
|
400
|
+
// ── SIGNAL 4: NEGATIVE VALENCE TOWARD THE AGENT'S ACTION ─────────────────────────────────────────
|
|
401
|
+
// A bare positive standing order is refused. "Always use tabs" is indistinguishable from a project
|
|
402
|
+
// convention being stated for the first time; a CORRECTION rejects something that already happened.
|
|
403
|
+
|
|
404
|
+
const VALENCE = [
|
|
405
|
+
['prohibition', /\b(?:never(?!\s*mind)|don'?t|do not|stop|quit|no longer|no more|cut it out|knock it off)\b/i],
|
|
406
|
+
['reproach', /\b(?:you keep|you always|i (?:told|asked) you|i already (?:told|asked|said)|you (?:were supposed to|should have|failed to|forgot to|didn'?t)|why (?:did|didn'?t|are|aren'?t) you|(?:second|third|fourth|fifth|\d+(?:st|nd|rd|th))\s+time)\b/i],
|
|
407
|
+
['rejection', /\b(?:wrong|incorrect|that'?s not|not what i (?:asked|wanted|said)|instead of|rather than|isn'?t what)\b/i],
|
|
408
|
+
['change-of-behaviour', TEMPORAL_SCOPE],
|
|
409
|
+
];
|
|
410
|
+
|
|
411
|
+
// ── TRIGGER INFERENCE ────────────────────────────────────────────────────────────────────────────
|
|
412
|
+
// TRIGGERS is a CLOSED enum in lesson-store.mjs, and deliberately so: gates scale with decision
|
|
413
|
+
// types, lessons scale with experience. A correction that maps to no trigger is prose, and the store
|
|
414
|
+
// throws on prose — so failing to map is a silence, never a default bucket.
|
|
415
|
+
//
|
|
416
|
+
// A TIE IS ALSO A SILENCE, and that rule was earned rather than assumed. Scoring the ten genuine
|
|
417
|
+
// corrections in the test suite produced a strict winner for nine of them and a three-way tie at 1
|
|
418
|
+
// for "Every time you finish you skip the tests. From now on run both suites before you tell me it
|
|
419
|
+
// works" — which really does fire at three different moments (`finish`, `claim-done`, `choose-work`)
|
|
420
|
+
// depending on how it is read. The first implementation broke that tie by list order, which is an
|
|
421
|
+
// arbitrary mechanism wearing the costume of a decision. A lesson filed at the wrong trigger
|
|
422
|
+
// interrupts at the wrong moment, and a gate that fires at the wrong moment is the one users learn
|
|
423
|
+
// to scroll past (ADR-030 §5). So ambiguity resolves the same way every other doubt in this module
|
|
424
|
+
// resolves: nothing is emitted. Measured cost: one of ten.
|
|
425
|
+
|
|
426
|
+
const TRIGGER_VOCAB = [
|
|
427
|
+
['relay-number', /\b(?:scores?|scored|scoring|benchmarks?|metrics?|percent(?:age)?|out of \d+|ratings?|grades?|stats?|subagent'?s? (?:result|number))\b/gi],
|
|
428
|
+
['finish', /\b(?:finish(?:ed|ing)?|close the issue|closing the issue|wrap up|when you'?re done|hand (?:it )?back|sign off)\b/gi],
|
|
429
|
+
// `push(?!\s+(?:through|back|forward))` — a corpus run filed "always push through them and do the
|
|
430
|
+
// careful planning" (persevere) as a shipping rule. The idiom is common and it is not `git push`.
|
|
431
|
+
['ship', /\b(?:push(?:ed|ing|es)?(?!\s+(?:through|back|forward|on))|publish(?:ed|ing)?|releas(?:e|ed|ing)|deploy(?:ed|ing|ment)?|ship(?:ped|ping)?|commit(?:ted|ting|s)?|versions?|bump(?:ed|ing)?|npm publish|merge[ds]?)\b/gi],
|
|
432
|
+
['write-code', /\b(?:code|functions?|files?|refactor(?:ed|ing)?|implement(?:ed|ing)?|hardcod(?:e|ed|ing)|wr(?:ote|ites?|iting|itten)|edit(?:ed|ing)?|hand[- ]?roll(?:ed|ing)?|modules?|scripts?)\b/gi],
|
|
433
|
+
['recommend-architecture', /\b(?:architect(?:ure|ural)?|designs?|designed|approach(?:es)?|recommend(?:ed|ation|ing)?|suggest(?:ed|ion|ing)?|tradeoffs?|propos(?:e|ed|al))\b/gi],
|
|
434
|
+
['mutate-machine', /\b(?:install(?:ed|ing)?|uninstall|delet(?:e|ed|ing)|keychain|globally|launchagent|outside (?:this|the) repo|my machine)\b/gi],
|
|
435
|
+
['claim-done', /\b(?:done|finished|works?|working|verif(?:y|ied|ying)|tested|completed?|proven?|all set|it'?s live)\b/gi],
|
|
436
|
+
['assert-fact', /\b(?:assert(?:ed|ing|ion)?|claim(?:ed|ing)?|assum(?:e|ed|ing|ption)|guess(?:ed|ing)?|memory|recall(?:ed)?|live source|source of truth|the api|look(?:ed)? it up)\b/gi],
|
|
437
|
+
['report-status', /\b(?:status|report(?:ed|ing)?|progress|summar(?:y|ise|ize|ised|ized)|where (?:we|things) (?:are|stand)|update me|tables?|prose|narrative|clickable|links?|urls?|paths?|show me|present(?:ed|ing)?)\b/gi],
|
|
438
|
+
['choose-work', /\b(?:choose|chose|pick(?:ed)?|priorit(?:y|ise|ize|ised|ized)|work on|backlog|skip(?:ped|ping)?)\b/gi],
|
|
439
|
+
];
|
|
440
|
+
|
|
441
|
+
// ── Helpers ──────────────────────────────────────────────────────────────────────────────────────
|
|
442
|
+
|
|
443
|
+
/**
|
|
444
|
+
* Redaction runs before anything is returned. ADR-033 names the transcript corpus as "a new
|
|
445
|
+
* secret-leakage surface"; a candidate is a durable artifact that a human will read and a store will
|
|
446
|
+
* keep, so a key pasted into a correction must not survive into it.
|
|
447
|
+
*/
|
|
448
|
+
function redact(text) {
|
|
449
|
+
return text
|
|
450
|
+
.replace(/-----BEGIN[^-]{0,40}-----[\s\S]*?-----END[^-]{0,40}-----/g, '[redacted-pem]')
|
|
451
|
+
.replace(/\b(?:sk|rk|pk)-[A-Za-z0-9_-]{16,}/g, '[redacted-key]')
|
|
452
|
+
.replace(/\bgh[pousr]_[A-Za-z0-9]{20,}/g, '[redacted-token]')
|
|
453
|
+
.replace(/\bAKIA[0-9A-Z]{16}\b/g, '[redacted-aws-key]')
|
|
454
|
+
.replace(/\bBearer\s+[A-Za-z0-9._-]{16,}/gi, 'Bearer [redacted]')
|
|
455
|
+
.replace(/\b[A-Fa-f0-9]{40,}\b/g, '[redacted-hex]')
|
|
456
|
+
.replace(/\b(password|passwd|secret|api[_-]?key|token)\s*[:=]\s*\S+/gi, '$1=[redacted]');
|
|
457
|
+
}
|
|
458
|
+
|
|
459
|
+
const sentencesOf = (text) =>
|
|
460
|
+
text
|
|
461
|
+
.split(/(?<=[.!?])\s+|\n+/)
|
|
462
|
+
.map((s) => s.trim())
|
|
463
|
+
.filter(Boolean);
|
|
464
|
+
|
|
465
|
+
const anyMatch = (patterns, text) => patterns.some((re) => re.test(text));
|
|
466
|
+
|
|
467
|
+
/**
|
|
468
|
+
* REPROACH-AS-QUESTION DISCRIMINATOR (found 2026-07-24: the corpus widened from 1 project to 9,
|
|
469
|
+
* 4,083 transcripts, 19 detections, three blind independent raters — holdout precision came back at
|
|
470
|
+
* 77.8%, below ADR-033 §2's ≥90% floor). Both holdout false positives, and every tune-half borderline,
|
|
471
|
+
* were the SAME named shape: "you keep telling me it's working and then it doesn't — why?", "You keep
|
|
472
|
+
* giving partial solutions. Do I need to restart?". These satisfy Signal 3 and Signal 4 on the SAME
|
|
473
|
+
* lexical fact — "you keep" (or an ordinal "that's the third time you've…") is simultaneously a
|
|
474
|
+
* BOUND_SECOND_PERSON hit and VALENCE's own "reproach" bucket — which is the identical double-hat
|
|
475
|
+
* shape the temporal-scope guard above already refuses, just with a different marker. A real recurring
|
|
476
|
+
* complaint like this states THAT something happened again; it states nothing about what should happen
|
|
477
|
+
* instead, and a human reads it as complaint, not instruction.
|
|
478
|
+
*
|
|
479
|
+
* This is deliberately NOT "suppress anything with a question mark". An utterance carrying both the
|
|
480
|
+
* reproach AND a stated rule ("…— from now on, never do that again") must still fire, and does: the
|
|
481
|
+
* moment ANY hit is a genuine forward directive — an explicit "always/never"-class bound quantifier,
|
|
482
|
+
* `stop <gerund>`, a clause-initial imperative, or a temporal-scope marker — isDirectiveHit is true for
|
|
483
|
+
* that hit and the guard below does not apply, regardless of how many question marks are nearby. It
|
|
484
|
+
* fires ONLY when every hit is a WEAK recurrence marker: bare "you keep"/"keep on", the copula "still"
|
|
485
|
+
* form, or an ordinal "Nth time" — none of which assert a universal or an imperative on their own —
|
|
486
|
+
* valence carries nothing but reproach, and the utterance asks a question somewhere. Two genuine true
|
|
487
|
+
* positives already in this file's test suite share the exact same weak-marker shape with NO question
|
|
488
|
+
* present ("You keep committing behaviour changes… Every push bumps it, same commit." / "That's the
|
|
489
|
+
* third time you've hardcoded the version in the script.") and must keep firing — the question-mark
|
|
490
|
+
* condition, not the marker, is what tells the two classes apart.
|
|
491
|
+
*/
|
|
492
|
+
const WEAK_RECURRENCE_MARKER = /\bkeep(?:\s+on)?\b|\bstill\b/;
|
|
493
|
+
const ORDINAL_TIME_MARKER = /\b(?:second|third|fourth|fifth|sixth|\d+(?:st|nd|rd|th))\s+time\b/;
|
|
494
|
+
function isDirectiveHit(hit) {
|
|
495
|
+
// stop-gerund / clause-initial-imperative / temporal-scope are stated rules by construction — only
|
|
496
|
+
// the second-person bucket mixes genuine universals ("you always/never") in with bare recurrence.
|
|
497
|
+
if (hit.binding !== 'second-person') return true;
|
|
498
|
+
return !(WEAK_RECURRENCE_MARKER.test(hit.marker) || ORDINAL_TIME_MARKER.test(hit.marker));
|
|
499
|
+
}
|
|
500
|
+
|
|
501
|
+
/**
|
|
502
|
+
* SIGNALS 2+3, as one predicate — see the header. Returns the markers that bind a quantifier to the
|
|
503
|
+
* agent within this sentence, or [] if the sentence quantifies over something else (a program, a
|
|
504
|
+
* spec, a third party) or does not quantify at all.
|
|
505
|
+
*/
|
|
506
|
+
function agentBoundQuantifiers(sentence) {
|
|
507
|
+
// A third-person subject governing the quantifier settles it: this is a claim about the world.
|
|
508
|
+
if (THIRD_PERSON_QUANT.test(sentence)) return [];
|
|
509
|
+
|
|
510
|
+
const hits = [];
|
|
511
|
+
for (const re of BOUND_SECOND_PERSON) {
|
|
512
|
+
const m = re.exec(sentence);
|
|
513
|
+
if (m) hits.push({ marker: m[0].toLowerCase().replace(/\s+/g, ' '), binding: 'second-person' });
|
|
514
|
+
}
|
|
515
|
+
const stop = STOP_GERUND.exec(sentence);
|
|
516
|
+
if (stop) hits.push({ marker: stop[0].toLowerCase(), binding: 'imperative' });
|
|
517
|
+
|
|
518
|
+
// Agent-directed imperative — a directive whose object is "you". GATED: on its own it is an
|
|
519
|
+
// ordinary request, so the detector fires on it only alongside strong rejection valence (see the
|
|
520
|
+
// directive-imperative gate). Recorded here as its own binding so that gate can find it.
|
|
521
|
+
for (const re of AGENT_DIRECTED_IMPERATIVE) {
|
|
522
|
+
const m = re.exec(sentence);
|
|
523
|
+
if (m) { hits.push({ marker: m[0].toLowerCase().replace(/\s+/g, ' '), binding: 'directive-imperative' }); break; }
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
for (const clause of sentence.split(CLAUSE_SPLIT)) {
|
|
527
|
+
const m = CLAUSE_INITIAL_QUANT.exec(clause.trim());
|
|
528
|
+
if (m) hits.push({ marker: m[1].toLowerCase(), binding: 'clause-initial-imperative' });
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
// Temporal scope quantifies over occasions but not over an agent — it counts only when the
|
|
532
|
+
// utterance actually addresses this agent.
|
|
533
|
+
if (TEMPORAL_SCOPE.test(sentence) && (ADDRESSES_AGENT.test(sentence) || hits.length)) {
|
|
534
|
+
hits.push({ marker: TEMPORAL_SCOPE.exec(sentence)[0].toLowerCase(), binding: 'temporal-scope' });
|
|
535
|
+
}
|
|
536
|
+
return hits;
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
function inferTrigger(text) {
|
|
540
|
+
const scored = TRIGGER_VOCAB
|
|
541
|
+
.map(([key, re]) => ({ key, score: (text.match(new RegExp(re.source, 'gi')) || []).length }))
|
|
542
|
+
.filter((s) => s.score > 0)
|
|
543
|
+
.sort((a, b) => b.score - a.score);
|
|
544
|
+
|
|
545
|
+
if (!scored.length) return null; // no trigger → prose → silence
|
|
546
|
+
if (scored.length > 1 && scored[1].score === scored[0].score) return null; // ambiguous → silence
|
|
547
|
+
return scored[0];
|
|
548
|
+
}
|
|
549
|
+
|
|
550
|
+
// ── The detector ─────────────────────────────────────────────────────────────────────────────────
|
|
551
|
+
|
|
552
|
+
/**
|
|
553
|
+
* @param {string} promptText the user's own words, from the UserPromptSubmit payload's `.prompt`
|
|
554
|
+
* (Sol C5: the authenticated utterance, never a transcript reconstruction)
|
|
555
|
+
* @param {object} context
|
|
556
|
+
* @param {object|null} context.precedingAssistantAction what the agent did immediately before —
|
|
557
|
+
* `{ tool?, summary? }`. REQUIRED: Signal 1 (adjacency). Its absence means there is nothing
|
|
558
|
+
* for the utterance to be correcting, which is how the harness-injected turn — the single
|
|
559
|
+
* highest-scoring hit in the corpus — is refused structurally rather than by blacklist.
|
|
560
|
+
* @param {string} [context.transcriptPath] provenance only, so a ruling can be checked
|
|
561
|
+
* @param {number} [context.turnIndex]
|
|
562
|
+
* @param {string} [context.timestamp]
|
|
563
|
+
* @returns {null | {statement, trigger, evidence, confidence, origin, status}}
|
|
564
|
+
*/
|
|
565
|
+
export function detectCorrection(promptText, context = {}) {
|
|
566
|
+
if (typeof promptText !== 'string') return null;
|
|
567
|
+
const raw = promptText.trim();
|
|
568
|
+
if (!raw || raw.length > MAX_UTTERANCE_CHARS || raw.length < MIN_STATEMENT_CHARS) return null;
|
|
569
|
+
|
|
570
|
+
// SIGNAL 1 — adjacency. No preceding agent action, no correction.
|
|
571
|
+
const prior = context.precedingAssistantAction;
|
|
572
|
+
const respondingTo = prior && (prior.summary || prior.tool) ? String(prior.summary || prior.tool).slice(0, 200) : null;
|
|
573
|
+
if (!respondingTo) return null;
|
|
574
|
+
|
|
575
|
+
// Hard exclusions — cheapest first, and every one of them is a named FP class.
|
|
576
|
+
if (anyMatch(HARNESS_TEMPLATES, raw)) return null; // not an utterance
|
|
577
|
+
if (anyMatch(DOCUMENT_MARKERS, raw)) return null; // pasted content, incl. this repo's own ADRs
|
|
578
|
+
if (anyMatch(USER_THIRD_PERSON, raw)) return null; // the injection tell
|
|
579
|
+
if (anyMatch(ATTRIBUTION, raw)) return null; // quoting a rule is not issuing one
|
|
580
|
+
if (anyMatch(DELEGATION, raw)) return null; // the subject is a third party
|
|
581
|
+
if (SPEC_FRAME.test(raw)) return null; // requirement about an artifact
|
|
582
|
+
if (raw.startsWith('/')) return null; // slash command, not speech
|
|
583
|
+
|
|
584
|
+
// SIGNALS 2+3 — a quantifier bound to the agent, in a sentence that is neither a question nor a
|
|
585
|
+
// hypothetical. Both filters are applied per sentence: "Always give me a link. Does that work?"
|
|
586
|
+
// must survive its own trailing question.
|
|
587
|
+
const bearing = [];
|
|
588
|
+
const signals = [];
|
|
589
|
+
for (const sentence of sentencesOf(raw)) {
|
|
590
|
+
if (sentence.endsWith('?')) continue;
|
|
591
|
+
if (HEDGE.test(sentence)) continue;
|
|
592
|
+
const hits = agentBoundQuantifiers(sentence);
|
|
593
|
+
if (!hits.length) continue;
|
|
594
|
+
bearing.push(sentence);
|
|
595
|
+
signals.push(...hits.map((h) => ({ ...h, span: sentence.slice(0, 120) })));
|
|
596
|
+
}
|
|
597
|
+
if (!bearing.length) return null;
|
|
598
|
+
|
|
599
|
+
// SIGNAL 4 — negative valence. A correction rejects something; a first-time convention does not.
|
|
600
|
+
const valence = VALENCE.filter(([, re]) => re.test(raw)).map(([name]) => name);
|
|
601
|
+
if (!valence.length) return null;
|
|
602
|
+
|
|
603
|
+
// NO SINGLE MARKER MAY SATISFY TWO INDEPENDENT SIGNALS. A temporal scope marker ("in the future")
|
|
604
|
+
// is the one token that appears in both tests — as a weak Signal 3 binding and as Signal 4's
|
|
605
|
+
// change-of-behaviour valence — so an utterance carrying nothing else clears a four-signal
|
|
606
|
+
// conjunction on the strength of one lexical fact wearing two hats.
|
|
607
|
+
//
|
|
608
|
+
// Found by running the real corpus, not by reasoning: both surviving false positives in 1,451
|
|
609
|
+
// turns were this exact shape, and both were statements of DESIRE rather than correction —
|
|
610
|
+
// "The point is I want you to be able to do it now and in the future." The genuine ones in the
|
|
611
|
+
// same shape ("From now on, close the issue yourself — don't ask me") always carry an independent
|
|
612
|
+
// prohibition. Requiring that independence drops both FPs and keeps every true positive.
|
|
613
|
+
const bindings = new Set(signals.map((s) => s.binding));
|
|
614
|
+
if (bindings.size === 1 && bindings.has('temporal-scope')
|
|
615
|
+
&& valence.length === 1 && valence[0] === 'change-of-behaviour') {
|
|
616
|
+
return null;
|
|
617
|
+
}
|
|
618
|
+
|
|
619
|
+
// AGENT-DIRECTED-IMPERATIVE GATE. A directive whose only signal is "you need to / I want you to"
|
|
620
|
+
// is an ordinary forward request unless it also REJECTS something. "You should add a test here" is
|
|
621
|
+
// not a correction; "You should have run the tests — you keep skipping them" is. So when the
|
|
622
|
+
// directive-imperative binding stands alone (no genuine quantifier beside it), require a STRONG
|
|
623
|
+
// valence class — a prohibition, reproach, or rejection — and refuse on a bare change-of-behaviour
|
|
624
|
+
// temporal marker, which any forward-looking request carries. This is the price of the recall the
|
|
625
|
+
// binding buys: it fires on directives that reject, never on directives that merely instruct.
|
|
626
|
+
const STRONG_VALENCE = new Set(['prohibition', 'reproach', 'rejection']);
|
|
627
|
+
if (bindings.size === 1 && bindings.has('directive-imperative')
|
|
628
|
+
&& !valence.some((v) => STRONG_VALENCE.has(v))) {
|
|
629
|
+
return null;
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
// REPROACH PHRASED AS A QUESTION — see isDirectiveHit's header comment for the full reasoning.
|
|
633
|
+
// Fires only when nothing anywhere in the utterance is a genuine forward directive, valence is
|
|
634
|
+
// reproach and nothing else, and the utterance asks a question somewhere: a recurring complaint
|
|
635
|
+
// with no stated rule, not a standing order.
|
|
636
|
+
const hasDirectiveSignal = signals.some(isDirectiveHit);
|
|
637
|
+
if (!hasDirectiveSignal && valence.length === 1 && valence[0] === 'reproach' && /\?/.test(raw)) {
|
|
638
|
+
return null;
|
|
639
|
+
}
|
|
640
|
+
|
|
641
|
+
const statement = redact(bearing.join(' ')).slice(0, MAX_STATEMENT_CHARS).trim();
|
|
642
|
+
if (statement.length < MIN_STATEMENT_CHARS) return null;
|
|
643
|
+
|
|
644
|
+
// The closed enum has the last word: no trigger, no lesson.
|
|
645
|
+
//
|
|
646
|
+
// Inferred from the CORRECTION SENTENCES, not from the whole turn. The first version scored the
|
|
647
|
+
// whole utterance and the corpus caught it out: a turn whose rule was "never give me an issue
|
|
648
|
+
// without a solution in the same line" (a status-reporting rule) filed as `ship`, because the
|
|
649
|
+
// user had mentioned "version numbers" two sentences earlier about something else entirely.
|
|
650
|
+
// Incidental vocabulary elsewhere in a turn should not decide when a rule interrupts — that is
|
|
651
|
+
// the same misfiling that makes the tie-break rule above refuse rather than guess.
|
|
652
|
+
const trigger = inferTrigger(statement);
|
|
653
|
+
if (!trigger) return null;
|
|
654
|
+
|
|
655
|
+
// Ordering only. Never a gate, never an auto-ratification threshold, and never 1.0.
|
|
656
|
+
const corroboration = new Set(signals.map((s) => s.binding)).size + valence.length;
|
|
657
|
+
const confidence = Math.min(0.9, 0.5 + 0.1 * (corroboration - 1));
|
|
658
|
+
|
|
659
|
+
return {
|
|
660
|
+
statement,
|
|
661
|
+
trigger: trigger.key,
|
|
662
|
+
|
|
663
|
+
// Sol C2: `makeLesson()` drops unknown top-level keys but passes `evidence[]` through whole, so
|
|
664
|
+
// the entire ratification surface rides inside it. `observed` is what lesson-gate.mjs prints.
|
|
665
|
+
evidence: [{
|
|
666
|
+
observed: `you said: "${redact(raw).slice(0, MAX_QUOTE_CHARS)}"`,
|
|
667
|
+
quote: redact(raw).slice(0, MAX_QUOTE_CHARS),
|
|
668
|
+
respondingTo,
|
|
669
|
+
signals,
|
|
670
|
+
valence,
|
|
671
|
+
detector: 'correction-detect',
|
|
672
|
+
source: {
|
|
673
|
+
transcriptPath: context.transcriptPath ?? null,
|
|
674
|
+
turnIndex: context.turnIndex ?? null,
|
|
675
|
+
timestamp: context.timestamp ?? null,
|
|
676
|
+
},
|
|
677
|
+
}],
|
|
678
|
+
|
|
679
|
+
confidence,
|
|
680
|
+
|
|
681
|
+
// ADR-033 §4, unconditional and with no override parameter — a caller cannot ask for anything
|
|
682
|
+
// else, because the only honest answer to "who put this row in the store" is: a machine.
|
|
683
|
+
origin: 'model-inferred',
|
|
684
|
+
status: 'candidate',
|
|
685
|
+
};
|
|
686
|
+
}
|