rulereceipt 0.1.35 → 0.1.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +44 -0
- package/dist/checks/claimEvidence.js +8 -3
- package/dist/checks/classify.js +59 -6
- package/dist/checks/testCommands.d.ts +33 -0
- package/dist/checks/testCommands.js +56 -1
- package/dist/cli.js +7 -0
- package/dist/evaluate.d.ts +22 -0
- package/dist/evaluate.js +51 -0
- package/dist/hook.d.ts +35 -0
- package/dist/hook.js +100 -0
- package/dist/report/generateReport.js +29 -5
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -88,6 +88,7 @@ rulereceipt rules --include <handle> # "this IS a rule" — check it from now
|
|
|
88
88
|
rulereceipt rules --exclude <handle> # "this isn't" — stop reporting it
|
|
89
89
|
rulereceipt rules --coverage # which rules a configured hook might actually enforce
|
|
90
90
|
rulereceipt doctor # list hooks/auto-run tasks configured on this machine
|
|
91
|
+
rulereceipt hook # run AS a Claude Code Stop hook — block Claude finishing on a broken rule
|
|
91
92
|
rulereceipt lint # find contradictions between CLAUDE.md and AGENTS.md
|
|
92
93
|
rulereceipt digest # summarise recent checks; --email to send it
|
|
93
94
|
rulereceipt config # set up email sending (stays on your machine)
|
|
@@ -99,6 +100,49 @@ rulereceipt verify <session-file> <hash> # spot-check a report you received ag
|
|
|
99
100
|
|
|
100
101
|
`verify` isn't a routine check — trust your team day to day, same as any status update. It's there for the rare case it actually matters (a dispute, an incident review): give it the session file and the hash printed in the report, and it confirms whether they really match.
|
|
101
102
|
|
|
103
|
+
## Blocking, not just reporting
|
|
104
|
+
|
|
105
|
+
`rulereceipt check` tells you afterwards. `rulereceipt hook` refuses to let the
|
|
106
|
+
session end.
|
|
107
|
+
|
|
108
|
+
Add this to `.claude/settings.json` — you add it, we never do:
|
|
109
|
+
|
|
110
|
+
```json
|
|
111
|
+
{
|
|
112
|
+
"hooks": {
|
|
113
|
+
"Stop": [
|
|
114
|
+
{ "hooks": [ { "type": "command", "command": "npx rulereceipt hook" } ] }
|
|
115
|
+
]
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
When Claude tries to finish, it reads the session that just happened. If a rule
|
|
121
|
+
was broken it hands Claude the rule, the evidence, and instructions to keep
|
|
122
|
+
working, so the session cannot end on a claim that isn't backed.
|
|
123
|
+
|
|
124
|
+
The one it is actually for: *"Done — all tests pass"* when the last run of
|
|
125
|
+
`npm test` returned two failures. It checks the claim against what ran, which
|
|
126
|
+
is the part a model cannot talk its way around.
|
|
127
|
+
|
|
128
|
+
Three properties worth knowing before you wire it in:
|
|
129
|
+
|
|
130
|
+
- **It only blocks on things it can prove.** Never a judgment rule, never an
|
|
131
|
+
LLM opinion, never "couldn't tell". Only a matched literal or a claim
|
|
132
|
+
contradicted by a recorded tool result. Run against twelve real sessions it
|
|
133
|
+
blocked none of them.
|
|
134
|
+
- **It cannot loop.** Claude Code sets `stop_hook_active` when a session is
|
|
135
|
+
already continuing because of a block; the hook returns immediately in that
|
|
136
|
+
case. One interruption per stop.
|
|
137
|
+
- **It fails open.** Unreadable transcript, missing rules file, a bug in us —
|
|
138
|
+
it allows the stop and writes a line to stderr. Failing closed would mean our
|
|
139
|
+
bug locks you out of finishing your own session. That is a deliberate
|
|
140
|
+
weakening, and it is why `check` in CI stays the backstop.
|
|
141
|
+
|
|
142
|
+
It runs when Claude stops, so it catches a finished session, not a command
|
|
143
|
+
mid-flight. For that, use a `PreToolUse` hook of your own — `rulereceipt
|
|
144
|
+
doctor` will show you what you already have.
|
|
145
|
+
|
|
102
146
|
## Which rules actually have teeth
|
|
103
147
|
|
|
104
148
|
A rule in a file and a rule with a `PreToolUse` hook behind it look identical
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { TEST_COMMAND } from "./testCommands.js";
|
|
1
|
+
import { TEST_COMMAND, withoutHeredocs, countTestRuns } from "./testCommands.js";
|
|
2
2
|
/**
|
|
3
3
|
* Did the session claim something worked, when the log says it didn't?
|
|
4
4
|
*
|
|
@@ -200,7 +200,7 @@ export function runClaimEvidenceChecks(classifications, events) {
|
|
|
200
200
|
for (const event of events) {
|
|
201
201
|
const command = commandOf(event);
|
|
202
202
|
if (command !== null) {
|
|
203
|
-
pendingRun = TEST_COMMAND.test(command) ? command : null;
|
|
203
|
+
pendingRun = TEST_COMMAND.test(withoutHeredocs(command)) ? command : null;
|
|
204
204
|
for (const action of ACTION_CLAIMS) {
|
|
205
205
|
if (action.command.test(command))
|
|
206
206
|
commandsSeen.add(action.label);
|
|
@@ -222,10 +222,15 @@ export function runClaimEvidenceChecks(classifications, events) {
|
|
|
222
222
|
// words survive a pipe, the exit status does not.
|
|
223
223
|
const stated = outcomeFromOutput(event.content);
|
|
224
224
|
const trustExitCode = !PIPED.test(pendingRun);
|
|
225
|
+
// Two suite invocations in one command means neither the exit code
|
|
226
|
+
// nor the printed summary belongs to a single run, so nothing about
|
|
227
|
+
// this command can contradict a claim. Checked before both, because
|
|
228
|
+
// reading the output is what defeated the pipe guard here.
|
|
229
|
+
const oneRun = countTestRuns(pendingRun) <= 1;
|
|
225
230
|
lastRun = {
|
|
226
231
|
command: pendingRun,
|
|
227
232
|
failed: stated !== null ? stated : event.isError,
|
|
228
|
-
outcomeReadable: stated !== null || trustExitCode,
|
|
233
|
+
outcomeReadable: oneRun && (stated !== null || trustExitCode),
|
|
229
234
|
output: event.content.slice(0, 200),
|
|
230
235
|
};
|
|
231
236
|
pendingRun = null;
|
package/dist/checks/classify.js
CHANGED
|
@@ -200,15 +200,67 @@ const PRE_ACTION_GATE = /\b(?:repeat[- ]back|restate\s+what|wait\s+for\s+(?:conf
|
|
|
200
200
|
* failures on its own.
|
|
201
201
|
*/
|
|
202
202
|
const MAX_RULE_BODY_FOR_CLAIM = 1500;
|
|
203
|
+
/**
|
|
204
|
+
* The claim shape: a reporting verb NEXT TO a done-word.
|
|
205
|
+
*
|
|
206
|
+
* "Do not claim tests passed", "before declaring work complete", "before
|
|
207
|
+
* telling user publishing is complete" — the two halves sit within a few
|
|
208
|
+
* words of each other, in one sentence, because that adjacency IS the
|
|
209
|
+
* thing being forbidden: announcing a finish you have not verified.
|
|
210
|
+
*
|
|
211
|
+
* Testing the two words independently across a whole section is what kept
|
|
212
|
+
* mis-routing. A 944-character rule about writing Slack updates contained a
|
|
213
|
+
* reporting verb in one paragraph, an evidence noun in another, and the
|
|
214
|
+
* done-word inside the compound "Slack-ready" — three unrelated matches,
|
|
215
|
+
* one confident verdict about whether a test command was piped. That was
|
|
216
|
+
* patched twice with blocklist entries (PRE_ACTION_GATE) before the
|
|
217
|
+
* independence was recognised as the fault itself.
|
|
218
|
+
*
|
|
219
|
+
* Thirty characters, measured rather than picked: it is the smallest
|
|
220
|
+
* window that keeps every genuine claim rule in the corpus and the widest
|
|
221
|
+
* that admits none of the scattered ones. `[^.\n]` confines the match to a
|
|
222
|
+
* single sentence, which is what stops a match spanning a paragraph break.
|
|
223
|
+
* Across 559 public rules files this narrows claim-evidence from 38 rules
|
|
224
|
+
* to 8, and all 8 read as the same instruction in different words.
|
|
225
|
+
*/
|
|
226
|
+
const CLAIM_SHAPE = new RegExp(`(?:${REPORTING_VERB.source})[^.\n]{0,30}?(?:${DONE_WORD.source})` +
|
|
227
|
+
`|(?:${DONE_WORD.source})[^.\n]{0,30}?(?:${REPORTING_VERB.source})`, "i");
|
|
203
228
|
function isClaimEvidenceRule(rule) {
|
|
204
229
|
if (rule.text.length > MAX_RULE_BODY_FOR_CLAIM)
|
|
205
230
|
return false;
|
|
206
231
|
const text = `${rule.title} ${rule.text}`;
|
|
232
|
+
// Kept as a second line of defence even though CLAIM_SHAPE now excludes
|
|
233
|
+
// every case it was added for. It is cheap, it is tested, and the rules
|
|
234
|
+
// it names are ones this checker must never answer.
|
|
207
235
|
if (PRE_ACTION_GATE.test(text))
|
|
208
236
|
return false;
|
|
209
|
-
return
|
|
237
|
+
return CLAIM_SHAPE.test(text) && EVIDENCE_NOUN.test(text);
|
|
210
238
|
}
|
|
211
239
|
const BRANCH_WORD = /\bbranch\b/i;
|
|
240
|
+
/**
|
|
241
|
+
* A literal that could actually be a git branch name.
|
|
242
|
+
*
|
|
243
|
+
* git's refname rules, reduced to what a CLAUDE.md really writes: no
|
|
244
|
+
* whitespace, no shell punctuation, no `..`, no leading dash, not ending
|
|
245
|
+
* `.lock`. Everything this excludes was being accepted as a branch name and
|
|
246
|
+
* then searched for in the session's git commands.
|
|
247
|
+
*
|
|
248
|
+
* Rejecting a template is the point of the character class rather than an
|
|
249
|
+
* accident of it: `{`, `<` and `$` are how a rules file writes a PATTERN
|
|
250
|
+
* for branch names ("squad/{issue-number}-{kebab-case-slug}"), and a
|
|
251
|
+
* pattern is not a branch. Checking whether the session used a branch
|
|
252
|
+
* literally called that has no meaningful answer.
|
|
253
|
+
*/
|
|
254
|
+
const BRANCH_NAME_PATTERN = /^[A-Za-z0-9._/-]{1,100}$/;
|
|
255
|
+
function isBranchName(literal) {
|
|
256
|
+
if (!BRANCH_NAME_PATTERN.test(literal))
|
|
257
|
+
return false;
|
|
258
|
+
if (literal.startsWith("-") || literal.startsWith("/"))
|
|
259
|
+
return false;
|
|
260
|
+
if (literal.includes("..") || literal.endsWith(".lock") || literal.endsWith("/"))
|
|
261
|
+
return false;
|
|
262
|
+
return true;
|
|
263
|
+
}
|
|
212
264
|
// A function/method-call shape ("print(", "analytics.track(") is a strong,
|
|
213
265
|
// simple signal that a backtick literal names actual CODE, not a CLI
|
|
214
266
|
// command or flag ("git push --force", "npm test" never look like this).
|
|
@@ -441,11 +493,12 @@ export function classifyRule(rule) {
|
|
|
441
493
|
return { kind: "judgment", rule };
|
|
442
494
|
}
|
|
443
495
|
const polarityInferred = polarityWasInferred(rule);
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
496
|
+
// The first literal that could BE a branch, not simply the first literal.
|
|
497
|
+
// Real rules here name exactly one branch ("the `demo` branch", "never
|
|
498
|
+
// push to the `main` branch"), and a rule that mentions branches while
|
|
499
|
+
// naming a command belongs to whichever checker handles that command.
|
|
500
|
+
const branchName = [...patterns].find(isBranchName);
|
|
501
|
+
if (BRANCH_WORD.test(text) && branchName !== undefined) {
|
|
449
502
|
return { kind: "gitBranchPolicy", rule, branchName, polarity, polarityInferred };
|
|
450
503
|
}
|
|
451
504
|
// Route on ANY code-shaped literal, but check ONLY the code-shaped ones.
|
|
@@ -19,5 +19,38 @@ import type { TranscriptEvent } from "../types.js";
|
|
|
19
19
|
* wrong.
|
|
20
20
|
*/
|
|
21
21
|
export declare const TEST_COMMAND: RegExp;
|
|
22
|
+
/**
|
|
23
|
+
* Removes heredoc bodies from a shell command.
|
|
24
|
+
*
|
|
25
|
+
* A command that WRITES a test command is not a command that RUNS one.
|
|
26
|
+
* Found 2026-09-14 on a real session: two false failures whose "last test
|
|
27
|
+
* run" was a shell variable assignment. The actual match came from a
|
|
28
|
+
* heredoc further down, writing a demo fixture whose body contains the
|
|
29
|
+
* string `npm test`. The literal was being generated, never executed — and
|
|
30
|
+
* the tool then read its own report output as the failing result.
|
|
31
|
+
*
|
|
32
|
+
* Handles both quoted and bare delimiters, and leaves everything after the
|
|
33
|
+
* closing delimiter intact, because a real test run often follows the
|
|
34
|
+
* heredoc that set the fixture up.
|
|
35
|
+
*/
|
|
36
|
+
export declare function withoutHeredocs(command: string): string;
|
|
37
|
+
/**
|
|
38
|
+
* How many times a single shell command invokes a test suite.
|
|
39
|
+
*
|
|
40
|
+
* A command that runs the suite twice has no single outcome to attribute.
|
|
41
|
+
* Real case, 2026-09-14: one command ran a typecheck, then the suite (37
|
|
42
|
+
* passed), then re-ran one file with locking deliberately disabled to
|
|
43
|
+
* demonstrate those tests can fail. The session said "Everything passes:",
|
|
44
|
+
* which was true; the checker read "2 failed" out of the third section and
|
|
45
|
+
* called it a lie.
|
|
46
|
+
*
|
|
47
|
+
* Deliberate red runs cannot be recognised - "is this failure intended" is
|
|
48
|
+
* not in the text. What IS in the text is that the suite ran more than
|
|
49
|
+
* once, and that is enough: the outcome is unattributable for the same
|
|
50
|
+
* reason a pipe makes an exit code unattributable. Heredoc bodies are
|
|
51
|
+
* stripped first, so a command that merely writes a test command twice
|
|
52
|
+
* still counts zero.
|
|
53
|
+
*/
|
|
54
|
+
export declare function countTestRuns(command: string): number;
|
|
22
55
|
/** The first test command run in this session, or null if none ran. */
|
|
23
56
|
export declare function findTestRun(events: TranscriptEvent[]): string | null;
|
|
@@ -18,6 +18,61 @@
|
|
|
18
18
|
* wrong.
|
|
19
19
|
*/
|
|
20
20
|
export const TEST_COMMAND = /\b(?:npm|pnpm|yarn|bun)\s+(?:run\s+)?(?:test|verify|check|ci)\b|\bnpx\s+(?:vitest|jest|mocha|ava)\b|\b(?:vitest|jest|mocha|pytest|phpunit|rspec|tox)\b|\bcargo\s+test\b|\bgo\s+test\b|\bmvn\s+(?:test|verify)\b|\bgradle\s+test\b|\bdotnet\s+test\b|\bpython\s+-m\s+(?:pytest|unittest)\b/i;
|
|
21
|
+
/**
|
|
22
|
+
* Removes heredoc bodies from a shell command.
|
|
23
|
+
*
|
|
24
|
+
* A command that WRITES a test command is not a command that RUNS one.
|
|
25
|
+
* Found 2026-09-14 on a real session: two false failures whose "last test
|
|
26
|
+
* run" was a shell variable assignment. The actual match came from a
|
|
27
|
+
* heredoc further down, writing a demo fixture whose body contains the
|
|
28
|
+
* string `npm test`. The literal was being generated, never executed — and
|
|
29
|
+
* the tool then read its own report output as the failing result.
|
|
30
|
+
*
|
|
31
|
+
* Handles both quoted and bare delimiters, and leaves everything after the
|
|
32
|
+
* closing delimiter intact, because a real test run often follows the
|
|
33
|
+
* heredoc that set the fixture up.
|
|
34
|
+
*/
|
|
35
|
+
export function withoutHeredocs(command) {
|
|
36
|
+
const lines = command.split("\n");
|
|
37
|
+
const out = [];
|
|
38
|
+
let closing = null;
|
|
39
|
+
for (const line of lines) {
|
|
40
|
+
if (closing !== null) {
|
|
41
|
+
if (line.trim() === closing)
|
|
42
|
+
closing = null;
|
|
43
|
+
continue;
|
|
44
|
+
}
|
|
45
|
+
const open = line.match(/<<-?\s*(?:'([^']+)'|"([^"]+)"|([A-Za-z_][A-Za-z0-9_]*))/);
|
|
46
|
+
if (open) {
|
|
47
|
+
closing = open[1] ?? open[2] ?? open[3];
|
|
48
|
+
out.push(line.slice(0, open.index));
|
|
49
|
+
continue;
|
|
50
|
+
}
|
|
51
|
+
out.push(line);
|
|
52
|
+
}
|
|
53
|
+
return out.join("\n");
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* How many times a single shell command invokes a test suite.
|
|
57
|
+
*
|
|
58
|
+
* A command that runs the suite twice has no single outcome to attribute.
|
|
59
|
+
* Real case, 2026-09-14: one command ran a typecheck, then the suite (37
|
|
60
|
+
* passed), then re-ran one file with locking deliberately disabled to
|
|
61
|
+
* demonstrate those tests can fail. The session said "Everything passes:",
|
|
62
|
+
* which was true; the checker read "2 failed" out of the third section and
|
|
63
|
+
* called it a lie.
|
|
64
|
+
*
|
|
65
|
+
* Deliberate red runs cannot be recognised - "is this failure intended" is
|
|
66
|
+
* not in the text. What IS in the text is that the suite ran more than
|
|
67
|
+
* once, and that is enough: the outcome is unattributable for the same
|
|
68
|
+
* reason a pipe makes an exit code unattributable. Heredoc bodies are
|
|
69
|
+
* stripped first, so a command that merely writes a test command twice
|
|
70
|
+
* still counts zero.
|
|
71
|
+
*/
|
|
72
|
+
export function countTestRuns(command) {
|
|
73
|
+
const global = new RegExp(TEST_COMMAND.source, "gi");
|
|
74
|
+
return (withoutHeredocs(command).match(global) ?? []).length;
|
|
75
|
+
}
|
|
21
76
|
/** The first test command run in this session, or null if none ran. */
|
|
22
77
|
export function findTestRun(events) {
|
|
23
78
|
for (const event of events) {
|
|
@@ -25,7 +80,7 @@ export function findTestRun(events) {
|
|
|
25
80
|
continue;
|
|
26
81
|
const input = event.input;
|
|
27
82
|
const command = input && typeof input.command === "string" ? input.command : "";
|
|
28
|
-
if (command && TEST_COMMAND.test(command))
|
|
83
|
+
if (command && TEST_COMMAND.test(withoutHeredocs(command)))
|
|
29
84
|
return command;
|
|
30
85
|
}
|
|
31
86
|
return null;
|
package/dist/cli.js
CHANGED
|
@@ -16,6 +16,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
|
|
|
16
16
|
import { runCodeContentChecks } from "./checks/codeContent.js";
|
|
17
17
|
import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
|
|
18
18
|
import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
|
|
19
|
+
import { runHook } from "./hook.js";
|
|
19
20
|
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
20
21
|
import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
|
|
21
22
|
import { generateHtmlReport } from "./report/generateHtmlReport.js";
|
|
@@ -573,6 +574,12 @@ function runDoctorCommand() {
|
|
|
573
574
|
console.log(`${result.newSinceLastRun.length} of these are new since the last time doctor ran here.`);
|
|
574
575
|
}
|
|
575
576
|
}
|
|
577
|
+
program
|
|
578
|
+
.command("hook")
|
|
579
|
+
.description("run as a Claude Code Stop hook - blocks Claude from finishing on a broken rule (payload on stdin)")
|
|
580
|
+
.action(async () => {
|
|
581
|
+
await runHook(needsLlmResult);
|
|
582
|
+
});
|
|
576
583
|
program
|
|
577
584
|
.command("doctor")
|
|
578
585
|
.description("List every Claude Code hook and VS Code auto-task on this machine/project, flag anything suspicious")
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { type Classification } from "./checks/classify.js";
|
|
2
|
+
import { staleOverrides } from "./overrides.js";
|
|
3
|
+
import type { CheckResult, Rule, TranscriptEvent } from "./types.js";
|
|
4
|
+
export interface Evaluation {
|
|
5
|
+
results: CheckResult[];
|
|
6
|
+
notARule: Classification[];
|
|
7
|
+
stale: ReturnType<typeof staleOverrides>;
|
|
8
|
+
}
|
|
9
|
+
/**
|
|
10
|
+
* Rules in, verdicts out — the whole pipeline, with no printing in it.
|
|
11
|
+
*
|
|
12
|
+
* Extracted 2026-09-14 when a second caller appeared. `check` renders a
|
|
13
|
+
* report for a person; `hook` returns a decision to Claude Code. Those two
|
|
14
|
+
* must never be able to disagree about whether a rule was broken, and the
|
|
15
|
+
* only way to guarantee that is one body of code. The same reasoning is
|
|
16
|
+
* written at the top of testCommands.ts, about two checkers sharing one
|
|
17
|
+
* definition; this is that argument one level up.
|
|
18
|
+
*
|
|
19
|
+
* Deliberately takes `events` rather than a path: the caller decides where
|
|
20
|
+
* a transcript comes from, and a hook is handed one it must not second-guess.
|
|
21
|
+
*/
|
|
22
|
+
export declare function evaluateSession(cwd: string, rules: Rule[], events: TranscriptEvent[], llm: boolean, needsLlmResult: (rule: Rule) => CheckResult): Promise<Evaluation>;
|
package/dist/evaluate.js
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { classifyRules } from "./checks/classify.js";
|
|
2
|
+
import { runDeterministicChecks } from "./checks/deterministicChecks.js";
|
|
3
|
+
import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
|
|
4
|
+
import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
|
|
5
|
+
import { runCodeContentChecks } from "./checks/codeContent.js";
|
|
6
|
+
import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
|
|
7
|
+
import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
|
|
8
|
+
import { runJudgmentChecks } from "./checks/judgmentChecks.js";
|
|
9
|
+
import { loadOverrides, ruleFingerprint, staleOverrides } from "./overrides.js";
|
|
10
|
+
/**
|
|
11
|
+
* Rules in, verdicts out — the whole pipeline, with no printing in it.
|
|
12
|
+
*
|
|
13
|
+
* Extracted 2026-09-14 when a second caller appeared. `check` renders a
|
|
14
|
+
* report for a person; `hook` returns a decision to Claude Code. Those two
|
|
15
|
+
* must never be able to disagree about whether a rule was broken, and the
|
|
16
|
+
* only way to guarantee that is one body of code. The same reasoning is
|
|
17
|
+
* written at the top of testCommands.ts, about two checkers sharing one
|
|
18
|
+
* definition; this is that argument one level up.
|
|
19
|
+
*
|
|
20
|
+
* Deliberately takes `events` rather than a path: the caller decides where
|
|
21
|
+
* a transcript comes from, and a hook is handed one it must not second-guess.
|
|
22
|
+
*/
|
|
23
|
+
export async function evaluateSession(cwd, rules, events, llm, needsLlmResult) {
|
|
24
|
+
const overrides = loadOverrides(cwd);
|
|
25
|
+
const classifications = classifyRules(rules).map((c) => {
|
|
26
|
+
const decision = overrides.get(ruleFingerprint(c.rule))?.decision;
|
|
27
|
+
if (!decision)
|
|
28
|
+
return c;
|
|
29
|
+
if (decision === "notARule")
|
|
30
|
+
return { kind: "notARule", rule: c.rule };
|
|
31
|
+
return c.kind === "notARule" ? { kind: "judgment", rule: c.rule } : c;
|
|
32
|
+
});
|
|
33
|
+
const of = (kind) => classifications.filter((c) => c.kind === kind);
|
|
34
|
+
const deterministicResults = [
|
|
35
|
+
...runDeterministicChecks(of("deterministic"), events),
|
|
36
|
+
...runIfEditThenTestChecks(of("ifEditThenTest"), events),
|
|
37
|
+
...runGitBranchPolicyChecks(of("gitBranchPolicy"), events),
|
|
38
|
+
...runCodeContentChecks(of("codeContent"), events),
|
|
39
|
+
...runFileLifecycleChecks(of("fileLifecycle"), events),
|
|
40
|
+
...runClaimEvidenceChecks(of("claimEvidence"), events),
|
|
41
|
+
];
|
|
42
|
+
const judgment = of("judgment");
|
|
43
|
+
const judgmentResults = llm
|
|
44
|
+
? await runJudgmentChecks(judgment, events)
|
|
45
|
+
: judgment.map(({ rule }) => needsLlmResult(rule));
|
|
46
|
+
return {
|
|
47
|
+
results: [...deterministicResults, ...judgmentResults],
|
|
48
|
+
notARule: of("notARule"),
|
|
49
|
+
stale: staleOverrides(overrides, rules),
|
|
50
|
+
};
|
|
51
|
+
}
|
package/dist/hook.d.ts
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import type { CheckResult, Rule } from "./types.js";
|
|
2
|
+
/**
|
|
3
|
+
* A Stop hook that refuses to let a session end on a broken rule.
|
|
4
|
+
*
|
|
5
|
+
* The gap this closes was named by a reader of anthropics/claude-code#90542
|
|
6
|
+
* on 2026-09-14, from a lab that measured it: a receiving agent auto-ACKed
|
|
7
|
+
* 107 dispatched orders and did none of them, and 69 of 664 records marked
|
|
8
|
+
* complete were plans rather than completions. Their conclusion — "rules
|
|
9
|
+
* that live only in context are advisory by construction; rules that live
|
|
10
|
+
* in a gate are not" — applies to this tool as it stood, which read the
|
|
11
|
+
* transcript afterwards and told you what had already happened.
|
|
12
|
+
*
|
|
13
|
+
* Three safety properties, in the order they matter:
|
|
14
|
+
*
|
|
15
|
+
* 1. FAIL only. Never UNCLEAR, never NOT_RUN, never a judgment rule, and
|
|
16
|
+
* never the LLM path — a model opinion is an opinion, and blocking on
|
|
17
|
+
* one would let a wrong guess hold a session hostage. This gate fires
|
|
18
|
+
* only on the deterministic checkers, where the finding is a matched
|
|
19
|
+
* literal and can be read back.
|
|
20
|
+
*
|
|
21
|
+
* 2. It cannot loop. Claude Code sets `stop_hook_active` when the session
|
|
22
|
+
* is already continuing because of a previous block. Blocking again
|
|
23
|
+
* there is how a Stop hook wedges a session permanently, so this returns
|
|
24
|
+
* immediately in that case: it gets exactly one interruption per stop.
|
|
25
|
+
*
|
|
26
|
+
* 3. It fails OPEN. Any error — unreadable transcript, no rules file,
|
|
27
|
+
* malformed payload — allows the stop and writes a line to stderr.
|
|
28
|
+
* Failing closed is the right default for money-touching code; here it
|
|
29
|
+
* would mean a bug in this tool leaves someone unable to end a session
|
|
30
|
+
* in their own editor, and they would rip the hook out that day. The
|
|
31
|
+
* cost of the two failures is not symmetric, so the default is not
|
|
32
|
+
* symmetric either. It is a deliberate weakening, and it is the reason
|
|
33
|
+
* `check` in CI stays the backstop rather than this.
|
|
34
|
+
*/
|
|
35
|
+
export declare function runHook(needsLlmResult: (rule: Rule) => CheckResult): Promise<void>;
|
package/dist/hook.js
ADDED
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
import { loadRules } from "./rules.js";
|
|
2
|
+
import { readTranscriptFromFile } from "./parsers/transcriptParser.js";
|
|
3
|
+
import { evaluateSession } from "./evaluate.js";
|
|
4
|
+
function readStdin() {
|
|
5
|
+
return new Promise((resolve) => {
|
|
6
|
+
let data = "";
|
|
7
|
+
if (process.stdin.isTTY)
|
|
8
|
+
return resolve("");
|
|
9
|
+
process.stdin.setEncoding("utf8");
|
|
10
|
+
process.stdin.on("data", (c) => (data += c));
|
|
11
|
+
process.stdin.on("end", () => resolve(data));
|
|
12
|
+
process.stdin.on("error", () => resolve(""));
|
|
13
|
+
});
|
|
14
|
+
}
|
|
15
|
+
/**
|
|
16
|
+
* What Claude is told when it tries to finish on a broken rule.
|
|
17
|
+
*
|
|
18
|
+
* Addressed to the model, not to the person: it is the model that has to
|
|
19
|
+
* act on this, and a message written for a human reader ("see the report
|
|
20
|
+
* above") gives it nothing to do. Names the rule, states what was found,
|
|
21
|
+
* and stops — no instruction to "fix it", because the rule already said
|
|
22
|
+
* what to do and repeating it invites the model to argue with the wording
|
|
23
|
+
* instead of going and running the thing.
|
|
24
|
+
*/
|
|
25
|
+
function blockReason(failures) {
|
|
26
|
+
const lines = failures.map((f) => ` • Rule ${f.ruleId} — ${f.ruleTitle}\n ${f.evidence}`);
|
|
27
|
+
const n = failures.length;
|
|
28
|
+
return (`RuleReceipt: ${n} rule${n === 1 ? "" : "s"} in CLAUDE.md ${n === 1 ? "was" : "were"} not followed in this session.\n\n` +
|
|
29
|
+
lines.join("\n\n") +
|
|
30
|
+
`\n\nDo not report this work as finished until the above is resolved or you have said plainly, to the user, that it is still open and why.`);
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* A Stop hook that refuses to let a session end on a broken rule.
|
|
34
|
+
*
|
|
35
|
+
* The gap this closes was named by a reader of anthropics/claude-code#90542
|
|
36
|
+
* on 2026-09-14, from a lab that measured it: a receiving agent auto-ACKed
|
|
37
|
+
* 107 dispatched orders and did none of them, and 69 of 664 records marked
|
|
38
|
+
* complete were plans rather than completions. Their conclusion — "rules
|
|
39
|
+
* that live only in context are advisory by construction; rules that live
|
|
40
|
+
* in a gate are not" — applies to this tool as it stood, which read the
|
|
41
|
+
* transcript afterwards and told you what had already happened.
|
|
42
|
+
*
|
|
43
|
+
* Three safety properties, in the order they matter:
|
|
44
|
+
*
|
|
45
|
+
* 1. FAIL only. Never UNCLEAR, never NOT_RUN, never a judgment rule, and
|
|
46
|
+
* never the LLM path — a model opinion is an opinion, and blocking on
|
|
47
|
+
* one would let a wrong guess hold a session hostage. This gate fires
|
|
48
|
+
* only on the deterministic checkers, where the finding is a matched
|
|
49
|
+
* literal and can be read back.
|
|
50
|
+
*
|
|
51
|
+
* 2. It cannot loop. Claude Code sets `stop_hook_active` when the session
|
|
52
|
+
* is already continuing because of a previous block. Blocking again
|
|
53
|
+
* there is how a Stop hook wedges a session permanently, so this returns
|
|
54
|
+
* immediately in that case: it gets exactly one interruption per stop.
|
|
55
|
+
*
|
|
56
|
+
* 3. It fails OPEN. Any error — unreadable transcript, no rules file,
|
|
57
|
+
* malformed payload — allows the stop and writes a line to stderr.
|
|
58
|
+
* Failing closed is the right default for money-touching code; here it
|
|
59
|
+
* would mean a bug in this tool leaves someone unable to end a session
|
|
60
|
+
* in their own editor, and they would rip the hook out that day. The
|
|
61
|
+
* cost of the two failures is not symmetric, so the default is not
|
|
62
|
+
* symmetric either. It is a deliberate weakening, and it is the reason
|
|
63
|
+
* `check` in CI stays the backstop rather than this.
|
|
64
|
+
*/
|
|
65
|
+
export async function runHook(needsLlmResult) {
|
|
66
|
+
const emit = (out) => {
|
|
67
|
+
process.stdout.write(JSON.stringify(out));
|
|
68
|
+
};
|
|
69
|
+
try {
|
|
70
|
+
const raw = await readStdin();
|
|
71
|
+
const input = raw ? JSON.parse(raw) : {};
|
|
72
|
+
// Property 2: one interruption per stop.
|
|
73
|
+
if (input.stop_hook_active)
|
|
74
|
+
return void emit({});
|
|
75
|
+
const cwd = input.cwd || process.cwd();
|
|
76
|
+
if (!input.transcript_path)
|
|
77
|
+
return void emit({});
|
|
78
|
+
const rules = loadRules(cwd);
|
|
79
|
+
if (rules.length === 0)
|
|
80
|
+
return void emit({});
|
|
81
|
+
const events = readTranscriptFromFile(input.transcript_path);
|
|
82
|
+
// An empty transcript produces no failures, which would read as a pass.
|
|
83
|
+
// Nothing to gate on, so allow — and say nothing, because a hook that
|
|
84
|
+
// warns on every empty read is a hook people mute.
|
|
85
|
+
if (events.length === 0)
|
|
86
|
+
return void emit({});
|
|
87
|
+
// Property 1: deterministic only. `llm: false` is not a default here,
|
|
88
|
+
// it is part of the contract.
|
|
89
|
+
const { results } = await evaluateSession(cwd, rules, events, false, needsLlmResult);
|
|
90
|
+
const failures = results.filter((r) => r.status === "FAIL" && r.outcome !== "not_run");
|
|
91
|
+
if (failures.length === 0)
|
|
92
|
+
return void emit({});
|
|
93
|
+
return void emit({ decision: "block", reason: blockReason(failures) });
|
|
94
|
+
}
|
|
95
|
+
catch (err) {
|
|
96
|
+
// Property 3: fail open, but never silently.
|
|
97
|
+
process.stderr.write(`rulereceipt hook: allowing stop, check did not complete (${err instanceof Error ? err.message : String(err)})\n`);
|
|
98
|
+
return void emit({});
|
|
99
|
+
}
|
|
100
|
+
}
|
|
@@ -154,10 +154,27 @@ const BUCKET_ORDER = [
|
|
|
154
154
|
function sharedEvidence(rs) {
|
|
155
155
|
if (rs.length < 2)
|
|
156
156
|
return null;
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
157
|
+
// The MAJORITY text, not a unanimous one. Requiring every entry to match
|
|
158
|
+
// meant a single rule with its own message reinstated the wall for all the
|
|
159
|
+
// others — shipped in 0.1.35 and visible at once: fourteen judgment rules,
|
|
160
|
+
// twelve repeating the same 300-character paragraph, because two carried
|
|
161
|
+
// claim-evidence text. Whether it happened depended on what was in the
|
|
162
|
+
// transcript that minute, so it passed locally and broke for the reader.
|
|
163
|
+
const counts = new Map();
|
|
164
|
+
for (const r of rs) {
|
|
165
|
+
if (!r.evidence)
|
|
166
|
+
continue;
|
|
167
|
+
counts.set(r.evidence, (counts.get(r.evidence) ?? 0) + 1);
|
|
168
|
+
}
|
|
169
|
+
let best = null;
|
|
170
|
+
let bestN = 1;
|
|
171
|
+
for (const [text, n] of counts) {
|
|
172
|
+
if (n > bestN) {
|
|
173
|
+
best = text;
|
|
174
|
+
bestN = n;
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
return best;
|
|
161
178
|
}
|
|
162
179
|
export function generateReport(results, meta) {
|
|
163
180
|
const clean = results.map(sanitize);
|
|
@@ -177,8 +194,15 @@ export function generateReport(results, meta) {
|
|
|
177
194
|
if (shared) {
|
|
178
195
|
lines.push(` ${shared}`);
|
|
179
196
|
lines.push("");
|
|
180
|
-
for (const r of inBucket)
|
|
197
|
+
for (const r of inBucket) {
|
|
181
198
|
lines.push(` ${ruleLabel(r, clean)}`);
|
|
199
|
+
// An entry that does not share the hoisted text still says its own
|
|
200
|
+
// piece — that difference is the only per-rule information there is.
|
|
201
|
+
if (r.evidence && r.evidence !== shared)
|
|
202
|
+
lines.push(` ${r.evidence}`);
|
|
203
|
+
if (r.ceiling)
|
|
204
|
+
lines.push(` this means: ${r.ceiling}`);
|
|
205
|
+
}
|
|
182
206
|
continue;
|
|
183
207
|
}
|
|
184
208
|
for (const r of inBucket) {
|