rulereceipt 0.1.35 → 0.1.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -88,6 +88,7 @@ rulereceipt rules --include <handle> # "this IS a rule" — check it from now
88
88
  rulereceipt rules --exclude <handle> # "this isn't" — stop reporting it
89
89
  rulereceipt rules --coverage # which rules a configured hook might actually enforce
90
90
  rulereceipt doctor # list hooks/auto-run tasks configured on this machine
91
+ rulereceipt hook # run AS a Claude Code Stop hook — block Claude finishing on a broken rule
91
92
  rulereceipt lint # find contradictions between CLAUDE.md and AGENTS.md
92
93
  rulereceipt digest # summarise recent checks; --email to send it
93
94
  rulereceipt config # set up email sending (stays on your machine)
@@ -99,6 +100,49 @@ rulereceipt verify <session-file> <hash> # spot-check a report you received ag
99
100
 
100
101
  `verify` isn't a routine check — trust your team day to day, same as any status update. It's there for the rare case it actually matters (a dispute, an incident review): give it the session file and the hash printed in the report, and it confirms whether they really match.
101
102
 
103
+ ## Blocking, not just reporting
104
+
105
+ `rulereceipt check` tells you afterwards. `rulereceipt hook` refuses to let the
106
+ session end.
107
+
108
+ Add this to `.claude/settings.json` — you add it, we never do:
109
+
110
+ ```json
111
+ {
112
+ "hooks": {
113
+ "Stop": [
114
+ { "hooks": [ { "type": "command", "command": "npx rulereceipt hook" } ] }
115
+ ]
116
+ }
117
+ }
118
+ ```
119
+
120
+ When Claude tries to finish, it reads the session that just happened. If a rule
121
+ was broken it hands Claude the rule, the evidence, and instructions to keep
122
+ working, so the session cannot end on a claim that isn't backed.
123
+
124
+ The one it is actually for: *"Done — all tests pass"* when the last run of
125
+ `npm test` returned two failures. It checks the claim against what ran, which
126
+ is the part a model cannot talk its way around.
127
+
128
+ Three properties worth knowing before you wire it in:
129
+
130
+ - **It only blocks on things it can prove.** Never a judgment rule, never an
131
+ LLM opinion, never "couldn't tell". Only a matched literal or a claim
132
+ contradicted by a recorded tool result. Run against twelve real sessions it
133
+ blocked none of them.
134
+ - **It cannot loop.** Claude Code sets `stop_hook_active` when a session is
135
+ already continuing because of a block; the hook returns immediately in that
136
+ case. One interruption per stop.
137
+ - **It fails open.** Unreadable transcript, missing rules file, a bug in us —
138
+ it allows the stop and writes a line to stderr. Failing closed would mean our
139
+ bug locks you out of finishing your own session. That is a deliberate
140
+ weakening, and it is why `check` in CI stays the backstop.
141
+
142
+ It runs when Claude stops, so it catches a finished session, not a command
143
+ mid-flight. For that, use a `PreToolUse` hook of your own — `rulereceipt
144
+ doctor` will show you what you already have.
145
+
102
146
  ## Which rules actually have teeth
103
147
 
104
148
  A rule in a file and a rule with a `PreToolUse` hook behind it look identical
@@ -1,4 +1,4 @@
1
- import { TEST_COMMAND } from "./testCommands.js";
1
+ import { TEST_COMMAND, withoutHeredocs, countTestRuns } from "./testCommands.js";
2
2
  /**
3
3
  * Did the session claim something worked, when the log says it didn't?
4
4
  *
@@ -200,7 +200,7 @@ export function runClaimEvidenceChecks(classifications, events) {
200
200
  for (const event of events) {
201
201
  const command = commandOf(event);
202
202
  if (command !== null) {
203
- pendingRun = TEST_COMMAND.test(command) ? command : null;
203
+ pendingRun = TEST_COMMAND.test(withoutHeredocs(command)) ? command : null;
204
204
  for (const action of ACTION_CLAIMS) {
205
205
  if (action.command.test(command))
206
206
  commandsSeen.add(action.label);
@@ -222,10 +222,15 @@ export function runClaimEvidenceChecks(classifications, events) {
222
222
  // words survive a pipe, the exit status does not.
223
223
  const stated = outcomeFromOutput(event.content);
224
224
  const trustExitCode = !PIPED.test(pendingRun);
225
+ // Two suite invocations in one command means neither the exit code
226
+ // nor the printed summary belongs to a single run, so nothing about
227
+ // this command can contradict a claim. Checked before both, because
228
+ // reading the output is what defeated the pipe guard here.
229
+ const oneRun = countTestRuns(pendingRun) <= 1;
225
230
  lastRun = {
226
231
  command: pendingRun,
227
232
  failed: stated !== null ? stated : event.isError,
228
- outcomeReadable: stated !== null || trustExitCode,
233
+ outcomeReadable: oneRun && (stated !== null || trustExitCode),
229
234
  output: event.content.slice(0, 200),
230
235
  };
231
236
  pendingRun = null;
@@ -200,15 +200,67 @@ const PRE_ACTION_GATE = /\b(?:repeat[- ]back|restate\s+what|wait\s+for\s+(?:conf
200
200
  * failures on its own.
201
201
  */
202
202
  const MAX_RULE_BODY_FOR_CLAIM = 1500;
203
+ /**
204
+ * The claim shape: a reporting verb NEXT TO a done-word.
205
+ *
206
+ * "Do not claim tests passed", "before declaring work complete", "before
207
+ * telling user publishing is complete" — the two halves sit within a few
208
+ * words of each other, in one sentence, because that adjacency IS the
209
+ * thing being forbidden: announcing a finish you have not verified.
210
+ *
211
+ * Testing the two words independently across a whole section is what kept
212
+ * mis-routing. A 944-character rule about writing Slack updates contained a
213
+ * reporting verb in one paragraph, an evidence noun in another, and the
214
+ * done-word inside the compound "Slack-ready" — three unrelated matches,
215
+ * one confident verdict about whether a test command was piped. That was
216
+ * patched twice with blocklist entries (PRE_ACTION_GATE) before the
217
+ * independence was recognised as the fault itself.
218
+ *
219
+ * Thirty characters, measured rather than picked: it is the smallest
220
+ * window that keeps every genuine claim rule in the corpus and the widest
221
+ * that admits none of the scattered ones. `[^.\n]` confines the match to a
222
+ * single sentence, which is what stops a match spanning a paragraph break.
223
+ * Across 559 public rules files this narrows claim-evidence from 38 rules
224
+ * to 8, and all 8 read as the same instruction in different words.
225
+ */
226
+ const CLAIM_SHAPE = new RegExp(`(?:${REPORTING_VERB.source})[^.\n]{0,30}?(?:${DONE_WORD.source})` +
227
+ `|(?:${DONE_WORD.source})[^.\n]{0,30}?(?:${REPORTING_VERB.source})`, "i");
203
228
  function isClaimEvidenceRule(rule) {
204
229
  if (rule.text.length > MAX_RULE_BODY_FOR_CLAIM)
205
230
  return false;
206
231
  const text = `${rule.title} ${rule.text}`;
232
+ // Kept as a second line of defence even though CLAIM_SHAPE now excludes
233
+ // every case it was added for. It is cheap, it is tested, and the rules
234
+ // it names are ones this checker must never answer.
207
235
  if (PRE_ACTION_GATE.test(text))
208
236
  return false;
209
- return REPORTING_VERB.test(text) && EVIDENCE_NOUN.test(text) && DONE_WORD.test(text);
237
+ return CLAIM_SHAPE.test(text) && EVIDENCE_NOUN.test(text);
210
238
  }
211
239
  const BRANCH_WORD = /\bbranch\b/i;
240
+ /**
241
+ * A literal that could actually be a git branch name.
242
+ *
243
+ * git's refname rules, reduced to what a CLAUDE.md really writes: no
244
+ * whitespace, no shell punctuation, no `..`, no leading dash, not ending
245
+ * `.lock`. Everything this excludes was being accepted as a branch name and
246
+ * then searched for in the session's git commands.
247
+ *
248
+ * Rejecting a template is the point of the character class rather than an
249
+ * accident of it: `{`, `<` and `$` are how a rules file writes a PATTERN
250
+ * for branch names ("squad/{issue-number}-{kebab-case-slug}"), and a
251
+ * pattern is not a branch. Checking whether the session used a branch
252
+ * literally called that has no meaningful answer.
253
+ */
254
+ const BRANCH_NAME_PATTERN = /^[A-Za-z0-9._/-]{1,100}$/;
255
+ function isBranchName(literal) {
256
+ if (!BRANCH_NAME_PATTERN.test(literal))
257
+ return false;
258
+ if (literal.startsWith("-") || literal.startsWith("/"))
259
+ return false;
260
+ if (literal.includes("..") || literal.endsWith(".lock") || literal.endsWith("/"))
261
+ return false;
262
+ return true;
263
+ }
212
264
  // A function/method-call shape ("print(", "analytics.track(") is a strong,
213
265
  // simple signal that a backtick literal names actual CODE, not a CLI
214
266
  // command or flag ("git push --force", "npm test" never look like this).
@@ -441,11 +493,12 @@ export function classifyRule(rule) {
441
493
  return { kind: "judgment", rule };
442
494
  }
443
495
  const polarityInferred = polarityWasInferred(rule);
444
- if (BRANCH_WORD.test(text)) {
445
- // first backtick literal is treated as the branch name — real rules
446
- // this targets name exactly one branch ("the `demo` branch", "never
447
- // push to `main`"), not a set of them
448
- const [branchName] = patterns;
496
+ // The first literal that could BE a branch, not simply the first literal.
497
+ // Real rules here name exactly one branch ("the `demo` branch", "never
498
+ // push to the `main` branch"), and a rule that mentions branches while
499
+ // naming a command belongs to whichever checker handles that command.
500
+ const branchName = [...patterns].find(isBranchName);
501
+ if (BRANCH_WORD.test(text) && branchName !== undefined) {
449
502
  return { kind: "gitBranchPolicy", rule, branchName, polarity, polarityInferred };
450
503
  }
451
504
  // Route on ANY code-shaped literal, but check ONLY the code-shaped ones.
@@ -19,5 +19,38 @@ import type { TranscriptEvent } from "../types.js";
19
19
  * wrong.
20
20
  */
21
21
  export declare const TEST_COMMAND: RegExp;
22
+ /**
23
+ * Removes heredoc bodies from a shell command.
24
+ *
25
+ * A command that WRITES a test command is not a command that RUNS one.
26
+ * Found 2026-09-14 on a real session: two false failures whose "last test
27
+ * run" was a shell variable assignment. The actual match came from a
28
+ * heredoc further down, writing a demo fixture whose body contains the
29
+ * string `npm test`. The literal was being generated, never executed — and
30
+ * the tool then read its own report output as the failing result.
31
+ *
32
+ * Handles both quoted and bare delimiters, and leaves everything after the
33
+ * closing delimiter intact, because a real test run often follows the
34
+ * heredoc that set the fixture up.
35
+ */
36
+ export declare function withoutHeredocs(command: string): string;
37
+ /**
38
+ * How many times a single shell command invokes a test suite.
39
+ *
40
+ * A command that runs the suite twice has no single outcome to attribute.
41
+ * Real case, 2026-09-14: one command ran a typecheck, then the suite (37
42
+ * passed), then re-ran one file with locking deliberately disabled to
43
+ * demonstrate those tests can fail. The session said "Everything passes:",
44
+ * which was true; the checker read "2 failed" out of the third section and
45
+ * called it a lie.
46
+ *
47
+ * Deliberate red runs cannot be recognised - "is this failure intended" is
48
+ * not in the text. What IS in the text is that the suite ran more than
49
+ * once, and that is enough: the outcome is unattributable for the same
50
+ * reason a pipe makes an exit code unattributable. Heredoc bodies are
51
+ * stripped first, so a command that merely writes a test command twice
52
+ * still counts zero.
53
+ */
54
+ export declare function countTestRuns(command: string): number;
22
55
  /** The first test command run in this session, or null if none ran. */
23
56
  export declare function findTestRun(events: TranscriptEvent[]): string | null;
@@ -18,6 +18,61 @@
18
18
  * wrong.
19
19
  */
20
20
  export const TEST_COMMAND = /\b(?:npm|pnpm|yarn|bun)\s+(?:run\s+)?(?:test|verify|check|ci)\b|\bnpx\s+(?:vitest|jest|mocha|ava)\b|\b(?:vitest|jest|mocha|pytest|phpunit|rspec|tox)\b|\bcargo\s+test\b|\bgo\s+test\b|\bmvn\s+(?:test|verify)\b|\bgradle\s+test\b|\bdotnet\s+test\b|\bpython\s+-m\s+(?:pytest|unittest)\b/i;
21
+ /**
22
+ * Removes heredoc bodies from a shell command.
23
+ *
24
+ * A command that WRITES a test command is not a command that RUNS one.
25
+ * Found 2026-09-14 on a real session: two false failures whose "last test
26
+ * run" was a shell variable assignment. The actual match came from a
27
+ * heredoc further down, writing a demo fixture whose body contains the
28
+ * string `npm test`. The literal was being generated, never executed — and
29
+ * the tool then read its own report output as the failing result.
30
+ *
31
+ * Handles both quoted and bare delimiters, and leaves everything after the
32
+ * closing delimiter intact, because a real test run often follows the
33
+ * heredoc that set the fixture up.
34
+ */
35
+ export function withoutHeredocs(command) {
36
+ const lines = command.split("\n");
37
+ const out = [];
38
+ let closing = null;
39
+ for (const line of lines) {
40
+ if (closing !== null) {
41
+ if (line.trim() === closing)
42
+ closing = null;
43
+ continue;
44
+ }
45
+ const open = line.match(/<<-?\s*(?:'([^']+)'|"([^"]+)"|([A-Za-z_][A-Za-z0-9_]*))/);
46
+ if (open) {
47
+ closing = open[1] ?? open[2] ?? open[3];
48
+ out.push(line.slice(0, open.index));
49
+ continue;
50
+ }
51
+ out.push(line);
52
+ }
53
+ return out.join("\n");
54
+ }
55
+ /**
56
+ * How many times a single shell command invokes a test suite.
57
+ *
58
+ * A command that runs the suite twice has no single outcome to attribute.
59
+ * Real case, 2026-09-14: one command ran a typecheck, then the suite (37
60
+ * passed), then re-ran one file with locking deliberately disabled to
61
+ * demonstrate those tests can fail. The session said "Everything passes:",
62
+ * which was true; the checker read "2 failed" out of the third section and
63
+ * called it a lie.
64
+ *
65
+ * Deliberate red runs cannot be recognised - "is this failure intended" is
66
+ * not in the text. What IS in the text is that the suite ran more than
67
+ * once, and that is enough: the outcome is unattributable for the same
68
+ * reason a pipe makes an exit code unattributable. Heredoc bodies are
69
+ * stripped first, so a command that merely writes a test command twice
70
+ * still counts zero.
71
+ */
72
+ export function countTestRuns(command) {
73
+ const global = new RegExp(TEST_COMMAND.source, "gi");
74
+ return (withoutHeredocs(command).match(global) ?? []).length;
75
+ }
21
76
  /** The first test command run in this session, or null if none ran. */
22
77
  export function findTestRun(events) {
23
78
  for (const event of events) {
@@ -25,7 +80,7 @@ export function findTestRun(events) {
25
80
  continue;
26
81
  const input = event.input;
27
82
  const command = input && typeof input.command === "string" ? input.command : "";
28
- if (command && TEST_COMMAND.test(command))
83
+ if (command && TEST_COMMAND.test(withoutHeredocs(command)))
29
84
  return command;
30
85
  }
31
86
  return null;
package/dist/cli.js CHANGED
@@ -16,6 +16,7 @@ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
16
16
  import { runCodeContentChecks } from "./checks/codeContent.js";
17
17
  import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
18
18
  import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
19
+ import { runHook } from "./hook.js";
19
20
  import { runJudgmentChecks } from "./checks/judgmentChecks.js";
20
21
  import { generateReport, generateMarkdownReport } from "./report/generateReport.js";
21
22
  import { generateHtmlReport } from "./report/generateHtmlReport.js";
@@ -573,6 +574,12 @@ function runDoctorCommand() {
573
574
  console.log(`${result.newSinceLastRun.length} of these are new since the last time doctor ran here.`);
574
575
  }
575
576
  }
577
+ program
578
+ .command("hook")
579
+ .description("run as a Claude Code Stop hook - blocks Claude from finishing on a broken rule (payload on stdin)")
580
+ .action(async () => {
581
+ await runHook(needsLlmResult);
582
+ });
576
583
  program
577
584
  .command("doctor")
578
585
  .description("List every Claude Code hook and VS Code auto-task on this machine/project, flag anything suspicious")
@@ -0,0 +1,22 @@
1
+ import { type Classification } from "./checks/classify.js";
2
+ import { staleOverrides } from "./overrides.js";
3
+ import type { CheckResult, Rule, TranscriptEvent } from "./types.js";
4
+ export interface Evaluation {
5
+ results: CheckResult[];
6
+ notARule: Classification[];
7
+ stale: ReturnType<typeof staleOverrides>;
8
+ }
9
+ /**
10
+ * Rules in, verdicts out — the whole pipeline, with no printing in it.
11
+ *
12
+ * Extracted 2026-09-14 when a second caller appeared. `check` renders a
13
+ * report for a person; `hook` returns a decision to Claude Code. Those two
14
+ * must never be able to disagree about whether a rule was broken, and the
15
+ * only way to guarantee that is one body of code. The same reasoning is
16
+ * written at the top of testCommands.ts, about two checkers sharing one
17
+ * definition; this is that argument one level up.
18
+ *
19
+ * Deliberately takes `events` rather than a path: the caller decides where
20
+ * a transcript comes from, and a hook is handed one it must not second-guess.
21
+ */
22
+ export declare function evaluateSession(cwd: string, rules: Rule[], events: TranscriptEvent[], llm: boolean, needsLlmResult: (rule: Rule) => CheckResult): Promise<Evaluation>;
@@ -0,0 +1,51 @@
1
+ import { classifyRules } from "./checks/classify.js";
2
+ import { runDeterministicChecks } from "./checks/deterministicChecks.js";
3
+ import { runIfEditThenTestChecks } from "./checks/ifEditThenTest.js";
4
+ import { runGitBranchPolicyChecks } from "./checks/gitBranchPolicy.js";
5
+ import { runCodeContentChecks } from "./checks/codeContent.js";
6
+ import { runFileLifecycleChecks } from "./checks/fileLifecycle.js";
7
+ import { runClaimEvidenceChecks } from "./checks/claimEvidence.js";
8
+ import { runJudgmentChecks } from "./checks/judgmentChecks.js";
9
+ import { loadOverrides, ruleFingerprint, staleOverrides } from "./overrides.js";
10
+ /**
11
+ * Rules in, verdicts out — the whole pipeline, with no printing in it.
12
+ *
13
+ * Extracted 2026-09-14 when a second caller appeared. `check` renders a
14
+ * report for a person; `hook` returns a decision to Claude Code. Those two
15
+ * must never be able to disagree about whether a rule was broken, and the
16
+ * only way to guarantee that is one body of code. The same reasoning is
17
+ * written at the top of testCommands.ts, about two checkers sharing one
18
+ * definition; this is that argument one level up.
19
+ *
20
+ * Deliberately takes `events` rather than a path: the caller decides where
21
+ * a transcript comes from, and a hook is handed one it must not second-guess.
22
+ */
23
+ export async function evaluateSession(cwd, rules, events, llm, needsLlmResult) {
24
+ const overrides = loadOverrides(cwd);
25
+ const classifications = classifyRules(rules).map((c) => {
26
+ const decision = overrides.get(ruleFingerprint(c.rule))?.decision;
27
+ if (!decision)
28
+ return c;
29
+ if (decision === "notARule")
30
+ return { kind: "notARule", rule: c.rule };
31
+ return c.kind === "notARule" ? { kind: "judgment", rule: c.rule } : c;
32
+ });
33
+ const of = (kind) => classifications.filter((c) => c.kind === kind);
34
+ const deterministicResults = [
35
+ ...runDeterministicChecks(of("deterministic"), events),
36
+ ...runIfEditThenTestChecks(of("ifEditThenTest"), events),
37
+ ...runGitBranchPolicyChecks(of("gitBranchPolicy"), events),
38
+ ...runCodeContentChecks(of("codeContent"), events),
39
+ ...runFileLifecycleChecks(of("fileLifecycle"), events),
40
+ ...runClaimEvidenceChecks(of("claimEvidence"), events),
41
+ ];
42
+ const judgment = of("judgment");
43
+ const judgmentResults = llm
44
+ ? await runJudgmentChecks(judgment, events)
45
+ : judgment.map(({ rule }) => needsLlmResult(rule));
46
+ return {
47
+ results: [...deterministicResults, ...judgmentResults],
48
+ notARule: of("notARule"),
49
+ stale: staleOverrides(overrides, rules),
50
+ };
51
+ }
package/dist/hook.d.ts ADDED
@@ -0,0 +1,35 @@
1
+ import type { CheckResult, Rule } from "./types.js";
2
+ /**
3
+ * A Stop hook that refuses to let a session end on a broken rule.
4
+ *
5
+ * The gap this closes was named by a reader of anthropics/claude-code#90542
6
+ * on 2026-09-14, from a lab that measured it: a receiving agent auto-ACKed
7
+ * 107 dispatched orders and did none of them, and 69 of 664 records marked
8
+ * complete were plans rather than completions. Their conclusion — "rules
9
+ * that live only in context are advisory by construction; rules that live
10
+ * in a gate are not" — applies to this tool as it stood, which read the
11
+ * transcript afterwards and told you what had already happened.
12
+ *
13
+ * Three safety properties, in the order they matter:
14
+ *
15
+ * 1. FAIL only. Never UNCLEAR, never NOT_RUN, never a judgment rule, and
16
+ * never the LLM path — a model opinion is an opinion, and blocking on
17
+ * one would let a wrong guess hold a session hostage. This gate fires
18
+ * only on the deterministic checkers, where the finding is a matched
19
+ * literal and can be read back.
20
+ *
21
+ * 2. It cannot loop. Claude Code sets `stop_hook_active` when the session
22
+ * is already continuing because of a previous block. Blocking again
23
+ * there is how a Stop hook wedges a session permanently, so this returns
24
+ * immediately in that case: it gets exactly one interruption per stop.
25
+ *
26
+ * 3. It fails OPEN. Any error — unreadable transcript, no rules file,
27
+ * malformed payload — allows the stop and writes a line to stderr.
28
+ * Failing closed is the right default for money-touching code; here it
29
+ * would mean a bug in this tool leaves someone unable to end a session
30
+ * in their own editor, and they would rip the hook out that day. The
31
+ * cost of the two failures is not symmetric, so the default is not
32
+ * symmetric either. It is a deliberate weakening, and it is the reason
33
+ * `check` in CI stays the backstop rather than this.
34
+ */
35
+ export declare function runHook(needsLlmResult: (rule: Rule) => CheckResult): Promise<void>;
package/dist/hook.js ADDED
@@ -0,0 +1,100 @@
1
+ import { loadRules } from "./rules.js";
2
+ import { readTranscriptFromFile } from "./parsers/transcriptParser.js";
3
+ import { evaluateSession } from "./evaluate.js";
4
+ function readStdin() {
5
+ return new Promise((resolve) => {
6
+ let data = "";
7
+ if (process.stdin.isTTY)
8
+ return resolve("");
9
+ process.stdin.setEncoding("utf8");
10
+ process.stdin.on("data", (c) => (data += c));
11
+ process.stdin.on("end", () => resolve(data));
12
+ process.stdin.on("error", () => resolve(""));
13
+ });
14
+ }
15
+ /**
16
+ * What Claude is told when it tries to finish on a broken rule.
17
+ *
18
+ * Addressed to the model, not to the person: it is the model that has to
19
+ * act on this, and a message written for a human reader ("see the report
20
+ * above") gives it nothing to do. Names the rule, states what was found,
21
+ * and stops — no instruction to "fix it", because the rule already said
22
+ * what to do and repeating it invites the model to argue with the wording
23
+ * instead of going and running the thing.
24
+ */
25
+ function blockReason(failures) {
26
+ const lines = failures.map((f) => ` • Rule ${f.ruleId} — ${f.ruleTitle}\n ${f.evidence}`);
27
+ const n = failures.length;
28
+ return (`RuleReceipt: ${n} rule${n === 1 ? "" : "s"} in CLAUDE.md ${n === 1 ? "was" : "were"} not followed in this session.\n\n` +
29
+ lines.join("\n\n") +
30
+ `\n\nDo not report this work as finished until the above is resolved or you have said plainly, to the user, that it is still open and why.`);
31
+ }
32
+ /**
33
+ * A Stop hook that refuses to let a session end on a broken rule.
34
+ *
35
+ * The gap this closes was named by a reader of anthropics/claude-code#90542
36
+ * on 2026-09-14, from a lab that measured it: a receiving agent auto-ACKed
37
+ * 107 dispatched orders and did none of them, and 69 of 664 records marked
38
+ * complete were plans rather than completions. Their conclusion — "rules
39
+ * that live only in context are advisory by construction; rules that live
40
+ * in a gate are not" — applies to this tool as it stood, which read the
41
+ * transcript afterwards and told you what had already happened.
42
+ *
43
+ * Three safety properties, in the order they matter:
44
+ *
45
+ * 1. FAIL only. Never UNCLEAR, never NOT_RUN, never a judgment rule, and
46
+ * never the LLM path — a model opinion is an opinion, and blocking on
47
+ * one would let a wrong guess hold a session hostage. This gate fires
48
+ * only on the deterministic checkers, where the finding is a matched
49
+ * literal and can be read back.
50
+ *
51
+ * 2. It cannot loop. Claude Code sets `stop_hook_active` when the session
52
+ * is already continuing because of a previous block. Blocking again
53
+ * there is how a Stop hook wedges a session permanently, so this returns
54
+ * immediately in that case: it gets exactly one interruption per stop.
55
+ *
56
+ * 3. It fails OPEN. Any error — unreadable transcript, no rules file,
57
+ * malformed payload — allows the stop and writes a line to stderr.
58
+ * Failing closed is the right default for money-touching code; here it
59
+ * would mean a bug in this tool leaves someone unable to end a session
60
+ * in their own editor, and they would rip the hook out that day. The
61
+ * cost of the two failures is not symmetric, so the default is not
62
+ * symmetric either. It is a deliberate weakening, and it is the reason
63
+ * `check` in CI stays the backstop rather than this.
64
+ */
65
+ export async function runHook(needsLlmResult) {
66
+ const emit = (out) => {
67
+ process.stdout.write(JSON.stringify(out));
68
+ };
69
+ try {
70
+ const raw = await readStdin();
71
+ const input = raw ? JSON.parse(raw) : {};
72
+ // Property 2: one interruption per stop.
73
+ if (input.stop_hook_active)
74
+ return void emit({});
75
+ const cwd = input.cwd || process.cwd();
76
+ if (!input.transcript_path)
77
+ return void emit({});
78
+ const rules = loadRules(cwd);
79
+ if (rules.length === 0)
80
+ return void emit({});
81
+ const events = readTranscriptFromFile(input.transcript_path);
82
+ // An empty transcript produces no failures, which would read as a pass.
83
+ // Nothing to gate on, so allow — and say nothing, because a hook that
84
+ // warns on every empty read is a hook people mute.
85
+ if (events.length === 0)
86
+ return void emit({});
87
+ // Property 1: deterministic only. `llm: false` is not a default here,
88
+ // it is part of the contract.
89
+ const { results } = await evaluateSession(cwd, rules, events, false, needsLlmResult);
90
+ const failures = results.filter((r) => r.status === "FAIL" && r.outcome !== "not_run");
91
+ if (failures.length === 0)
92
+ return void emit({});
93
+ return void emit({ decision: "block", reason: blockReason(failures) });
94
+ }
95
+ catch (err) {
96
+ // Property 3: fail open, but never silently.
97
+ process.stderr.write(`rulereceipt hook: allowing stop, check did not complete (${err instanceof Error ? err.message : String(err)})\n`);
98
+ return void emit({});
99
+ }
100
+ }
@@ -154,10 +154,27 @@ const BUCKET_ORDER = [
154
154
  function sharedEvidence(rs) {
155
155
  if (rs.length < 2)
156
156
  return null;
157
- const first = rs[0].evidence;
158
- if (!first)
159
- return null;
160
- return rs.every((r) => r.evidence === first) ? first : null;
157
+ // The MAJORITY text, not a unanimous one. Requiring every entry to match
158
+ // meant a single rule with its own message reinstated the wall for all the
159
+ // others — shipped in 0.1.35 and visible at once: fourteen judgment rules,
160
+ // twelve repeating the same 300-character paragraph, because two carried
161
+ // claim-evidence text. Whether it happened depended on what was in the
162
+ // transcript that minute, so it passed locally and broke for the reader.
163
+ const counts = new Map();
164
+ for (const r of rs) {
165
+ if (!r.evidence)
166
+ continue;
167
+ counts.set(r.evidence, (counts.get(r.evidence) ?? 0) + 1);
168
+ }
169
+ let best = null;
170
+ let bestN = 1;
171
+ for (const [text, n] of counts) {
172
+ if (n > bestN) {
173
+ best = text;
174
+ bestN = n;
175
+ }
176
+ }
177
+ return best;
161
178
  }
162
179
  export function generateReport(results, meta) {
163
180
  const clean = results.map(sanitize);
@@ -177,8 +194,15 @@ export function generateReport(results, meta) {
177
194
  if (shared) {
178
195
  lines.push(` ${shared}`);
179
196
  lines.push("");
180
- for (const r of inBucket)
197
+ for (const r of inBucket) {
181
198
  lines.push(` ${ruleLabel(r, clean)}`);
199
+ // An entry that does not share the hoisted text still says its own
200
+ // piece — that difference is the only per-rule information there is.
201
+ if (r.evidence && r.evidence !== shared)
202
+ lines.push(` ${r.evidence}`);
203
+ if (r.ceiling)
204
+ lines.push(` this means: ${r.ceiling}`);
205
+ }
182
206
  continue;
183
207
  }
184
208
  for (const r of inBucket) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "rulereceipt",
3
- "version": "0.1.35",
3
+ "version": "0.1.36",
4
4
  "description": "Checks whether a Claude Code session actually followed your CLAUDE.md / AGENTS.md rules, with evidence.",
5
5
  "repository": {
6
6
  "type": "git",